From b9850740b965b0c6ff126400f6ac52a0ec577ae1 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 4 Oct 2026 11:48:47 +0200 Subject: [PATCH 1/8] Add regression test for converting json to std::variant (#5595) * Add regression test for converting json to std::variant (#5066) With 3.10.5, get>() was well-formed through the string from_json overload, so the implicit conversion operator was a candidate when converting json to std::variant, and MSVC picked it over the variant's converting constructor. The tightened constraints from #3427 and #3604 (3.11.0) removed that path; this test guards against regressions. Signed-off-by: Niels Lohmann * Fix clang-tidy and clang 6 in the #5066 regression test ci_clang_tidy asked for emplace_back instead of push_back. The push_back is the point of the test: #5066 is about the implicit conversion from json to the vector's value type, which emplace_back would bypass. Silence the check on that line. clang 5 and 6 cannot instantiate std::variant from libstdc++ 10's ("cannot cast private base class"), which broke ci_test_compilers_clang (6). Tested with the CI images: clang 7 to 11 compile and pass, including clang 11 with libstdc++ 10. Skip the runtime check for clang before 7; the static_assert still runs. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- tests/src/unit-regression2.cpp | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index 1ea5ac595..79ca5a770 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -808,6 +808,26 @@ TEST_CASE("regression tests 2") CHECK(j == k); } +#ifdef JSON_HAS_CPP_17 + SECTION("issue #5066 - MSVC converts json to std::variant via the conversion operator") + { + // std::variant must not be retrievable via get<>(), because otherwise the + // implicit conversion operator becomes a candidate that MSVC picks over the variant's + // converting constructor, routing a number through the string from_json overload + static_assert(!nlohmann::detail::is_detected>::value, + "std::variant must not be retrievable via get<>()"); + + // clang before 7 cannot instantiate libstdc++'s std::variant +#if !(defined(__clang__) && __clang_major__ < 7) + // push_back, not emplace_back: #5066 needs the implicit conversion + // from json to the vector's value type + std::vector> v; + v.push_back(json(1)); // NOLINT(hicpp-use-emplace,modernize-use-emplace) + CHECK(std::get<0>(v[0]) == 1); +#endif + } +#endif + SECTION("issue #3669 - invalid use of incomplete type with optional member and to_json") { const Issue3669Holder h{}; From 38a2db260c48efc4887613b412a8c941594988fa Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 4 Oct 2026 11:50:42 +0200 Subject: [PATCH 2/8] Remove dead metaprogramming and duplicated code in traits and pointers (#5728) * Remove unused is_sax and is_detected_convertible detail::is_sax had no user: the parser and the binary reader only use is_sax_static_asserts, so is_sax was a second, unchecked copy of the SAX event list. is_sax_static_asserts asserted boolean(bool) twice in a row, and detail::is_detected_convertible was never used anywhere. Remove all three and include for size_t instead of . Only names in nlohmann::detail are removed; behavior, public API and ABI are unchanged. The diagnostics for an incomplete SAX handler are the same, apart from the duplicated boolean() message. Part of #5708 Signed-off-by: Niels Lohmann * Replace meta/logic.hpp with a disjunction trait meta/logic.hpp added a second set of type-level boolean helpers (cxpr_and, cxpr_or, cxpr_not, ...) next to the existing conjunction and negation in type_traits.hpp. It was used only by one static_assert in from_json_tuple_impl, two of its templates were never used, and it was the only header without the license banner and relied on transitive includes for . Add the missing disjunction next to conjunction and negation, use the three in the static_assert, and delete logic.hpp together with its BUILD.bazel entry. same_sign now uses disjunction as well, which resolves the 2022 TODO waiting for such a trait. The static_assert accepts and rejects the same types as before. Only names in nlohmann::detail change; behavior, public API and ABI are unchanged. Part of #5708 Signed-off-by: Niels Lohmann * Remove unused would_call_std_* from NLOHMANN_CAN_CALL_STD_FUNC_IMPL Besides detail::result_of_begin/end, which is_range and iterator_t use, the macro defined a namespace detail2 with a tag type, a catch-all overload and would_call_std_begin/end, plus would_call_std_begin/end structs directly in namespace nlohmann. Nothing has used them since they were added in #3020. Reduce the macro to its detail part. Without the trailing struct the ';' after the two invocations would be an empty declaration that -Wextra-semi flags, so drop it. macro_scope.hpp included meta/detected.hpp only for this macro; all users of detected.hpp include it (or type_traits.hpp) themselves, so remove the include. Behavior and ABI are unchanged. The undocumented, untested and unused names nlohmann::would_call_std_begin, nlohmann::would_call_std_end and namespace nlohmann::detail2 are no longer declared. Part of #5708 Signed-off-by: Niels Lohmann * Simplify is_ordered_map to reuse has_capacity is_ordered_map re-detected capacity() with a C++03 sizeof/vararg trick right after has_capacity did the same detection through is_detected. For ordered_map, the old trick took the address of std::vector::capacity, which [namespace.std]/6 makes unspecified. Reuse has_capacity instead, which removes the unspecified-behavior pointer-to-std-member and two NOLINT suppressions. Part of #5708 Signed-off-by: Niels Lohmann * Remove duplicate const overload of json_pointer::get_checked The const and non-const get_checked() overloads had byte-identical 50-line bodies, differing only in the signature. The remaining template deduces a const-qualified BasicJsonType for const callers, so at(), the out_of_range::create() calls and the bounds check all still work. Part of #5708 Signed-off-by: Niels Lohmann * Fix tautological clause in iter_impl's iterator category assertion The static_assert meant to check the LegacyBidirectionalIterator named requirement had a first clause comparing std::bidirectional_iterator_tag to itself, which is always true and checks nothing; only array_t::iterator was actually being checked, despite the message claiming object iterators were checked too. Drop the tautological clause, reword the message to describe what is actually checked, and note that object_t may use a forward-only iterator as long as reverse iteration and operator-- are unused. The check is intentionally not extended to object_t::iterator, since that would reject object types with forward-only iterators that compile and work correctly today. Part of #5708 Signed-off-by: Niels Lohmann * Fix misplaced and stale comments in JSON_HAS_RANGES and conversions The JSON_HAS_RANGES feature-detection block had its libc++ comment sitting above the clang+libstdc++ branch it does not describe, leaving the libc++ branch uncommented and the clang+libstdc++ branch without its own rationale. Move each comment to sit under its own branch, and give the clang+libstdc++ branch (added in issue 5161) its own one-line reason referencing that issue instead of reusing the libc++ branch's comment. Also fix a duplicated-word typo ("in large in large cpp files") in from_json.hpp, drop two unanswered 2017 design questions left as comments in type_traits.hpp and from_json.hpp that no longer reflect open questions, and correct NLOHMANN_JSON_SERIALIZE_ENUM_STRICT's @since tag from 3.12.0 to 3.13.0, the release it was actually introduced in. Part of #5708 Signed-off-by: Niels Lohmann * Support any-rank C arrays in from_json, not just rank 1-4 from_json() for C arrays had four hand-unrolled overloads (rank 1-4, added incrementally in #4262), each with its own nested loops. to_json() already handles any rank recursively, so a rank-5+ C array could be serialized but not read back with get_to()/get<>(). Replace the four overloads with one from_json() SFINAE-constrained on get::type>() existing, forwarding to a pair of mutually recursive from_json_c_array_element() helpers: one assigns a non-array element via get(), the other loops over a array element and recurses one dimension at a time. Each dimension still goes through at(), so type_error.304/out_of_range.401 stay unchanged; ranks 1-4 keep their existing behavior and semantics. Adds rank-5 round-trip and mismatched-shape tests to unit-conversions.cpp. Public API: additive only (rank 5+ C arrays become readable). Signed-off-by: Niels Lohmann #5708 item 1 * Move templated_json_throw into nlohmann::detail templated_json_throw() was defined in macro_scope.hpp, which is included outside NLOHMANN_JSON_NAMESPACE_BEGIN, so the helper leaked into the global namespace as ::templated_json_throw with no ABI tag. Unqualified lookup in NLOHMANN_JSON_SERIALIZE_ENUM_STRICT could then bind to a same-named function declared in the user's own namespace instead, which fails to compile with Clang ("does not name a template"). Move the helper next to the exception classes in exceptions.hpp, inside nlohmann::detail, and call it qualified as ::nlohmann::detail::templated_json_throw<...>(...) from both macro expansion sites. Rewrite the doc comment to give the real reason for the helper (JSON_THROW may expand to code that discards its argument, e.g. when exceptions are disabled) and fix the "supress" typo. templated_json_throw was never released (added by #5151 after v3.12.0), so it can be moved freely. Adds a regression test that expands NLOHMANN_JSON_SERIALIZE_ENUM_STRICT inside a namespace declaring its own templated_json_throw. Public API: no change (::templated_json_throw was an unreleased, unintentional global-namespace leak with no callers relying on its location). Overlaps #5698, which rewrites the same two macro call lines; the overlapping hunks are small and should be trivial to reconcile on rebase. Signed-off-by: Niels Lohmann #5708 item 2 * Factor the repeated JSON_HAS_RANGES/MinGW guard into one macro The std::ranges view conversion (excluded on MinGW because of its incomplete C++20 ranges support, #4916) was gated by the same #if JSON_HAS_RANGES && !defined(__MINGW32__) condition at seven independent sites in to_json.hpp and type_traits.hpp, with the MinGW rationale duplicated in two of them and missing from the rest. Since the sites come in matching pairs (one enables is_compatible_range_view and a view-based overload, the other adds the exclusion to the plain-array-type overload), a drift between any pair would produce an ambiguous or missing overload on exactly one platform. Add JSON_HAS_RANGE_VIEW_CONVERSION next to JSON_HAS_RANGES in macro_scope.hpp, combining both conditions with the #4916 reasoning in one place, #undef it in macro_unscope.hpp, and use it at all seven sites. This does not fold the MinGW check into JSON_HAS_RANGES itself: JSON_HAS_RANGES is user-overridable and also gates the enable_borrowed_range specialization in iteration_proxy.hpp, which is not excluded on MinGW. No behavior or public API change: JSON_HAS_RANGE_VIEW_CONVERSION expands to exactly the condition that was previously written out at each site. Overlaps #5585, #5600 and #3575, which touch the same to_json.hpp and type_traits.hpp lines; the change here is a mechanical search-and-replace of the guard condition and should rebase cleanly. Signed-off-by: Niels Lohmann #5708 item 11 * De-duplicate from_json.hpp's map and array-fallback bodies Several from_json() overload pairs in from_json.hpp were copies of each other, so a fix has to be applied twice (as #5681 already does): - from_json(..., std::map&) and from_json(..., std::unordered_map&) for non-string keys had identical 16-line bodies: array check, m.clear(), pair check loop, m.emplace(...). Route both through a new from_json_pair_array_to_map(j, m) helper. - The from_json_array_impl priority_tag<1> and priority_tag<0> fallbacks ran the same std::transform/std::inserter loop, differing only in ret.reserve(j.size()). Merge them into one body and, modeled on the existing from_json_object_reserve, add a from_json_array_reserve pair so the reserve() call is only made for ConstructibleArrayType that support it. Error ids (type_error.302), messages, diagnostic paths ((at(0)/at(1)) and behavior for types with/without reserve() are unchanged; only the duplication is removed. Public API: no change. Overlaps #5681, which changes the "&j" to "&p" line in both map bodies; the shared helper here should make that a one-line change instead of two on rebase. Signed-off-by: Niels Lohmann #5708 item 5 * Unify json_pointer's three array-index parsers array_index(), contains() and get_checked_or_null() each re-implemented the RFC 6901 array-index rules and the size_type range check: array_index() does the canonical parse and throws; contains() (which must not throw, #5395) re-validates every digit by hand and runs its own strtoull/ERANGE check before calling array_index() anyway, parsing every array token twice; get_checked_or_null() wraps array_index() in JSON_TRY/ JSON_INTERNAL_CATCH (detail::out_of_range&) to turn an unrepresentable index into "not found". Add a single private, noexcept parse_array_index(s, idx) returning an array_index_status (ok / leading_zero / not_a_number / unresolved / exceeds_size_type). array_index() becomes a thin wrapper mapping each status to the existing parse_error.106/109 or out_of_range.404/410; contains() and get_checked_or_null() switch on the status directly. This removes contains()'s digit-validation loop and its second strtoull call, and get_checked_or_null()'s JSON_TRY/JSON_INTERNAL_CATCH. Bugfix as a consequence: get_checked_or_null()'s JSON_TRY/ JSON_INTERNAL_CATCH was dead code under JSON_NOEXCEPTION (JSON_TRY expands to "if(true)" and the catch to "if(false)", so JSON_THROW's std::abort() ran unconditionally), meaning value() and contains() would abort instead of returning the default/false for an out-of-range-sized or oversized array index when exceptions are disabled (#5672). Switching on parse_array_index()'s return value instead of relying on an actual throw/catch fixes this: get_checked_or_null() now returns nullptr for array_index_status::unresolved/exceeds_size_type in every build configuration, and still calls JSON_THROW (aborting under JSON_NOEXCEPTION, as before) only for a malformed index (leading_zero/not_a_number), matching its documented @throw list. All existing error ids, messages and diagnostic paths are unchanged; a few reference tokens that used to fail contains()'s manual per-character validation (e.g. "1a") now fail via array_index_status::unresolved instead, with no observable difference since contains() only returns bool. Adds regression tests to unit-element_access2.cpp's "access on array type" section covering value() with an index that exceeds size_type and one with a trailing non-digit, both of which must yield the default value rather than abort/throw. Public API: no change. Overlaps #5700, #5614 and #5692, which touch the contains() and get_checked_or_null() array hunks; this change replaces those hunks with calls into the new shared parser, so a rebase will need to re-apply their token-handling changes (e.g. the empty-token case) on top of the switch statements here. Signed-off-by: Niels Lohmann #5708 item 4 * Regenerate single_include after merging develop The merge commit kept develop's single_include/nlohmann/json.hpp because make amalgamate saw it as up to date. Signed-off-by: Niels Lohmann * Address review: switch in array_index, drop redundant inline - json_pointer::array_index() dispatches on array_index_status with a switch, matching the other parse_array_index() caller - drop `inline` from the function templates this PR adds or moves in from_json.hpp - reword a comment that described the change rather than the code Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- BUILD.bazel | 1 - .../nlohmann/detail/conversions/from_json.hpp | 144 +-- .../nlohmann/detail/conversions/to_json.hpp | 10 +- include/nlohmann/detail/exceptions.hpp | 21 + .../nlohmann/detail/iterators/iter_impl.hpp | 8 +- include/nlohmann/detail/json_pointer.hpp | 252 +++--- include/nlohmann/detail/macro_scope.hpp | 60 +- include/nlohmann/detail/macro_unscope.hpp | 1 + .../nlohmann/detail/meta/call_std/begin.hpp | 2 +- include/nlohmann/detail/meta/call_std/end.hpp | 2 +- include/nlohmann/detail/meta/is_sax.hpp | 35 +- include/nlohmann/detail/meta/logic.hpp | 54 -- include/nlohmann/detail/meta/type_traits.hpp | 35 +- single_include/nlohmann/json.hpp | 819 +++++++----------- tests/src/unit-conversions.cpp | 59 ++ tests/src/unit-type_traits.cpp | 12 + 16 files changed, 597 insertions(+), 918 deletions(-) delete mode 100644 include/nlohmann/detail/meta/logic.hpp diff --git a/BUILD.bazel b/BUILD.bazel index 03f73fa92..a9b09fd79 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -53,7 +53,6 @@ cc_library( "include/nlohmann/detail/meta/detected.hpp", "include/nlohmann/detail/meta/identity_tag.hpp", "include/nlohmann/detail/meta/is_sax.hpp", - "include/nlohmann/detail/meta/logic.hpp", "include/nlohmann/detail/meta/std_fs.hpp", "include/nlohmann/detail/meta/type_traits.hpp", "include/nlohmann/detail/meta/void_t.hpp", diff --git a/include/nlohmann/detail/conversions/from_json.hpp b/include/nlohmann/detail/conversions/from_json.hpp index d5c3328ca..5274b9857 100644 --- a/include/nlohmann/detail/conversions/from_json.hpp +++ b/include/nlohmann/detail/conversions/from_json.hpp @@ -27,7 +27,6 @@ #include #include #include -#include #include #include @@ -211,62 +210,29 @@ inline void from_json(const BasicJsonType& j, std::valarray& l) }); } +// element is not itself a C array: read it directly +template +auto from_json_c_array_element(const BasicJsonType& j, T& e) +-> decltype(e = j.template get(), void()) +{ + e = j.template get(); +} + +// element is itself a C array: recurse one dimension at a time, so any rank is supported template -auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) +void from_json_c_array_element(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { for (std::size_t i = 0; i < N; ++i) { - arr[i] = j.at(i).template get(); + from_json_c_array_element(j.at(i), arr[i]); } } -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) +template +auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) +-> decltype(j.template get::type>(), void()) { - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - arr[i1][i2] = j.at(i1).at(i2).template get(); - } - } -} - -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) -{ - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - for (std::size_t i3 = 0; i3 < N3; ++i3) - { - arr[i1][i2][i3] = j.at(i1).at(i2).at(i3).template get(); - } - } - } -} - -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3][N4]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) -{ - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - for (std::size_t i3 = 0; i3 < N3; ++i3) - { - for (std::size_t i4 = 0; i4 < N4; ++i4) - { - arr[i1][i2][i3][i4] = j.at(i1).at(i2).at(i3).at(i4).template get(); - } - } - } - } + from_json_c_array_element(j, arr); } template @@ -286,20 +252,33 @@ auto from_json_array_impl(const BasicJsonType& j, std::array& arr, } } +// reserve() is called through this pair (modeled on from_json_object_reserve) +// so from_json_array_impl below has a single body for both ConstructibleArrayType +// that support reserve() and those that don't. +template +auto from_json_array_reserve(ConstructibleArrayType& arr, typename ConstructibleArrayType::size_type size, priority_tag<1> /*unused*/) +-> decltype(arr.reserve(size), void()) +{ + arr.reserve(size); +} + +template +void from_json_array_reserve(ConstructibleArrayType& /*arr*/, std::size_t /*size*/, priority_tag<0> /*unused*/) +{} + template::value, int> = 0> auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, priority_tag<1> /*unused*/) -> decltype( - arr.reserve(std::declval()), j.template get(), void()) { using std::end; ConstructibleArrayType ret; - ret.reserve(j.size()); + from_json_array_reserve(ret, j.size(), priority_tag<1> {}); std::transform(j.begin(), j.end(), std::inserter(ret, end(ret)), [](const BasicJsonType & i) { @@ -310,27 +289,6 @@ auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, p arr = std::move(ret); } -template::value, - int> = 0> -inline void from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, - priority_tag<0> /*unused*/) -{ - using std::end; - - ConstructibleArrayType ret; - std::transform( - j.begin(), j.end(), std::inserter(ret, end(ret)), - [](const BasicJsonType & i) - { - // get() returns *this, this won't call a from_json - // method when value_type is BasicJsonType - return i.template get(); - }); - arr = std::move(ret); -} - template < typename BasicJsonType, typename ConstructibleArrayType, enable_if_t < is_constructible_array_type::value&& @@ -433,9 +391,7 @@ inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj) } // overload for arithmetic types, not chosen for basic_json template arguments -// (BooleanType, etc.); note: Is it really necessary to provide explicit -// overloads for boolean_t etc. in case of a custom BooleanType which is not -// an arithmetic type? +// (BooleanType, etc.) template < typename BasicJsonType, typename ArithmeticType, enable_if_t < std::is_arithmetic::value&& @@ -531,7 +487,7 @@ inline void from_json_tuple_impl(const BasicJsonType& j, std::pair& p, p template std::tuple from_json_tuple_impl(const BasicJsonType& j, identity_tag> /*unused*/, priority_tag<2> /*unused*/) { - static_assert(cxpr_and>, is_compatible_reference_type>...>::value, + static_assert(conjunction>, is_compatible_reference_type>...>::value, "Can not return a tuple containing references to types not contained in a Json, try Json::get_to()"); return from_json_tuple_impl_base<1, Args...>(j, index_sequence_for {}); } @@ -554,10 +510,10 @@ auto from_json(const BasicJsonType& j, TupleRelated&& t) return from_json_tuple_impl(j, std::forward(t), priority_tag<3> {}); } -template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator, - typename = enable_if_t < !std::is_constructible < - typename BasicJsonType::string_t, Key >::value >> -inline void from_json(const BasicJsonType& j, std::map& m) +// shared body for std::map/std::unordered_map with a non-string Key: both +// containers are read from an array of [key, value] pairs the same way +template +void from_json_pair_array_to_map(const BasicJsonType& j, MapType& m) { if (JSON_HEDLEY_UNLIKELY(!j.is_array())) { @@ -570,33 +526,29 @@ inline void from_json(const BasicJsonType& j, std::map(), p.at(1).template get()); + m.emplace(p.at(0).template get(), p.at(1).template get()); } } +template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator, + typename = enable_if_t < !std::is_constructible < + typename BasicJsonType::string_t, Key >::value >> +void from_json(const BasicJsonType& j, std::map& m) +{ + from_json_pair_array_to_map(j, m); +} + template < typename BasicJsonType, typename Key, typename Value, typename Hash, typename KeyEqual, typename Allocator, typename = enable_if_t < !std::is_constructible < typename BasicJsonType::string_t, Key >::value >> -inline void from_json(const BasicJsonType& j, std::unordered_map& m) +void from_json(const BasicJsonType& j, std::unordered_map& m) { - if (JSON_HEDLEY_UNLIKELY(!j.is_array())) - { - JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); - } - m.clear(); - for (const auto& p : j) - { - if (JSON_HEDLEY_UNLIKELY(!p.is_array())) - { - JSON_THROW(type_error::create(302, concat("type must be array, but is ", p.type_name()), &p)); - } - m.emplace(p.at(0).template get(), p.at(1).template get()); - } + from_json_pair_array_to_map(j, m); } #if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM -// Workaround for MSVC 19.51 (and possibly later): in large in large cpp files, the compiler may fail to resolve with generic has_from_json (issue #4996) +// Workaround for MSVC 19.51 (and possibly later): in large cpp files, the compiler may fail to resolve with generic has_from_json (issue #4996) template struct has_from_json : std::true_type {}; diff --git a/include/nlohmann/detail/conversions/to_json.hpp b/include/nlohmann/detail/conversions/to_json.hpp index 7b97068f3..2dba6c163 100644 --- a/include/nlohmann/detail/conversions/to_json.hpp +++ b/include/nlohmann/detail/conversions/to_json.hpp @@ -178,7 +178,7 @@ struct external_constructor template < typename BasicJsonType, typename CompatibleArrayType, enable_if_t < !std::is_same::value -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif , int > = 0 > @@ -222,9 +222,7 @@ struct external_constructor j.assert_invariant(); } - // std::ranges does not work properly on MinGW due to incomplete C++20 support - // see https://github.com/nlohmann/json/issues/4916 -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template>::value, int> = 0> static void construct(BasicJsonType& j, CompatibleArrayType && arr) @@ -384,7 +382,7 @@ template < typename BasicJsonType, typename CompatibleArrayType, !std::is_same::value&& !is_compatible_binary_type::value&& !is_basic_json::value -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif , @@ -394,7 +392,7 @@ inline void to_json(BasicJsonType& j, const CompatibleArrayType& arr) external_constructor::construct(j, arr); } -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template < typename BasicJsonType, typename T, enable_if_t < is_compatible_range_view>::value && !is_compatible_string_type>::value diff --git a/include/nlohmann/detail/exceptions.hpp b/include/nlohmann/detail/exceptions.hpp index 808a3b6bf..155f4a4e5 100644 --- a/include/nlohmann/detail/exceptions.hpp +++ b/include/nlohmann/detail/exceptions.hpp @@ -286,6 +286,27 @@ class other_error : public exception other_error(int id_, const char* what_arg) : exception(id_, what_arg) {} }; +/*! +@brief helper function to call JSON_THROW from a template +@note JSON_THROW is a macro that, depending on the JSON_THROW_USER / + JSON_TRY_USER / JSON_NOEXCEPTION configuration, may expand to code + that does not reference its argument (e.g. `std::abort()`), which + would trigger a compilation error if the argument's type depends on + a template parameter that is otherwise unused. Wrapping the call in + a templated function avoids this and gives the compiler a single + place to see the (possibly unused) parameter. +*/ +template +void templated_json_throw(ExceptionType exception) +{ + JSON_THROW(exception); + + // JSON_THROW may expand to code that discards its argument (e.g. when + // exceptions are disabled) - the cast below avoids an unused-parameter + // warning with -Werror in that case + (void)exception; +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/iterators/iter_impl.hpp b/include/nlohmann/detail/iterators/iter_impl.hpp index 2115b6aeb..84fdd07f8 100644 --- a/include/nlohmann/detail/iterators/iter_impl.hpp +++ b/include/nlohmann/detail/iterators/iter_impl.hpp @@ -60,9 +60,11 @@ class iter_impl // NOLINT(cppcoreguidelines-special-member-functions,hicpp-speci static_assert(is_basic_json::type>::value, "iter_impl only accepts (const) basic_json"); // superficial check for the LegacyBidirectionalIterator named requirement - static_assert(std::is_base_of::value - && std::is_base_of::iterator_category>::value, - "basic_json iterator assumes array and object type iterators satisfy the LegacyBidirectionalIterator named requirement."); + // note: only array_t::iterator is checked here; object_t::iterator may be + // a forward-only iterator as long as reverse iteration and operator-- + // are never used on it + static_assert(std::is_base_of::iterator_category>::value, + "basic_json iterator assumes array type iterators satisfy the LegacyBidirectionalIterator named requirement."); public: /// The std::iterator class template (used as a base class to provide typedefs) is deprecated in C++17. diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index a10f4f39f..788af1047 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -240,6 +240,72 @@ class json_pointer } private: + /*! + @brief result of @ref parse_array_index + + @ref array_index maps each value to the corresponding parse_error/out_of_range + exception; @ref contains and @ref get_checked_or_null, which must not throw for + an out-of-range or unrepresentable index, switch on it directly instead. + */ + enum class array_index_status + { + ok, ///< @a s is a valid, representable array index + leading_zero, ///< @a s begins with '0' but has more than one character + not_a_number, ///< @a s does not begin with a digit + unresolved, ///< @a s could not be converted to an integer + exceeds_size_type ///< @a s converts to an integer that exceeds size_type + }; + + /*! + @param[in] s reference token to be converted into an array index + @param[out] idx the integer representation of @a s if @ref array_index_status::ok + is returned; left unchanged otherwise + + @return whether @a s is a valid array index, and if not, why + + @note this function never throws; @ref array_index and the callers that must not + throw (@ref contains, @ref get_checked_or_null) build on it instead of each + re-implementing the RFC 6901 digit rules and the @a size_type range check + */ + template + static array_index_status parse_array_index(const string_t& s, typename BasicJsonType::size_type& idx) noexcept + { + using size_type = typename BasicJsonType::size_type; + + // error condition (cf. RFC 6901, Sect. 4) + if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0')) + { + return array_index_status::leading_zero; + } + + // error condition (cf. RFC 6901, Sect. 4) + if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9'))) + { + return array_index_status::not_a_number; + } + + const char* p = s.data(); + char* p_end = nullptr; // NOLINT(misc-const-correctness) + errno = 0; // strtoull doesn't reset errno + const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) + if (p == p_end // invalid input or empty string + || errno == ERANGE // out of range + || JSON_HEDLEY_UNLIKELY(static_cast(p_end - p) != s.size())) // incomplete read + { + return array_index_status::unresolved; + } + + // the index does not fit into size_type; on 64-bit platforms this is + // only SIZE_MAX itself (see #2203 and #5395) + if (res >= static_cast((std::numeric_limits::max)())) // NOLINT(runtime/int) + { + return array_index_status::exceeds_size_type; + } + + idx = static_cast(res); + return array_index_status::ok; + } + /*! @param[in] s reference token to be converted into an array index @@ -253,39 +319,23 @@ class json_pointer template static typename BasicJsonType::size_type array_index(const string_t& s) { - using size_type = typename BasicJsonType::size_type; - - // error condition (cf. RFC 6901, Sect. 4) - if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0')) + typename BasicJsonType::size_type idx{}; + switch (parse_array_index(s, idx)) { - JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr)); + case array_index_status::leading_zero: + JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr)); + case array_index_status::not_a_number: + JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); + case array_index_status::unresolved: + JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr)); + case array_index_status::exceeds_size_type: + JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); + case array_index_status::ok: + default: + break; } - // error condition (cf. RFC 6901, Sect. 4) - if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9'))) - { - JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); - } - - const char* p = s.data(); - char* p_end = nullptr; // NOLINT(misc-const-correctness) - errno = 0; // strtoull doesn't reset errno - const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) - if (p == p_end // invalid input or empty string - || errno == ERANGE // out of range - || JSON_HEDLEY_UNLIKELY(static_cast(p_end - p) != s.size())) // incomplete read - { - JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr)); - } - - // the index does not fit into size_type; on 64-bit platforms this is - // only SIZE_MAX itself (see #2203 and #5395) - if (res >= static_cast((std::numeric_limits::max)())) // NOLINT(runtime/int) - { - JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); - } - - return static_cast(res); + return idx; } JSON_PRIVATE_UNLESS_TESTED: @@ -590,63 +640,6 @@ class json_pointer return *ptr; } - /*! - @throw parse_error.106 if an array index begins with '0' - @throw parse_error.109 if an array index was not a number - @throw out_of_range.402 if the array index '-' is used - @throw out_of_range.404 if the JSON pointer can not be resolved - */ - template - const BasicJsonType& get_checked(const BasicJsonType* ptr) const - { - for (const auto& reference_token : reference_tokens) - { - switch (ptr->type()) - { - case detail::value_t::object: - { - // note: at performs range check - ptr = &ptr->at(reference_token); - break; - } - - case detail::value_t::array: - { - if (JSON_HEDLEY_UNLIKELY(reference_token == "-")) - { - // "-" always fails the range check - JSON_THROW(detail::out_of_range::create(402, detail::concat( - "array index '-' (", std::to_string(ptr->m_data.m_value.array->size()), - ") is out of range"), ptr)); - } - - const auto idx = array_index(reference_token); - // Bounds check before access to avoid exception with JSON_NOEXCEPTION - if (JSON_HEDLEY_UNLIKELY(idx >= ptr->m_data.m_value.array->size())) - { - JSON_THROW(detail::out_of_range::create(401, detail::concat( - "array index ", std::to_string(idx), " is out of range"), ptr)); - } - ptr = &ptr->operator[](idx); - break; - } - - case detail::value_t::null: - case detail::value_t::string: - case detail::value_t::boolean: - case detail::value_t::number_integer: - case detail::value_t::number_unsigned: - case detail::value_t::number_float: - case detail::value_t::binary: - case detail::value_t::discarded: - default: - JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", reference_token, "'"), ptr)); - } - } - - return *ptr; - } - /*! @brief return a pointer to the pointed to value, or `nullptr` if the pointer cannot be resolved because a key is missing, an array @@ -685,29 +678,24 @@ class json_pointer return nullptr; } - // tokens that array_index() rejects with parse_error.106/109 - // are passed on to it; all other tokens that it would reject - // with out_of_range.404/410 are detected here, so that this - // also works without exceptions - if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1 && !(reference_token[0] >= '1' && reference_token[0] <= '9'))) + // a malformed index throws parse_error.106/109; an + // index that is syntactically valid but cannot be + // represented (out_of_range.404/410) is treated like an + // out-of-range index below + typename BasicJsonType::size_type idx{}; + switch (parse_array_index(reference_token, idx)) { - static_cast(array_index(reference_token)); // throws parse_error.106/109 + case array_index_status::leading_zero: + JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", reference_token, "' must not begin with '0'"), nullptr)); + case array_index_status::not_a_number: + JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", reference_token, "' is not a number"), nullptr)); + case array_index_status::unresolved: + case array_index_status::exceeds_size_type: + return nullptr; + case array_index_status::ok: + default: + break; } - if (JSON_HEDLEY_UNLIKELY(reference_token.empty() || !std::all_of(reference_token.begin(), reference_token.end(), [](const char c) - { - return c >= '0' && c <= '9'; - }))) - { - return nullptr; - } - errno = 0; // strtoull() does not reset errno on success - char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) - if (JSON_HEDLEY_UNLIKELY(errno == ERANGE || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) - { - return nullptr; - } - const auto idx = static_cast(magnitude); if (JSON_HEDLEY_UNLIKELY(idx >= ptr->m_data.m_value.array->size())) { @@ -734,8 +722,8 @@ class json_pointer } /*! - @throw parse_error.106 if an array index begins with '0' - @throw parse_error.109 if an array index was not a number + @note unlike array_index(), this never throws: a malformed or unrepresentable + array index reference token is treated like a missing key (see #5395) */ template bool contains(const BasicJsonType* ptr) const @@ -763,49 +751,17 @@ class json_pointer // "-" always fails the range check return false; } - if (JSON_HEDLEY_UNLIKELY(reference_token.empty())) - { - // an empty reference token is not an array index; array_index() - // would throw out_of_range.404 -- contains() must not throw (see #5395) - return false; - } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9'))) - { - // invalid char - return false; - } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1)) - { - if (JSON_HEDLEY_UNLIKELY(!('1' <= reference_token[0] && reference_token[0] <= '9'))) - { - // the first char should be between '1' and '9' - return false; - } - for (std::size_t i = 1; i < reference_token.size(); i++) - { - if (JSON_HEDLEY_UNLIKELY(!('0' <= reference_token[i] && reference_token[i] <= '9'))) - { - // other char should be between '0' and '9' - return false; - } - } - } - // the reference token consists only of digits at this point (cf. checks - // above); however, its numeric value might not be representable, in which - // case array_index() would throw out_of_range.404/410 -- contains() must - // not throw (see #5395), so such a reference token is treated as "not found" - errno = 0; // strtoull() does not reset errno on success - char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) - if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX - || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) + // any parse failure (malformed index, or one that is syntactically + // valid but not representable as size_type) means the reference + // token cannot denote an existing array element -- contains() must + // not throw (see #5395), so it is treated as "not found" + typename BasicJsonType::size_type idx{}; + if (JSON_HEDLEY_UNLIKELY(parse_array_index(reference_token, idx) != array_index_status::ok)) { - // the array index cannot be represented as size_type return false; } - const auto idx = array_index(reference_token); if (idx >= ptr->size()) { // index out of range diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 3a5f7e606..17118927c 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -9,7 +9,6 @@ #pragma once #include // declval, pair -#include #include // This file contains all internal macro definitions (except those affecting ABI) @@ -140,10 +139,12 @@ // libstdc++ < 11 has incomplete C++20 ranges (issue #4440) #elif defined(_GLIBCXX_RELEASE) && _GLIBCXX_RELEASE < 11 #define JSON_HAS_RANGES 0 - // libc++ < 16 has incomplete C++20 ranges (issue #4440) + // clang < 16 with libstdc++ does not implement the ranges customization + // points libstdc++ declares, so its C++20 ranges support is incomplete (issue #5161) #elif defined(__clang__) && !defined(__apple_build_version__) \ && __clang_major__ < 16 && defined(__GLIBCXX__) #define JSON_HAS_RANGES 0 + // libc++ < 16 has incomplete C++20 ranges (issue #4440) #elif defined(_LIBCPP_VERSION) && _LIBCPP_VERSION < 160000 #define JSON_HAS_RANGES 0 // nvcc CUDA 12.0/12.1 chokes on the enable_borrowed_range variable-template @@ -158,6 +159,18 @@ #endif #endif +// std::ranges view conversion (to_json/is_compatible_array_type_impl) additionally +// needs to be disabled on MinGW, whose std::ranges support is incomplete +// (issue #4916); this macro combines both conditions so the check and its +// reason are not duplicated at every use site. +#ifndef JSON_HAS_RANGE_VIEW_CONVERSION + #if JSON_HAS_RANGES && !defined(__MINGW32__) + #define JSON_HAS_RANGE_VIEW_CONVERSION 1 + #else + #define JSON_HAS_RANGE_VIEW_CONVERSION 0 + #endif +#endif + #ifndef JSON_HAS_STD_FORMAT #if defined(JSON_HAS_CPP_20) && defined(__cpp_lib_format) #define JSON_HAS_STD_FORMAT 1 @@ -279,21 +292,6 @@ -/*! -@brief function to wrap JSON_THROW_MACRO - there can be compilation errors about - there being no arguments to JSON_THROW that depend on template arguments - if this is not used to call JSON_THROW -*/ -template -void templated_json_throw(ExceptionType exception) -{ - JSON_THROW(exception); - - /* JSON_THROW(exception) discards exception and aborts - void cast needed to supress - compilation error if compiled with -Werror and Wunused-parameter */ - (void)exception; -} - /*! @brief macro to briefly define a mapping between an enum and JSON with exception on invalid input @@ -314,7 +312,7 @@ void templated_json_throw(ExceptionType exception) return ej_pair.first == e; \ }); \ if (it != std::end(m)) j = it->second; \ - else templated_json_throw(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE, nullptr)); \ + else ::nlohmann::detail::templated_json_throw(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE, nullptr)); \ } \ template \ inline void from_json(const BasicJsonType& j, ENUM_TYPE& e) \ @@ -329,7 +327,7 @@ void templated_json_throw(ExceptionType exception) return ej_pair.second == j; \ }); \ if (it != std::end(m)) e = it->first; \ - else templated_json_throw(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \ + else ::nlohmann::detail::templated_json_throw(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \ } // Ugly macros to avoid uglier copy-paste when specializing basic_json. They @@ -874,30 +872,6 @@ void templated_json_throw(ExceptionType exception) \ template \ using result_of_##std_name = decltype(std_name(std::declval()...)); \ - } \ - \ - namespace detail2 { \ - struct std_name##_tag \ - { \ - }; \ - \ - template \ - std_name##_tag std_name(T&&...); \ - \ - template \ - using result_of_##std_name = decltype(std_name(std::declval()...)); \ - \ - template \ - struct would_call_std_##std_name \ - { \ - static constexpr auto const value = ::nlohmann::detail:: \ - is_detected_exact::value; \ - }; \ - } /* namespace detail2 */ \ - \ - template \ - struct would_call_std_##std_name : detail2::would_call_std_##std_name \ - { \ } #ifndef JSON_USE_IMPLICIT_CONVERSIONS diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index 1e6e6cce6..8e1d49842 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -35,6 +35,7 @@ #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM #undef JSON_HAS_THREE_WAY_COMPARISON #undef JSON_HAS_RANGES + #undef JSON_HAS_RANGE_VIEW_CONVERSION #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON diff --git a/include/nlohmann/detail/meta/call_std/begin.hpp b/include/nlohmann/detail/meta/call_std/begin.hpp index a086a5f3d..d12bbf7da 100644 --- a/include/nlohmann/detail/meta/call_std/begin.hpp +++ b/include/nlohmann/detail/meta/call_std/begin.hpp @@ -12,6 +12,6 @@ NLOHMANN_JSON_NAMESPACE_BEGIN -NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin); +NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin) NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/meta/call_std/end.hpp b/include/nlohmann/detail/meta/call_std/end.hpp index 40c9942b9..dc1895fc4 100644 --- a/include/nlohmann/detail/meta/call_std/end.hpp +++ b/include/nlohmann/detail/meta/call_std/end.hpp @@ -12,6 +12,6 @@ NLOHMANN_JSON_NAMESPACE_BEGIN -NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end); +NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end) NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/meta/is_sax.hpp b/include/nlohmann/detail/meta/is_sax.hpp index 8e8a0de24..ac238b0c2 100644 --- a/include/nlohmann/detail/meta/is_sax.hpp +++ b/include/nlohmann/detail/meta/is_sax.hpp @@ -8,7 +8,7 @@ #pragma once -#include // size_t +#include // size_t #include // declval #include // string @@ -70,37 +70,6 @@ using parse_error_function_t = decltype(std::declval().parse_error( std::declval(), std::declval(), std::declval())); -template -struct is_sax -{ - private: - static_assert(is_basic_json::value, - "BasicJsonType must be of type basic_json<...>"); - - using number_integer_t = typename BasicJsonType::number_integer_t; - using number_unsigned_t = typename BasicJsonType::number_unsigned_t; - using number_float_t = typename BasicJsonType::number_float_t; - using string_t = typename BasicJsonType::string_t; - using binary_t = typename BasicJsonType::binary_t; - using exception_t = typename BasicJsonType::exception; - - public: - static constexpr bool value = - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value; -}; - template struct is_sax_static_asserts { @@ -120,8 +89,6 @@ struct is_sax_static_asserts "Missing/invalid function: bool null()"); static_assert(is_detected_exact::value, "Missing/invalid function: bool boolean(bool)"); - static_assert(is_detected_exact::value, - "Missing/invalid function: bool boolean(bool)"); static_assert( is_detected_exact::value, diff --git a/include/nlohmann/detail/meta/logic.hpp b/include/nlohmann/detail/meta/logic.hpp deleted file mode 100644 index cb50a5d19..000000000 --- a/include/nlohmann/detail/meta/logic.hpp +++ /dev/null @@ -1,54 +0,0 @@ -#pragma once - -#include - -NLOHMANN_JSON_NAMESPACE_BEGIN -namespace detail -{ -#ifdef JSON_HAS_CPP_17 - -template -struct cxpr_or_impl : std::integral_constant < bool, (Booleans || ...) > {}; - -template -struct cxpr_and_impl : std::integral_constant < bool, (Booleans &&...) > {}; - -#else - -template -struct cxpr_or_impl : std::false_type {}; - -template -struct cxpr_or_impl : std::true_type {}; - -template -struct cxpr_or_impl : cxpr_or_impl {}; - -template -struct cxpr_and_impl : std::true_type {}; - -template -struct cxpr_and_impl : cxpr_and_impl {}; - -template -struct cxpr_and_impl : std::false_type {}; - -#endif - -template -struct cxpr_not : std::integral_constant < bool, !Boolean::value > {}; - -template -struct cxpr_or : cxpr_or_impl {}; - -template -struct cxpr_or_c : cxpr_or_impl {}; - -template -struct cxpr_and : cxpr_and_impl {}; - -template -struct cxpr_and_c : cxpr_and_impl {}; - -} // namespace detail -NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/meta/type_traits.hpp b/include/nlohmann/detail/meta/type_traits.hpp index 14029c0fc..68bf4ad2c 100644 --- a/include/nlohmann/detail/meta/type_traits.hpp +++ b/include/nlohmann/detail/meta/type_traits.hpp @@ -314,6 +314,13 @@ template struct conjunction : std::conditional(B::value), conjunction, B>::type {}; +// https://en.cppreference.com/w/cpp/types/disjunction +template struct disjunction : std::false_type { }; +template struct disjunction : B { }; +template +struct disjunction +: std::conditional(B::value), B, disjunction>::type {}; + // https://en.cppreference.com/w/cpp/types/negation template struct negation : std::integral_constant < bool, !B::value > { }; @@ -508,9 +515,7 @@ template struct is_range_view_optional_type> : std: template struct is_range_view_optional_type : std::false_type {}; #endif -// std::ranges does not work properly on MinGW due to incomplete C++20 support -// see https://github.com/nlohmann/json/issues/4916 -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION // SafeToCheck guards against types that trigger circular constraints when // std::ranges::view is evaluated on GCC 12 / libstdc++ 12: @@ -549,7 +554,7 @@ struct is_compatible_array_type_impl < // filter_view) can match BOTH this iterator-based specialization AND the view-based one // below, causing ambiguity. Exclude views here so the two specializations are mutually // exclusive: this one handles plain iterable containers, the other handles views. -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif >> @@ -559,7 +564,7 @@ struct is_compatible_array_type_impl < range_value_t>::value; }; -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template struct is_compatible_array_type_impl < BasicJsonType, CompatibleArrayType, @@ -635,7 +640,6 @@ struct is_compatible_integer_type_impl < std::is_integral::value&& !std::is_same::value >> { - // is there an assert somewhere on overflows? using RealLimits = std::numeric_limits; using CompatibleLimits = std::numeric_limits; @@ -863,20 +867,7 @@ struct has_capacity : std::integral_constant -struct is_ordered_map -{ - using one = char; - - struct two - { - char x[2]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - }; - - template static one test( decltype(&C::capacity) ) ; - template static two test(...); - - enum { value = sizeof(test(nullptr)) == sizeof(char) }; // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg,cppcoreguidelines-use-enum-class) -}; +struct is_ordered_map : has_capacity {}; // to avoid useless casts (see https://github.com/nlohmann/json/issues/2893#issuecomment-889152324) template < typename T, typename U, enable_if_t < !std::is_same::value, int > = 0 > @@ -900,10 +891,8 @@ using all_signed = conjunction...>; template using all_unsigned = conjunction...>; -// there's a disjunction trait in another PR; replace when merged template -using same_sign = std::integral_constant < bool, - all_signed::value || all_unsigned::value >; +using same_sign = disjunction, all_unsigned>; template using never_out_of_range = std::integral_constant < bool, diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index a91bd9361..f8fc9fdb9 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -276,104 +276,6 @@ #include // declval, pair -// #include -// __ _____ _____ _____ -// __| | __| | | | JSON for Modern C++ -// | | |__ | | | | | | version 3.12.0 -// |_____|_____|_____|_|___| https://github.com/nlohmann/json -// -// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann -// SPDX-License-Identifier: MIT - - - -#include - -// #include -// __ _____ _____ _____ -// __| | __| | | | JSON for Modern C++ -// | | |__ | | | | | | version 3.12.0 -// |_____|_____|_____|_|___| https://github.com/nlohmann/json -// -// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann -// SPDX-License-Identifier: MIT - - - -// #include - - -NLOHMANN_JSON_NAMESPACE_BEGIN -namespace detail -{ - -template struct make_void -{ - using type = void; -}; -template using void_t = typename make_void::type; - -} // namespace detail -NLOHMANN_JSON_NAMESPACE_END - - -NLOHMANN_JSON_NAMESPACE_BEGIN -namespace detail -{ - -// https://en.cppreference.com/w/cpp/experimental/is_detected -struct nonesuch -{ - nonesuch() = delete; - ~nonesuch() = delete; - nonesuch(nonesuch const&) = delete; - nonesuch(nonesuch const&&) = delete; - void operator=(nonesuch const&) = delete; - void operator=(nonesuch&&) = delete; -}; - -template class Op, - class... Args> -struct detector -{ - using value_t = std::false_type; - using type = Default; -}; - -template class Op, class... Args> -struct detector>, Op, Args...> -{ - using value_t = std::true_type; - using type = Op; -}; - -template class Op, class... Args> -using is_detected = typename detector::value_t; - -template class Op, class... Args> -struct is_detected_lazy : is_detected { }; - -template class Op, class... Args> -using detected_t = typename detector::type; - -template class Op, class... Args> -using detected_or = detector; - -template class Op, class... Args> -using detected_or_t = typename detected_or::type; - -template class Op, class... Args> -using is_detected_exact = std::is_same>; - -template class Op, class... Args> -using is_detected_convertible = - std::is_convertible, To>; - -} // namespace detail -NLOHMANN_JSON_NAMESPACE_END - // #include @@ -2552,10 +2454,12 @@ JSON_HEDLEY_DIAGNOSTIC_POP // libstdc++ < 11 has incomplete C++20 ranges (issue #4440) #elif defined(_GLIBCXX_RELEASE) && _GLIBCXX_RELEASE < 11 #define JSON_HAS_RANGES 0 - // libc++ < 16 has incomplete C++20 ranges (issue #4440) + // clang < 16 with libstdc++ does not implement the ranges customization + // points libstdc++ declares, so its C++20 ranges support is incomplete (issue #5161) #elif defined(__clang__) && !defined(__apple_build_version__) \ && __clang_major__ < 16 && defined(__GLIBCXX__) #define JSON_HAS_RANGES 0 + // libc++ < 16 has incomplete C++20 ranges (issue #4440) #elif defined(_LIBCPP_VERSION) && _LIBCPP_VERSION < 160000 #define JSON_HAS_RANGES 0 // nvcc CUDA 12.0/12.1 chokes on the enable_borrowed_range variable-template @@ -2570,6 +2474,18 @@ JSON_HEDLEY_DIAGNOSTIC_POP #endif #endif +// std::ranges view conversion (to_json/is_compatible_array_type_impl) additionally +// needs to be disabled on MinGW, whose std::ranges support is incomplete +// (issue #4916); this macro combines both conditions so the check and its +// reason are not duplicated at every use site. +#ifndef JSON_HAS_RANGE_VIEW_CONVERSION + #if JSON_HAS_RANGES && !defined(__MINGW32__) + #define JSON_HAS_RANGE_VIEW_CONVERSION 1 + #else + #define JSON_HAS_RANGE_VIEW_CONVERSION 0 + #endif +#endif + #ifndef JSON_HAS_STD_FORMAT #if defined(JSON_HAS_CPP_20) && defined(__cpp_lib_format) #define JSON_HAS_STD_FORMAT 1 @@ -2691,21 +2607,6 @@ JSON_HEDLEY_DIAGNOSTIC_POP -/*! -@brief function to wrap JSON_THROW_MACRO - there can be compilation errors about - there being no arguments to JSON_THROW that depend on template arguments - if this is not used to call JSON_THROW -*/ -template -void templated_json_throw(ExceptionType exception) -{ - JSON_THROW(exception); - - /* JSON_THROW(exception) discards exception and aborts - void cast needed to supress - compilation error if compiled with -Werror and Wunused-parameter */ - (void)exception; -} - /*! @brief macro to briefly define a mapping between an enum and JSON with exception on invalid input @@ -2726,7 +2627,7 @@ void templated_json_throw(ExceptionType exception) return ej_pair.first == e; \ }); \ if (it != std::end(m)) j = it->second; \ - else templated_json_throw(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE, nullptr)); \ + else ::nlohmann::detail::templated_json_throw(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE, nullptr)); \ } \ template \ inline void from_json(const BasicJsonType& j, ENUM_TYPE& e) \ @@ -2741,7 +2642,7 @@ void templated_json_throw(ExceptionType exception) return ej_pair.second == j; \ }); \ if (it != std::end(m)) e = it->first; \ - else templated_json_throw(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \ + else ::nlohmann::detail::templated_json_throw(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \ } // Ugly macros to avoid uglier copy-paste when specializing basic_json. They @@ -3286,30 +3187,6 @@ void templated_json_throw(ExceptionType exception) \ template \ using result_of_##std_name = decltype(std_name(std::declval()...)); \ - } \ - \ - namespace detail2 { \ - struct std_name##_tag \ - { \ - }; \ - \ - template \ - std_name##_tag std_name(T&&...); \ - \ - template \ - using result_of_##std_name = decltype(std_name(std::declval()...)); \ - \ - template \ - struct would_call_std_##std_name \ - { \ - static constexpr auto const value = ::nlohmann::detail:: \ - is_detected_exact::value; \ - }; \ - } /* namespace detail2 */ \ - \ - template \ - struct would_call_std_##std_name : detail2::would_call_std_##std_name \ - { \ } #ifndef JSON_USE_IMPLICIT_CONVERSIONS @@ -3851,6 +3728,31 @@ NLOHMANN_JSON_NAMESPACE_END // #include // #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +template struct make_void +{ + using type = void; +}; +template using void_t = typename make_void::type; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END // #include @@ -3922,7 +3824,7 @@ NLOHMANN_JSON_NAMESPACE_END NLOHMANN_JSON_NAMESPACE_BEGIN -NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin); +NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin) NLOHMANN_JSON_NAMESPACE_END @@ -3942,13 +3844,84 @@ NLOHMANN_JSON_NAMESPACE_END NLOHMANN_JSON_NAMESPACE_BEGIN -NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end); +NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end) NLOHMANN_JSON_NAMESPACE_END // #include // #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +// https://en.cppreference.com/w/cpp/experimental/is_detected +struct nonesuch +{ + nonesuch() = delete; + ~nonesuch() = delete; + nonesuch(nonesuch const&) = delete; + nonesuch(nonesuch const&&) = delete; + void operator=(nonesuch const&) = delete; + void operator=(nonesuch&&) = delete; +}; + +template class Op, + class... Args> +struct detector +{ + using value_t = std::false_type; + using type = Default; +}; + +template class Op, class... Args> +struct detector>, Op, Args...> +{ + using value_t = std::true_type; + using type = Op; +}; + +template class Op, class... Args> +using is_detected = typename detector::value_t; + +template class Op, class... Args> +struct is_detected_lazy : is_detected { }; + +template class Op, class... Args> +using detected_t = typename detector::type; + +template class Op, class... Args> +using detected_or = detector; + +template class Op, class... Args> +using detected_or_t = typename detected_or::type; + +template class Op, class... Args> +using is_detected_exact = std::is_same>; + +template class Op, class... Args> +using is_detected_convertible = + std::is_convertible, To>; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END // #include // __ _____ _____ _____ @@ -4314,6 +4287,13 @@ template struct conjunction : std::conditional(B::value), conjunction, B>::type {}; +// https://en.cppreference.com/w/cpp/types/disjunction +template struct disjunction : std::false_type { }; +template struct disjunction : B { }; +template +struct disjunction +: std::conditional(B::value), B, disjunction>::type {}; + // https://en.cppreference.com/w/cpp/types/negation template struct negation : std::integral_constant < bool, !B::value > { }; @@ -4508,9 +4488,7 @@ template struct is_range_view_optional_type> : std: template struct is_range_view_optional_type : std::false_type {}; #endif -// std::ranges does not work properly on MinGW due to incomplete C++20 support -// see https://github.com/nlohmann/json/issues/4916 -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION // SafeToCheck guards against types that trigger circular constraints when // std::ranges::view is evaluated on GCC 12 / libstdc++ 12: @@ -4549,7 +4527,7 @@ struct is_compatible_array_type_impl < // filter_view) can match BOTH this iterator-based specialization AND the view-based one // below, causing ambiguity. Exclude views here so the two specializations are mutually // exclusive: this one handles plain iterable containers, the other handles views. -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif >> @@ -4559,7 +4537,7 @@ struct is_compatible_array_type_impl < range_value_t>::value; }; -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template struct is_compatible_array_type_impl < BasicJsonType, CompatibleArrayType, @@ -4635,7 +4613,6 @@ struct is_compatible_integer_type_impl < std::is_integral::value&& !std::is_same::value >> { - // is there an assert somewhere on overflows? using RealLimits = std::numeric_limits; using CompatibleLimits = std::numeric_limits; @@ -4863,20 +4840,7 @@ struct has_capacity : std::integral_constant -struct is_ordered_map -{ - using one = char; - - struct two - { - char x[2]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - }; - - template static one test( decltype(&C::capacity) ) ; - template static two test(...); - - enum { value = sizeof(test(nullptr)) == sizeof(char) }; // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg,cppcoreguidelines-use-enum-class) -}; +struct is_ordered_map : has_capacity {}; // to avoid useless casts (see https://github.com/nlohmann/json/issues/2893#issuecomment-889152324) template < typename T, typename U, enable_if_t < !std::is_same::value, int > = 0 > @@ -4900,10 +4864,8 @@ using all_signed = conjunction...>; template using all_unsigned = conjunction...>; -// there's a disjunction trait in another PR; replace when merged template -using same_sign = std::integral_constant < bool, - all_signed::value || all_unsigned::value >; +using same_sign = disjunction, all_unsigned>; template using never_out_of_range = std::integral_constant < bool, @@ -5452,6 +5414,27 @@ class other_error : public exception other_error(int id_, const char* what_arg) : exception(id_, what_arg) {} }; +/*! +@brief helper function to call JSON_THROW from a template +@note JSON_THROW is a macro that, depending on the JSON_THROW_USER / + JSON_TRY_USER / JSON_NOEXCEPTION configuration, may expand to code + that does not reference its argument (e.g. `std::abort()`), which + would trigger a compilation error if the argument's type depends on + a template parameter that is otherwise unused. Wrapping the call in + a templated function avoids this and gives the compiler a single + place to see the (possibly unused) parameter. +*/ +template +void templated_json_throw(ExceptionType exception) +{ + JSON_THROW(exception); + + // JSON_THROW may expand to code that discards its argument (e.g. when + // exceptions are disabled) - the cast below avoids an unused-parameter + // warning with -Werror in that case + (void)exception; +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -5521,63 +5504,6 @@ NLOHMANN_JSON_NAMESPACE_END // #include -// #include - - -// #include - - -NLOHMANN_JSON_NAMESPACE_BEGIN -namespace detail -{ -#ifdef JSON_HAS_CPP_17 - -template -struct cxpr_or_impl : std::integral_constant < bool, (Booleans || ...) > {}; - -template -struct cxpr_and_impl : std::integral_constant < bool, (Booleans &&...) > {}; - -#else - -template -struct cxpr_or_impl : std::false_type {}; - -template -struct cxpr_or_impl : std::true_type {}; - -template -struct cxpr_or_impl : cxpr_or_impl {}; - -template -struct cxpr_and_impl : std::true_type {}; - -template -struct cxpr_and_impl : cxpr_and_impl {}; - -template -struct cxpr_and_impl : std::false_type {}; - -#endif - -template -struct cxpr_not : std::integral_constant < bool, !Boolean::value > {}; - -template -struct cxpr_or : cxpr_or_impl {}; - -template -struct cxpr_or_c : cxpr_or_impl {}; - -template -struct cxpr_and : cxpr_and_impl {}; - -template -struct cxpr_and_c : cxpr_and_impl {}; - -} // namespace detail -NLOHMANN_JSON_NAMESPACE_END - // #include // #include @@ -5763,62 +5689,29 @@ inline void from_json(const BasicJsonType& j, std::valarray& l) }); } +// element is not itself a C array: read it directly +template +auto from_json_c_array_element(const BasicJsonType& j, T& e) +-> decltype(e = j.template get(), void()) +{ + e = j.template get(); +} + +// element is itself a C array: recurse one dimension at a time, so any rank is supported template -auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) +void from_json_c_array_element(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { for (std::size_t i = 0; i < N; ++i) { - arr[i] = j.at(i).template get(); + from_json_c_array_element(j.at(i), arr[i]); } } -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) +template +auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) +-> decltype(j.template get::type>(), void()) { - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - arr[i1][i2] = j.at(i1).at(i2).template get(); - } - } -} - -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) -{ - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - for (std::size_t i3 = 0; i3 < N3; ++i3) - { - arr[i1][i2][i3] = j.at(i1).at(i2).at(i3).template get(); - } - } - } -} - -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3][N4]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) -{ - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - for (std::size_t i3 = 0; i3 < N3; ++i3) - { - for (std::size_t i4 = 0; i4 < N4; ++i4) - { - arr[i1][i2][i3][i4] = j.at(i1).at(i2).at(i3).at(i4).template get(); - } - } - } - } + from_json_c_array_element(j, arr); } template @@ -5838,20 +5731,33 @@ auto from_json_array_impl(const BasicJsonType& j, std::array& arr, } } +// reserve() is called through this pair (modeled on from_json_object_reserve) +// so from_json_array_impl below has a single body for both ConstructibleArrayType +// that support reserve() and those that don't. +template +auto from_json_array_reserve(ConstructibleArrayType& arr, typename ConstructibleArrayType::size_type size, priority_tag<1> /*unused*/) +-> decltype(arr.reserve(size), void()) +{ + arr.reserve(size); +} + +template +void from_json_array_reserve(ConstructibleArrayType& /*arr*/, std::size_t /*size*/, priority_tag<0> /*unused*/) +{} + template::value, int> = 0> auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, priority_tag<1> /*unused*/) -> decltype( - arr.reserve(std::declval()), j.template get(), void()) { using std::end; ConstructibleArrayType ret; - ret.reserve(j.size()); + from_json_array_reserve(ret, j.size(), priority_tag<1> {}); std::transform(j.begin(), j.end(), std::inserter(ret, end(ret)), [](const BasicJsonType & i) { @@ -5862,27 +5768,6 @@ auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, p arr = std::move(ret); } -template::value, - int> = 0> -inline void from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, - priority_tag<0> /*unused*/) -{ - using std::end; - - ConstructibleArrayType ret; - std::transform( - j.begin(), j.end(), std::inserter(ret, end(ret)), - [](const BasicJsonType & i) - { - // get() returns *this, this won't call a from_json - // method when value_type is BasicJsonType - return i.template get(); - }); - arr = std::move(ret); -} - template < typename BasicJsonType, typename ConstructibleArrayType, enable_if_t < is_constructible_array_type::value&& @@ -5985,9 +5870,7 @@ inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj) } // overload for arithmetic types, not chosen for basic_json template arguments -// (BooleanType, etc.); note: Is it really necessary to provide explicit -// overloads for boolean_t etc. in case of a custom BooleanType which is not -// an arithmetic type? +// (BooleanType, etc.) template < typename BasicJsonType, typename ArithmeticType, enable_if_t < std::is_arithmetic::value&& @@ -6083,7 +5966,7 @@ inline void from_json_tuple_impl(const BasicJsonType& j, std::pair& p, p template std::tuple from_json_tuple_impl(const BasicJsonType& j, identity_tag> /*unused*/, priority_tag<2> /*unused*/) { - static_assert(cxpr_and>, is_compatible_reference_type>...>::value, + static_assert(conjunction>, is_compatible_reference_type>...>::value, "Can not return a tuple containing references to types not contained in a Json, try Json::get_to()"); return from_json_tuple_impl_base<1, Args...>(j, index_sequence_for {}); } @@ -6106,10 +5989,10 @@ auto from_json(const BasicJsonType& j, TupleRelated&& t) return from_json_tuple_impl(j, std::forward(t), priority_tag<3> {}); } -template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator, - typename = enable_if_t < !std::is_constructible < - typename BasicJsonType::string_t, Key >::value >> -inline void from_json(const BasicJsonType& j, std::map& m) +// shared body for std::map/std::unordered_map with a non-string Key: both +// containers are read from an array of [key, value] pairs the same way +template +void from_json_pair_array_to_map(const BasicJsonType& j, MapType& m) { if (JSON_HEDLEY_UNLIKELY(!j.is_array())) { @@ -6122,33 +6005,29 @@ inline void from_json(const BasicJsonType& j, std::map(), p.at(1).template get()); + m.emplace(p.at(0).template get(), p.at(1).template get()); } } +template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator, + typename = enable_if_t < !std::is_constructible < + typename BasicJsonType::string_t, Key >::value >> +void from_json(const BasicJsonType& j, std::map& m) +{ + from_json_pair_array_to_map(j, m); +} + template < typename BasicJsonType, typename Key, typename Value, typename Hash, typename KeyEqual, typename Allocator, typename = enable_if_t < !std::is_constructible < typename BasicJsonType::string_t, Key >::value >> -inline void from_json(const BasicJsonType& j, std::unordered_map& m) +void from_json(const BasicJsonType& j, std::unordered_map& m) { - if (JSON_HEDLEY_UNLIKELY(!j.is_array())) - { - JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); - } - m.clear(); - for (const auto& p : j) - { - if (JSON_HEDLEY_UNLIKELY(!p.is_array())) - { - JSON_THROW(type_error::create(302, concat("type must be array, but is ", p.type_name()), &p)); - } - m.emplace(p.at(0).template get(), p.at(1).template get()); - } + from_json_pair_array_to_map(j, m); } #if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM -// Workaround for MSVC 19.51 (and possibly later): in large in large cpp files, the compiler may fail to resolve with generic has_from_json (issue #4996) +// Workaround for MSVC 19.51 (and possibly later): in large cpp files, the compiler may fail to resolve with generic has_from_json (issue #4996) template struct has_from_json : std::true_type {}; @@ -6844,7 +6723,7 @@ struct external_constructor template < typename BasicJsonType, typename CompatibleArrayType, enable_if_t < !std::is_same::value -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif , int > = 0 > @@ -6888,9 +6767,7 @@ struct external_constructor j.assert_invariant(); } - // std::ranges does not work properly on MinGW due to incomplete C++20 support - // see https://github.com/nlohmann/json/issues/4916 -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template>::value, int> = 0> static void construct(BasicJsonType& j, CompatibleArrayType && arr) @@ -7050,7 +6927,7 @@ template < typename BasicJsonType, typename CompatibleArrayType, !std::is_same::value&& !is_compatible_binary_type::value&& !is_basic_json::value -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif , @@ -7060,7 +6937,7 @@ inline void to_json(BasicJsonType& j, const CompatibleArrayType& arr) external_constructor::construct(j, arr); } -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template < typename BasicJsonType, typename T, enable_if_t < is_compatible_range_view>::value && !is_compatible_string_type>::value @@ -13449,7 +13326,7 @@ NLOHMANN_JSON_NAMESPACE_END -#include // size_t +#include // size_t #include // declval #include // string @@ -13514,37 +13391,6 @@ using parse_error_function_t = decltype(std::declval().parse_error( std::declval(), std::declval(), std::declval())); -template -struct is_sax -{ - private: - static_assert(is_basic_json::value, - "BasicJsonType must be of type basic_json<...>"); - - using number_integer_t = typename BasicJsonType::number_integer_t; - using number_unsigned_t = typename BasicJsonType::number_unsigned_t; - using number_float_t = typename BasicJsonType::number_float_t; - using string_t = typename BasicJsonType::string_t; - using binary_t = typename BasicJsonType::binary_t; - using exception_t = typename BasicJsonType::exception; - - public: - static constexpr bool value = - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value; -}; - template struct is_sax_static_asserts { @@ -13564,8 +13410,6 @@ struct is_sax_static_asserts "Missing/invalid function: bool null()"); static_assert(is_detected_exact::value, "Missing/invalid function: bool boolean(bool)"); - static_assert(is_detected_exact::value, - "Missing/invalid function: bool boolean(bool)"); static_assert( is_detected_exact::value, @@ -18725,9 +18569,11 @@ class iter_impl // NOLINT(cppcoreguidelines-special-member-functions,hicpp-speci static_assert(is_basic_json::type>::value, "iter_impl only accepts (const) basic_json"); // superficial check for the LegacyBidirectionalIterator named requirement - static_assert(std::is_base_of::value - && std::is_base_of::iterator_category>::value, - "basic_json iterator assumes array and object type iterators satisfy the LegacyBidirectionalIterator named requirement."); + // note: only array_t::iterator is checked here; object_t::iterator may be + // a forward-only iterator as long as reverse iteration and operator-- + // are never used on it + static_assert(std::is_base_of::iterator_category>::value, + "basic_json iterator assumes array type iterators satisfy the LegacyBidirectionalIterator named requirement."); public: /// The std::iterator class template (used as a base class to provide typedefs) is deprecated in C++17. @@ -19869,6 +19715,72 @@ class json_pointer } private: + /*! + @brief result of @ref parse_array_index + + @ref array_index maps each value to the corresponding parse_error/out_of_range + exception; @ref contains and @ref get_checked_or_null, which must not throw for + an out-of-range or unrepresentable index, switch on it directly instead. + */ + enum class array_index_status + { + ok, ///< @a s is a valid, representable array index + leading_zero, ///< @a s begins with '0' but has more than one character + not_a_number, ///< @a s does not begin with a digit + unresolved, ///< @a s could not be converted to an integer + exceeds_size_type ///< @a s converts to an integer that exceeds size_type + }; + + /*! + @param[in] s reference token to be converted into an array index + @param[out] idx the integer representation of @a s if @ref array_index_status::ok + is returned; left unchanged otherwise + + @return whether @a s is a valid array index, and if not, why + + @note this function never throws; @ref array_index and the callers that must not + throw (@ref contains, @ref get_checked_or_null) build on it instead of each + re-implementing the RFC 6901 digit rules and the @a size_type range check + */ + template + static array_index_status parse_array_index(const string_t& s, typename BasicJsonType::size_type& idx) noexcept + { + using size_type = typename BasicJsonType::size_type; + + // error condition (cf. RFC 6901, Sect. 4) + if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0')) + { + return array_index_status::leading_zero; + } + + // error condition (cf. RFC 6901, Sect. 4) + if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9'))) + { + return array_index_status::not_a_number; + } + + const char* p = s.data(); + char* p_end = nullptr; // NOLINT(misc-const-correctness) + errno = 0; // strtoull doesn't reset errno + const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) + if (p == p_end // invalid input or empty string + || errno == ERANGE // out of range + || JSON_HEDLEY_UNLIKELY(static_cast(p_end - p) != s.size())) // incomplete read + { + return array_index_status::unresolved; + } + + // the index does not fit into size_type; on 64-bit platforms this is + // only SIZE_MAX itself (see #2203 and #5395) + if (res >= static_cast((std::numeric_limits::max)())) // NOLINT(runtime/int) + { + return array_index_status::exceeds_size_type; + } + + idx = static_cast(res); + return array_index_status::ok; + } + /*! @param[in] s reference token to be converted into an array index @@ -19882,39 +19794,23 @@ class json_pointer template static typename BasicJsonType::size_type array_index(const string_t& s) { - using size_type = typename BasicJsonType::size_type; - - // error condition (cf. RFC 6901, Sect. 4) - if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0')) + typename BasicJsonType::size_type idx{}; + switch (parse_array_index(s, idx)) { - JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr)); + case array_index_status::leading_zero: + JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr)); + case array_index_status::not_a_number: + JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); + case array_index_status::unresolved: + JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr)); + case array_index_status::exceeds_size_type: + JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); + case array_index_status::ok: + default: + break; } - // error condition (cf. RFC 6901, Sect. 4) - if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9'))) - { - JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); - } - - const char* p = s.data(); - char* p_end = nullptr; // NOLINT(misc-const-correctness) - errno = 0; // strtoull doesn't reset errno - const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) - if (p == p_end // invalid input or empty string - || errno == ERANGE // out of range - || JSON_HEDLEY_UNLIKELY(static_cast(p_end - p) != s.size())) // incomplete read - { - JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr)); - } - - // the index does not fit into size_type; on 64-bit platforms this is - // only SIZE_MAX itself (see #2203 and #5395) - if (res >= static_cast((std::numeric_limits::max)())) // NOLINT(runtime/int) - { - JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); - } - - return static_cast(res); + return idx; } JSON_PRIVATE_UNLESS_TESTED: @@ -20219,63 +20115,6 @@ class json_pointer return *ptr; } - /*! - @throw parse_error.106 if an array index begins with '0' - @throw parse_error.109 if an array index was not a number - @throw out_of_range.402 if the array index '-' is used - @throw out_of_range.404 if the JSON pointer can not be resolved - */ - template - const BasicJsonType& get_checked(const BasicJsonType* ptr) const - { - for (const auto& reference_token : reference_tokens) - { - switch (ptr->type()) - { - case detail::value_t::object: - { - // note: at performs range check - ptr = &ptr->at(reference_token); - break; - } - - case detail::value_t::array: - { - if (JSON_HEDLEY_UNLIKELY(reference_token == "-")) - { - // "-" always fails the range check - JSON_THROW(detail::out_of_range::create(402, detail::concat( - "array index '-' (", std::to_string(ptr->m_data.m_value.array->size()), - ") is out of range"), ptr)); - } - - const auto idx = array_index(reference_token); - // Bounds check before access to avoid exception with JSON_NOEXCEPTION - if (JSON_HEDLEY_UNLIKELY(idx >= ptr->m_data.m_value.array->size())) - { - JSON_THROW(detail::out_of_range::create(401, detail::concat( - "array index ", std::to_string(idx), " is out of range"), ptr)); - } - ptr = &ptr->operator[](idx); - break; - } - - case detail::value_t::null: - case detail::value_t::string: - case detail::value_t::boolean: - case detail::value_t::number_integer: - case detail::value_t::number_unsigned: - case detail::value_t::number_float: - case detail::value_t::binary: - case detail::value_t::discarded: - default: - JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", reference_token, "'"), ptr)); - } - } - - return *ptr; - } - /*! @brief return a pointer to the pointed to value, or `nullptr` if the pointer cannot be resolved because a key is missing, an array @@ -20314,29 +20153,24 @@ class json_pointer return nullptr; } - // tokens that array_index() rejects with parse_error.106/109 - // are passed on to it; all other tokens that it would reject - // with out_of_range.404/410 are detected here, so that this - // also works without exceptions - if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1 && !(reference_token[0] >= '1' && reference_token[0] <= '9'))) + // a malformed index throws parse_error.106/109; an + // index that is syntactically valid but cannot be + // represented (out_of_range.404/410) is treated like an + // out-of-range index below + typename BasicJsonType::size_type idx{}; + switch (parse_array_index(reference_token, idx)) { - static_cast(array_index(reference_token)); // throws parse_error.106/109 + case array_index_status::leading_zero: + JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", reference_token, "' must not begin with '0'"), nullptr)); + case array_index_status::not_a_number: + JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", reference_token, "' is not a number"), nullptr)); + case array_index_status::unresolved: + case array_index_status::exceeds_size_type: + return nullptr; + case array_index_status::ok: + default: + break; } - if (JSON_HEDLEY_UNLIKELY(reference_token.empty() || !std::all_of(reference_token.begin(), reference_token.end(), [](const char c) - { - return c >= '0' && c <= '9'; - }))) - { - return nullptr; - } - errno = 0; // strtoull() does not reset errno on success - char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) - if (JSON_HEDLEY_UNLIKELY(errno == ERANGE || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) - { - return nullptr; - } - const auto idx = static_cast(magnitude); if (JSON_HEDLEY_UNLIKELY(idx >= ptr->m_data.m_value.array->size())) { @@ -20363,8 +20197,8 @@ class json_pointer } /*! - @throw parse_error.106 if an array index begins with '0' - @throw parse_error.109 if an array index was not a number + @note unlike array_index(), this never throws: a malformed or unrepresentable + array index reference token is treated like a missing key (see #5395) */ template bool contains(const BasicJsonType* ptr) const @@ -20392,49 +20226,17 @@ class json_pointer // "-" always fails the range check return false; } - if (JSON_HEDLEY_UNLIKELY(reference_token.empty())) - { - // an empty reference token is not an array index; array_index() - // would throw out_of_range.404 -- contains() must not throw (see #5395) - return false; - } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9'))) - { - // invalid char - return false; - } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1)) - { - if (JSON_HEDLEY_UNLIKELY(!('1' <= reference_token[0] && reference_token[0] <= '9'))) - { - // the first char should be between '1' and '9' - return false; - } - for (std::size_t i = 1; i < reference_token.size(); i++) - { - if (JSON_HEDLEY_UNLIKELY(!('0' <= reference_token[i] && reference_token[i] <= '9'))) - { - // other char should be between '0' and '9' - return false; - } - } - } - // the reference token consists only of digits at this point (cf. checks - // above); however, its numeric value might not be representable, in which - // case array_index() would throw out_of_range.404/410 -- contains() must - // not throw (see #5395), so such a reference token is treated as "not found" - errno = 0; // strtoull() does not reset errno on success - char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) - if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX - || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) + // any parse failure (malformed index, or one that is syntactically + // valid but not representable as size_type) means the reference + // token cannot denote an existing array element -- contains() must + // not throw (see #5395), so it is treated as "not found" + typename BasicJsonType::size_type idx{}; + if (JSON_HEDLEY_UNLIKELY(parse_array_index(reference_token, idx) != array_index_status::ok)) { - // the array index cannot be represented as size_type return false; } - const auto idx = array_index(reference_token); if (idx >= ptr->size()) { // index out of range @@ -34096,6 +33898,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM #undef JSON_HAS_THREE_WAY_COMPARISON #undef JSON_HAS_RANGES + #undef JSON_HAS_RANGE_VIEW_CONVERSION #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON diff --git a/tests/src/unit-conversions.cpp b/tests/src/unit-conversions.cpp index df8685529..5f83a9253 100644 --- a/tests/src/unit-conversions.cpp +++ b/tests/src/unit-conversions.cpp @@ -430,6 +430,37 @@ TEST_CASE("value conversion") CHECK(std::equal(std::begin(nbs[0][0][0]), std::end(nbs[1][1][1]), std::begin(nbs2[0][0][0]))); } + SECTION("built-in arrays: 5D") + { + // NOLINTBEGIN(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + const int nbs[1][1][1][2][2] = {{{{{0, 1}, {2, 3}}}}}; + int nbs2[1][1][1][2][2] = {{{{{0, 0}, {0, 0}}}}}; + // NOLINTEND(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + + const json j2 = nbs; + j2.get_to(nbs2); + CHECK(std::equal(std::begin(nbs[0][0][0][0]), std::end(nbs[0][0][0][1]), std::begin(nbs2[0][0][0][0]))); + } + + SECTION("built-in arrays: mismatched shape") + { + // NOLINTBEGIN(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + int nbs2[2][3] = {{0, 0, 0}, {0, 0, 0}}; + // NOLINTEND(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + + SECTION("not an array") + { + const json j2 = 42; + CHECK_THROWS_WITH_AS(j2.get_to(nbs2), "[json.exception.type_error.304] cannot use at() with number", json::type_error&); + } + + SECTION("too few elements") + { + const json j2 = {{0, 1, 2}}; + CHECK_THROWS_WITH_AS(j2.get_to(nbs2), "[json.exception.out_of_range.401] array index 1 is out of range", json::out_of_range&); + } + } + SECTION("std::deque") { std::deque a{"previous", "value"}; @@ -1748,6 +1779,34 @@ NLOHMANN_JSON_SERIALIZE_ENUM_STRICT(StrictTaskState, {STRICT_TS_COMPLETED, "completed"}, }) +// regression test for #5708 item 2: NLOHMANN_JSON_SERIALIZE_ENUM_STRICT must not rely on +// unqualified lookup of a helper name that a user's own namespace may also declare +namespace ns_with_colliding_name +{ +// NOLINTNEXTLINE(misc-use-internal-linkage) - used to shadow the library's internal helper name +inline void templated_json_throw(int /*unused*/) {} + +enum class colliding_enum { a, b }; + +// NOLINTNEXTLINE(misc-use-internal-linkage,misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - false positive +NLOHMANN_JSON_SERIALIZE_ENUM_STRICT(colliding_enum, +{ + {colliding_enum::a, "a"}, + {colliding_enum::b, "b"} +}) +} // namespace ns_with_colliding_name + +TEST_CASE("NLOHMANN_JSON_SERIALIZE_ENUM_STRICT in a namespace with a colliding name") +{ + using ns_with_colliding_name::colliding_enum; + + CHECK(json(colliding_enum::a) == "a"); + CHECK(colliding_enum::b == json("b")); + + json _; + CHECK_THROWS_WITH_AS(_ = json("nope").get(), "[json.exception.out_of_range.410] enum value out of range for colliding_enum: \"nope\"", json::out_of_range&); +} + TEST_CASE("Strict JSON to enum mapping") { SECTION("enum class") diff --git a/tests/src/unit-type_traits.cpp b/tests/src/unit-type_traits.cpp index 6dc166d03..a3937bf2e 100644 --- a/tests/src/unit-type_traits.cpp +++ b/tests/src/unit-type_traits.cpp @@ -10,10 +10,14 @@ #if JSON_TEST_USING_MULTIPLE_HEADERS #include + #include #else #include #endif +#include +#include + TEST_CASE("type traits") { SECTION("is_c_string") @@ -83,4 +87,12 @@ TEST_CASE("type traits") } } } + + SECTION("is_ordered_map") + { + using nlohmann::detail::is_ordered_map; + + CHECK(is_ordered_map>::value); + CHECK_FALSE(is_ordered_map>::value); + } } From a0b71e272006c400703daeb2f4dc3248e47ddb01 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 4 Oct 2026 11:57:41 +0200 Subject: [PATCH 3/8] Deduplicate basic_json internals; make insert(pos, json&&) move (#5727) * Remove unused private aliases from basic_json The private aliases primitive_iterator_t, internal_iterator and output_adapter_t are not used anywhere: iter_impl, binary_writer and the tests refer to the detail:: names directly. As the aliases are private, no user or derived class can depend on them. The internal_iterator.hpp include stays because iter_impl needs it. Part of #5724 Signed-off-by: Niels Lohmann * Fix meta()'s dead, syntactically invalid HP aCC branch The HP aCC branch of basic_json::meta() was missing a semicolon and has therefore never compiled; adding only a semicolon would also make it throw type_error.305, since it assigned a plain string to result["compiler"] and then indexed into it like the other branches do into an object. Make the branch consistent with the others by assigning an object with "family" and "version" keys, narrow the condition to __HP_aCC (a C compiler cannot build this header-only library), and fix meta.md, which documented the old (impossible) plain-string behavior. Part of #5724 Signed-off-by: Niels Lohmann * Stop the noexcept null constructor delegating to a throwing one basic_json(std::nullptr_t) delegated to basic_json(value_t), whose underlying json_value(value_t) constructor allocates for other types and can therefore throw, which is why the noexcept had a NOLINT(bugprone-exception-escape). The delegated-to constructor also called assert_invariant() a second time. The default member initializers of data already produce the same null state (a value-initialized, i.e. zeroed, union with object == nullptr), so the delegation and its NOLINT can simply be dropped. Part of #5724 Signed-off-by: Niels Lohmann * Remove four cppcheck accessForwarded suppressions in the move constructor basic_json(basic_json&&) built its base subobject with std::forward(other), so cppcheck saw the whole of other as forwarded and flagged every subsequent access to it as accessForwarded, three of them still marked "TODO check". Only the base subobject is actually moved from; cast explicitly to the base type instead, the way ordered_map already does, so cppcheck can tell the two are unrelated. Behavior is unchanged: for a non-reference T, std::forward(x) is defined as static_cast(x). Part of #5724 Signed-off-by: Niels Lohmann * Remove stale cppcheck suppressions and name the local parser in parse() Running the pinned cppcheck (ci_cppcheck's invocation) without --inline-suppr across all configurations reports no syntaxError, no ignoredReturnValue and no assertWithSideEffect, so the corresponding suppressions in json_fwd.hpp, string_concat.hpp and assert_invariant() no longer match anything (json_fwd.hpp's is kept, since downstream users who run an older cppcheck against it could still hit the warning it once silenced). The three basic_json::parse() overloads still trigger a false-positive accessMoved/accessForwarded because they build a temporary parser and call .parse() on it in the same expression; giving that parser a name makes the warning go away without changing behavior, and removes the last of the inline suppressions on these functions. Part of #5724 Signed-off-by: Niels Lohmann * Deduplicate the string/binary cleanup in the two erase() overloads erase(pos) and erase(first, last) each carried a byte-identical 14-line block that destroys and deallocates a string or binary value before resetting the type to null. That reimplements the string/binary cases of json_value::destroy(), so any future change to how those values are freed would have to be made in three places instead of one. Both overloads now just call destroy() and reset the union; for the other primitive types (boolean, numbers) destroy() is a no-op, so behavior is unchanged. Also fix erase(first, last)'s error-path branch hint, which used JSON_HEDLEY_LIKELY where erase(pos), the iterator-range constructor, and every other error path in the class use JSON_HEDLEY_UNLIKELY. This only affects code layout, not semantics. Part of #5724 Signed-off-by: Niels Lohmann * Fix stale and copy-pasted comments in basic_json Several comments no longer match the code: the class invariant and assert_invariant()'s doc still named the members m_value/m_type (now m_data.m_value/m_data.m_type) and did not mention the binary invariant that assert_invariant() already checks; the json_value note and the get() @tparam list omitted binary_t even though binary is a variable-length, pointer-stored type like the others; the key-based value() overload's brief said "via JSON Pointer", which is the other overload; and swap(binary_t&)/swap(binary_t::container_type&) both carried "swap only works for strings", copied from swap(string_t&). Comment-only change; behavior, the public API and the ABI are unchanged. The private get_impl() doxygen and the emplace() comments that border #5585's hunk are intentionally left alone. Part of #5724 Signed-off-by: Niels Lohmann * Deduplicate the 16 copied from_cbor/msgpack/ubjson/bjdata/bon8/bson bodies Each of the 16 binary deserialization overloads (from_cbor, from_msgpack, from_ubjson, from_bjdata, from_bon8, from_bson, each in an InputType&& and an iterator/sentinel version, plus the deprecated span overloads of from_cbor/from_msgpack/from_ubjson/from_bson) had the same body, differing only in the input_format_t value. Every copy built a temporary binary_reader from std::move(ia) and called sax_parse on it in the same expression, which also produced a false-positive cppcheck accessMoved on all 16 lines and needed a NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) on the four span overloads. Add a private from_binary_impl() helper that builds the reader as a named local instead, and make each of the 16 overloads a one-line forward to it. All public signatures, default arguments, JSON_HEDLEY_WARN_UNUSED_RESULT and JSON_HEDLEY_DEPRECATED_FOR attributes are unchanged, tag_handler keeps defaulting to cbor_tag_handler_t::error for the non-CBOR formats (matching binary_reader::sax_parse's own default), and the helper is placed in the existing private section before the binary section banner rather than between the from_* overloads, so from_binary_impl() itself does not collide with #5688's insertion point. Collapsing the from_bjdata/from_bon8 bodies into one-line forwards does rewrite the "return result; }" context lines that #5688 inserts its two deprecated overloads after, so that PR will need a small manual rebase (reinserting its overloads after the new one-line bodies) rather than applying cleanly. Part of #5724 Signed-off-by: Niels Lohmann * Make insert(pos, basic_json&&) move its argument instead of copying it insert(const_iterator pos, basic_json&& val) delegated to insert(pos, val), but val is a named rvalue reference, so inside the function it is an lvalue: the call always resolved to insert(const_iterator, const basic_json&) and deep-copied the value. This has been the case since the overload was introduced, in every release. push_back(basic_json&&), by contrast, already moves. Give the rvalue overload its own body with the same two checks (type_error.309, invalid_iterator.202), then move the argument into a local before inserting it. Moving into a local first, rather than inserting std::move(val) directly, keeps this safe even when val aliases an element of the same array (e.g. arr.insert(arr.begin(), std::move(arr[1]))), since std::vector::insert(pos, T&&) is not guaranteed to handle an argument that aliases one of its own elements. This is a deliberate, small behavior change: the moved-from argument now ends up null afterwards, the same as after push_back(&&), instead of keeping its old value unchanged. No signature changes, so the public API and ABI are unaffected. Add unit-modifiers coverage for the moved-from state and for self-aliasing insertion, both with and without reallocation of the underlying array. Part of #5724 Signed-off-by: Niels Lohmann * Deduplicate object key lookup and checked at() access at()/find()/count()/contains()/const operator[]/erase_internal() each repeated the raw object lookup (m_value.object->find(key)), and the six at() overloads additionally repeated the type_error.304 check and out_of_range.401/403 throw. Route them all through two new private helpers, object_lookup()/object_at() (plus array_at() for the index overloads of at()), templated on the constness of the receiver so one body serves both the const and non-const overload. count() is left untouched, since it already goes through object_t::count() rather than a second find(). The at(KeyType&&) overloads used to forward the same key twice: once into object->find() and again, on the not-found path, into the string_t() conversion for the exception message. object_at() now forwards it only into the lookup and reuses the (unmoved) key for the message. clang-tidy 22 (Docker silkeh/clang:22) still flags that reuse under bugprone-use-after-move/hicpp-invalid-access-moved even with the single forward, since it cannot see that object_t::find() (a plain std::map or ordered_map) never actually moves from its argument; add a NOLINTNEXTLINE with that reasoning rather than avoid the pattern. No signature, exception id/message, or set_parent() behavior changes. Overlaps #5689, #5705, #5606, #5687 and #5585, which touch the same hunks; whichever of this commit and those PRs lands second will need a small rebase. Signed-off-by: Niels Lohmann * Deduplicate the lookup/default/throw body of value() The six non-deprecated value() overloads each held a full copy of the same body: the four key-based overloads looked up the key and either returned the found element converted to the requested type or the default value (throwing type_error.306 if this is not an object), and the two json_pointer overloads did the same via ptr.get_checked_or_null(), throwing type_error.306 unless is_structured(). Replace the duplicated bodies with two private helpers, value_member() and value_pointee(), that return a const basic_json* (null when not found) and do the type check/throw once each. Every value() overload now just picks between the found pointer's get() and the default. Same signatures, template parameters, SFINAE conditions, exception id, message and this context on every overload. Overlaps #5689 (routes find() through lookup_key()) and #5705 (adds a deleted integral-key value() next to these overloads); whichever of this commit and those PRs lands second will need a small rebase. #5724 item 4 Signed-off-by: Niels Lohmann * Deduplicate the null-to-container conversion into convert_null_to() Nine sites wrote out the same "turn a null value into an empty array or object" logic with two different idioms: operator[](size_type), operator[](key_type), operator[](KeyType&&) and update() set m_type then assigned m_value.array/object directly via create(), while the three push_back() overloads, emplace_back() and emplace() set m_type then assigned m_value = value_t::array/object (going through json_value's converting constructor and a temporary). Both idioms end up calling create() and produce the same state, just via a different path; both also share a latent exception-safety bug, since m_type is written before the (possibly throwing) allocation, so a throwing allocator leaves m_type == array/object with a null pointer behind it, violating the class invariant and crashing on the next access to, or destruction of, the value. Add a private convert_null_to(value_t) helper and call it from all nine sites. Unlike the idioms it replaces, it allocates the container first and only then writes m_type, so a throwing allocation leaves the value as a valid null instead of a mistyped, half-constructed one; verified with a throwing allocator (see unit-allocator.cpp's bad_allocator) that j["x"] = ... on a null j now stays null, and no longer trips assert_invariant()/crashes, when create() throws. Same allocator usage and assert_invariant() call as before, otherwise. Overlaps #5585, which reorders these same nine blocks for exception safety; whichever of this commit and that PR lands second will need a small rebase. Signed-off-by: Niels Lohmann * Deduplicate the linear key search in ordered_map emplace, at, erase(key), count and find each repeated the same "for (auto it = begin(); it != end(); ++it) if (m_compare(it->first, key)) ..." loop (15 copies across their key_type and transparent KeyType&& overloads), and both erase(key) overloads additionally repeated the exception-sensitive in-place reconstruction (destroy, placement-new, pop_back) used to remove an element while keeping the const Key non-movable. Add two private helpers: find_impl(Self&, KeyType&&), a static member template that runs the search once for either constness of the receiver, and erase_at(iterator), which keeps the existing pop_back-based reconstruction instead of switching to erase()/resize() (which would add a DefaultInsertable requirement). Route find, at, count, emplace, insert(const value_type&) and both erase(key) overloads through them. Same signatures, is_usable_as_key_type constraints and exception messages/types. Overlaps #5609 and #5685, which both rewrite emplace (#5609 also touches insert and adds private members at the end of the class); whichever of this commit and those PRs lands second will need a rebase. #5724 item 6 Signed-off-by: Niels Lohmann * Re-enable bugprone-use-after-move/hicpp-invalid-access-moved These two checks (and portability-template-virtual-member-function) were disabled in #4489 (November 2024) "only removed to get the CI going". portability-template-virtual-member-function is a separate, still-open cleanup (#5725 item 3 on its own branch) and stays disabled here; this commit only re-enables the move/forward checks and cleans up what they flag on this branch. The move constructor (json.hpp) already casts to the base type instead of forwarding the whole object (#5724 item 9), so it no longer trips either check. The at(KeyType&&) double-forward this check used to flag was reduced to a single forward with the now-unforwarded reuse annotated by a NOLINTNEXTLINE in #5724 item 3's object_at() helper (clang-tidy 22 still flags that reuse even after a single forward; see that commit's message). What is left here: - from_json_inplace_array_impl(), from_json_tuple_impl_base() and the std::pair overload of from_json_tuple_impl() forwarded j into every j.at(...) call in a pack expansion or a pair of calls. at() has no ref-qualified overloads, so the forward was a no-op; call j.at(...) directly. - container_input_adapter_factory::create() forwards container twice on purpose, into begin() and end(), so both see the same value category and produce matching iterator types. Annotate it with NOLINTNEXTLINE and a comment instead of changing it. - unit-class_parser.cpp's "move constructor resets the moved-from value to npos" test still pointed at the pre-static_cast move constructor by line number and mentioned the cppcheck-suppress annotation that #5724 item 9 already removed; update the comment. No behavior change anywhere in include/. Verified with clang-tidy 22.1.8 (Docker silkeh/clang:22, --platform linux/amd64) against a TU including json.hpp with the repo's .clang-tidy: bugprone-use-after-move and hicpp-invalid-access-moved report nothing unsuppressed. Overlaps #5737 (open PR for the rest of #5725 item 3: the from_json.hpp/input_adapters.hpp cleanup above, and portability-template-virtual-member-function), which currently keeps both checks disabled pending this move-constructor change; whichever of this commit and that PR lands second will need a small rebase of .clang-tidy. Signed-off-by: Niels Lohmann * Make convert_null_to() take the container type as a template argument Passing array_t or object_t instead of a value_t makes an invalid target a compile error instead of a runtime assertion, and removes the branch. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .clang-tidy | 9 - docs/mkdocs/docs/api/basic_json/meta.md | 2 +- .../nlohmann/detail/input/input_adapters.hpp | 2 +- include/nlohmann/detail/string_concat.hpp | 1 - include/nlohmann/json.hpp | 644 +++++++----------- include/nlohmann/ordered_map.hpp | 183 ++--- tests/src/unit-class_parser.cpp | 8 +- tests/src/unit-modifiers.cpp | 43 ++ 8 files changed, 366 insertions(+), 526 deletions(-) diff --git a/.clang-tidy b/.clang-tidy index fa3e03ae3..a16136785 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -1,18 +1,9 @@ -# bugprone-use-after-move (hicpp-invalid-access-moved is its alias) still flags -# the basic_json move constructor, which forwards the whole object to its base -# class (#5724), and two forwards in the error-message construction of -# at(KeyType&&) (json.hpp, both overloads: find(std::forward(key)) -# followed by string_t(std::forward(key)) in the throw), which #5689 -# rewrites. Re-enable both checks once those changes have landed. # portability-avoid-pragma-once: kept disabled on purpose. #pragma once is accepted # by every supported compiler, and tools/amalgamate/amalgamate.py strips it from # single_include, so there is nothing left to fix here. Checks: '*, - -bugprone-use-after-move, - -hicpp-invalid-access-moved, - -altera-id-dependent-backward-branch, -altera-struct-pack-align, -altera-unroll-loops, diff --git a/docs/mkdocs/docs/api/basic_json/meta.md b/docs/mkdocs/docs/api/basic_json/meta.md index 55476c5f3..9b1ac2a4e 100644 --- a/docs/mkdocs/docs/api/basic_json/meta.md +++ b/docs/mkdocs/docs/api/basic_json/meta.md @@ -13,7 +13,7 @@ JSON object holding version information | key | description | |-------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| -| `compiler` | Information on the used compiler. It is an object with the following keys: `c++` (the used C++ standard), `family` (the compiler family; possible values are `clang`, `icc`, `gcc`, `ilecpp`, `msvc`, `pgcpp`, `sunpro`, and `unknown`), and `version` (the compiler version). On HP aCC compilers, `compiler` is instead the plain string `hp`. | +| `compiler` | Information on the used compiler. It is an object with the following keys: `c++` (the used C++ standard), `family` (the compiler family; possible values are `clang`, `icc`, `gcc`, `hp`, `ilecpp`, `msvc`, `pgcpp`, `sunpro`, and `unknown`), and `version` (the compiler version). | | `copyright` | The copyright line for the library as string. | | `name` | The name of the library as string. | | `platform` | The used platform as string. Possible values are `win32`, `linux`, `apple`, `unix`, and `unknown`. | diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index 09f0ded14..7f2f79cbf 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -746,7 +746,7 @@ struct container_input_adapter_factory< ContainerType, { // container is forwarded twice on purpose: the resulting begin/end // iterator types must match adapter_type, computed the same way - // NOLINTNEXTLINE(bugprone-use-after-move) + // NOLINTNEXTLINE(bugprone-use-after-move,hicpp-invalid-access-moved) return input_adapter(begin(std::forward(container)), end(std::forward(container))); } }; diff --git a/include/nlohmann/detail/string_concat.hpp b/include/nlohmann/detail/string_concat.hpp index a39dd5e63..74002c878 100644 --- a/include/nlohmann/detail/string_concat.hpp +++ b/include/nlohmann/detail/string_concat.hpp @@ -39,7 +39,6 @@ inline std::size_t concat_length(const char /*c*/, const Args& ... rest) template inline std::size_t concat_length(const char* cstr, const Args& ... rest) { - // cppcheck-suppress ignoredReturnValue return ::strlen(cstr) + concat_length(rest...); } diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 87e1dbc04..89ba13233 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -115,11 +115,12 @@ struct is_std_optional> : std::true_type {}; @brief a class to store JSON values @internal -@invariant The member variables @a m_value and @a m_type have the following -relationship: -- If `m_type == value_t::object`, then `m_value.object != nullptr`. -- If `m_type == value_t::array`, then `m_value.array != nullptr`. -- If `m_type == value_t::string`, then `m_value.string != nullptr`. +@invariant The member variables @a m_data.m_value and @a m_data.m_type have +the following relationship: +- If `m_data.m_type == value_t::object`, then `m_data.m_value.object != nullptr`. +- If `m_data.m_type == value_t::array`, then `m_data.m_value.array != nullptr`. +- If `m_data.m_type == value_t::string`, then `m_data.m_value.string != nullptr`. +- If `m_data.m_type == value_t::binary`, then `m_data.m_value.binary != nullptr`. The invariants are checked by member function assert_invariant(). @note ObjectType trick from https://stackoverflow.com/a/9860911 @@ -182,18 +183,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } private: - using primitive_iterator_t = ::nlohmann::detail::primitive_iterator_t; - template - using internal_iterator = ::nlohmann::detail::internal_iterator; template using iter_impl = ::nlohmann::detail::iter_impl; template using iteration_proxy = ::nlohmann::detail::iteration_proxy; template using json_reverse_iterator = ::nlohmann::detail::json_reverse_iterator; - template - using output_adapter_t = ::nlohmann::detail::output_adapter_t; - template using binary_reader = ::nlohmann::detail::binary_reader; template using binary_writer = ::nlohmann::detail::binary_writer; @@ -334,8 +329,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec std::to_string(__GNUC_PATCHLEVEL__)) } }; -#elif defined(__HP_cc) || defined(__HP_aCC) - result["compiler"] = "hp" +#elif defined(__HP_aCC) + result["compiler"] = {{"family", "hp"}, {"version", __HP_aCC}}; #elif defined(__IBMCPP__) result["compiler"] = {{"family", "ilecpp"}, {"version", __IBMCPP__}}; #elif defined(_MSC_VER) @@ -477,9 +472,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec binary | binary | pointer to @ref binary_t null | null | *no value is stored* - @note Variable-length types (objects, arrays, and strings) are stored as - pointers. The size of the union should not exceed 64 bits if the default - value types are used. + @note Variable-length types (objects, arrays, strings, and binary + values) are stored as pointers. The size of the union should not exceed + 64 bits if the default value types are used. @since version 1.0.0 */ @@ -731,7 +726,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec end of every constructor to make sure that created objects respect the invariant. Furthermore, it has to be called each time the type of a JSON value is changed, because the invariant expresses a relationship between - @a m_type and @a m_value. + @a m_data.m_type and @a m_data.m_value. Furthermore, the parent relation is checked for arrays and objects: If @a check_parents true and the value is an array or object, then the @@ -752,7 +747,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec #if JSON_DIAGNOSTICS JSON_TRY { - // cppcheck-suppress assertWithSideEffect JSON_ASSERT(!check_parents || !is_structured() || std::all_of(begin(), end(), [this](const basic_json & j) { return j.m_parent == this; @@ -1924,8 +1918,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief create a null object /// @sa https://json.nlohmann.me/api/basic_json/basic_json/ - basic_json(std::nullptr_t = nullptr) noexcept // NOLINT(bugprone-exception-escape) - : basic_json(value_t::null) + basic_json(std::nullptr_t = nullptr) noexcept { assert_invariant(); } @@ -2289,15 +2282,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief move constructor /// @sa https://json.nlohmann.me/api/basic_json/basic_json/ basic_json(basic_json&& other) noexcept - : json_base_class_t(std::forward(other)), - m_data(std::move(other.m_data)) // cppcheck-suppress[accessForwarded] TODO check + : json_base_class_t(std::move(static_cast(other))), + m_data(std::move(other.m_data)) #if JSON_DIAGNOSTIC_POSITIONS - , start_position(other.start_position) // cppcheck-suppress[accessForwarded] TODO check - , end_position(other.end_position) // cppcheck-suppress[accessForwarded] TODO check + , start_position(other.start_position) + , end_position(other.end_position) #endif { // check that the passed value is valid - other.assert_invariant(false); // cppcheck-suppress[accessForwarded] + other.assert_invariant(false); // invalidate payload other.m_data.m_type = value_t::null; @@ -2864,7 +2857,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec @tparam PointerType pointer type; must be a pointer to @ref array_t, @ref object_t, @ref string_t, @ref boolean_t, @ref number_integer_t, - @ref number_unsigned_t, or @ref number_float_t. + @ref number_unsigned_t, @ref number_float_t, or @ref binary_t. @return pointer to the internally stored JSON value if the requested pointer type @a PointerType fits to the JSON value; `nullptr` otherwise @@ -3032,6 +3025,90 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @} + private: + /// @brief look up @a key in the object held by @a j, for either constness of @a j + /// @note the single place that performs a (possibly transparent) object key lookup + template + static auto object_lookup(Self& j, KeyType&& key) + -> decltype(j.m_data.m_value.object->find(lookup_key(std::forward(key)))) + { + return j.m_data.m_value.object->find(lookup_key(std::forward(key))); + } + + /// @brief checked object element access used by the at() overloads taking a key + /// @throw type_error.304 if @a j is not an object + /// @throw out_of_range.403 if @a key is not found + template + static auto object_at(Self& j, KeyType&& key) + -> decltype((object_lookup(j, std::forward(key))->second)) + { + // at only works for objects + if (JSON_HEDLEY_UNLIKELY(!j.is_object())) + { + JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", j.type_name()), &j)); + } + + auto it = object_lookup(j, std::forward(key)); + if (it == j.m_data.m_value.object->end()) + { + // key is only forwarded into the lookup above: object_t::find() (a plain + // std::map or ordered_map) never moves from its argument, so key is still + // valid here regardless of whether KeyType was deduced as an rvalue reference + // NOLINTNEXTLINE(bugprone-use-after-move,hicpp-invalid-access-moved) + JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(key), "' not found"), &j)); + } + return it->second; + } + + /// @brief checked array element access used by the at() overloads taking an index + /// @throw type_error.304 if @a j is not an array + /// @throw out_of_range.401 if @a idx is out of range + template + static auto array_at(Self& j, size_type idx) + -> decltype((*j.m_data.m_value.array)[idx]) + { + // at only works for arrays + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", j.type_name()), &j)); + } + + if (JSON_HEDLEY_UNLIKELY(idx >= j.m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), &j)); + } + + return (*j.m_data.m_value.array)[idx]; + } + + /// @brief convert a null value to an empty container of type @a Container + /// @tparam Container array_t or object_t; any other type does not compile + template + void convert_null_to() + { + JSON_ASSERT(is_null()); + // create the container before touching the type, so a throwing + // allocation leaves this value as a valid null rather than a type + // tag with a dangling/null pointer behind it + set_container(create()); + assert_invariant(); + } + + /// @brief store a freshly created array and set the matching type + void set_container(array_t* array) noexcept + { + m_data.m_value.array = array; + m_data.m_type = value_t::array; + } + + /// @brief store a freshly created object and set the matching type + void set_container(object_t* object) noexcept + { + m_data.m_value.object = object; + m_data.m_type = value_t::object; + } + + public: //////////////////// // element access // //////////////////// @@ -3044,54 +3121,21 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/at/ reference at(size_type idx) { - // at only works for arrays - if (JSON_HEDLEY_UNLIKELY(!is_array())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) - { - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } - - return set_parent((*m_data.m_value.array)[idx]); + return set_parent(array_at(*this, idx)); } /// @brief access specified array element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ const_reference at(size_type idx) const { - // at only works for arrays - if (JSON_HEDLEY_UNLIKELY(!is_array())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) - { - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } - - return (*m_data.m_value.array)[idx]; + return array_at(*this, idx); } /// @brief access specified object element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ reference at(const typename object_t::key_type& key) { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(key); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", key, "' not found"), this)); - } - return set_parent(it->second); + return set_parent(object_at(*this, key)); } /// @brief access specified object element with bounds checking @@ -3100,36 +3144,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_usable_as_basic_json_key_type::value, int> = 0> reference at(KeyType && key) { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(std::forward(key)), "' not found"), this)); - } - return set_parent(it->second); + return set_parent(object_at(*this, std::forward(key))); } /// @brief access specified object element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ const_reference at(const typename object_t::key_type& key) const { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(key); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", key, "' not found"), this)); - } - return it->second; + return object_at(*this, key); } /// @brief access specified object element with bounds checking @@ -3138,18 +3160,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_usable_as_basic_json_key_type::value, int> = 0> const_reference at(KeyType && key) const { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(std::forward(key)), "' not found"), this)); - } - return it->second; + return object_at(*this, std::forward(key)); } /// @brief access specified array element @@ -3159,9 +3170,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value.array = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for arrays @@ -3227,9 +3236,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value.object = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for objects @@ -3249,7 +3256,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // const operator[] only works for objects if (JSON_HEDLEY_LIKELY(is_object())) { - auto it = m_data.m_value.object->find(key); + auto it = object_lookup(*this, key); JSON_ASSERT(it != m_data.m_value.object->end()); return it->second; } @@ -3280,9 +3287,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value.object = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for objects @@ -3304,7 +3309,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // const operator[] only works for objects if (JSON_HEDLEY_LIKELY(is_object())) { - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); + auto it = object_lookup(*this, std::forward(key)); JSON_ASSERT(it != m_data.m_value.object->end()); return it->second; } @@ -3324,6 +3329,36 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_c_string_uncvref::value, string_t, typename std::decay::type >; + /// @brief look up @a key for value(), the single place shared by all key-based overloads + /// @throw type_error.306 if this is not an object + /// @return a pointer to the found value, or `nullptr` if @a key was not found + template + const basic_json* value_member(KeyType&& key) const + { + // value only works for objects + if (JSON_HEDLEY_UNLIKELY(!is_object())) + { + JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + } + + const auto it = find(std::forward(key)); + return it != end() ? &*it : nullptr; + } + + /// @brief resolve @a ptr for value(), the single place shared by both json_pointer overloads + /// @throw type_error.306 if this is not an array or object + /// @return a pointer to the resolved value, or `nullptr` if @a ptr does not resolve + const basic_json* value_pointee(const json_pointer& ptr) const + { + // value only works for arrays and objects + if (JSON_HEDLEY_UNLIKELY(!is_structured())) + { + JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + } + + return ptr.get_checked_or_null(this); + } + public: // an integer literal 0 would otherwise convert to a null const char* and from there to key_type template::value, int> = 0> @@ -3337,20 +3372,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(const typename object_t::key_type& key, const ValueType& default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(key); - if (it != end()) - { - return it->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(key); + return found != nullptr ? found->template get() : default_value; } /// @brief access specified object element with default value @@ -3362,20 +3386,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(const typename object_t::key_type& key, ValueType && default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(key); - if (it != end()) - { - return it->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(key); + return found != nullptr ? found->template get() : std::forward(default_value); } /// @brief access specified object element with default value @@ -3388,23 +3401,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(KeyType && key, const ValueType& default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(std::forward(key)); - if (it != end()) - { - return it->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(std::forward(key)); + return found != nullptr ? found->template get() : default_value; } - /// @brief access specified object element via JSON Pointer with default value + /// @brief access specified object element with default value /// @sa https://json.nlohmann.me/api/basic_json/value/ template < class ValueType, class KeyType, class ReturnType = typename value_return_type::type, detail::enable_if_t < @@ -3415,20 +3417,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(KeyType && key, ValueType && default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(std::forward(key)); - if (it != end()) - { - return it->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(std::forward(key)); + return found != nullptr ? found->template get() : std::forward(default_value); } /// @brief access specified object element via JSON Pointer with default value @@ -3438,21 +3429,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(const json_pointer& ptr, const ValueType& default_value) const { - // value only works for arrays and objects - if (JSON_HEDLEY_LIKELY(is_structured())) - { - // If the pointer resolves to a value, return it. Otherwise, return - // 'default_value'. - const auto* res = ptr.get_checked_or_null(this); - if (JSON_HEDLEY_LIKELY(res != nullptr)) - { - return res->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If the pointer resolves to a value, return it. Otherwise, return + // 'default_value'. + const auto* found = value_pointee(ptr); + return found != nullptr ? found->template get() : default_value; } /// @brief access specified object element via JSON Pointer with default value @@ -3463,21 +3443,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(const json_pointer& ptr, ValueType && default_value) const { - // value only works for arrays and objects - if (JSON_HEDLEY_LIKELY(is_structured())) - { - // If the pointer resolves to a value, return it. Otherwise, return - // 'default_value'. - const auto* res = ptr.get_checked_or_null(this); - if (JSON_HEDLEY_LIKELY(res != nullptr)) - { - return res->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If the pointer resolves to a value, return it. Otherwise, return + // 'default_value'. + const auto* found = value_pointee(ptr); + return found != nullptr ? found->template get() : std::forward(default_value); } template < class ValueType, class BasicJsonType, detail::enable_if_t < @@ -3562,21 +3531,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(invalid_iterator::create(205, "iterator out of range", this)); } - if (is_string()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.string); - std::allocator_traits::deallocate(alloc, m_data.m_value.string, 1); - m_data.m_value.string = nullptr; - } - else if (is_binary()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.binary); - std::allocator_traits::deallocate(alloc, m_data.m_value.binary, 1); - m_data.m_value.binary = nullptr; - } - + m_data.m_value.destroy(m_data.m_type); + m_data.m_value = {}; m_data.m_type = value_t::null; assert_invariant(); break; @@ -3628,27 +3584,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::string: case value_t::binary: { - if (JSON_HEDLEY_LIKELY(!first.m_it.primitive_iterator.is_begin() - || !last.m_it.primitive_iterator.is_end())) + if (JSON_HEDLEY_UNLIKELY(!first.m_it.primitive_iterator.is_begin() + || !last.m_it.primitive_iterator.is_end())) { JSON_THROW(invalid_iterator::create(204, "iterators out of range", this)); } - if (is_string()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.string); - std::allocator_traits::deallocate(alloc, m_data.m_value.string, 1); - m_data.m_value.string = nullptr; - } - else if (is_binary()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.binary); - std::allocator_traits::deallocate(alloc, m_data.m_value.binary, 1); - m_data.m_value.binary = nullptr; - } - + m_data.m_value.destroy(m_data.m_type); + m_data.m_value = {}; m_data.m_type = value_t::null; assert_invariant(); break; @@ -3704,7 +3647,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - const auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); + const auto it = object_lookup(*this, std::forward(key)); if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); @@ -3781,7 +3724,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(key); + result.m_it.object_iterator = object_lookup(*this, key); } return result; @@ -3795,7 +3738,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(key); + result.m_it.object_iterator = object_lookup(*this, key); } return result; @@ -3811,7 +3754,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(lookup_key(std::forward(key))); + result.m_it.object_iterator = object_lookup(*this, std::forward(key)); } return result; @@ -3827,7 +3770,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(lookup_key(std::forward(key))); + result.m_it.object_iterator = object_lookup(*this, std::forward(key)); } return result; @@ -3858,7 +3801,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(const typename object_t::key_type& key) const { - return is_object() && m_data.m_value.object->find(key) != m_data.m_value.object->end(); + return is_object() && object_lookup(*this, key) != m_data.m_value.object->end(); } /// @brief check the existence of an element in a JSON object @@ -3868,7 +3811,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(KeyType && key) const { - return is_object() && m_data.m_value.object->find(lookup_key(std::forward(key))) != m_data.m_value.object->end(); + return is_object() && object_lookup(*this, std::forward(key)) != m_data.m_value.object->end(); } /// @brief check the existence of an element in a JSON object given a JSON pointer @@ -4233,9 +4176,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array (move semantics) @@ -4266,9 +4207,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array @@ -4298,9 +4237,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value = value_t::object; - assert_invariant(); + convert_null_to(); } // add the element to the object @@ -4354,9 +4291,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array (perfect forwarding) @@ -4379,9 +4314,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value = value_t::object; - assert_invariant(); + convert_null_to(); } // add the element to the array (perfect forwarding) @@ -4441,7 +4374,22 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/insert/ iterator insert(const_iterator pos, basic_json&& val) // NOLINT(performance-unnecessary-value-param) { - return insert(std::move(pos), val); + // insert only works for arrays + if (JSON_HEDLEY_LIKELY(is_array())) + { + // check if iterator pos fits to this JSON value + if (JSON_HEDLEY_UNLIKELY(pos.m_object != this)) + { + JSON_THROW(invalid_iterator::create(202, "iterator does not fit current value", this)); + } + + // moving into a local first keeps this safe even if val aliases + // an element of this array + basic_json tmp(std::move(val)); + return insert_iterator(pos, std::move(tmp)); + } + + JSON_THROW(type_error::create(309, detail::concat("cannot use insert() with ", type_name()), this)); } /// @brief inserts copies of element into array @@ -4617,14 +4565,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// value is an object; called first by both @ref update overloads void prepare_update() { - // implicitly convert a null value to an empty object; create the - // object before setting the type, so a throwing allocation leaves - // this value null + // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_value.object = create(); - m_data.m_type = value_t::object; - assert_invariant(); + convert_null_to(); } if (JSON_HEDLEY_UNLIKELY(!is_object())) @@ -4833,7 +4777,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(binary_t& other) // NOLINT(bugprone-exception-escape,cppcoreguidelines-noexcept-swap,performance-noexcept-swap) { - // swap only works for strings + // swap only works for binary values if (JSON_HEDLEY_LIKELY(is_binary())) { using std::swap; @@ -4849,7 +4793,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(typename binary_t::container_type& other) // NOLINT(bugprone-exception-escape) { - // swap only works for strings + // swap only works for binary values if (JSON_HEDLEY_LIKELY(is_binary())) { using std::swap; @@ -5350,7 +5294,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(detail::input_adapter(std::forward(i)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved,accessForwarded] + auto p = parser(detail::input_adapter(std::forward(i)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -5367,7 +5312,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(detail::input_adapter(std::move(first), std::move(last)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved] + auto p = parser(detail::input_adapter(std::move(first), std::move(last)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -5380,7 +5326,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(i.get(), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved] + auto p = parser(i.get(), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -5607,6 +5554,31 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #endif + private: + /*! + @brief shared implementation of the binary from_cbor/from_msgpack/ + from_ubjson/from_bjdata/from_bon8/from_bson overloads + + Building the @ref binary_reader as a named local, rather than as a + temporary that @ref detail::json_sax_dom_parser::parse is called on in + the same expression, avoids a false-positive cppcheck accessMoved + warning in each of the 16 callers. + */ + template + static basic_json from_binary_impl(InputAdapterType ia, const input_format_t format, + const bool strict, const bool allow_exceptions, + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + { + basic_json result; + detail::json_sax_dom_parser sdp(result, allow_exceptions); + binary_reader reader(std::move(ia), format); + if (!reader.sax_parse(&sdp, strict, tag_handler)) + { + result = value_t::discarded; + } + return result; + } + ////////////////////////////////////////// // binary serialization/deserialization // ////////////////////////////////////////// @@ -5779,14 +5751,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, tag_handler); } /// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5799,14 +5764,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, tag_handler); } template @@ -5827,15 +5785,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, tag_handler); } /// @brief create a JSON value from an input in MessagePack format @@ -5846,14 +5796,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions); } /// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5865,14 +5808,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions); } template @@ -5891,15 +5827,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::msgpack, strict, allow_exceptions); } /// @brief create a JSON value from an input in UBJSON format @@ -5910,14 +5838,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions); } /// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5929,14 +5850,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions); } template @@ -5955,15 +5869,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::ubjson, strict, allow_exceptions); } /// @brief create a JSON value from an input in BJData format @@ -5974,14 +5880,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions); } /// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5993,14 +5892,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions); } /// @brief create a JSON value from an input in BON8 format @@ -6011,14 +5903,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bon8, strict, allow_exceptions); } /// @brief create a JSON value from an input in BON8 format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -6030,14 +5915,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bon8, strict, allow_exceptions); } /// @brief create a JSON value from an input in BSON format @@ -6048,14 +5926,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions); } /// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -6067,14 +5938,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions); } template @@ -6093,15 +5957,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::bson, strict, allow_exceptions); } /// @} diff --git a/include/nlohmann/ordered_map.hpp b/include/nlohmann/ordered_map.hpp index d1c247483..eb521114e 100644 --- a/include/nlohmann/ordered_map.hpp +++ b/include/nlohmann/ordered_map.hpp @@ -74,16 +74,43 @@ template , return *this; } +private: + /// @brief find the entry for @a key, for either constness of @a self + /// @note the single place that performs the linear key search + template + static auto find_impl(Self& self, KeyType&& key) -> decltype(self.begin()) + { + for (auto it = self.begin(); it != self.end(); ++it) + { + if (self.m_compare(it->first, key)) + { + return it; + } + } + return self.end(); + } + + /// @brief remove the entry @a it points to, preserving order + /// @note keys are not movable, so the tail is destroyed and re-constructed in place + void erase_at(iterator it) + { + for (auto next = it; ++next != this->end(); ++it) + { + it->~value_type(); // Destroy but keep allocation + new (&*it) value_type{std::move(*next)}; + } + Container::pop_back(); + } + +public: template::value, int> = 0> std::pair emplace(const key_type& key, V && t) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - return {it, false}; - } + return {it, false}; } append(key, std::forward(t)); return {std::prev(this->end()), true}; @@ -94,12 +121,10 @@ template , detail::is_constructible>::value, int> = 0> std::pair emplace(KeyType && key, V && t) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - return {it, false}; - } + return {it, false}; } append(std::forward(key), std::forward(t)); return {std::prev(this->end()), true}; @@ -131,75 +156,55 @@ template , T& at(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } template::value, int> = 0> T & at(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } const T& at(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } template::value, int> = 0> const T & at(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } size_type erase(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - // Since we cannot move const Keys, re-construct them in place - for (auto next = it; ++next != this->end(); ++it) - { - it->~value_type(); // Destroy but keep allocation - new (&*it) value_type{std::move(*next)}; - } - Container::pop_back(); - return 1; - } + erase_at(it); + return 1; } return 0; } @@ -208,19 +213,11 @@ template , detail::is_usable_as_key_type::value, int> = 0> size_type erase(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - // Since we cannot move const Keys, re-construct them in place - for (auto next = it; ++next != this->end(); ++it) - { - it->~value_type(); // Destroy but keep allocation - new (&*it) value_type{std::move(*next)}; - } - Container::pop_back(); - return 1; - } + erase_at(it); + return 1; } return 0; } @@ -285,80 +282,38 @@ template , size_type count(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return 1; - } - } - return 0; + return find_impl(*this, key) != this->end() ? 1 : 0; } template::value, int> = 0> size_type count(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return 1; - } - } - return 0; + return find_impl(*this, key) != this->end() ? 1 : 0; } iterator find(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } template::value, int> = 0> iterator find(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } const_iterator find(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } template::value, int> = 0> const_iterator find(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } std::pair insert( value_type&& value ) @@ -368,12 +323,10 @@ template , std::pair insert( const value_type& value ) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, value.first); + if (it != this->end()) { - if (m_compare(it->first, value.first)) - { - return {it, false}; - } + return {it, false}; } append(value); return {--this->end(), true}; diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index 75f3757e8..0ae5e382b 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -2684,12 +2684,10 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") SECTION("move constructor resets the moved-from value to npos") { - // basic_json(basic_json&&) (json.hpp, around line 1951) copies + // basic_json(basic_json&&) copies // other's start_position/end_position into *this and then resets - // other's to npos (see the cppcheck-suppress[accessForwarded] - // annotation there, which flags this reset as worth a second - // look). Only the top-level moved-from value is affected; its - // (moved-away) children are gone along with it. + // other's to npos. Only the top-level moved-from value is + // affected; its (moved-away) children are gone along with it. const std::string s = R"({"a":1,"b":[1,2,3]})"; json a = json::parse(s); const auto a_start = a.start_pos(); diff --git a/tests/src/unit-modifiers.cpp b/tests/src/unit-modifiers.cpp index ee81120cf..532283f95 100644 --- a/tests/src/unit-modifiers.cpp +++ b/tests/src/unit-modifiers.cpp @@ -630,6 +630,49 @@ TEST_CASE("modifiers") } } + SECTION("rvalue at position moves rather than copies") + { + // regression test: insert(pos, basic_json&&) used to forward to + // insert(pos, const basic_json&) because the named rvalue + // reference parameter is itself an lvalue, so it always + // deep-copied its argument instead of moving it + json j_big = std::string(1000, 'x'); + const auto* const original_buffer = j_big.get_ref().data(); + + auto it = j_array.insert(j_array.begin(), std::move(j_big)); + CHECK(j_array.size() == 5); + CHECK(*it == json(std::string(1000, 'x'))); + CHECK((*it).get_ref().data() == original_buffer); + + // the moved-from value is null, the same as after push_back(&&) + CHECK(j_big.is_null()); // NOLINT(bugprone-use-after-move,hicpp-invalid-access-moved) + } + + SECTION("self-aliasing insertion") + { + SECTION("without reallocation") + { + json j_self = {1, 2, 3, 4}; + j_self.get_ref().reserve(j_self.size() + 1); + + auto it = j_self.insert(j_self.begin(), std::move(j_self[1])); + CHECK(j_self.size() == 5); + CHECK(*it == json(2)); + CHECK(j_self == json({2, 1, nullptr, 3, 4})); + } + + SECTION("with reallocation") + { + json j_self = {1, 2, 3, 4}; + j_self.get_ref().shrink_to_fit(); + + auto it = j_self.insert(j_self.begin(), std::move(j_self[1])); + CHECK(j_self.size() == 5); + CHECK(*it == json(2)); + CHECK(j_self == json({2, 1, nullptr, 3, 4})); + } + } + SECTION("copies at position") { SECTION("insert before begin()") From c5650eaa3c91c3ff145e055c4937ee7411cc03d2 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 4 Oct 2026 11:59:03 +0200 Subject: [PATCH 4/8] Reject malformed UTF-16/UTF-32 units in wide-string input (#5704) The wide-string input adapter (used for std::u16string, std::u32string, std::wstring, and iterators over 2- or 4-byte character types) passed some malformed code units on to the lexer as values that are neither a byte (0x00..0xFF) nor char_traits::eof(). As a result: - A lone UTF-16 surrogate inside true/false/null was accepted if its low byte matched the expected letter, or ended the input silently if it was the last unit. - A high surrogate followed by a unit that is not its low surrogate swallowed that unit; if the swallowed unit was the newline ending a // comment, the comment silently extended over the next line. - Where wint_t is a signed int (macOS, the BSDs), a negative wchar_t collided with char_traits::eof() (ending the input early) or was truncated to its low byte, depending on its value. The UTF-32 helper now converts the code unit to std::uint32_t before the range checks, so a negative unit reaches the same "emit 0xFF" branch already used for code points above U+10FFFF. The UTF-16 helper now peeks at the next unit before consuming it, and emits 0xFF instead of the raw surrogate when no valid pair is found, matching how ill-formed UTF-8 bytes are rejected elsewhere in the lexer. Fixes #5645. Signed-off-by: Niels Lohmann --- .../nlohmann/detail/input/input_adapters.hpp | 13 +- single_include/nlohmann/json.hpp | 843 +++++++----------- tests/src/unit-wstring.cpp | 60 +- 3 files changed, 393 insertions(+), 523 deletions(-) diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index 7f2f79cbf..9578e275a 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -453,8 +453,10 @@ struct wide_string_input_helper } else { - // get the current character - const auto wc = input.get_character(); + // get the current character; converted to an unsigned type so that + // a negative unit (wint_t is signed on some platforms) is not + // mistaken for an ASCII character or for EOF + const auto wc = static_cast(input.get_character()); if (wc <= 0x10FFFF) { @@ -522,9 +524,11 @@ struct wide_string_input_helper bool valid_pair = false; if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty())) { - const auto wc2 = static_cast(input.get_character()); + // only consume the next unit if it completes the pair + const auto wc2 = static_cast(*input.current); if (0xDC00 <= wc2 && wc2 <= 0xDFFF) { + input.get_character(); const auto charcode = 0x10000u + (((static_cast(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); utf8_bytes_filled = 0; encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) @@ -537,7 +541,8 @@ struct wide_string_input_helper if (!valid_pair) { - utf8_bytes[0] = static_cast::int_type>(wc); + // emit a byte that is never valid UTF-8 (see the UTF-32 case) + utf8_bytes[0] = 0xFF; utf8_bytes_filled = 1; } } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index f8fc9fdb9..3b74fd67f 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -5046,7 +5046,6 @@ inline std::size_t concat_length(const char /*c*/, const Args& ... rest) template inline std::size_t concat_length(const char* cstr, const Args& ... rest) { - // cppcheck-suppress ignoredReturnValue return ::strlen(cstr) + concat_length(rest...); } @@ -8025,8 +8024,10 @@ struct wide_string_input_helper } else { - // get the current character - const auto wc = input.get_character(); + // get the current character; converted to an unsigned type so that + // a negative unit (wint_t is signed on some platforms) is not + // mistaken for an ASCII character or for EOF + const auto wc = static_cast(input.get_character()); if (wc <= 0x10FFFF) { @@ -8094,9 +8095,11 @@ struct wide_string_input_helper bool valid_pair = false; if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty())) { - const auto wc2 = static_cast(input.get_character()); + // only consume the next unit if it completes the pair + const auto wc2 = static_cast(*input.current); if (0xDC00 <= wc2 && wc2 <= 0xDFFF) { + input.get_character(); const auto charcode = 0x10000u + (((static_cast(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); utf8_bytes_filled = 0; encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) @@ -8109,7 +8112,8 @@ struct wide_string_input_helper if (!valid_pair) { - utf8_bytes[0] = static_cast::int_type>(wc); + // emit a byte that is never valid UTF-8 (see the UTF-32 case) + utf8_bytes[0] = 0xFF; utf8_bytes_filled = 1; } } @@ -8318,7 +8322,7 @@ struct container_input_adapter_factory< ContainerType, { // container is forwarded twice on purpose: the resulting begin/end // iterator types must match adapter_type, computed the same way - // NOLINTNEXTLINE(bugprone-use-after-move) + // NOLINTNEXTLINE(bugprone-use-after-move,hicpp-invalid-access-moved) return input_adapter(begin(std::forward(container)), end(std::forward(container))); } }; @@ -26273,16 +26277,43 @@ template , return *this; } +private: + /// @brief find the entry for @a key, for either constness of @a self + /// @note the single place that performs the linear key search + template + static auto find_impl(Self& self, KeyType&& key) -> decltype(self.begin()) + { + for (auto it = self.begin(); it != self.end(); ++it) + { + if (self.m_compare(it->first, key)) + { + return it; + } + } + return self.end(); + } + + /// @brief remove the entry @a it points to, preserving order + /// @note keys are not movable, so the tail is destroyed and re-constructed in place + void erase_at(iterator it) + { + for (auto next = it; ++next != this->end(); ++it) + { + it->~value_type(); // Destroy but keep allocation + new (&*it) value_type{std::move(*next)}; + } + Container::pop_back(); + } + +public: template::value, int> = 0> std::pair emplace(const key_type& key, V && t) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - return {it, false}; - } + return {it, false}; } append(key, std::forward(t)); return {std::prev(this->end()), true}; @@ -26293,12 +26324,10 @@ template , detail::is_constructible>::value, int> = 0> std::pair emplace(KeyType && key, V && t) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - return {it, false}; - } + return {it, false}; } append(std::forward(key), std::forward(t)); return {std::prev(this->end()), true}; @@ -26330,75 +26359,55 @@ template , T& at(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } template::value, int> = 0> T & at(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } const T& at(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } template::value, int> = 0> const T & at(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } size_type erase(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - // Since we cannot move const Keys, re-construct them in place - for (auto next = it; ++next != this->end(); ++it) - { - it->~value_type(); // Destroy but keep allocation - new (&*it) value_type{std::move(*next)}; - } - Container::pop_back(); - return 1; - } + erase_at(it); + return 1; } return 0; } @@ -26407,19 +26416,11 @@ template , detail::is_usable_as_key_type::value, int> = 0> size_type erase(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - // Since we cannot move const Keys, re-construct them in place - for (auto next = it; ++next != this->end(); ++it) - { - it->~value_type(); // Destroy but keep allocation - new (&*it) value_type{std::move(*next)}; - } - Container::pop_back(); - return 1; - } + erase_at(it); + return 1; } return 0; } @@ -26484,80 +26485,38 @@ template , size_type count(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return 1; - } - } - return 0; + return find_impl(*this, key) != this->end() ? 1 : 0; } template::value, int> = 0> size_type count(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return 1; - } - } - return 0; + return find_impl(*this, key) != this->end() ? 1 : 0; } iterator find(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } template::value, int> = 0> iterator find(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } const_iterator find(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } template::value, int> = 0> const_iterator find(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } std::pair insert( value_type&& value ) @@ -26567,12 +26526,10 @@ template , std::pair insert( const value_type& value ) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, value.first); + if (it != this->end()) { - if (m_compare(it->first, value.first)) - { - return {it, false}; - } + return {it, false}; } append(value); return {--this->end(), true}; @@ -26695,11 +26652,12 @@ struct is_std_optional> : std::true_type {}; @brief a class to store JSON values @internal -@invariant The member variables @a m_value and @a m_type have the following -relationship: -- If `m_type == value_t::object`, then `m_value.object != nullptr`. -- If `m_type == value_t::array`, then `m_value.array != nullptr`. -- If `m_type == value_t::string`, then `m_value.string != nullptr`. +@invariant The member variables @a m_data.m_value and @a m_data.m_type have +the following relationship: +- If `m_data.m_type == value_t::object`, then `m_data.m_value.object != nullptr`. +- If `m_data.m_type == value_t::array`, then `m_data.m_value.array != nullptr`. +- If `m_data.m_type == value_t::string`, then `m_data.m_value.string != nullptr`. +- If `m_data.m_type == value_t::binary`, then `m_data.m_value.binary != nullptr`. The invariants are checked by member function assert_invariant(). @note ObjectType trick from https://stackoverflow.com/a/9860911 @@ -26762,18 +26720,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } private: - using primitive_iterator_t = ::nlohmann::detail::primitive_iterator_t; - template - using internal_iterator = ::nlohmann::detail::internal_iterator; template using iter_impl = ::nlohmann::detail::iter_impl; template using iteration_proxy = ::nlohmann::detail::iteration_proxy; template using json_reverse_iterator = ::nlohmann::detail::json_reverse_iterator; - template - using output_adapter_t = ::nlohmann::detail::output_adapter_t; - template using binary_reader = ::nlohmann::detail::binary_reader; template using binary_writer = ::nlohmann::detail::binary_writer; @@ -26914,8 +26866,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec std::to_string(__GNUC_PATCHLEVEL__)) } }; -#elif defined(__HP_cc) || defined(__HP_aCC) - result["compiler"] = "hp" +#elif defined(__HP_aCC) + result["compiler"] = {{"family", "hp"}, {"version", __HP_aCC}}; #elif defined(__IBMCPP__) result["compiler"] = {{"family", "ilecpp"}, {"version", __IBMCPP__}}; #elif defined(_MSC_VER) @@ -27057,9 +27009,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec binary | binary | pointer to @ref binary_t null | null | *no value is stored* - @note Variable-length types (objects, arrays, and strings) are stored as - pointers. The size of the union should not exceed 64 bits if the default - value types are used. + @note Variable-length types (objects, arrays, strings, and binary + values) are stored as pointers. The size of the union should not exceed + 64 bits if the default value types are used. @since version 1.0.0 */ @@ -27311,7 +27263,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec end of every constructor to make sure that created objects respect the invariant. Furthermore, it has to be called each time the type of a JSON value is changed, because the invariant expresses a relationship between - @a m_type and @a m_value. + @a m_data.m_type and @a m_data.m_value. Furthermore, the parent relation is checked for arrays and objects: If @a check_parents true and the value is an array or object, then the @@ -27332,7 +27284,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec #if JSON_DIAGNOSTICS JSON_TRY { - // cppcheck-suppress assertWithSideEffect JSON_ASSERT(!check_parents || !is_structured() || std::all_of(begin(), end(), [this](const basic_json & j) { return j.m_parent == this; @@ -28504,8 +28455,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief create a null object /// @sa https://json.nlohmann.me/api/basic_json/basic_json/ - basic_json(std::nullptr_t = nullptr) noexcept // NOLINT(bugprone-exception-escape) - : basic_json(value_t::null) + basic_json(std::nullptr_t = nullptr) noexcept { assert_invariant(); } @@ -28869,15 +28819,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief move constructor /// @sa https://json.nlohmann.me/api/basic_json/basic_json/ basic_json(basic_json&& other) noexcept - : json_base_class_t(std::forward(other)), - m_data(std::move(other.m_data)) // cppcheck-suppress[accessForwarded] TODO check + : json_base_class_t(std::move(static_cast(other))), + m_data(std::move(other.m_data)) #if JSON_DIAGNOSTIC_POSITIONS - , start_position(other.start_position) // cppcheck-suppress[accessForwarded] TODO check - , end_position(other.end_position) // cppcheck-suppress[accessForwarded] TODO check + , start_position(other.start_position) + , end_position(other.end_position) #endif { // check that the passed value is valid - other.assert_invariant(false); // cppcheck-suppress[accessForwarded] + other.assert_invariant(false); // invalidate payload other.m_data.m_type = value_t::null; @@ -29444,7 +29394,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec @tparam PointerType pointer type; must be a pointer to @ref array_t, @ref object_t, @ref string_t, @ref boolean_t, @ref number_integer_t, - @ref number_unsigned_t, or @ref number_float_t. + @ref number_unsigned_t, @ref number_float_t, or @ref binary_t. @return pointer to the internally stored JSON value if the requested pointer type @a PointerType fits to the JSON value; `nullptr` otherwise @@ -29612,6 +29562,90 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @} + private: + /// @brief look up @a key in the object held by @a j, for either constness of @a j + /// @note the single place that performs a (possibly transparent) object key lookup + template + static auto object_lookup(Self& j, KeyType&& key) + -> decltype(j.m_data.m_value.object->find(lookup_key(std::forward(key)))) + { + return j.m_data.m_value.object->find(lookup_key(std::forward(key))); + } + + /// @brief checked object element access used by the at() overloads taking a key + /// @throw type_error.304 if @a j is not an object + /// @throw out_of_range.403 if @a key is not found + template + static auto object_at(Self& j, KeyType&& key) + -> decltype((object_lookup(j, std::forward(key))->second)) + { + // at only works for objects + if (JSON_HEDLEY_UNLIKELY(!j.is_object())) + { + JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", j.type_name()), &j)); + } + + auto it = object_lookup(j, std::forward(key)); + if (it == j.m_data.m_value.object->end()) + { + // key is only forwarded into the lookup above: object_t::find() (a plain + // std::map or ordered_map) never moves from its argument, so key is still + // valid here regardless of whether KeyType was deduced as an rvalue reference + // NOLINTNEXTLINE(bugprone-use-after-move,hicpp-invalid-access-moved) + JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(key), "' not found"), &j)); + } + return it->second; + } + + /// @brief checked array element access used by the at() overloads taking an index + /// @throw type_error.304 if @a j is not an array + /// @throw out_of_range.401 if @a idx is out of range + template + static auto array_at(Self& j, size_type idx) + -> decltype((*j.m_data.m_value.array)[idx]) + { + // at only works for arrays + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", j.type_name()), &j)); + } + + if (JSON_HEDLEY_UNLIKELY(idx >= j.m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), &j)); + } + + return (*j.m_data.m_value.array)[idx]; + } + + /// @brief convert a null value to an empty container of type @a Container + /// @tparam Container array_t or object_t; any other type does not compile + template + void convert_null_to() + { + JSON_ASSERT(is_null()); + // create the container before touching the type, so a throwing + // allocation leaves this value as a valid null rather than a type + // tag with a dangling/null pointer behind it + set_container(create()); + assert_invariant(); + } + + /// @brief store a freshly created array and set the matching type + void set_container(array_t* array) noexcept + { + m_data.m_value.array = array; + m_data.m_type = value_t::array; + } + + /// @brief store a freshly created object and set the matching type + void set_container(object_t* object) noexcept + { + m_data.m_value.object = object; + m_data.m_type = value_t::object; + } + + public: //////////////////// // element access // //////////////////// @@ -29624,54 +29658,21 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/at/ reference at(size_type idx) { - // at only works for arrays - if (JSON_HEDLEY_UNLIKELY(!is_array())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) - { - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } - - return set_parent((*m_data.m_value.array)[idx]); + return set_parent(array_at(*this, idx)); } /// @brief access specified array element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ const_reference at(size_type idx) const { - // at only works for arrays - if (JSON_HEDLEY_UNLIKELY(!is_array())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) - { - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } - - return (*m_data.m_value.array)[idx]; + return array_at(*this, idx); } /// @brief access specified object element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ reference at(const typename object_t::key_type& key) { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(key); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", key, "' not found"), this)); - } - return set_parent(it->second); + return set_parent(object_at(*this, key)); } /// @brief access specified object element with bounds checking @@ -29680,36 +29681,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_usable_as_basic_json_key_type::value, int> = 0> reference at(KeyType && key) { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(std::forward(key)), "' not found"), this)); - } - return set_parent(it->second); + return set_parent(object_at(*this, std::forward(key))); } /// @brief access specified object element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ const_reference at(const typename object_t::key_type& key) const { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(key); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", key, "' not found"), this)); - } - return it->second; + return object_at(*this, key); } /// @brief access specified object element with bounds checking @@ -29718,18 +29697,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_usable_as_basic_json_key_type::value, int> = 0> const_reference at(KeyType && key) const { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(std::forward(key)), "' not found"), this)); - } - return it->second; + return object_at(*this, std::forward(key)); } /// @brief access specified array element @@ -29739,9 +29707,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value.array = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for arrays @@ -29807,9 +29773,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value.object = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for objects @@ -29829,7 +29793,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // const operator[] only works for objects if (JSON_HEDLEY_LIKELY(is_object())) { - auto it = m_data.m_value.object->find(key); + auto it = object_lookup(*this, key); JSON_ASSERT(it != m_data.m_value.object->end()); return it->second; } @@ -29860,9 +29824,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value.object = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for objects @@ -29884,7 +29846,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // const operator[] only works for objects if (JSON_HEDLEY_LIKELY(is_object())) { - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); + auto it = object_lookup(*this, std::forward(key)); JSON_ASSERT(it != m_data.m_value.object->end()); return it->second; } @@ -29904,6 +29866,36 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_c_string_uncvref::value, string_t, typename std::decay::type >; + /// @brief look up @a key for value(), the single place shared by all key-based overloads + /// @throw type_error.306 if this is not an object + /// @return a pointer to the found value, or `nullptr` if @a key was not found + template + const basic_json* value_member(KeyType&& key) const + { + // value only works for objects + if (JSON_HEDLEY_UNLIKELY(!is_object())) + { + JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + } + + const auto it = find(std::forward(key)); + return it != end() ? &*it : nullptr; + } + + /// @brief resolve @a ptr for value(), the single place shared by both json_pointer overloads + /// @throw type_error.306 if this is not an array or object + /// @return a pointer to the resolved value, or `nullptr` if @a ptr does not resolve + const basic_json* value_pointee(const json_pointer& ptr) const + { + // value only works for arrays and objects + if (JSON_HEDLEY_UNLIKELY(!is_structured())) + { + JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + } + + return ptr.get_checked_or_null(this); + } + public: // an integer literal 0 would otherwise convert to a null const char* and from there to key_type template::value, int> = 0> @@ -29917,20 +29909,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(const typename object_t::key_type& key, const ValueType& default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(key); - if (it != end()) - { - return it->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(key); + return found != nullptr ? found->template get() : default_value; } /// @brief access specified object element with default value @@ -29942,20 +29923,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(const typename object_t::key_type& key, ValueType && default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(key); - if (it != end()) - { - return it->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(key); + return found != nullptr ? found->template get() : std::forward(default_value); } /// @brief access specified object element with default value @@ -29968,23 +29938,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(KeyType && key, const ValueType& default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(std::forward(key)); - if (it != end()) - { - return it->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(std::forward(key)); + return found != nullptr ? found->template get() : default_value; } - /// @brief access specified object element via JSON Pointer with default value + /// @brief access specified object element with default value /// @sa https://json.nlohmann.me/api/basic_json/value/ template < class ValueType, class KeyType, class ReturnType = typename value_return_type::type, detail::enable_if_t < @@ -29995,20 +29954,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(KeyType && key, ValueType && default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(std::forward(key)); - if (it != end()) - { - return it->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(std::forward(key)); + return found != nullptr ? found->template get() : std::forward(default_value); } /// @brief access specified object element via JSON Pointer with default value @@ -30018,21 +29966,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(const json_pointer& ptr, const ValueType& default_value) const { - // value only works for arrays and objects - if (JSON_HEDLEY_LIKELY(is_structured())) - { - // If the pointer resolves to a value, return it. Otherwise, return - // 'default_value'. - const auto* res = ptr.get_checked_or_null(this); - if (JSON_HEDLEY_LIKELY(res != nullptr)) - { - return res->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If the pointer resolves to a value, return it. Otherwise, return + // 'default_value'. + const auto* found = value_pointee(ptr); + return found != nullptr ? found->template get() : default_value; } /// @brief access specified object element via JSON Pointer with default value @@ -30043,21 +29980,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(const json_pointer& ptr, ValueType && default_value) const { - // value only works for arrays and objects - if (JSON_HEDLEY_LIKELY(is_structured())) - { - // If the pointer resolves to a value, return it. Otherwise, return - // 'default_value'. - const auto* res = ptr.get_checked_or_null(this); - if (JSON_HEDLEY_LIKELY(res != nullptr)) - { - return res->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If the pointer resolves to a value, return it. Otherwise, return + // 'default_value'. + const auto* found = value_pointee(ptr); + return found != nullptr ? found->template get() : std::forward(default_value); } template < class ValueType, class BasicJsonType, detail::enable_if_t < @@ -30142,21 +30068,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(invalid_iterator::create(205, "iterator out of range", this)); } - if (is_string()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.string); - std::allocator_traits::deallocate(alloc, m_data.m_value.string, 1); - m_data.m_value.string = nullptr; - } - else if (is_binary()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.binary); - std::allocator_traits::deallocate(alloc, m_data.m_value.binary, 1); - m_data.m_value.binary = nullptr; - } - + m_data.m_value.destroy(m_data.m_type); + m_data.m_value = {}; m_data.m_type = value_t::null; assert_invariant(); break; @@ -30208,27 +30121,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::string: case value_t::binary: { - if (JSON_HEDLEY_LIKELY(!first.m_it.primitive_iterator.is_begin() - || !last.m_it.primitive_iterator.is_end())) + if (JSON_HEDLEY_UNLIKELY(!first.m_it.primitive_iterator.is_begin() + || !last.m_it.primitive_iterator.is_end())) { JSON_THROW(invalid_iterator::create(204, "iterators out of range", this)); } - if (is_string()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.string); - std::allocator_traits::deallocate(alloc, m_data.m_value.string, 1); - m_data.m_value.string = nullptr; - } - else if (is_binary()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.binary); - std::allocator_traits::deallocate(alloc, m_data.m_value.binary, 1); - m_data.m_value.binary = nullptr; - } - + m_data.m_value.destroy(m_data.m_type); + m_data.m_value = {}; m_data.m_type = value_t::null; assert_invariant(); break; @@ -30284,7 +30184,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - const auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); + const auto it = object_lookup(*this, std::forward(key)); if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); @@ -30361,7 +30261,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(key); + result.m_it.object_iterator = object_lookup(*this, key); } return result; @@ -30375,7 +30275,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(key); + result.m_it.object_iterator = object_lookup(*this, key); } return result; @@ -30391,7 +30291,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(lookup_key(std::forward(key))); + result.m_it.object_iterator = object_lookup(*this, std::forward(key)); } return result; @@ -30407,7 +30307,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(lookup_key(std::forward(key))); + result.m_it.object_iterator = object_lookup(*this, std::forward(key)); } return result; @@ -30438,7 +30338,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(const typename object_t::key_type& key) const { - return is_object() && m_data.m_value.object->find(key) != m_data.m_value.object->end(); + return is_object() && object_lookup(*this, key) != m_data.m_value.object->end(); } /// @brief check the existence of an element in a JSON object @@ -30448,7 +30348,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(KeyType && key) const { - return is_object() && m_data.m_value.object->find(lookup_key(std::forward(key))) != m_data.m_value.object->end(); + return is_object() && object_lookup(*this, std::forward(key)) != m_data.m_value.object->end(); } /// @brief check the existence of an element in a JSON object given a JSON pointer @@ -30813,9 +30713,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array (move semantics) @@ -30846,9 +30744,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array @@ -30878,9 +30774,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value = value_t::object; - assert_invariant(); + convert_null_to(); } // add the element to the object @@ -30934,9 +30828,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array (perfect forwarding) @@ -30959,9 +30851,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value = value_t::object; - assert_invariant(); + convert_null_to(); } // add the element to the array (perfect forwarding) @@ -31021,7 +30911,22 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/insert/ iterator insert(const_iterator pos, basic_json&& val) // NOLINT(performance-unnecessary-value-param) { - return insert(std::move(pos), val); + // insert only works for arrays + if (JSON_HEDLEY_LIKELY(is_array())) + { + // check if iterator pos fits to this JSON value + if (JSON_HEDLEY_UNLIKELY(pos.m_object != this)) + { + JSON_THROW(invalid_iterator::create(202, "iterator does not fit current value", this)); + } + + // moving into a local first keeps this safe even if val aliases + // an element of this array + basic_json tmp(std::move(val)); + return insert_iterator(pos, std::move(tmp)); + } + + JSON_THROW(type_error::create(309, detail::concat("cannot use insert() with ", type_name()), this)); } /// @brief inserts copies of element into array @@ -31197,14 +31102,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// value is an object; called first by both @ref update overloads void prepare_update() { - // implicitly convert a null value to an empty object; create the - // object before setting the type, so a throwing allocation leaves - // this value null + // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_value.object = create(); - m_data.m_type = value_t::object; - assert_invariant(); + convert_null_to(); } if (JSON_HEDLEY_UNLIKELY(!is_object())) @@ -31413,7 +31314,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(binary_t& other) // NOLINT(bugprone-exception-escape,cppcoreguidelines-noexcept-swap,performance-noexcept-swap) { - // swap only works for strings + // swap only works for binary values if (JSON_HEDLEY_LIKELY(is_binary())) { using std::swap; @@ -31429,7 +31330,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(typename binary_t::container_type& other) // NOLINT(bugprone-exception-escape) { - // swap only works for strings + // swap only works for binary values if (JSON_HEDLEY_LIKELY(is_binary())) { using std::swap; @@ -31930,7 +31831,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(detail::input_adapter(std::forward(i)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved,accessForwarded] + auto p = parser(detail::input_adapter(std::forward(i)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -31947,7 +31849,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(detail::input_adapter(std::move(first), std::move(last)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved] + auto p = parser(detail::input_adapter(std::move(first), std::move(last)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -31960,7 +31863,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(i.get(), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved] + auto p = parser(i.get(), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -32187,6 +32091,31 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #endif + private: + /*! + @brief shared implementation of the binary from_cbor/from_msgpack/ + from_ubjson/from_bjdata/from_bon8/from_bson overloads + + Building the @ref binary_reader as a named local, rather than as a + temporary that @ref detail::json_sax_dom_parser::parse is called on in + the same expression, avoids a false-positive cppcheck accessMoved + warning in each of the 16 callers. + */ + template + static basic_json from_binary_impl(InputAdapterType ia, const input_format_t format, + const bool strict, const bool allow_exceptions, + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + { + basic_json result; + detail::json_sax_dom_parser sdp(result, allow_exceptions); + binary_reader reader(std::move(ia), format); + if (!reader.sax_parse(&sdp, strict, tag_handler)) + { + result = value_t::discarded; + } + return result; + } + ////////////////////////////////////////// // binary serialization/deserialization // ////////////////////////////////////////// @@ -32359,14 +32288,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, tag_handler); } /// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32379,14 +32301,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, tag_handler); } template @@ -32407,15 +32322,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, tag_handler); } /// @brief create a JSON value from an input in MessagePack format @@ -32426,14 +32333,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions); } /// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32445,14 +32345,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions); } template @@ -32471,15 +32364,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::msgpack, strict, allow_exceptions); } /// @brief create a JSON value from an input in UBJSON format @@ -32490,14 +32375,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions); } /// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32509,14 +32387,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions); } template @@ -32535,15 +32406,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::ubjson, strict, allow_exceptions); } /// @brief create a JSON value from an input in BJData format @@ -32554,14 +32417,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions); } /// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32573,14 +32429,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions); } /// @brief create a JSON value from an input in BON8 format @@ -32591,14 +32440,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bon8, strict, allow_exceptions); } /// @brief create a JSON value from an input in BON8 format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32610,14 +32452,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bon8, strict, allow_exceptions); } /// @brief create a JSON value from an input in BSON format @@ -32628,14 +32463,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions); } /// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32647,14 +32475,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions); } template @@ -32673,15 +32494,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::bson, strict, allow_exceptions); } /// @} diff --git a/tests/src/unit-wstring.cpp b/tests/src/unit-wstring.cpp index b553ce246..c71684002 100644 --- a/tests/src/unit-wstring.cpp +++ b/tests/src/unit-wstring.cpp @@ -8,6 +8,7 @@ #include "doctest_compatibility.h" +#include #include using nlohmann::json; @@ -68,15 +69,15 @@ TEST_CASE("wide strings") CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&); // a lone low surrogate cannot start a pair - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // a high surrogate followed by a non-low-surrogate unit is invalid - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // ... also when the unit is above the low surrogates - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // a lone low surrogate must not swallow the following unit: pairing // it with any second unit would produce valid UTF-8, so the error // has to report an ill-formed byte at the surrogate's own position - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // a valid surrogate pair is still decoded (U+1F600) CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get() == "\xF0\x9F\x98\x80"); } @@ -104,4 +105,55 @@ TEST_CASE("wide strings") // the same unit inside a string is reported as an ill-formed byte CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); } + + SECTION("malformed wide-string input outside strings (#5645)") + { + json _; + + // a lone low surrogate inside a literal must not be truncated to its + // low byte and mistaken for the letter the literal expects next + // (0xDC72 truncates to 'r', which is what "true" expects after 't') + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast(0xDC72), u'u', u'e'}), + "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&); + + // ... also when the lone surrogate is the last unit of the input + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'f', u'a', u'l', u's', static_cast(0xDD65)}), + "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'fals\xFF'", json::parse_error&); + + // a high surrogate followed by a unit that is not its low surrogate + // must not silently swallow that unit + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast(0xD872), u'X', u'u', u'e'}), + "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&); + + // ... in particular, if the swallowed unit is the newline that ends a + // // comment, the comment must not extend over the following line + CHECK(json::parse(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast(0xD800), u'\n', + u',', u'2', u' ', u'/', u'/', u'\n', u']'}, + nullptr, true, /*ignore_comments*/true) == json::parse("[1,2]")); + CHECK(json::accept(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast(0xD800), u'\n', + u',', u'2', u' ', u'/', u'/', u'\n', u']'}, /*ignore_comments*/true)); + + // cases 5 and 6 use a 32-bit wchar_t (Linux, macOS, the BSDs) to reach + // the UTF-32 helper tested above via u32string; the 16-bit wchar_t of + // Windows goes through the UTF-16 helper instead, already covered by + // the u16string cases above +#if WCHAR_MAX > 0xFFFFu + // a negative wchar_t must not be mistaken for + // char_traits::eof() and silently end the input, letting + // trailing garbage pass the strict end-of-input check (only observable + // where wint_t is signed, e.g. macOS/the BSDs; on Linux wint_t is + // unsigned and this was already handled by #5348) + std::wstring w = L"[1]"; + w.push_back(static_cast(-1)); + w += L"garbage"; + CHECK(!json::accept(w)); + CHECK_THROWS_WITH_AS(_ = json::parse(w), + "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&); + + // other negative wchar_t units must not be truncated to their low + // byte (0xFFFFFF72 truncates to 'r', as in the u16string case above) + CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L't', static_cast(0xFFFFFF72), L'u', L'e'}), + "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&); +#endif + } } From 1a77948c25eee071319f2f07751455f884e25859 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 4 Oct 2026 11:59:34 +0200 Subject: [PATCH 5/8] Document the macros that preview version 4.0 in the roadmap (#5593) * Document the macros that preview version 4.0 in the roadmap List the macros that guard breaking changes planned to become the default in version 4.0.0, and explain that 4.0.0 will be the sum of these opt-in flags, which can be tried on the 3.x release train. Signed-off-by: Niels Lohmann * List the deprecated functions removed in 4.0 in the roadmap The roadmap lists what will be removed; the migration guide keeps the examples for how to replace each item. Also mention the deprecated (ptr, len) overloads of the from_* functions in the migration guide. Signed-off-by: Niels Lohmann * Wrap overlong line in cbor_tag_handler_t documentation The line added in #5559 exceeds the 160-character limit enforced by the documentation style check. Signed-off-by: Niels Lohmann * Add JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON to the 4.0 example Signed-off-by: Niels Lohmann * Add JSON_STRICT_BINARY_UTF8 to the 4.0 roadmap The macro comes from #5741: the CBOR, UBJSON, BJData, and BSON writers keep writing ill-formed UTF-8 unchanged in 3.x and are planned to check it by default in 4.0. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../macros/json_brace_init_copy_semantics.md | 1 + docs/mkdocs/docs/community/roadmap.md | 70 +++++++++++++++++-- .../docs/integration/migration_guide.md | 7 +- 3 files changed, 71 insertions(+), 7 deletions(-) diff --git a/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md b/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md index 459818c87..71b8bb28f 100644 --- a/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md +++ b/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md @@ -115,3 +115,4 @@ The default value is `0` (disabled — existing behavior is preserved). ## Version history - Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/community/roadmap.md b/docs/mkdocs/docs/community/roadmap.md index e8c407d3f..48977b295 100644 --- a/docs/mkdocs/docs/community/roadmap.md +++ b/docs/mkdocs/docs/community/roadmap.md @@ -14,7 +14,7 @@ work items are tracked in the [GitHub milestones](https://github.com/nlohmann/js opt-in. - **Keep the 3.x public API stable.** Releases follow [semantic versioning](https://semver.org). Changes that would break existing code are only added behind a feature macro, so users can opt in and test their code before a next - major release. + major release, see [Version 4.0](#version-40). - **Support a broad range of compilers and platforms.** The [CI](quality_assurance.md) keeps testing old and new versions of GCC, Clang, MSVC, and other compilers on Linux, macOS, and Windows. - **Keep the quality assurance up.** Every change keeps the test coverage at 100%, passes the static and dynamic @@ -37,7 +37,67 @@ work items are tracked in the [GitHub milestones](https://github.com/nlohmann/js ## Version 4.0 -There is no decision yet on whether or when a version 4.0 with breaking changes will be released. Proposals that need -a major version, for instance stricter type conversions, are collected in issue -[#3453](https://github.com/nlohmann/json/issues/3453). Until then, such changes are only added as opt-in behavior -behind feature macros. +There is no release date for version 4.0 yet. Proposals that need a major version, for instance stricter type +conversions, are collected in issue [#3453](https://github.com/nlohmann/json/issues/3453). + +!!! note "Not final" + + The plan for version 4.0 described below is not final and may still change: macros may be added to or removed from + the list, and planned defaults may be revised. Any such change will be documented on this page. + +### Trying out 4.0 today + +Version 4.0 will not be developed on a separate branch. Instead, every breaking change is first added to a 3.x release +behind a macro whose default keeps the 3.x behavior. Version 4.0 then switches the defaults and removes the macros. +Version 4.0 is therefore the sum of these macros: you can try it on the 3.x release train today by defining each macro +to its 4.0 value and fixing what no longer compiles or behaves differently. Once your code works with all of them, it +is ready for version 4.0. + +The following macros guard changes that are planned to become the default in version 4.0: + +| Macro | 3.x default | 4.0 behavior | CMake option | Added | +|------------------------------------------------------------------------------------------------------------------|-------------|-------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------|--------| +| [`JSON_USE_IMPLICIT_CONVERSIONS`](../api/macros/json_use_implicit_conversions.md) | `1` | `0`: no implicit conversions from `basic_json` to other types; use [`get`](../api/basic_json/get.md) instead | [`JSON_ImplicitConversions`](../integration/cmake.md#json_implicitconversions) | 3.9.0 | +| [`JSON_USE_GLOBAL_UDLS`](../api/macros/json_use_global_udls.md) | `1` | `0`: the string literals `_json` and `_json_pointer` are only available in namespace `nlohmann::literals` | [`JSON_GlobalUDLs`](../integration/cmake.md#json_globaludls) | 3.11.0 | +| [`JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md) | `0` | removed: the deprecated legacy comparison of discarded values can no longer be enabled | [`JSON_LegacyDiscardedValueComparison`](../integration/cmake.md#json_legacydiscardedvaluecomparison) | 3.11.0 | +| [`JSON_BRACE_INIT_COPY_SEMANTICS`](../api/macros/json_brace_init_copy_semantics.md) | `0` | `1`: single-element brace initialization such as `#!cpp json j{obj};` copies the element instead of creating an array | – | 3.13.0 | +| [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) | `0` | `1`: reading from a stream does not consume the character after a number | – | 3.13.0 | +| [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) | `0` | `1`: a NUL byte in the input is a parse error instead of the end of input | [`JSON_StrictNulHandling`](../integration/cmake.md#json_strictnulhandling) | 3.13.0 | +| [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) | `0` | `1`: `to_cbor`, `to_ubjson`, `to_bjdata`, and `to_bson` throw for strings that are not valid UTF-8 by default | [`JSON_StrictBinaryUTF8`](../integration/cmake.md#json_strictbinaryutf8) | 3.13.0 | + +For example, the following makes a 3.x release behave like version 4.0 with respect to these changes: + +```cpp +#define JSON_USE_IMPLICIT_CONVERSIONS 0 +#define JSON_USE_GLOBAL_UDLS 0 +#define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 +#define JSON_BRACE_INIT_COPY_SEMANTICS 1 +#define JSON_PRECISE_STREAM_POSITION 1 +#define JSON_STRICT_NUL_HANDLING 1 +#define JSON_STRICT_BINARY_UTF8 1 +#include +``` + +The macros must be defined before the library header is included; setting them once in the build system is the easiest +way to achieve this. + +### Removal of deprecated functions + +Version 4.0 will remove all deprecated functions. Compiling with deprecation warnings enabled shows which of them your +code still uses. The [migration guide](../integration/migration_guide.md#replace-deprecated-functions) shows how to +replace each of them. + +| Deprecated | Since | Migration | +|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------|----------------------------------------------------------------------------------| +| `#!cpp operator<<(basic_json&, std::istream&)` | 3.0.0 | [Parsing](../integration/migration_guide.md#parsing) | +| `#!cpp operator>>(const basic_json&, std::ostream&)` | 3.0.0 | [Miscellaneous functions](../integration/migration_guide.md#miscellaneous-functions) | +| `iterator_wrapper` | 3.1.0 | [Miscellaneous functions](../integration/migration_guide.md#miscellaneous-functions) | +| [`parse`](../api/basic_json/parse.md), [`accept`](../api/basic_json/accept.md), and [`sax_parse`](../api/basic_json/sax_parse.md) with an initializer list `{ptr, len}` or `{first, last}` | 3.8.0 | [Parsing](../integration/migration_guide.md#parsing) | +| [`from_bson`](../api/basic_json/from_bson.md), [`from_cbor`](../api/basic_json/from_cbor.md), [`from_msgpack`](../api/basic_json/from_msgpack.md), and [`from_ubjson`](../api/basic_json/from_ubjson.md) with `(ptr, len)` or an initializer list | 3.8.0 | [Parsing](../integration/migration_guide.md#parsing) | +| [`json_pointer::operator string_t`](../api/json_pointer/operator_string_t.md) | 3.11.0 | [JSON Pointers](../integration/migration_guide.md#json-pointers) | +| [`json_pointer`](../api/json_pointer/index.md) with a `basic_json` type as template argument, and the overloads of `value`, `contains`, `operator[]`, and `at` accepting such a pointer | 3.11.0 | [JSON Pointers](../integration/migration_guide.md#json-pointers) | +| Comparing a [`json_pointer`](../api/json_pointer/index.md) with a string via [`operator==`](../api/json_pointer/operator_eq.md) or [`operator!=`](../api/json_pointer/operator_ne.md) | 3.11.2 | [JSON Pointers](../integration/migration_guide.md#json-pointers) | + +The deprecated legacy comparison of discarded values is controlled by a macro and therefore listed in the table above. + +New breaking changes will follow the same path: they are added to these tables when they land in a 3.x release. diff --git a/docs/mkdocs/docs/integration/migration_guide.md b/docs/mkdocs/docs/integration/migration_guide.md index 29d54b6df..8a07d48a3 100644 --- a/docs/mkdocs/docs/integration/migration_guide.md +++ b/docs/mkdocs/docs/integration/migration_guide.md @@ -2,11 +2,14 @@ This page collects some guidelines on how to future-proof your code for future versions of this library. For how to add the library to your project in the first place, see [Integration](index.md), [CMake](cmake.md), or -[Package Managers](package_managers.md). +[Package Managers](package_managers.md). The [roadmap](../community/roadmap.md#version-40) lists what will change in +version 4.0, including the macros that let you try its behavior with a 3.x release; this page describes how to adjust +your code. ## Replace deprecated functions -The following functions have been deprecated and will be removed in the next major version (i.e., 4.0.0). All +The following functions have been deprecated and will be removed in the next major version (i.e., 4.0.0), see the +[roadmap](../community/roadmap.md#removal-of-deprecated-functions) for an overview. All deprecations are annotated with [`HEDLEY_DEPRECATED_FOR`](https://nemequ.github.io/hedley/api-reference.html#HEDLEY_DEPRECATED_FOR) to report which function to use instead. From 40021f38fb306fc04b7b52173e8b2f6d1203c27d Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 4 Oct 2026 12:00:31 +0200 Subject: [PATCH 6/8] Add basic_json::as_base_class and document name conflicts with custom base classes (#5589) * Add basic_json::as_base_class and document name conflicts with custom base classes Members of basic_json hide members of a custom base class with the same name, and future releases may add members that hide ones accessible today. Document this in json_base_class_t and add as_base_class() to reach hidden members without spelling out the cast. Also make json_base_class_t a public member type. It was documented since 3.12.0, but declared private, so users could not name it. Supersedes #3899. Co-authored-by: Raphael Grimm <1005058+barcode@users.noreply.github.com> Signed-off-by: Niels Lohmann * Add as_base_class to the docset search index New public members get an entry in docs/docset/docSet.sql (as done for to_bon8/from_bon8 in #2998). Without it, the Dash/Zeal docset built from the documentation cannot find basic_json::as_base_class. Signed-off-by: Niels Lohmann * Silence clang-tidy for the hidden type_name() in the base class test ci_clang_tidy failed with readability-convert-member-functions-to-static on base_class_with_hidden_members::type_name(). It must stay a non-static member: the test shows that it is hidden by the non-static basic_json::type_name() and reachable through as_base_class(). Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann Co-authored-by: Raphael Grimm <1005058+barcode@users.noreply.github.com> --- docs/docset/docSet.sql | 1 + .../docs/api/basic_json/as_base_class.md | 53 ++++++++++++++ docs/mkdocs/docs/api/basic_json/index.md | 1 + .../docs/api/basic_json/json_base_class_t.md | 14 ++++ docs/mkdocs/docs/examples/as_base_class.cpp | 41 +++++++++++ .../mkdocs/docs/examples/as_base_class.output | 2 + docs/mkdocs/mkdocs.yml | 1 + include/nlohmann/json.hpp | 18 ++++- single_include/nlohmann/json.hpp | 18 ++++- tests/src/unit-custom-base-class.cpp | 70 +++++++++++++++++++ 10 files changed, 217 insertions(+), 2 deletions(-) create mode 100644 docs/mkdocs/docs/api/basic_json/as_base_class.md create mode 100644 docs/mkdocs/docs/examples/as_base_class.cpp create mode 100644 docs/mkdocs/docs/examples/as_base_class.output diff --git a/docs/docset/docSet.sql b/docs/docset/docSet.sql index 477b2f9fe..09802e43e 100644 --- a/docs/docset/docSet.sql +++ b/docs/docset/docSet.sql @@ -19,6 +19,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('format_as', 'Function', 'api/ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::accept', 'Function', 'api/basic_json/accept/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::array', 'Function', 'api/basic_json/array/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::array_t', 'Type', 'api/basic_json/array_t/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::as_base_class', 'Method', 'api/basic_json/as_base_class/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::at', 'Method', 'api/basic_json/at/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::back', 'Method', 'api/basic_json/back/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::basic_json', 'Constructor', 'api/basic_json/basic_json/index.html'); diff --git a/docs/mkdocs/docs/api/basic_json/as_base_class.md b/docs/mkdocs/docs/api/basic_json/as_base_class.md new file mode 100644 index 000000000..769707140 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json/as_base_class.md @@ -0,0 +1,53 @@ +# nlohmann::basic_json::as_base_class + +```cpp +json_base_class_t& as_base_class() noexcept; +const json_base_class_t& as_base_class() const noexcept; +``` + +Returns a reference to this object as its custom base class [`json_base_class_t`](json_base_class_t.md). No copy is +made. + +Since `basic_json` derives from `json_base_class_t`, a member of `basic_json` hides any member of the custom base class +with the same name. This function makes such hidden members accessible again. + +## Return value + +reference to this object as [`json_base_class_t`](json_base_class_t.md) + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +The function is equivalent to `static_cast(j)` (or `static_cast(j)`). + +## Examples + +??? example + + The example shows how to use `as_base_class` to access members of the custom base class that are hidden by members + of `basic_json`. + + ```cpp + --8<-- "examples/as_base_class.cpp" + ``` + + Output: + + ```json + --8<-- "examples/as_base_class.output" + ``` + +## See also + +- [json_base_class_t](json_base_class_t.md) - type of the custom base class + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/index.md b/docs/mkdocs/docs/api/basic_json/index.md index 866bda67a..fc934ca93 100644 --- a/docs/mkdocs/docs/api/basic_json/index.md +++ b/docs/mkdocs/docs/api/basic_json/index.md @@ -200,6 +200,7 @@ Direct access to the stored value of a JSON value. - [**get_ref**](get_ref.md) - get a reference value - [**operator ValueType**](operator_ValueType.md) - get a value - [**get_binary**](get_binary.md) - get a binary value +- [**as_base_class**](as_base_class.md) - access the custom base class ### Element access diff --git a/docs/mkdocs/docs/api/basic_json/json_base_class_t.md b/docs/mkdocs/docs/api/basic_json/json_base_class_t.md index 0d1abc9d4..6f0026558 100644 --- a/docs/mkdocs/docs/api/basic_json/json_base_class_t.md +++ b/docs/mkdocs/docs/api/basic_json/json_base_class_t.md @@ -27,6 +27,18 @@ A `CustomBaseClass` with non-static data members forfeits `basic_json`'s [standard layout](https://en.cppreference.com/w/cpp/named_req/StandardLayoutType) guarantee. See [Template Parameter Requirements](../../features/types/template_parameters.md#custombaseclass). +#### Name conflicts + +Since `basic_json` derives from `CustomBaseClass`, members of `basic_json` hide members of `CustomBaseClass` with the +same name. Hidden members remain accessible via [`as_base_class`](as_base_class.md) or by casting the value to +`json_base_class_t`. + +!!! warning "Avoid generic member names" + + Future versions of the library may add members to `basic_json` that hide members of `CustomBaseClass` that are + accessible today. To reduce the risk of such conflicts, avoid generic names for the members of `CustomBaseClass`, + for instance by using a distinctive prefix. + ## Examples ??? example @@ -45,8 +57,10 @@ A `CustomBaseClass` with non-static data members forfeits `basic_json`'s ## See also +- [as_base_class](as_base_class.md) - access the custom base class - [Template Parameter Requirements](../../features/types/template_parameters.md#custombaseclass) - the requirements for `CustomBaseClass` ## Version history - Added in version 3.12.0. +- Made a public member type in version 3.13.0; it was private before, so it could not be named outside the class. diff --git a/docs/mkdocs/docs/examples/as_base_class.cpp b/docs/mkdocs/docs/examples/as_base_class.cpp new file mode 100644 index 000000000..48357b045 --- /dev/null +++ b/docs/mkdocs/docs/examples/as_base_class.cpp @@ -0,0 +1,41 @@ +#include +#include + +class base_class_with_hidden_members +{ + public: + const char* type_name() const noexcept + { + return "my_type_name"; + } + + std::size_t size() const noexcept + { + return 42; + } +}; + +using json = nlohmann::basic_json < + std::map, + std::vector, + std::string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer, + std::vector, + base_class_with_hidden_members + >; + +int main() +{ + json j = {1, 2, 3}; + + // the members of basic_json hide the members of the base class + std::cout << j.type_name() << ' ' << j.size() << '\n'; + + // access the hidden members of the base class + std::cout << j.as_base_class().type_name() << ' ' << j.as_base_class().size() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/as_base_class.output b/docs/mkdocs/docs/examples/as_base_class.output new file mode 100644 index 000000000..5ca62a673 --- /dev/null +++ b/docs/mkdocs/docs/examples/as_base_class.output @@ -0,0 +1,2 @@ +array 3 +my_type_name 42 diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index b4daf34c3..7b5512084 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -116,6 +116,7 @@ nav: - 'accept': api/basic_json/accept.md - 'array': api/basic_json/array.md - 'array_t': api/basic_json/array_t.md + - 'as_base_class': api/basic_json/as_base_class.md - 'at': api/basic_json/at.md - 'back': api/basic_json/back.md - 'begin': api/basic_json/begin.md diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 89ba13233..07b625a89 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -162,7 +162,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// workaround type for MSVC using basic_json_t = NLOHMANN_BASIC_JSON_TPL; - using json_base_class_t = ::nlohmann::detail::json_base_class; JSON_PRIVATE_UNLESS_TESTED: // convenience aliases for types residing in namespace detail; @@ -216,6 +215,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec using cbor_tag_handler_t = detail::cbor_tag_handler_t; /// how to encode BJData using bjdata_version_t = detail::bjdata_version_t; + /// base class used to inject custom functionality into each instance of basic_json + /// @sa https://json.nlohmann.me/api/basic_json/json_base_class_t/ + using json_base_class_t = ::nlohmann::detail::json_base_class; /// helper type for initializer lists of basic_json values using initializer_list_t = std::initializer_list>; @@ -3023,6 +3025,20 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return *get_ptr(); } + /// @brief access the custom base class + /// @sa https://json.nlohmann.me/api/basic_json/as_base_class/ + json_base_class_t& as_base_class() noexcept + { + return static_cast(*this); + } + + /// @brief access the custom base class + /// @sa https://json.nlohmann.me/api/basic_json/as_base_class/ + const json_base_class_t& as_base_class() const noexcept + { + return static_cast(*this); + } + /// @} private: diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 3b74fd67f..16e87cc86 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -26699,7 +26699,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// workaround type for MSVC using basic_json_t = NLOHMANN_BASIC_JSON_TPL; - using json_base_class_t = ::nlohmann::detail::json_base_class; JSON_PRIVATE_UNLESS_TESTED: // convenience aliases for types residing in namespace detail; @@ -26753,6 +26752,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec using cbor_tag_handler_t = detail::cbor_tag_handler_t; /// how to encode BJData using bjdata_version_t = detail::bjdata_version_t; + /// base class used to inject custom functionality into each instance of basic_json + /// @sa https://json.nlohmann.me/api/basic_json/json_base_class_t/ + using json_base_class_t = ::nlohmann::detail::json_base_class; /// helper type for initializer lists of basic_json values using initializer_list_t = std::initializer_list>; @@ -29560,6 +29562,20 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return *get_ptr(); } + /// @brief access the custom base class + /// @sa https://json.nlohmann.me/api/basic_json/as_base_class/ + json_base_class_t& as_base_class() noexcept + { + return static_cast(*this); + } + + /// @brief access the custom base class + /// @sa https://json.nlohmann.me/api/basic_json/as_base_class/ + const json_base_class_t& as_base_class() const noexcept + { + return static_cast(*this); + } + /// @} private: diff --git a/tests/src/unit-custom-base-class.cpp b/tests/src/unit-custom-base-class.cpp index a6b9b9ea4..38b665793 100644 --- a/tests/src/unit-custom-base-class.cpp +++ b/tests/src/unit-custom-base-class.cpp @@ -10,6 +10,8 @@ #include #include #include +#include +#include #include #include "doctest_compatibility.h" @@ -406,6 +408,74 @@ TEST_CASE("JSON Visit Node") CHECK(expected.empty()); } +// Test accessing members of a custom base class that are hidden by members of nlohmann::basic_json +class base_class_with_hidden_members +{ + public: + const char* type_name() const noexcept // NOLINT(readability-convert-member-functions-to-static) + { + return "custom type_name"; + } + + std::size_t size() const noexcept + { + return m_size; + } + + std::size_t m_size = 42; +}; + +using json_with_hidden_base_members = + nlohmann::basic_json < + std::map, + std::vector, + std::string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer, + std::vector, + base_class_with_hidden_members + >; + +TEST_CASE("JSON Node as_base_class") +{ + using json = json_with_hidden_base_members; + + static_assert(std::is_same().as_base_class()), json::json_base_class_t&>::value, ""); + static_assert(std::is_same().as_base_class()), const json::json_base_class_t&>::value, ""); + static_assert(noexcept(std::declval().as_base_class()), ""); + static_assert(noexcept(std::declval().as_base_class()), ""); + + SECTION("non-const") + { + json j = {1, 2, 3}; + + CHECK(std::string(j.type_name()) == "array"); + CHECK(j.size() == 3); + CHECK(std::string(j.as_base_class().type_name()) == "custom type_name"); + CHECK(j.as_base_class().size() == 42); + CHECK(&j.as_base_class() == &static_cast(j)); + + j.as_base_class().m_size = 7; + CHECK(j.as_base_class().size() == 7); + CHECK(j.size() == 3); + } + + SECTION("const") + { + const json j = {1, 2, 3}; + + CHECK(std::string(j.type_name()) == "array"); + CHECK(j.size() == 3); + CHECK(std::string(j.as_base_class().type_name()) == "custom type_name"); + CHECK(j.as_base_class().size() == 42); + CHECK(&j.as_base_class() == &static_cast(j)); + } +} + // A custom base class with a const member: copy-constructible (initializing a // const member works fine), but not copy-/move-assignable (assigning one does // not). Used to check that copy construction never requires more than that. From f56b418c56dcbf7c05fe6cca72e2a4de2d072b99 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 4 Oct 2026 12:13:48 +0200 Subject: [PATCH 7/8] Follow each binary format's UTF-8 rule: strict writers (CBOR/UBJSON/BJData/BSON), lenient readers (#5741) Signed-off-by: Niels Lohmann --- CMakeLists.txt | 6 + cmake/ci.cmake | 2 +- docs/mkdocs/docs/api/basic_json/to_bjdata.md | 7 +- docs/mkdocs/docs/api/basic_json/to_bson.md | 6 + docs/mkdocs/docs/api/basic_json/to_cbor.md | 8 ++ docs/mkdocs/docs/api/basic_json/to_ubjson.md | 5 + docs/mkdocs/docs/api/macros/index.md | 2 + .../api/macros/json_strict_binary_utf8.md | 97 +++++++++++++++ .../docs/features/binary_formats/bjdata.md | 16 +++ .../docs/features/binary_formats/bson.md | 16 +-- .../docs/features/binary_formats/cbor.md | 17 +-- .../features/binary_formats/messagepack.md | 15 +-- .../docs/features/binary_formats/ubjson.md | 16 +++ docs/mkdocs/docs/features/macros.md | 14 +++ docs/mkdocs/docs/features/namespace.md | 1 + docs/mkdocs/docs/home/exceptions.md | 13 ++- docs/mkdocs/docs/integration/cmake.md | 5 + docs/mkdocs/mkdocs.yml | 1 + include/nlohmann/detail/abi_macros.hpp | 19 ++- .../nlohmann/detail/input/binary_reader.hpp | 29 ++--- include/nlohmann/detail/macro_unscope.hpp | 1 + .../nlohmann/detail/output/binary_writer.hpp | 70 ++++++++++- include/nlohmann/detail/string_utils.hpp | 44 +------ single_include/nlohmann/json_fwd.hpp | 19 ++- tests/abi/config/default.cpp | 4 + tests/abi/config/noversion.cpp | 4 + tests/src/unit-binary_utf8_strict.cpp | 110 ++++++++++++++++++ tests/src/unit-bjdata.cpp | 37 ++++++ tests/src/unit-bson.cpp | 37 ++++++ tests/src/unit-cbor.cpp | 85 +++++++++++--- tests/src/unit-msgpack.cpp | 36 ++++-- tests/src/unit-ubjson.cpp | 37 ++++++ 32 files changed, 651 insertions(+), 128 deletions(-) create mode 100644 docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md create mode 100644 tests/src/unit-binary_utf8_strict.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index f0a6771dc..b5092ae49 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -61,6 +61,7 @@ option(JSON_Install "Install CMake targets during install option(JSON_MultipleHeaders "Use non-amalgamated version of the library." ON) option(JSON_SystemInclude "Include as system headers (skip for clang-tidy)." OFF) option(JSON_StrictNulHandling "Build with strict NUL-byte handling enabled." OFF) +option(JSON_StrictBinaryUTF8 "Build with UTF-8 checks in the CBOR, UBJSON, BJData, and BSON writers enabled." OFF) if (JSON_CI) include(ci) @@ -118,6 +119,10 @@ if (JSON_StrictNulHandling) message(STATUS "Strict NUL-byte handling enabled (JSON_STRICT_NUL_HANDLING=1)") endif() +if (JSON_StrictBinaryUTF8) + message(STATUS "Strict UTF-8 checks in binary writers enabled (JSON_STRICT_BINARY_UTF8=1)") +endif() + if (JSON_Diagnostic_Positions) message(STATUS "Diagnostic positions enabled (JSON_DIAGNOSTIC_POSITIONS=1)") endif() @@ -153,6 +158,7 @@ target_compile_definitions( $<$:JSON_DIAGNOSTIC_POSITIONS=1> $<$:JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1> $<$:JSON_STRICT_NUL_HANDLING=1> + $<$:JSON_STRICT_BINARY_UTF8=1> ) target_include_directories( diff --git a/cmake/ci.cmake b/cmake/ci.cmake index e67aba6d3..2aceb5afd 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -701,7 +701,7 @@ ci_get_cmake(4.0.0 CMAKE_4_0_0_BINARY) # the tests require CMake 3.13 or later, so they are excluded for CMake 3.5.0 set(JSON_CMAKE_FLAGS_3_5_0 JSON_Diagnostics JSON_Diagnostic_Positions JSON_GlobalUDLs JSON_ImplicitConversions JSON_DisableEnumSerialization JSON_LegacyDiscardedValueComparison JSON_Install JSON_MultipleHeaders JSON_SystemInclude JSON_Valgrind - JSON_StrictNulHandling) + JSON_StrictNulHandling JSON_StrictBinaryUTF8) set(JSON_CMAKE_FLAGS_3_31_6 JSON_BuildTests ${JSON_CMAKE_FLAGS_3_5_0}) set(JSON_CMAKE_FLAGS_4_0_0 JSON_BuildTests ${JSON_CMAKE_FLAGS_3_5_0}) diff --git a/docs/mkdocs/docs/api/basic_json/to_bjdata.md b/docs/mkdocs/docs/api/basic_json/to_bjdata.md index 44cc399e1..63dc5379e 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/to_bjdata.md @@ -56,6 +56,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false, and `j` contains a non-empty array, object, or binary value. +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not + valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged ## Complexity @@ -104,4 +107,6 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.11.0. -- BJData version parameter (for draft3 binary encoding) added in version 3.12.0. \ No newline at end of file +- BJData version parameter (for draft3 binary encoding) added in version 3.12.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. \ No newline at end of file diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index c79f39ffc..e3104b13c 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -46,6 +46,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the BSON binary subtype; example: `"subtype 70000 is too large for the BSON binary subtype (max 255)"` +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is not valid + UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged ## Complexity @@ -98,3 +101,6 @@ pass before anything is written. - Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. - Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0. - `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0. +- Throwing `type_error.316` for a string value or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, detected before anything is written, + added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/to_cbor.md b/docs/mkdocs/docs/api/basic_json/to_cbor.md index 3bbd9c7d3..cac8a0917 100644 --- a/docs/mkdocs/docs/api/basic_json/to_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/to_cbor.md @@ -35,6 +35,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ Strong guarantee: if an exception is thrown, there are no changes in the JSON value. +## Exceptions + +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not + valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged + ## Complexity Linear in the size of the JSON value `j`. @@ -68,3 +74,5 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Compact representation of floating-point numbers added in version 3.8.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/to_ubjson.md b/docs/mkdocs/docs/api/basic_json/to_ubjson.md index 1b7f7767e..8f2ea5eb9 100644 --- a/docs/mkdocs/docs/api/basic_json/to_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/to_ubjson.md @@ -49,6 +49,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false, and `j` contains a non-empty array, object, or binary value. +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not + valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged ## Complexity @@ -97,3 +100,5 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.1.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index 3c9b42af2..872d6cb8c 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -18,6 +18,8 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_PRECISE_STREAM_POSITION**](json_precise_stream_position.md) - opt in to leaving an input stream positioned right after a parsed number +- [**JSON_STRICT_BINARY_UTF8**](json_strict_binary_utf8.md) - opt in to checking strings for valid UTF-8 in the CBOR, + UBJSON, BJData, and BSON writers - [**JSON_STRICT_NUL_HANDLING**](json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of treating it as end of input diff --git a/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md new file mode 100644 index 000000000..7fe6d9a58 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md @@ -0,0 +1,97 @@ +# JSON_STRICT_BINARY_UTF8 + +```cpp +#define JSON_STRICT_BINARY_UTF8 /* value */ +``` + +When defined to `1`, the binary writers [`to_cbor`](../basic_json/to_cbor.md), [`to_ubjson`](../basic_json/to_ubjson.md), +[`to_bjdata`](../basic_json/to_bjdata.md), and [`to_bson`](../basic_json/to_bson.md) check every string value and +object key for valid UTF-8 and throw [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for +ill-formed UTF-8, like [`dump`](../basic_json/dump.md) does. Without it, they write the bytes unchanged. + +The macro does not affect: + +- [`to_msgpack`](../basic_json/to_msgpack.md): the MessagePack specification allows a `str` value to contain bytes that + are not valid UTF-8, so it always writes them unchanged. +- [`to_bon8`](../basic_json/to_bon8.md): BON8 always checks, because the UTF-8 lead bytes mark where a string ends. +- The binary readers ([`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md), + [`from_ubjson`](../basic_json/from_ubjson.md), [`from_bjdata`](../basic_json/from_bjdata.md), + [`from_bson`](../basic_json/from_bson.md)): none of these formats requires a decoder to reject ill-formed UTF-8, so + they always return the bytes unchanged. + +## Default definition + +The default value is `0` (disabled, the behavior of version 3.12.0 and earlier is preserved). + +```cpp +#define JSON_STRICT_BINARY_UTF8 0 +``` + +## Notes + +!!! note "Background" + + CBOR, UBJSON, BJData, and BSON all require strings to be UTF-8. Up to version 3.12.0, the writers did not check + this, so they could produce output that other decoders reject. Checking by default would break code that stores + other encodings (for instance ISO 8859-1) in a string and only ever writes it to a binary format, so this macro + offers the check as an opt-in ahead of version 4.0.0, where it is planned to become the default (see + [#5529](https://github.com/nlohmann/json/issues/5529) and [#5651](https://github.com/nlohmann/json/issues/5651)). + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no + effect. + +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_sbu8`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + +## Examples + +??? example "Default behavior (macro not defined)" + + Without the macro, the bytes are written unchanged: + + ```cpp + #include + + using json = nlohmann::json; + + int main() + { + auto v = json::to_cbor(json("\xFF")); + // v is {0x61, 0xFF} + } + ``` + +??? example "Opt-in check (macro defined to 1)" + + With the macro, ill-formed UTF-8 is rejected: + + ```cpp + #define JSON_STRICT_BINARY_UTF8 1 + #include + + using json = nlohmann::json; + + int main() + { + auto v = json::to_cbor(json("\xFF")); + // throws type_error.316: invalid UTF-8 byte at index 0: 0xFF + } + ``` + +## See also + +- [**to_cbor**](../basic_json/to_cbor.md) - create a CBOR serialization of a JSON value +- [**to_ubjson**](../basic_json/to_ubjson.md) - create a UBJSON serialization of a JSON value +- [**to_bjdata**](../basic_json/to_bjdata.md) - create a BJData serialization of a JSON value +- [**to_bson**](../basic_json/to_bson.md) - create a BSON serialization of a JSON value +- [**error_handler_t**](../basic_json/error_handler_t.md) - how [`dump`](../basic_json/dump.md) treats ill-formed UTF-8 + +## Version history + +- Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index 8e3ed41f6..12d1af51b 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -63,6 +63,13 @@ The library uses the following mapping from JSON values types to BJData types ac - strings with more than 18446744073709551615 bytes, i.e., 264-1 bytes (theoretical) +!!! warning "UTF-8 validation of string values and object keys" + + BJData strings must use UTF-8 encoding. By default, `to_bjdata()` writes the bytes of string values and object keys + unchanged, even if they are not valid UTF-8. If + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead. + !!! info "Unused BJData markers" The following markers are not used in the conversion: @@ -208,6 +215,15 @@ The library maps BJData types to JSON value types as follows: The mapping is **complete** in the sense that any BJData value can be converted to a JSON value. +!!! warning "Ill-formed UTF-8 in string values and object keys" + + BJData strings must use UTF-8 encoding, but this is not enforced on read: `from_bjdata()` accepts a string + value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error + handler is passed that replaces or ignores the ill-formed bytes. By default, `to_bjdata()` writes such a value + back unchanged (see above). + !!! info "Round trips" A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with diff --git a/docs/mkdocs/docs/features/binary_formats/bson.md b/docs/mkdocs/docs/features/binary_formats/bson.md index cca11451e..245ed3ebf 100644 --- a/docs/mkdocs/docs/features/binary_formats/bson.md +++ b/docs/mkdocs/docs/features/binary_formats/bson.md @@ -109,14 +109,16 @@ The library maps BSON record types to JSON value types as follows: If BSON input must be validated for strict specification compliance, validate it separately before passing it to `from_bson()`. -!!! warning "UTF-8 validation of string values" +!!! warning "Ill-formed UTF-8 in string values" - The BSON specification requires `string` values (type `0x02`) to be valid UTF-8. This library validates the - bytes of every such string at decode time and rejects ill-formed UTF-8 with a - [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions` - set to `false`, a discarded value), rather than only failing later when the resulting value is dumped. Element - (key) names and `binary` values (type `0x05`) are unaffected and are never validated, since they are read - byte-by-byte as a C string, or are not required to hold text, respectively. + The BSON specification requires `string` values (type `0x02`) to be valid UTF-8, but this is not required of a + decoder. `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back unchanged. + However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error handler is + passed that replaces or ignores the ill-formed bytes. By default, `to_bson()` writes such a string value or element + (key) name unchanged; if [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it + throws the same exception instead. Element (key) names are never validated on read, since they are read byte-by-byte + as a C string. `binary` values (type `0x05`) are unaffected, since they are not required to hold text. ??? example "Example: deserialize a JSON value from BSON" diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index eb7bc7f41..66300c5fa 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -189,15 +189,16 @@ The library maps CBOR types to JSON value types as follows: ([RFC 8392](https://www.rfc-editor.org/rfc/rfc8392.html)), cannot be read with this library and need a general-purpose CBOR library instead. -!!! warning "UTF-8 validation of text strings" +!!! warning "Ill-formed UTF-8 in text strings" - [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings - (major type 3) to be valid UTF-8. This library validates the bytes of every text string (object keys included) at - decode time and rejects ill-formed UTF-8 with a - [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with - `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value is - dumped. Byte strings (major type 2) are unaffected and are never validated, since they are not required to hold - text. + [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings (major + type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this. This library does not: + `from_cbor()` accepts a text string (object keys included) whose bytes are not valid UTF-8 and hands them back + unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error handler is + passed that replaces or ignores the ill-formed bytes. By default, `to_cbor()` writes such a value back unchanged; if + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws the same exception + instead. Byte strings (major type 2) are unaffected, since they are not required to hold text. !!! warning "Tagged items" diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index 2e674252a..047944852 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -153,14 +153,15 @@ The library maps MessagePack types to JSON value types as follows: This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed on. Such input needs a general-purpose MessagePack library instead. -!!! warning "UTF-8 validation of string values" +!!! warning "Ill-formed UTF-8 in string values" - The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8. - This library validates the bytes of every such string (object keys included) at decode time and rejects - ill-formed UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, - with `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting - value is dumped. `bin`/`ext`/`fixext` values are unaffected and are never validated, since they are not required - to hold text. + The MessagePack specification explicitly allows a `str` value (`fixstr`, `str 8`, `str 16`, `str 32`) to contain + a byte sequence that is not valid UTF-8, and expects a deserializer to hand the original bytes back unchanged. + This library follows that: `from_msgpack()` reads `str` bytes (object keys included) as-is, without validating + them, and `to_msgpack()` writes them back as-is, so such a value round-trips through `from_msgpack(to_msgpack(j))` + byte for byte. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way, unless an + error handler is passed that replaces or ignores the ill-formed bytes. ??? example "Example: deserialize a JSON value from MessagePack" diff --git a/docs/mkdocs/docs/features/binary_formats/ubjson.md b/docs/mkdocs/docs/features/binary_formats/ubjson.md index 37aa069e2..dbdff6e6c 100644 --- a/docs/mkdocs/docs/features/binary_formats/ubjson.md +++ b/docs/mkdocs/docs/features/binary_formats/ubjson.md @@ -47,6 +47,13 @@ The library uses the following mapping from JSON values types to UBJSON types ac - strings with more than 9223372036854775807 bytes (theoretical) +!!! warning "UTF-8 validation of string values and object keys" + + UBJSON's required string encoding is UTF-8. By default, `to_ubjson()` writes the bytes of string values and object + keys unchanged, even if they are not valid UTF-8. If + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead. + !!! info "Unused UBJSON markers" The following markers are not used in the conversion: @@ -120,6 +127,15 @@ The library maps UBJSON types to JSON value types as follows: The mapping is **complete** in the sense that any UBJSON value can be converted to a JSON value. +!!! warning "Ill-formed UTF-8 in string values and object keys" + + UBJSON's required string encoding is UTF-8, but this is not enforced on read: `from_ubjson()` accepts a string + value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error + handler is passed that replaces or ignores the ill-formed bytes. By default, `to_ubjson()` writes such a value + back unchanged (see above). + ??? example "Example: deserialize a JSON value from UBJSON" ```cpp diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index 90bb352c3..681980315 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -138,6 +138,20 @@ using the library with compilers that do not fully support C++11 and may only wo See [full documentation of `JSON_SKIP_UNSUPPORTED_COMPILER_CHECK`](../api/macros/json_skip_unsupported_compiler_check.md). +## `JSON_STRICT_BINARY_UTF8` + +When defined to `1`, [`to_cbor`](../api/basic_json/to_cbor.md), [`to_ubjson`](../api/basic_json/to_ubjson.md), +[`to_bjdata`](../api/basic_json/to_bjdata.md), and [`to_bson`](../api/basic_json/to_bson.md) throw +[`type_error.316`](../home/exceptions.md#jsonexceptiontype_error316) for a string value or object key that is not +valid UTF-8. The default value is `0`, which writes the bytes unchanged as before version 3.13.0; this is planned to +become the default in version 4.0.0. + +The check can also be enabled with the CMake option +[`JSON_StrictBinaryUTF8`](../integration/cmake.md#json_strictbinaryutf8) (`OFF` by default) which sets +`JSON_STRICT_BINARY_UTF8` accordingly. + +See [full documentation of `JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md). + ## `JSON_STRICT_NUL_HANDLING` When defined to `1`, a `'\0'` (NUL) byte anywhere in the input is rejected with `parse_error.101`, like any other diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 577f5e221..dbddac13d 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -20,6 +20,7 @@ The complete default namespace name is derived as follows: `_bics`. - [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) defined non-zero appends `_psp`. - [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) defined non-zero appends `_snul`. + - [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) defined non-zero appends `_sbu8`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index c78aeaa68..a3797cf04 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -340,8 +340,9 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde ### json.exception.parse_error.113 A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a -string was read where one was required (for instance as a map key), the string's length specification is invalid, or -the string's bytes are not valid UTF-8. +string was read where one was required (for instance as a map key), or the string's length specification is invalid. +The bytes of a string itself are not checked for valid UTF-8 on read; see the ill-formed UTF-8 notes on the +individual [binary format](../features/binary_formats/index.md) pages for how such a string is handled afterward. CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other type (for instance integers or `null`) are therefore not supported; see the notes on @@ -364,9 +365,6 @@ type (for instance integers or `null`) are therefore not supported; see the note ``` [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData string: string length must not be negative ``` - ``` - [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte - ``` ### json.exception.parse_error.114 @@ -749,6 +747,11 @@ The [`unflatten()`](../api/basic_json/unflatten.md) function only works for an o The [`dump()`](../api/basic_json/dump.md) function only works with UTF-8 encoded strings; that is, if you assign a `std::string` to a JSON value, make sure it is UTF-8 encoded. See the FAQ entry on [serializing untrusted or invalid UTF-8](faq.md#serializing-untrusted-or-invalid-utf-8) for background and the recommended fix. +If [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled, the binary writers +[`to_cbor()`](../api/basic_json/to_cbor.md), [`to_ubjson()`](../api/basic_json/to_ubjson.md), +[`to_bjdata()`](../api/basic_json/to_bjdata.md), and [`to_bson()`](../api/basic_json/to_bson.md) throw this exception +for a string value or object key that is not valid UTF-8 as well. + !!! failure "Example message" Calling `dump()` on a JSON value containing an ISO 8859-1 encoded string: diff --git a/docs/mkdocs/docs/integration/cmake.md b/docs/mkdocs/docs/integration/cmake.md index 785fc3ed6..9151d12be 100644 --- a/docs/mkdocs/docs/integration/cmake.md +++ b/docs/mkdocs/docs/integration/cmake.md @@ -212,6 +212,11 @@ Use the non-amalgamated version of the library. This option is `ON` by default. Treat the library headers like system headers (i.e., adding `SYSTEM` to the [`target_include_directories`](https://cmake.org/cmake/help/latest/command/target_include_directories.html) call) to check for this library by tools like Clang-Tidy. This option is `OFF` by default. +### `JSON_StrictBinaryUTF8` + +Check string values and object keys for valid UTF-8 in the CBOR, UBJSON, BJData, and BSON writers, by defining the +macro [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md). This option is `OFF` by default. + ### `JSON_StrictNulHandling` Reject a `'\0'` (NUL) byte in the input instead of treating it as end of input, by defining the macro diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 7b5512084..3cb25ba9b 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -307,6 +307,7 @@ nav: - 'JSON_PRECISE_STREAM_POSITION': api/macros/json_precise_stream_position.md - 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md - 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md + - 'JSON_STRICT_BINARY_UTF8': api/macros/json_strict_binary_utf8.md - 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md - 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md - 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index 0bace616a..0153c8706 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -46,6 +46,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -82,14 +86,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -98,7 +108,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 86f00bef6..3d79ad41b 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -4044,28 +4044,13 @@ class binary_reader const NumberType len, string_t& result) { - // get_bytes() appends to result, and CBOR indefinite-length strings - // collect all their chunks in the same result; validating only the - // newly read bytes keeps the check linear in the input size - const std::size_t old_size = result.size(); - if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) - { - return false; - } - - // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications - // all require text strings to be valid UTF-8; reject anything else - // right here so malformed input is caught at decode time instead of - // only surfacing later as a type_error.316 when the value is dumped - // (which would defeat allow_exceptions=false / strict discarding). - if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) - { - return sax->parse_error(chars_read, get_token_string(), - parse_error::create(113, chars_read, - exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)); - } - - return true; + // Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the + // choice to the decoder), MessagePack (whose spec explicitly allows + // a str object to contain an invalid byte sequence), UBJSON, BJData, + // or BSON requires a decoder to reject ill-formed UTF-8. The bytes + // are kept unchanged; dump() and the binary writers are the ones + // that check them and report type_error.316 if they are not valid. + return get_bytes(format, len, "string", result); } /*! diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index 8e1d49842..55b2ac99e 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -42,6 +42,7 @@ #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION #undef JSON_STRICT_NUL_HANDLING + #undef JSON_STRICT_BINARY_UTF8 #endif #include diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 9d844d85e..da5aa0cf4 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -115,6 +115,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -145,6 +147,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -211,6 +215,8 @@ class binary_writer case value_t::string: { + check_text_utf8(*j.m_data.m_value.string, j); + // step 1: write control byte and the string length write_cbor_head(0x60, j.m_data.m_value.string->size()); @@ -287,6 +293,11 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // el.first is checked here, against the object as + // diagnostics context, because write_cbor(el.first) + // converts it to a temporary basic_json that would be + // used as the context instead + check_text_utf8(el.first, j); write_cbor(el.first); write_cbor(el.second); } @@ -629,6 +640,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -678,6 +691,8 @@ class binary_writer case value_t::string: { + check_text_utf8(*j.m_data.m_value.string, j); + if (add_prefix) { oa.write_character(to_char_type('S')); @@ -840,6 +855,7 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { + check_text_utf8(el.first, j); write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(el.first.data()), @@ -884,6 +900,10 @@ class binary_writer /*! @return The size of a BSON document entry header, including the id marker and the entry name size (and its null-terminator). + @throw out_of_range.409 if @a name contains U+0000, before anything is + written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a name is + not valid UTF-8, before anything is written */ static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { @@ -893,7 +913,8 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - static_cast(j); + check_text_utf8(name, j); + return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; } @@ -949,9 +970,21 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a value + is not valid UTF-8, before anything is written + + @note The UTF-8 check is skipped if @a value is already too long for the + 32-bit BSON length field (@ref to_bson_length rejects it later, once + the size of the whole document is known); this also keeps the check + from reading past a StringType that reports a size larger than what + it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value) + static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { + if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) + { + check_text_utf8(value, j); + } return sizeof(std::int32_t) + value.size() + 1ul; } @@ -1080,6 +1113,8 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a j is a + string that is not valid UTF-8, before anything is written */ static std::size_t calc_bson_value_size(const BasicJsonType& j) { @@ -1101,7 +1136,7 @@ class binary_writer return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string, j); case value_t::null: return 0ul; @@ -1214,6 +1249,8 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or a key is not valid UTF-8, before anything is written */ static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { @@ -2092,7 +2129,7 @@ class binary_writer */ void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) { - check_bon8_utf8(s, context); + check_utf8(s, context); // a string that follows another string terminates it if (string_open) @@ -2122,7 +2159,7 @@ class binary_writer @throw type_error.316 if @a s is not valid UTF-8; the message names the first byte of the first invalid or incomplete sequence */ - static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + static void check_utf8(const string_t& s, const BasicJsonType& context) { static_cast(context); // only used when exceptions are enabled const auto* data = reinterpret_cast(s.data()); @@ -2133,6 +2170,29 @@ class binary_writer } } + /*! + @brief check a CBOR, UBJSON, BJData, or BSON text string for valid UTF-8 + + The check only happens if JSON_STRICT_BINARY_UTF8 is enabled. Otherwise, + the bytes are written unchanged, as before version 3.13.0. MessagePack + always writes the bytes as is, and BON8 always checks them (see + @ref check_utf8). + + @param[in] s the string to check + @param[in] context the value that holds @a s (for diagnostics) + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a s is + not valid UTF-8 + */ + static void check_text_utf8(const string_t& s, const BasicJsonType& context) + { +#if JSON_STRICT_BINARY_UTF8 + check_utf8(s, context); +#else + static_cast(s); + static_cast(context); +#endif + } + /*! @brief write an integer in the shortest encoding diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index 7c40f7395..2b6864d0d 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -117,13 +117,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally written by Björn Hoehrmann. See http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. -The library checks UTF-8 well-formedness (RFC 3629, section 4) in four +The library checks UTF-8 well-formedness (RFC 3629, section 4) in three places, which differ in speed, diagnostics, and how they read the input: -- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in - strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR, - MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in - text strings at decode time). +- decode() below: the serializer, to escape and, in strict mode, reject + ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON, + UBJSON and BJData readers do not use it: none of those specs requires a + decoder to reject ill-formed UTF-8 in text strings, so the readers keep + the bytes as is and leave the check to dump() and the binary writers. - the per-lead-byte switch in lexer::scan_string(): JSON text, with a diagnostic for each kind of error. - validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's @@ -178,38 +179,5 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std: return state; } -/*! -@brief check whether a string consists solely of valid UTF-8 - -Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text -strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the -MessagePack/BSON specifications all require text strings to be UTF-8), so -that malformed input is caught immediately instead of only surfacing later -as a type_error.316 when the resulting value is dumped. - -@param[in] s the string to check -@param[in] first index of the first byte to check; the bytes before it are - assumed to have been validated already and to end on a - code point boundary -@return whether @a s (from index @a first on) is valid UTF-8 -*/ -template -inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept -{ - std::uint8_t state = UTF8_ACCEPT; - std::uint32_t codepoint = 0; - - for (std::size_t i = first; i < s.size(); ++i) - { - decode(state, codepoint, static_cast(s[i])); - if (state == UTF8_REJECT) - { - return false; - } - } - - return state == UTF8_ACCEPT; -} - } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index 3ee7afa73..7cadc9b1c 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -63,6 +63,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -99,14 +103,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -115,7 +125,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index 879322dd0..e4c627060 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -44,6 +44,10 @@ TEST_CASE("default namespace") expected += "_snul"; #endif +#if JSON_STRICT_BINARY_UTF8 + expected += "_sbu8"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 4b1eb6ee4..858964695 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -45,6 +45,10 @@ TEST_CASE("default namespace without version component") expected += "_snul"; #endif +#if JSON_STRICT_BINARY_UTF8 + expected += "_sbu8"; +#endif + expected += "::basic_json"; // fallback for Clang diff --git a/tests/src/unit-binary_utf8_strict.cpp b/tests/src/unit-binary_utf8_strict.cpp new file mode 100644 index 000000000..c83b5938c --- /dev/null +++ b/tests/src/unit-binary_utf8_strict.cpp @@ -0,0 +1,110 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// The binary writers check strings and object keys for valid UTF-8 only if +// JSON_STRICT_BINARY_UTF8 is enabled (planned to be the default in 4.0.0). +// Without it, they write the bytes unchanged, as before version 3.13.0; the +// tests for that are next to the other tests of each format. +#ifdef JSON_STRICT_BINARY_UTF8 + #undef JSON_STRICT_BINARY_UTF8 +#endif + +#define JSON_STRICT_BINARY_UTF8 1 + +#include +using nlohmann::json; + +#include +#include + +TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)") +{ + SECTION("CBOR") + { + // a string value with ill-formed UTF-8 is rejected + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + + // binary values are not text and are unaffected + CHECK_NOTHROW(json::to_cbor(json::binary(std::vector({0xFF})))); + + // a value read back from CBOR with ill-formed bytes cannot be written + // back either (the reader is lenient regardless of the macro) + const json j = json::from_cbor(std::vector({0x62, 0xc0, 0xae})); + CHECK_THROWS_WITH_AS(json::to_cbor(j), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + } + + SECTION("UBJSON") + { + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("BJData") + { + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("BSON") + { + // to_bson() rejects the same kind of ill-formed string value, before + // any bytes reach the output adapter (the BSON document length + // prefix must be known up front, so nothing is written incrementally) + std::vector out{0x42}; // a sentinel byte the writer must not touch + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + CHECK(out == std::vector {0x42}); + + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected as well; unlike + // the reader (which never validates element names), the writer + // checks both string values and object keys + CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("MessagePack and BON8 are unaffected") + { + // MessagePack allows any bytes in a str, so to_msgpack() writes them as + // is; BON8 always checks, because the lead bytes mark where strings end + CHECK(json::to_msgpack(json("\xFF")) == std::vector({0xa1, 0xff})); + CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&); + } +} diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index b6be66c9e..f34df1717 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -3906,6 +3906,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK(json::to_bjdata(j) == v); CHECK(json::from_bjdata(v) == j); } + + SECTION("ill-formed UTF-8 (see #5529, #5651)") + { + // none of the binary format specs requires a decoder to reject + // ill-formed UTF-8 in a text string, so a value whose bytes are + // not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') + // round-trips byte for byte as a string value; to_bjdata() writes + // the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp) + const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; + json j; + CHECK_NOTHROW(j = json::from_bjdata(v)); + REQUIRE(j.is_string()); + CHECK(j.get_ref() == std::string("\xc0\xae")); + CHECK_THROWS_AS(j.dump(), json::type_error&); + CHECK(json::from_bjdata(json::to_bjdata(j)) == j); + + // the same bytes as an object key round-trip as well + const std::vector v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'}; + json j_key; + CHECK_NOTHROW(j_key = json::from_bjdata(v_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_bjdata(json::to_bjdata(j_key)) == j_key); + + CHECK(json::from_bjdata(json::to_bjdata(json("\xFF"))) == json("\xFF")); + // a truncated multi-byte sequence + CHECK(json::from_bjdata(json::to_bjdata(json("\xC3"))) == json("\xC3")); + // an encoded surrogate half (U+D800) + CHECK(json::from_bjdata(json::to_bjdata(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); + // an overlong encoding of '.' + CHECK(json::from_bjdata(json::to_bjdata(json("\xC0\xAF"))) == json("\xC0\xAF")); + + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_bjdata(json::to_bjdata(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + } } SECTION("Array Type") diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 2f9a727ab..92a14e6fe 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -154,6 +154,43 @@ TEST_CASE("BSON") #endif } + SECTION("ill-formed UTF-8 (see #5529, #5651)") + { + // a BSON document {"s": "\xC0\xAE"} (0xC0 0xAE is an overlong + // encoding of '.'); the BSON spec does not require a decoder to + // reject ill-formed UTF-8 in a string value, so the reader hands the + // bytes back unchanged + const std::vector v = + { + 0x0F, 0x00, 0x00, 0x00, // document length + 0x02, 's', 0x00, // type 0x02 (string), key "s" + 0x03, 0x00, 0x00, 0x00, // string length (including null) + 0xc0, 0xae, 0x00, // string content and its null terminator + 0x00 // document terminator + }; + json j; + CHECK_NOTHROW(j = json::from_bson(v)); + REQUIRE(j.is_object()); + REQUIRE(j.contains("s")); + CHECK(j["s"].get_ref() == std::string("\xc0\xae")); + // dump() still requires valid UTF-8 and throws for such a value + CHECK_THROWS_AS(j.dump(), json::type_error&); + // to_bson() writes the bytes back unchanged, as before 3.13.0, + // unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp) + CHECK(json::from_bson(json::to_bson(j)) == j); + + CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}})) == json{{"s", "\xFF"}}); + // a truncated multi-byte sequence + CHECK(json::from_bson(json::to_bson(json{{"s", "\xC3"}})) == json{{"s", "\xC3"}}); + // an encoded surrogate half (U+D800) + CHECK(json::from_bson(json::to_bson(json{{"s", "\xED\xA0\x80"}})) == json{{"s", "\xED\xA0\x80"}}); + // an overlong encoding of '.' + CHECK(json::from_bson(json::to_bson(json{{"s", "\xC0\xAF"}})) == json{{"s", "\xC0\xAF"}}); + + // an object key with ill-formed UTF-8 is kept as well + CHECK(json::from_bson(json::to_bson(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + } + SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON") { // out_of_range.412 is thrown from a single shared helper diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index c3a425a16..633f50fea 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -1801,19 +1801,41 @@ TEST_CASE("CBOR") CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&); } - SECTION("invalid UTF-8 in string (see #5529)") + SECTION("ill-formed UTF-8 in string (see #5529, #5651)") { + // RFC 8949 §3.1 leaves it up to the decoder whether to reject + // ill-formed UTF-8 in a text string; this library does not, and + // hands the original bytes back unchanged, matching the + // MessagePack reader and the behavior before #5185/#5531 (not in + // any release) + // a two-character text string (major type 3) whose bytes are not - // valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be - // rejected at decode time, matching every other kind of - // malformed binary input, rather than only failing later when - // the resulting value is dumped - json _; - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_cbor(std::vector({0x62, 0xc0, 0xae}), true, false).is_discarded()); + // valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') round-trips + // byte for byte as a string value + const std::vector ill_formed_value = {0x62, 0xc0, 0xae}; + json j_value; + CHECK_NOTHROW(j_value = json::from_cbor(ill_formed_value)); + REQUIRE(j_value.is_string()); + CHECK(j_value.get_ref() == std::string("\xc0\xae")); + // dump() still requires valid UTF-8 and throws for such a value, + // unless an error handler that replaces or ignores the bytes is + // passed + CHECK_THROWS_AS(j_value.dump(), json::type_error&); + // to_cbor() writes the bytes back unchanged, as before 3.13.0, + // unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp) + CHECK(json::from_cbor(json::to_cbor(j_value)) == j_value); + + // the same bytes as an object key round-trip as well + const std::vector ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01}; + json j_key; + CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_cbor(json::to_cbor(j_key)) == j_key); // a CBOR byte string (major type 2) with the very same bytes is // NOT text and must still be accepted as-is + json _; CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x42, 0xc0, 0xae}))); CHECK(_ == json::binary(std::vector({0xc0, 0xae}))); @@ -1822,17 +1844,47 @@ TEST_CASE("CBOR") CHECK(json::from_cbor(json::to_cbor(j)) == j); } - SECTION("invalid UTF-8 in indefinite-length string") + SECTION("to_cbor keeps ill-formed UTF-8 (see #5651)") + { + // to_cbor() writes the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp); from_cbor() reads them back as is + CHECK(json::from_cbor(json::to_cbor(json("\xFF"))) == json("\xFF")); + // a truncated multi-byte sequence + CHECK(json::from_cbor(json::to_cbor(json("\xC3"))) == json("\xC3")); + // an encoded surrogate half (U+D800) + CHECK(json::from_cbor(json::to_cbor(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); + // an overlong encoding of '.' + CHECK(json::from_cbor(json::to_cbor(json("\xC0\xAF"))) == json("\xC0\xAF")); + + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_cbor(json::to_cbor(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + + // binary values are not text and are unaffected + CHECK_NOTHROW(json::to_cbor(json::binary(std::vector({0xFF})))); + } + + SECTION("ill-formed UTF-8 in indefinite-length string") { json _; - // every chunk must be valid UTF-8 on its own (RFC 8949, Section - // 3.2.3), so a code point split across two chunks is rejected - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded()); + // the chunks are concatenated as is, without checking that each + // chunk is valid UTF-8 on its own (RFC 8949, Section 3.2.3), so + // a code point split across two chunks yields a valid string + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}))); + CHECK(_ == "\xc3\xa9"); + CHECK(_.dump() == "\"\xc3\xa9\""); - // an ill-formed later chunk is rejected after valid ones - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + // a truncated code point is kept as is + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0xff}))); + CHECK(_ == "\xc3"); + CHECK_THROWS_AS(_.dump(), json::type_error&); + CHECK(json::from_cbor(json::to_cbor(_)) == _); + + // an ill-formed later chunk is kept after valid ones + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff}))); + CHECK(_ == "\xc3\xa9\xc0\xae"); + CHECK_THROWS_AS(_.dump(), json::type_error&); // valid multi-byte chunks are accepted CHECK(json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6"); @@ -1840,9 +1892,6 @@ TEST_CASE("CBOR") SECTION("many chunks in indefinite-length string") { - // only the newly read chunk is validated, not the whole string - // collected so far; validating the latter made this input take - // quadratic time (about ten seconds for 100000 chunks) constexpr std::size_t chunks = 100000; std::vector v{0x7f}; for (std::size_t i = 0; i < chunks; ++i) diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index afabb85c6..a92496144 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -1540,19 +1540,39 @@ TEST_CASE("MessagePack") CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&); } - SECTION("invalid UTF-8 in string (see #5529)") + SECTION("ill-formed UTF-8 in string (see #5529, #5651)") { + // the MessagePack specification explicitly allows a str object to + // contain a byte sequence that is not valid UTF-8 and expects a + // deserializer to hand the original bytes back unchanged; this + // library follows that, unlike CBOR/UBJSON/BJData/BSON, whose + // specifications require text strings to be valid UTF-8 + // a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8 - // (0xC0 0xAE is an overlong encoding of '.') must be rejected at - // decode time, matching every other kind of malformed binary - // input, rather than only failing later when the resulting - // value is dumped - json _; - CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0xa2, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing MessagePack string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_msgpack(std::vector({0xa2, 0xc0, 0xae}), true, false).is_discarded()); + // (0xC0 0xAE is an overlong encoding of '.') round-trips byte for + // byte as a string value + const std::vector ill_formed_value = {0xa2, 0xc0, 0xae}; + json j_value; + CHECK_NOTHROW(j_value = json::from_msgpack(ill_formed_value)); + REQUIRE(j_value.is_string()); + CHECK(j_value.get_ref() == std::string("\xc0\xae")); + CHECK(json::from_msgpack(json::to_msgpack(j_value)) == j_value); + // dump() still requires valid UTF-8 and throws for such a value, + // unless an error handler that replaces or ignores the bytes is + // passed + CHECK_THROWS_AS(j_value.dump(), json::type_error&); + + // the same bytes as an object key round-trip as well + const std::vector ill_formed_key = {0x81, 0xa2, 0xc0, 0xae, 0x01}; + json j_key; + CHECK_NOTHROW(j_key = json::from_msgpack(ill_formed_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_msgpack(json::to_msgpack(j_key)) == j_key); // a MessagePack bin8 blob with the very same bytes is NOT text // and must still be accepted as-is + json _; CHECK_NOTHROW(_ = json::from_msgpack(std::vector({0xc4, 0x02, 0xc0, 0xae}))); CHECK(_ == json::binary(std::vector({0xc0, 0xae}))); diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index 9a9710b48..450882a12 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -2505,6 +2505,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK(json::to_ubjson(j) == v); CHECK(json::from_ubjson(v) == j); } + + SECTION("ill-formed UTF-8 (see #5529, #5651)") + { + // none of the binary format specs requires a decoder to reject + // ill-formed UTF-8 in a text string, so a value whose bytes are + // not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') + // round-trips byte for byte as a string value; to_ubjson() writes + // the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp) + const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; + json j; + CHECK_NOTHROW(j = json::from_ubjson(v)); + REQUIRE(j.is_string()); + CHECK(j.get_ref() == std::string("\xc0\xae")); + CHECK_THROWS_AS(j.dump(), json::type_error&); + CHECK(json::from_ubjson(json::to_ubjson(j)) == j); + + // the same bytes as an object key round-trip as well + const std::vector v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'}; + json j_key; + CHECK_NOTHROW(j_key = json::from_ubjson(v_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_ubjson(json::to_ubjson(j_key)) == j_key); + + CHECK(json::from_ubjson(json::to_ubjson(json("\xFF"))) == json("\xFF")); + // a truncated multi-byte sequence + CHECK(json::from_ubjson(json::to_ubjson(json("\xC3"))) == json("\xC3")); + // an encoded surrogate half (U+D800) + CHECK(json::from_ubjson(json::to_ubjson(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); + // an overlong encoding of '.' + CHECK(json::from_ubjson(json::to_ubjson(json("\xC0\xAF"))) == json("\xC0\xAF")); + + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_ubjson(json::to_ubjson(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + } } SECTION("Array Type") From 73e9eae3c135e262dac3fe3e7978b46694e531d3 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 4 Oct 2026 12:13:49 +0200 Subject: [PATCH 8/8] Add an error_handler parameter for UTF-8 to the binary readers and writers (#5746) Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/dump.md | 9 +- .../docs/api/basic_json/error_handler_t.md | 32 +- .../mkdocs/docs/api/basic_json/from_bjdata.md | 15 +- docs/mkdocs/docs/api/basic_json/from_bson.md | 15 +- docs/mkdocs/docs/api/basic_json/from_cbor.md | 18 +- .../docs/api/basic_json/from_msgpack.md | 18 +- .../mkdocs/docs/api/basic_json/from_ubjson.md | 15 +- docs/mkdocs/docs/api/basic_json/to_bjdata.md | 26 +- docs/mkdocs/docs/api/basic_json/to_bson.md | 28 +- docs/mkdocs/docs/api/basic_json/to_cbor.md | 26 +- docs/mkdocs/docs/api/basic_json/to_msgpack.md | 20 +- docs/mkdocs/docs/api/basic_json/to_ubjson.md | 26 +- .../api/macros/json_strict_binary_utf8.md | 18 +- docs/mkdocs/docs/examples/error_handler_t.cpp | 3 +- .../docs/examples/error_handler_t.output | 1 + .../docs/features/binary_formats/bjdata.md | 24 +- .../docs/features/binary_formats/bson.md | 19 +- .../docs/features/binary_formats/cbor.md | 17 +- .../features/binary_formats/messagepack.md | 18 +- .../docs/features/binary_formats/ubjson.md | 24 +- docs/mkdocs/docs/home/exceptions.md | 18 +- .../nlohmann/detail/input/binary_reader.hpp | 114 ++- .../nlohmann/detail/output/binary_writer.hpp | 213 +++-- .../nlohmann/detail/output/error_handler.hpp | 50 ++ include/nlohmann/detail/output/serializer.hpp | 72 +- include/nlohmann/detail/string_utils.hpp | 130 ++++ include/nlohmann/json.hpp | 135 ++-- single_include/nlohmann/json.hpp | 734 ++++++++++++++---- tests/src/unit-binary_utf8_error_handler.cpp | 362 +++++++++ tests/src/unit-binary_utf8_strict.cpp | 16 +- 30 files changed, 1794 insertions(+), 422 deletions(-) create mode 100644 include/nlohmann/detail/output/error_handler.hpp create mode 100644 tests/src/unit-binary_utf8_error_handler.cpp diff --git a/docs/mkdocs/docs/api/basic_json/dump.md b/docs/mkdocs/docs/api/basic_json/dump.md index a3c9db3a8..d805f3db1 100644 --- a/docs/mkdocs/docs/api/basic_json/dump.md +++ b/docs/mkdocs/docs/api/basic_json/dump.md @@ -25,10 +25,12 @@ and `ensure_ascii` parameters. result consists of ASCII characters only. `error_handler` (in) -: how to react on decoding errors; there are three possible values (see [`error_handler_t`](error_handler_t.md): +: how to react on decoding errors; there are four possible values (see [`error_handler_t`](error_handler_t.md): `strict` (throws an exception in case a decoding error occurs; default), `replace` (replace invalid UTF-8 sequences - with U+FFFD), and `ignore` (ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the - output unchanged, and invalid bytes are dropped)). + with U+FFFD), `ignore` (ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the + output unchanged, and invalid bytes are dropped), and `keep` (write the ill-formed bytes to the output as is, + without escaping them, even if `ensure_ascii` is `#!cpp true`; the result is then not valid UTF-8, but equals the + input bytes exactly, and well-formed characters around the ill-formed bytes are still escaped as usual)). ## Return value @@ -94,3 +96,4 @@ Binary values are serialized as an object containing two keys: - Indentation character `indent_char`, option `ensure_ascii` and exceptions added in version 3.0.0. - Error handlers added in version 3.4.0. - Serialization of binary values added in version 3.8.0. +- Error handler `keep` added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/error_handler_t.md b/docs/mkdocs/docs/api/basic_json/error_handler_t.md index 51dc6510f..17327fe26 100644 --- a/docs/mkdocs/docs/api/basic_json/error_handler_t.md +++ b/docs/mkdocs/docs/api/basic_json/error_handler_t.md @@ -4,15 +4,31 @@ enum class error_handler_t { strict, replace, - ignore + ignore, + keep }; ``` -This enumeration is used in the [`dump`](dump.md) function to choose how to treat decoding errors while serializing a -`basic_json` value. Three values are differentiated: +This enumeration is used to choose how to treat ill-formed UTF-8 in a string value or object key: + +- [`dump`](dump.md) uses it while serializing a `basic_json` value to text. +- [`to_cbor`](to_cbor.md), [`to_msgpack`](to_msgpack.md), [`to_ubjson`](to_ubjson.md), [`to_bjdata`](to_bjdata.md), + and [`to_bson`](to_bson.md) use it while serializing a `basic_json` value to that binary format. Their default is + `keep`, as no binary writer checked before this parameter was added. CBOR, UBJSON, BJData, and BSON require valid + UTF-8, so for these four the default is `strict` if [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) + is enabled; MessagePack's specification explicitly allows a string to contain ill-formed UTF-8, so `to_msgpack` + stays at `keep`. `to_bon8` does not take this parameter: BON8 always validates, since UTF-8 lead bytes are + structural to that format. +- [`from_cbor`](from_cbor.md), [`from_msgpack`](from_msgpack.md), [`from_ubjson`](from_ubjson.md), + [`from_bjdata`](from_bjdata.md), and [`from_bson`](from_bson.md) use it while parsing that binary format, to decide + whether to check a string value or object key for well-formed UTF-8 at all; by default (`keep`) they do not, as no + binary reader did before this parameter was added. `from_bon8` does not take this parameter, for the same reason + `to_bon8` does not. + +Four values are differentiated: strict -: throw a `type_error` exception in case of invalid UTF-8 +: throw a `type_error`/`parse_error` exception in case of invalid UTF-8 replace : replace invalid UTF-8 sequences with U+FFFD (� REPLACEMENT CHARACTER) @@ -20,6 +36,12 @@ replace ignore : ignore invalid UTF-8 sequences; all valid bytes are copied to the output unchanged, and invalid bytes are dropped +keep +: keep invalid UTF-8 sequences unchanged; only meaningful for the binary formats mentioned above, since [`dump`] + (dump.md) itself must produce text, and `keep` there writes the ill-formed bytes to the output as is, so the + result is then not valid UTF-8 (but still equals the input bytes exactly, including around any well-formed + characters, which are still escaped as usual) + ## Examples ??? example @@ -45,3 +67,5 @@ ignore ## Version history - Added in version 3.4.0. +- Added `keep`, and made this enumeration apply to the binary readers and writers in addition to `dump`, in version + 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/from_bjdata.md b/docs/mkdocs/docs/api/basic_json/from_bjdata.md index 9df21ab30..a45e00ad5 100644 --- a/docs/mkdocs/docs/api/basic_json/from_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/from_bjdata.md @@ -5,12 +5,14 @@ template static basic_json from_bjdata(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_bjdata(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the BJData (Binary JData) serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + BJData does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -73,7 +81,7 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if a parse error occurs - Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string could not be parsed - successfully + successfully, or if a string value or object key is not valid UTF-8 and `error_handler` is `strict` - Throws [out_of_range.408](../../home/exceptions.md#jsonexceptionout_of_range408) if the size of an optimized container or n-dimensional array cannot be represented by `std::size_t` @@ -111,3 +119,4 @@ Linear in the size of the input. - Added in version 3.11.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/from_bson.md b/docs/mkdocs/docs/api/basic_json/from_bson.md index 9dfd9dc18..b037e8e07 100644 --- a/docs/mkdocs/docs/api/basic_json/from_bson.md +++ b/docs/mkdocs/docs/api/basic_json/from_bson.md @@ -5,12 +5,14 @@ template static basic_json from_bson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_bson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the BSON (Binary JSON) serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + BSON does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -75,6 +83,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va invalid string or byte array length) - Throws [`parse_error.114`](../../home/exceptions.md#jsonexceptionparse_error114) if an unsupported BSON record type is encountered +- Throws [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) if a string value or object key is + not valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -111,6 +121,7 @@ Linear in the size of the input. - Added in version 3.4.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/from_cbor.md b/docs/mkdocs/docs/api/basic_json/from_cbor.md index 791c183bb..7d7df08a5 100644 --- a/docs/mkdocs/docs/api/basic_json/from_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/from_cbor.md @@ -6,14 +6,16 @@ template static basic_json from_cbor(InputType&& i, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error); + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_cbor(IteratorType first, SentinelType last, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error); + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the CBOR (Concise Binary Object Representation) serialization format. @@ -65,6 +67,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ : how to treat CBOR tags (optional, `error` by default); see [`cbor_tag_handler_t`](cbor_tag_handler_t.md) for more information +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + CBOR does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -80,8 +88,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from CBOR were used in the given input or if the input is not valid CBOR -- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other - types are not supported, as JSON object keys are always strings) or a string is malformed +- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of + other types are not supported, as JSON object keys are always strings), or if a string value or object key is not + valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -121,6 +130,7 @@ Linear in the size of the input. - Added `tag_handler` parameter in version 3.9.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/from_msgpack.md b/docs/mkdocs/docs/api/basic_json/from_msgpack.md index 395512acb..2e7fa5062 100644 --- a/docs/mkdocs/docs/api/basic_json/from_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/from_msgpack.md @@ -5,12 +5,14 @@ template static basic_json from_msgpack(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_msgpack(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the MessagePack serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + MessagePack's specification explicitly allows ill-formed UTF-8, so checking is opt-in: the default, `keep`, does + not check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -73,8 +81,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from MessagePack were used in the given input or if the input is not valid MessagePack -- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other - types are not supported, as JSON object keys are always strings) or a string is malformed +- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of + other types are not supported, as JSON object keys are always strings), or if a string value or object key is not + valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -113,6 +122,7 @@ Linear in the size of the input. - Added `allow_exceptions` parameter in version 3.2.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/from_ubjson.md b/docs/mkdocs/docs/api/basic_json/from_ubjson.md index 1ad076588..ac9404d4c 100644 --- a/docs/mkdocs/docs/api/basic_json/from_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/from_ubjson.md @@ -5,12 +5,14 @@ template static basic_json from_ubjson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_ubjson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the UBJSON (Universal Binary JSON) serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + UBJSON does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -73,7 +81,7 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if a parse error occurs - Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string could not be parsed - successfully + successfully, or if a string value or object key is not valid UTF-8 and `error_handler` is `strict` - Throws [out_of_range.408](../../home/exceptions.md#jsonexceptionout_of_range408) if the size of an optimized container or n-dimensional array cannot be represented by `std::size_t` @@ -112,6 +120,7 @@ Linear in the size of the input. - Added `allow_exceptions` parameter in version 3.2.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/to_bjdata.md b/docs/mkdocs/docs/api/basic_json/to_bjdata.md index 63dc5379e..df3b69004 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/to_bjdata.md @@ -5,15 +5,18 @@ static std::vector to_bjdata(const basic_json& j, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2); + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = error_handler_t::keep); // (2) static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2); + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = error_handler_t::keep); static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2); + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the BJData (Binary JData) serialization format. BJData aims to @@ -43,6 +46,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ : which version of BJData to use (see note on "Binary values" on [BJData](../../features/binary_formats/bjdata.md)); optional, `#!cpp bjdata_version_t::draft2` by default. +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_bjdata` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. BJData serialization as byte vector @@ -56,9 +65,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false, and `j` contains a non-empty array, object, or binary value. -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not - valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are - written unchanged +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -108,5 +117,6 @@ Linear in the size of the JSON value `j`. - Added in version 3.11.0. - BJData version parameter (for draft3 binary encoding) added in version 3.12.0. -- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if - [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. \ No newline at end of file +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`. \ No newline at end of file diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index e3104b13c..ad974e048 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -2,11 +2,14 @@ ```cpp // (1) -static std::vector to_bson(const basic_json& j); +static std::vector to_bson(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep); // (2) -static void to_bson(const basic_json& j, detail::output_adapter o); -static void to_bson(const basic_json& j, detail::output_adapter o); +static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); +static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); ``` BSON (Binary JSON) is a binary format in which zero or more ordered key/value pairs are stored as a single entity (a @@ -25,6 +28,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `o` (in) : output adapter to write serialization to +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_bson` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. BSON serialization as a byte vector @@ -46,9 +55,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the BSON binary subtype; example: `"subtype 70000 is too large for the BSON binary subtype (max 255)"` -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is not valid - UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are - written unchanged +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -101,6 +110,7 @@ pass before anything is written. - Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. - Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0. - `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0. -- Throwing `type_error.316` for a string value or object key that is not valid UTF-8 if - [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, detected before anything is written, - added in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316` before anything + is written. diff --git a/docs/mkdocs/docs/api/basic_json/to_cbor.md b/docs/mkdocs/docs/api/basic_json/to_cbor.md index cac8a0917..c72f310bb 100644 --- a/docs/mkdocs/docs/api/basic_json/to_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/to_cbor.md @@ -2,11 +2,14 @@ ```cpp // (1) -static std::vector to_cbor(const basic_json& j); +static std::vector to_cbor(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep); // (2) -static void to_cbor(const basic_json& j, detail::output_adapter o); -static void to_cbor(const basic_json& j, detail::output_adapter o); +static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); +static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the CBOR (Concise Binary Object Representation) serialization @@ -26,6 +29,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `o` (in) : output adapter to write serialization to +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_cbor` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. CBOR serialization as a byte vector @@ -37,9 +46,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va ## Exceptions -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not - valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are - written unchanged +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -74,5 +83,6 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Compact representation of floating-point numbers added in version 3.8.0. -- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if - [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`. diff --git a/docs/mkdocs/docs/api/basic_json/to_msgpack.md b/docs/mkdocs/docs/api/basic_json/to_msgpack.md index 707de9514..17b9fe29c 100644 --- a/docs/mkdocs/docs/api/basic_json/to_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/to_msgpack.md @@ -2,11 +2,14 @@ ```cpp // (1) -static std::vector to_msgpack(const basic_json& j); +static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep); // (2) -static void to_msgpack(const basic_json& j, detail::output_adapter o); -static void to_msgpack(const basic_json& j, detail::output_adapter o); +static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); +static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the MessagePack serialization format. MessagePack is a binary @@ -25,6 +28,13 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `o` (in) : output adapter to write serialization to +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_msgpack` did before + this parameter was added and as the MessagePack specification allows; `strict` throws; `replace`/`ignore` sanitize + it the same way [`dump`](dump.md) would. Unlike the other binary writers, the default stays `keep` even if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled. + ## Return value 1. MessagePack serialization as a byte vector @@ -42,6 +52,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the MessagePack ext type; example: `"subtype 70000 is too large for the MessagePack ext type (max 255)"` +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -91,6 +103,8 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before. - Fixed in version 3.13.0 to serialize `number_integer_t`/`number_unsigned_t` pairs of different width correctly; before, integers could be serialized with the wrong value if `number_integer_t` was narrower than `number_unsigned_t`. diff --git a/docs/mkdocs/docs/api/basic_json/to_ubjson.md b/docs/mkdocs/docs/api/basic_json/to_ubjson.md index 8f2ea5eb9..6860c01c6 100644 --- a/docs/mkdocs/docs/api/basic_json/to_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/to_ubjson.md @@ -4,13 +4,16 @@ // (1) static std::vector to_ubjson(const basic_json& j, const bool use_size = false, - const bool use_type = false); + const bool use_type = false, + const error_handler_t error_handler = error_handler_t::keep); // (2) static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false); + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = error_handler_t::keep); static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false); + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the UBJSON (Universal Binary JSON) serialization format. UBJSON @@ -36,6 +39,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ : whether to add type annotations to container types (must be combined with `#!cpp use_size = true`); optional, `#!cpp false` by default. +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_ubjson` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. UBJSON serialization as a byte vector @@ -49,9 +58,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false, and `j` contains a non-empty array, object, or binary value. -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not - valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are - written unchanged +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -100,5 +109,6 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.1.0. -- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if - [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`. diff --git a/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md index 7fe6d9a58..ed7bbe6b3 100644 --- a/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md +++ b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md @@ -4,15 +4,18 @@ #define JSON_STRICT_BINARY_UTF8 /* value */ ``` -When defined to `1`, the binary writers [`to_cbor`](../basic_json/to_cbor.md), [`to_ubjson`](../basic_json/to_ubjson.md), -[`to_bjdata`](../basic_json/to_bjdata.md), and [`to_bson`](../basic_json/to_bson.md) check every string value and -object key for valid UTF-8 and throw [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for -ill-formed UTF-8, like [`dump`](../basic_json/dump.md) does. Without it, they write the bytes unchanged. +When defined to `1`, the `error_handler` parameter of the binary writers [`to_cbor`](../basic_json/to_cbor.md), +[`to_ubjson`](../basic_json/to_ubjson.md), [`to_bjdata`](../basic_json/to_bjdata.md), and +[`to_bson`](../basic_json/to_bson.md) defaults to [`error_handler_t::strict`](../basic_json/error_handler_t.md) instead +of `error_handler_t::keep`. These writers then check every string value and object key for valid UTF-8 and throw +[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, like +[`dump`](../basic_json/dump.md) does. Without it, they write the bytes unchanged. An `error_handler` passed explicitly +always takes precedence. The macro does not affect: - [`to_msgpack`](../basic_json/to_msgpack.md): the MessagePack specification allows a `str` value to contain bytes that - are not valid UTF-8, so it always writes them unchanged. + are not valid UTF-8, so its `error_handler` always defaults to `keep`. - [`to_bon8`](../basic_json/to_bon8.md): BON8 always checks, because the UTF-8 lead bytes mark where a string ends. - The binary readers ([`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md), [`from_ubjson`](../basic_json/from_ubjson.md), [`from_bjdata`](../basic_json/from_bjdata.md), @@ -33,8 +36,9 @@ The default value is `0` (disabled, the behavior of version 3.12.0 and earlier i CBOR, UBJSON, BJData, and BSON all require strings to be UTF-8. Up to version 3.12.0, the writers did not check this, so they could produce output that other decoders reject. Checking by default would break code that stores - other encodings (for instance ISO 8859-1) in a string and only ever writes it to a binary format, so this macro - offers the check as an opt-in ahead of version 4.0.0, where it is planned to become the default (see + other encodings (for instance ISO 8859-1) in a string and only ever writes it to a binary format. You can pass + `error_handler_t::strict` to each call, or use this macro to check by default ahead of version 4.0.0, where + `strict` is planned to become the default (see [#5529](https://github.com/nlohmann/json/issues/5529) and [#5651](https://github.com/nlohmann/json/issues/5651)). !!! warning "Opt-in only" diff --git a/docs/mkdocs/docs/examples/error_handler_t.cpp b/docs/mkdocs/docs/examples/error_handler_t.cpp index b4718d7e6..bf035cea3 100644 --- a/docs/mkdocs/docs/examples/error_handler_t.cpp +++ b/docs/mkdocs/docs/examples/error_handler_t.cpp @@ -20,5 +20,6 @@ int main() << j_invalid.dump(-1, ' ', false, json::error_handler_t::replace) << "\nstring with ignored invalid characters: " << j_invalid.dump(-1, ' ', false, json::error_handler_t::ignore) - << '\n'; + << "\nstring with the invalid byte kept as is (" << j_invalid.dump(-1, ' ', false, json::error_handler_t::keep).size() + << " bytes, not valid UTF-8 itself)\n"; } diff --git a/docs/mkdocs/docs/examples/error_handler_t.output b/docs/mkdocs/docs/examples/error_handler_t.output index 718d62bee..37cae62a2 100644 --- a/docs/mkdocs/docs/examples/error_handler_t.output +++ b/docs/mkdocs/docs/examples/error_handler_t.output @@ -1,3 +1,4 @@ [json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9 string with replaced invalid characters: "ä�ü" string with ignored invalid characters: "äü" +string with the invalid byte kept as is (7 bytes, not valid UTF-8 itself) diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index 12d1af51b..3e3d8e885 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -65,10 +65,12 @@ The library uses the following mapping from JSON values types to BJData types ac !!! warning "UTF-8 validation of string values and object keys" - BJData strings must use UTF-8 encoding. By default, `to_bjdata()` writes the bytes of string values and object keys - unchanged, even if they are not valid UTF-8. If - [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead. + BJData strings must use UTF-8 encoding. By default (the [`error_handler`](../../api/basic_json/to_bjdata.md) + parameter left at `keep`), `to_bjdata()` writes the bytes of string values and object keys unchanged, even if they + are not valid UTF-8. With `error_handler_t::strict`, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead; + `replace`/`ignore` sanitize the string. [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) + makes `strict` the default. !!! info "Unused BJData markers" @@ -217,12 +219,16 @@ The library maps BJData types to JSON value types as follows: !!! warning "Ill-formed UTF-8 in string values and object keys" - BJData strings must use UTF-8 encoding, but this is not enforced on read: `from_bjdata()` accepts a string - value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, + BJData strings must use UTF-8 encoding, but checking it on read is opt-in: with the + [`error_handler`](../../api/basic_json/from_bjdata.md) parameter left at `keep` (the default), `from_bjdata()` + accepts a string value or object key whose bytes are not valid UTF-8 and hands them back unchanged. Passing + `error_handler_t::strict` makes `from_bjdata()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error - handler is passed that replaces or ignores the ill-formed bytes. By default, `to_bjdata()` writes such a value - back unchanged (see above). + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read with the default + `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. `to_bjdata()`'s + own `error_handler` parameter defaults to `keep` (see above), so such a value is written back unchanged. !!! info "Round trips" diff --git a/docs/mkdocs/docs/features/binary_formats/bson.md b/docs/mkdocs/docs/features/binary_formats/bson.md index 245ed3ebf..f205cba17 100644 --- a/docs/mkdocs/docs/features/binary_formats/bson.md +++ b/docs/mkdocs/docs/features/binary_formats/bson.md @@ -112,13 +112,18 @@ The library maps BSON record types to JSON value types as follows: !!! warning "Ill-formed UTF-8 in string values" The BSON specification requires `string` values (type `0x02`) to be valid UTF-8, but this is not required of a - decoder. `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back unchanged. - However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error handler is - passed that replaces or ignores the ill-formed bytes. By default, `to_bson()` writes such a string value or element - (key) name unchanged; if [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it - throws the same exception instead. Element (key) names are never validated on read, since they are read byte-by-byte - as a C string. `binary` values (type `0x05`) are unaffected, since they are not required to hold text. + decoder, so checking is opt-in: with the [`error_handler`](../../api/basic_json/from_bson.md) parameter left at + `keep` (the default), `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back + unchanged. Passing `error_handler_t::strict` makes `from_bson()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md) + still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a + value read with the default `keep` handler, unless an error handler is passed that replaces or ignores the + ill-formed bytes. `to_bson()`'s own `error_handler` parameter defaults to `keep`, so such a string value or element + (key) name is written unchanged; with `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled), it throws the same exception + instead. Element (key) names are never validated on read, since they are read byte-by-byte as a C string. `binary` + values (type `0x05`) are unaffected, since they are not required to hold text. ??? example "Example: deserialize a JSON value from BSON" diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index 66300c5fa..7b5be4631 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -192,12 +192,17 @@ The library maps CBOR types to JSON value types as follows: !!! warning "Ill-formed UTF-8 in text strings" [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings (major - type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this. This library does not: - `from_cbor()` accepts a text string (object keys included) whose bytes are not valid UTF-8 and hands them back - unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error handler is - passed that replaces or ignores the ill-formed bytes. By default, `to_cbor()` writes such a value back unchanged; if - [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws the same exception + type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this, so checking is opt-in: with the + [`error_handler`](../../api/basic_json/from_cbor.md) parameter left at `keep` (the default), `from_cbor()` accepts a + text string (object keys included) whose bytes are not valid UTF-8 and hands them back unchanged. Passing + `error_handler_t::strict` makes `from_cbor()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md) + still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a + value read with the default `keep` handler, unless an error handler is passed that replaces or ignores the + ill-formed bytes. `to_cbor()`'s own [`error_handler`](../../api/basic_json/to_cbor.md) parameter defaults to `keep`, + so such a value is written back unchanged; with `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled), it throws the same exception instead. Byte strings (major type 2) are unaffected, since they are not required to hold text. !!! warning "Tagged items" diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index 047944852..24eeab139 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -157,11 +157,19 @@ The library maps MessagePack types to JSON value types as follows: The MessagePack specification explicitly allows a `str` value (`fixstr`, `str 8`, `str 16`, `str 32`) to contain a byte sequence that is not valid UTF-8, and expects a deserializer to hand the original bytes back unchanged. - This library follows that: `from_msgpack()` reads `str` bytes (object keys included) as-is, without validating - them, and `to_msgpack()` writes them back as-is, so such a value round-trips through `from_msgpack(to_msgpack(j))` - byte for byte. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way, unless an - error handler is passed that replaces or ignores the ill-formed bytes. + This library follows that by default: with its + [`error_handler`](../../api/basic_json/from_msgpack.md) parameter left at `keep` (the default), + `from_msgpack()` reads `str` bytes (object keys included) as-is, without validating them, so such a value + round-trips through `from_msgpack(to_msgpack(j))` byte for byte. Passing `error_handler_t::strict` makes + `from_msgpack()` check anyway and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. `to_msgpack()` also writes `str` bytes as-is by + default, since the specification permits it; its [`error_handler`](../../api/basic_json/to_msgpack.md) parameter + can be set to `strict` to throw [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) instead, or + to `replace`/`ignore` to sanitize the string, for instance for a decoder that rejects ill-formed UTF-8. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way with the + default `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. ??? example "Example: deserialize a JSON value from MessagePack" diff --git a/docs/mkdocs/docs/features/binary_formats/ubjson.md b/docs/mkdocs/docs/features/binary_formats/ubjson.md index dbdff6e6c..f7a14d855 100644 --- a/docs/mkdocs/docs/features/binary_formats/ubjson.md +++ b/docs/mkdocs/docs/features/binary_formats/ubjson.md @@ -49,10 +49,12 @@ The library uses the following mapping from JSON values types to UBJSON types ac !!! warning "UTF-8 validation of string values and object keys" - UBJSON's required string encoding is UTF-8. By default, `to_ubjson()` writes the bytes of string values and object - keys unchanged, even if they are not valid UTF-8. If - [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead. + UBJSON's required string encoding is UTF-8. By default (the [`error_handler`](../../api/basic_json/to_ubjson.md) + parameter left at `keep`), `to_ubjson()` writes the bytes of string values and object keys unchanged, even if they + are not valid UTF-8. With `error_handler_t::strict`, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead; + `replace`/`ignore` sanitize the string. [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) + makes `strict` the default. !!! info "Unused UBJSON markers" @@ -129,12 +131,16 @@ The library maps UBJSON types to JSON value types as follows: !!! warning "Ill-formed UTF-8 in string values and object keys" - UBJSON's required string encoding is UTF-8, but this is not enforced on read: `from_ubjson()` accepts a string - value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, + UBJSON's required string encoding is UTF-8, but checking it on read is opt-in: with the + [`error_handler`](../../api/basic_json/from_ubjson.md) parameter left at `keep` (the default), `from_ubjson()` + accepts a string value or object key whose bytes are not valid UTF-8 and hands them back unchanged. Passing + `error_handler_t::strict` makes `from_ubjson()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error - handler is passed that replaces or ignores the ill-formed bytes. By default, `to_ubjson()` writes such a value - back unchanged (see above). + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read with the default + `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. `to_ubjson()`'s + own `error_handler` parameter defaults to `keep` (see above), so such a value is written back unchanged. ??? example "Example: deserialize a JSON value from UBJSON" diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index a3797cf04..e7eb8fd03 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -340,9 +340,11 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde ### json.exception.parse_error.113 A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a -string was read where one was required (for instance as a map key), or the string's length specification is invalid. -The bytes of a string itself are not checked for valid UTF-8 on read; see the ill-formed UTF-8 notes on the -individual [binary format](../features/binary_formats/index.md) pages for how such a string is handled afterward. +string was read where one was required (for instance as a map key), the string's length specification is invalid, or +the string's bytes are not valid UTF-8 and the `error_handler` parameter of the corresponding `from_*` function is +set to `strict`. By default (`error_handler_t::keep`), the bytes of a string are not checked for valid UTF-8 on read; +see the ill-formed UTF-8 notes on the individual [binary format](../features/binary_formats/index.md) pages for how +such a string is handled depending on `error_handler`. CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other type (for instance integers or `null`) are therefore not supported; see the notes on @@ -365,6 +367,9 @@ type (for instance integers or `null`) are therefore not supported; see the note ``` [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData string: string length must not be negative ``` + ``` + [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte + ``` ### json.exception.parse_error.114 @@ -747,10 +752,11 @@ The [`unflatten()`](../api/basic_json/unflatten.md) function only works for an o The [`dump()`](../api/basic_json/dump.md) function only works with UTF-8 encoded strings; that is, if you assign a `std::string` to a JSON value, make sure it is UTF-8 encoded. See the FAQ entry on [serializing untrusted or invalid UTF-8](faq.md#serializing-untrusted-or-invalid-utf-8) for background and the recommended fix. -If [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled, the binary writers -[`to_cbor()`](../api/basic_json/to_cbor.md), [`to_ubjson()`](../api/basic_json/to_ubjson.md), +The binary writers [`to_cbor()`](../api/basic_json/to_cbor.md), [`to_ubjson()`](../api/basic_json/to_ubjson.md), [`to_bjdata()`](../api/basic_json/to_bjdata.md), and [`to_bson()`](../api/basic_json/to_bson.md) throw this exception -for a string value or object key that is not valid UTF-8 as well. +as well for a string value or object key that is not valid UTF-8 if their `error_handler` is `strict` (the default if +[`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled). So does +[`to_msgpack()`](../api/basic_json/to_msgpack.md) if `error_handler_t::strict` is passed. !!! failure "Example message" diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 3d79ad41b..1a78a6502 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -31,6 +31,7 @@ #include #include #include +#include #include #include #include @@ -108,8 +109,16 @@ class binary_reader @brief create a binary reader @param[in] adapter input adapter to read from + @param[in] format the binary format to parse + @param[in] error_handler how to treat text strings and object keys that + are not well-formed UTF-8; none of the supported formats + requires a decoder to reject those, so the default is to + @ref error_handler_t::keep them unchanged, as every binary + reader did before this parameter existed */ - explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json) noexcept : ia(std::move(adapter)), input_format(format) + explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json, + const error_handler_t error_handler = error_handler_t::keep) noexcept + : ia(std::move(adapter)), input_format(format), error_handler(error_handler) { (void)detail::is_sax_static_asserts {}; } @@ -428,7 +437,7 @@ class binary_reader { if (get_bson_cstr_bulk(result, std::integral_constant {})) { - return true; + return check_string_utf8(result, "key"); } auto out = std::back_inserter(result); @@ -441,7 +450,7 @@ class binary_reader } if (current == 0x00) { - return true; + return check_string_utf8(result, "key"); } *out++ = static_cast(current); } @@ -522,7 +531,7 @@ class binary_reader "string"), nullptr)); } - return true; + return check_string_utf8(result, "string"); } /*! @@ -1149,7 +1158,7 @@ class binary_reader @return whether string creation completed */ - bool get_cbor_string(string_t& result) + bool get_cbor_string(string_t& result, const char* context = "string") { // number of indefinite-length strings that have been opened and not // closed yet. RFC 8949, Section 3.2.3 does not permit nesting them, @@ -1179,7 +1188,7 @@ class binary_reader { if (--open == 0) { - return true; + return check_string_utf8(result, context); } get(); continue; @@ -1192,7 +1201,7 @@ class binary_reader if (open == 0) { - return true; + return check_string_utf8(result, context); } get(); @@ -1216,7 +1225,7 @@ class binary_reader // EOF and major type 3 (text string) are left to get_cbor_string if (current == char_traits::eof() || (static_cast(current) & 0xE0u) == 0x60u) { - return get_cbor_string(result); + return get_cbor_string(result, "key"); } const char* found = nullptr; @@ -2004,7 +2013,7 @@ class binary_reader @return whether string creation completed */ - bool get_msgpack_string(string_t& result) + bool get_msgpack_string(string_t& result, const char* context = "string") { if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "string"))) { @@ -2047,25 +2056,25 @@ class binary_reader case 0xBE: case 0xBF: { - return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result); + return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result) && check_string_utf8(result, context); } case 0xD9: // str 8 { std::uint8_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDA: // str 16 { std::uint16_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDB: // str 32 { std::uint32_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } default: @@ -2143,7 +2152,7 @@ class binary_reader // byte 0xC1 are left to get_msgpack_string if (current == char_traits::eof()) { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } if (current <= 0x7F || current >= 0xE0) { @@ -2159,7 +2168,7 @@ class binary_reader } else { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } break; } @@ -2405,7 +2414,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, true, "key") || !sax->key(key))) { return false; } @@ -2427,7 +2436,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false, "key") || !sax->key(key))) { return false; } @@ -2495,7 +2504,7 @@ class binary_reader @return whether string creation completed */ - bool get_ubjson_string(string_t& result, const bool get_char = true) + bool get_ubjson_string(string_t& result, const bool get_char = true, const char* context = "string") { if (get_char) { @@ -2516,31 +2525,31 @@ class binary_reader case 'U': { std::uint8_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'i': { std::int8_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'I': { std::int16_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'l': { std::int32_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'L': { std::int64_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'u': @@ -2550,7 +2559,7 @@ class binary_reader break; } std::uint16_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'm': @@ -2560,7 +2569,7 @@ class binary_reader break; } std::uint32_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'M': @@ -2570,7 +2579,7 @@ class binary_reader break; } std::uint64_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } default: @@ -4044,15 +4053,53 @@ class binary_reader const NumberType len, string_t& result) { - // Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the - // choice to the decoder), MessagePack (whose spec explicitly allows - // a str object to contain an invalid byte sequence), UBJSON, BJData, - // or BSON requires a decoder to reject ill-formed UTF-8. The bytes - // are kept unchanged; dump() and the binary writers are the ones - // that check them and report type_error.316 if they are not valid. + // Strings are taken as is by default: none of CBOR (RFC 8949 §3.1 + // leaves the choice to the decoder), MessagePack (whose spec + // explicitly allows a str object to contain an invalid byte + // sequence), UBJSON, BJData, or BSON requires a decoder to reject + // ill-formed UTF-8. Checking (and, with @ref error_handler_t::strict, + // rejecting, or with `replace`/`ignore`, sanitizing) is opt-in via + // @ref error_handler, applied once the whole string (all chunks of + // an indefinite-length CBOR string included) has been assembled, by + // @ref check_string_utf8 at the call site. return get_bytes(format, len, "string", result); } + /*! + @brief validate a decoded text string (value or object key) against @ref error_handler + + None of the binary formats requires a decoder to reject ill-formed UTF-8 + in a text string (see @ref get_string), so by default + (@ref error_handler_t::keep) this does nothing. A stricter + @ref error_handler opts into the same well-formedness check @ref + serializer::dump_escaped_impl applies when dumping a string: + @ref error_handler_t::strict rejects ill-formed input with + parse_error.113 (honoring `allow_exceptions` via @a sax), while + @ref error_handler_t::replace / @ref error_handler_t::ignore sanitize + @a result in place, using the exact same rules. + + @param[in,out] result the already assembled string to check + @param[in] context further context information (for diagnostics) + @return whether @a result is acceptable (always true for `keep`) + */ + bool check_string_utf8(string_t& result, const char* context) + { + if (error_handler == error_handler_t::keep || is_valid_utf8(result)) + { + return true; + } + + if (error_handler == error_handler_t::strict) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format, "invalid string: ill-formed UTF-8 byte", context), nullptr)); + } + + result = sanitize_utf8(result, error_handler); + return true; + } + /*! @brief create a byte array by reading bytes from the input @@ -4226,6 +4273,9 @@ class binary_reader /// input format const input_format_t input_format = input_format_t::json; + /// how to treat text strings/object keys that are not well-formed UTF-8 + const error_handler_t error_handler = error_handler_t::keep; + /// the SAX parser json_sax_t* sax = nullptr; diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index da5aa0cf4..e74671db2 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -26,6 +26,7 @@ #include #include #include +#include #include #include #include @@ -93,8 +94,12 @@ class binary_writer @param[in] sink output sink to write to (a value-type sink such as output_vector_sink, or output_adapter_sink wrapping a type-erased output adapter) + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ - explicit binary_writer(OutputSinkType sink) : oa(std::move(sink)) + explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(std::move(sink)), error_handler(error_handler_) {} /*! @@ -107,16 +112,20 @@ class binary_writer from one. @param[in] adapter output adapter to write to + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ template < typename SinkType = OutputSinkType, typename std::enable_if < std::is_constructible>::value, int >::type = 0 > - explicit binary_writer(output_adapter_t adapter) : oa(SinkType(std::move(adapter))) + explicit binary_writer(output_adapter_t adapter, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(SinkType(std::move(adapter))), error_handler(error_handler_) {} /*! @param[in] j JSON value to serialize - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string - value or an object key is not valid UTF-8 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -147,8 +156,8 @@ class binary_writer /*! @param[in] j JSON value to serialize - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string - value or an object key is not valid UTF-8 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -215,15 +224,16 @@ class binary_writer case value_t::string: { - check_text_utf8(*j.m_data.m_value.string, j); + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); // step 1: write control byte and the string length - write_cbor_head(0x60, j.m_data.m_value.string->size()); + write_cbor_head(0x60, value.size()); // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -296,8 +306,14 @@ class binary_writer // el.first is checked here, against the object as // diagnostics context, because write_cbor(el.first) // converts it to a temporary basic_json that would be - // used as the context instead - check_text_utf8(el.first, j); + // used as the context instead; for error_handler_t::keep + // and ::replace/::ignore the recursive write_cbor(el.first) + // call below handles the key like any other string, so no + // separate check is needed here for those + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_cbor(el.first); write_cbor(el.second); } @@ -445,8 +461,11 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); + const auto N = to_msgpack_length(value.size(), j); if (N <= 31) { // fixstr @@ -473,8 +492,8 @@ class binary_writer // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -621,6 +640,13 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // as in write_cbor, el.first is checked here against the + // object as diagnostics context; the recursive call below + // handles keep/replace/ignore like any other string + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_msgpack(el.first); write_msgpack(el.second); } @@ -640,8 +666,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string - value or an object key is not valid UTF-8 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -691,16 +717,17 @@ class binary_writer case value_t::string: { - check_text_utf8(*j.m_data.m_value.string, j); + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); if (add_prefix) { oa.write_character(to_char_type('S')); } - write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); + write_number_with_ubjson_prefix(value.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -855,11 +882,12 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { - check_text_utf8(el.first, j); - write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); + string_t storage; + const string_t& key = sanitize_utf8_for_write(el.first, j, storage); + write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(el.first.data()), - el.first.size()); + reinterpret_cast(key.data()), + key.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } @@ -902,10 +930,10 @@ class binary_writer and the entry name size (and its null-terminator). @throw out_of_range.409 if @a name contains U+0000, before anything is written - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a name is - not valid UTF-8, before anything is written + @throw type_error.316 if @a name is not valid UTF-8, before anything is + written */ - static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) + std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { const auto it = name.find(static_cast(0)); if (JSON_HEDLEY_UNLIKELY(it != BasicJsonType::string_t::npos)) @@ -913,9 +941,10 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - check_text_utf8(name, j); + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(name, j, storage); - return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; + return /*id*/ 1ul + sanitized.size() + /*zero-terminator*/1u; } /*! @@ -935,14 +964,28 @@ class binary_writer /*! @brief Writes the given @a element_type and @a name to the output adapter + + @a name has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_entry_header_size during the earlier + size pass, so only @ref error_handler_t::replace / @ref + error_handler_t::ignore need to sanitize it again here, to actually write + the bytes that size was computed from. */ void write_bson_entry_header(const string_t& name, const std::uint8_t element_type) { oa.write_character(to_char_type(element_type)); - oa.write_characters( - reinterpret_cast(name.data()), - name.size()); + + if (error_handler == error_handler_t::keep || error_handler == error_handler_t::strict || is_valid_utf8(name)) + { + oa.write_characters(reinterpret_cast(name.data()), name.size()); + } + else + { + const string_t sanitized = sanitize_utf8(name, error_handler); + oa.write_characters(reinterpret_cast(sanitized.data()), sanitized.size()); + } + // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -970,8 +1013,8 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a value - is not valid UTF-8, before anything is written + @throw type_error.316 if @a value is not valid UTF-8, before anything is + written @note The UTF-8 check is skipped if @a value is already too long for the 32-bit BSON length field (@ref to_bson_length rejects it later, once @@ -979,27 +1022,41 @@ class binary_writer from reading past a StringType that reports a size larger than what it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) + std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) { - check_text_utf8(value, j); + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, j, storage); + return sizeof(std::int32_t) + sanitized.size() + 1ul; } return sizeof(std::int32_t) + value.size() + 1ul; } /*! @brief Writes a BSON element with key @a name and string value @a value + + @a value has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_string_size during the earlier size + pass, so only @ref error_handler_t::replace / @ref error_handler_t::ignore + need to sanitize it again here, to actually write the bytes that size was + computed from. */ void write_bson_string(const string_t& name, const string_t& value) { write_bson_entry_header(name, 0x02); - write_number(to_bson_length(value.size() + 1ul), true); + const bool sanitize = error_handler != error_handler_t::keep + && error_handler != error_handler_t::strict + && !is_valid_utf8(value); + const string_t sanitized = sanitize ? sanitize_utf8(value, error_handler) : string_t{}; + const string_t& written = sanitize ? sanitized : value; + + write_number(to_bson_length(written.size() + 1ul), true); oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + reinterpret_cast(written.data()), + written.size()); // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -1113,10 +1170,10 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a j is a - string that is not valid UTF-8, before anything is written + @throw type_error.316 if @a j is a string that is not valid UTF-8, before + anything is written */ - static std::size_t calc_bson_value_size(const BasicJsonType& j) + std::size_t calc_bson_value_size(const BasicJsonType& j) { switch (j.type()) { @@ -1249,10 +1306,10 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string - value or a key is not valid UTF-8, before anything is written + @throw type_error.316 if a string value or a key is not valid UTF-8, + before anything is written */ - static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) + std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { // the object or array whose entries are being sized, and the ones it // is in; nothing is allocated unless the document nests @@ -2171,26 +2228,54 @@ class binary_writer } /*! - @brief check a CBOR, UBJSON, BJData, or BSON text string for valid UTF-8 + @brief return @a s as it should be written, honoring @ref error_handler - The check only happens if JSON_STRICT_BINARY_UTF8 is enabled. Otherwise, - the bytes are written unchanged, as before version 3.13.0. MessagePack - always writes the bytes as is, and BON8 always checks them (see - @ref check_utf8). + Used by @ref write_cbor, @ref write_msgpack, @ref write_ubjson (and so + @ref write_bjdata), and the BSON writing functions for string values and + object keys; never by @ref write_bon8, which always validates, since UTF-8 + lead bytes are structural there. - @param[in] s the string to check - @param[in] context the value that holds @a s (for diagnostics) - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a s is - not valid UTF-8 + - @ref error_handler_t::keep: @a s is returned unchanged, without even + checking it (the behavior of release 3.12.0 and earlier). + - @ref error_handler_t::strict: @ref check_utf8 is called, which throws + type_error.316 if @a s is not valid UTF-8. + - @ref error_handler_t::replace / @ref error_handler_t::ignore: @a s is + sanitized into @a storage with exactly the rules @ref + serializer::dump_escaped_impl uses, so that parsing what @ref + basic_json::dump produces for the same string and the same handler + yields the same result. + + Well-formed input is never copied: this returns a reference to @a s + itself in every case but a sanitized `replace`/`ignore` one, so @a + storage must outlive the returned reference only then. + + @param[in] s the string (value or object key) to write + @param[in] context the value @a s belongs to (for diagnostics) + @param[out] storage backing storage for a sanitized copy + + @return a reference to @a s, or to @a storage once it holds a sanitized copy */ - static void check_text_utf8(const string_t& s, const BasicJsonType& context) + const string_t& sanitize_utf8_for_write(const string_t& s, const BasicJsonType& context, string_t& storage) const { -#if JSON_STRICT_BINARY_UTF8 - check_utf8(s, context); -#else - static_cast(s); - static_cast(context); -#endif + switch (error_handler) + { + case error_handler_t::keep: + return s; + + case error_handler_t::strict: + check_utf8(s, context); + return s; + + case error_handler_t::replace: + case error_handler_t::ignore: + default: + if (is_valid_utf8(s)) + { + return s; + } + storage = sanitize_utf8(s, error_handler); + return storage; + } } /*! @@ -2521,6 +2606,10 @@ class binary_writer /// the output OutputSinkType oa; + + /// how to treat a string value or object key that is not valid UTF-8 + /// (CBOR, MessagePack, UBJSON, BJData, and BSON; not BON8) + const error_handler_t error_handler = binary_writer_default_error_handler(); }; } // namespace detail diff --git a/include/nlohmann/detail/output/error_handler.hpp b/include/nlohmann/detail/output/error_handler.hpp new file mode 100644 index 000000000..b95d70fce --- /dev/null +++ b/include/nlohmann/detail/output/error_handler.hpp @@ -0,0 +1,50 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// how to treat decoding errors +/// +/// @ref basic_json::dump uses this to decide what to do with ill-formed +/// UTF-8 while escaping a string, and the binary writers (@ref +/// basic_json::to_cbor, @ref basic_json::to_ubjson, @ref +/// basic_json::to_bjdata, @ref basic_json::to_bson) use it the same way for +/// string values and object keys. The binary readers (@ref +/// basic_json::from_cbor, @ref basic_json::from_msgpack, @ref +/// basic_json::from_ubjson, @ref basic_json::from_bjdata, @ref +/// basic_json::from_bson) use it to decide whether to check text strings +/// and object keys for well-formed UTF-8 at all, since none of those +/// formats requires a decoder to do so. +enum class error_handler_t +{ + strict, ///< throw a type_error/parse_error exception in case of invalid UTF-8 + replace, ///< replace invalid UTF-8 sequences with U+FFFD + ignore, ///< ignore invalid UTF-8 sequences + keep ///< keep invalid UTF-8 sequences unchanged +}; + +/// the default error handler of the CBOR, UBJSON, BJData, and BSON writers: +/// error_handler_t::strict if JSON_STRICT_BINARY_UTF8 is enabled, otherwise +/// error_handler_t::keep (the behavior before version 3.13.0) +constexpr error_handler_t binary_writer_default_error_handler() noexcept +{ +#if JSON_STRICT_BINARY_UTF8 + return error_handler_t::strict; +#else + return error_handler_t::keep; +#endif +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index 1c519d4c7..dcaae8055 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -27,6 +27,7 @@ #include #include #include +#include #include #include #include @@ -41,14 +42,6 @@ namespace detail // serialization // /////////////////// -/// how to treat decoding errors -enum class error_handler_t -{ - strict, ///< throw a type_error exception in case of invalid UTF-8 - replace, ///< replace invalid UTF-8 sequences with U+FFFD - ignore ///< ignore invalid UTF-8 sequences -}; - template class serializer { @@ -839,6 +832,16 @@ class serializer // EnsureAscii parameter is used, non-ASCII characters if ((codepoint <= 0x1F) || (EnsureAscii && (codepoint >= 0x7F))) { + if (EnsureAscii && error_handler == error_handler_t::keep) + { + // this character was buffered as raw bytes + // below in case it turned out to be part of + // an ill-formed sequence (which is kept as + // is); now that it decoded to a well-formed + // code point, undo that and \u-escape it + // like any other character instead + bytes = bytes_after_last_accept; + } if (codepoint <= 0xFFFF) { write_u_escape(bytes, static_cast(codepoint)); @@ -937,6 +940,44 @@ class serializer break; } + case error_handler_t::keep: + { + // the bytes of this (now abandoned) ill-formed + // sequence seen so far are already buffered below + // and are kept unchanged in the output + if (undumped_chars > 0) + { + // the byte that ended the sequence may be OK + // for itself (e.g., a quote that must still be + // escaped, or the lead byte of a well-formed + // code point), so read it again + --i; + } + else + { + // a byte that cannot start a sequence (e.g., + // 0xFF or a stray continuation byte) is kept + // as well + string_buffer[bytes++] = s[i]; + } + + // write buffer and reset index; there must be 13 bytes + // left, as this is the maximal number of bytes to be + // written ("\uxxxx\uxxxx\0") for one code point + if (string_buffer.size() - bytes < 13) + { + put_buffer(string_buffer, bytes); + bytes = 0; + } + + bytes_after_last_accept = bytes; + undumped_chars = 0; + + // continue processing the string + state = UTF8_ACCEPT; + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -945,9 +986,12 @@ class serializer default: // decode found yet incomplete multibyte code point { - if (!EnsureAscii) + if (!EnsureAscii || error_handler == error_handler_t::keep) { - // code point will not be escaped - copy byte to buffer + // code point will not be escaped (or will be kept as + // is if it turns out to be ill-formed) - copy byte to + // buffer; dropped again above if it decodes to a + // well-formed code point that needs \u-escaping string_buffer[bytes++] = s[i]; } ++undumped_chars; @@ -998,6 +1042,14 @@ class serializer break; } + case error_handler_t::keep: + { + // write the ill-formed trailing bytes as is; they were + // buffered above regardless of EnsureAscii + put_buffer(string_buffer, bytes); + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index 2b6864d0d..4ef748f13 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -16,6 +16,7 @@ #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -179,5 +180,134 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std: return state; } +/*! +@brief check a string for well-formed UTF-8 (RFC 3629, section 4) + +Used by the binary readers (CBOR, MessagePack, UBJSON, BJData, BSON) when an +@ref error_handler_t other than `keep` is requested for a text string value +or object key: none of those formats requires a decoder to reject ill-formed +UTF-8 on its own, so the check is opt-in there, unlike the JSON lexer and the +serializer's @ref decode -based escaping, which always run it. + +@param[in] s the string to check +@param[in] first the index to start checking at +@return whether `s.substr(first)` is well-formed UTF-8 + +@sa @ref decode +*/ +template +inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept +{ + std::uint8_t state = UTF8_ACCEPT; + std::uint32_t codepoint = 0; + + for (std::size_t i = first; i < s.size(); ++i) + { + decode(state, codepoint, static_cast(s[i])); + if (state == UTF8_REJECT) + { + return false; + } + } + + return state == UTF8_ACCEPT; +} + +/*! +@brief sanitize a string with ill-formed UTF-8 for @ref error_handler_t::replace or @ref error_handler_t::ignore + +Replaces every maximal ill-formed subsequence with U+FFFD (`replace`) or +drops it (`ignore`), using exactly the same boundaries @ref +serializer::dump_escaped_impl uses while escaping a string: a byte that does +not extend the sequence started by the previous byte(s) is reread as the +start of a new one, instead of being swallowed along with them. + +@pre @a error_handler is @ref error_handler_t::replace or @ref error_handler_t::ignore +@note Well-formed input is copied through unchanged, including bytes (e.g. + control characters or quotes) that @ref serializer::dump_escaped_impl + would itself escape; this function only concerns itself with + well-formedness, not with producing valid JSON text. + +@param[in] s the string to sanitize +@param[in] error_handler @ref error_handler_t::replace or @ref error_handler_t::ignore + +@return @a s with every ill-formed subsequence replaced or removed + +@sa @ref decode +*/ +template +inline StringType sanitize_utf8(const StringType& s, const error_handler_t error_handler) +{ + JSON_ASSERT(error_handler == error_handler_t::replace || error_handler == error_handler_t::ignore); + + StringType result; + result.reserve(s.size()); + + std::uint32_t codepoint = 0; + std::uint8_t state = UTF8_ACCEPT; + // length of result after the last accepted code point + std::size_t result_len_after_last_accept = 0; + // whether bytes of an as yet unresolved sequence were already appended + bool pending = false; + + for (std::size_t i = 0; i < s.size(); ++i) + { + switch (decode(state, codepoint, static_cast(s[i]))) + { + case UTF8_ACCEPT: // decode found a well-formed code point + { + result.push_back(s[i]); + result_len_after_last_accept = result.size(); + pending = false; + break; + } + + case UTF8_REJECT: // decode found an ill-formed byte + { + // in case we saw this byte for the first time, read it again, + // because it may be fine for itself, just not for the + // sequence that came before it + if (pending) + { + --i; + } + + // drop the bytes of the ill-formed sequence buffered below + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + result_len_after_last_accept = result.size(); + } + + pending = false; + state = UTF8_ACCEPT; + break; + } + + default: // decode found yet incomplete multibyte code point + { + result.push_back(s[i]); + pending = true; + break; + } + } + } + + // the string ended with an incomplete sequence + if (state != UTF8_ACCEPT) + { + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + } + } + + return result; +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 07b625a89..421feda50 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -195,9 +195,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // used by the vector-returning to_* overloads template using vector_binary_writer = ::nlohmann::detail::binary_writer>; - template static vector_binary_writer vector_writer(std::vector& v) + template static vector_binary_writer vector_writer( + std::vector& v, const ::nlohmann::detail::error_handler_t error_handler = ::nlohmann::detail::binary_writer_default_error_handler()) { - return vector_binary_writer(::nlohmann::detail::output_vector_sink(v)); + return vector_binary_writer(::nlohmann::detail::output_vector_sink(v), error_handler); } JSON_PRIVATE_UNLESS_TESTED: @@ -5583,11 +5584,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template static basic_json from_binary_impl(InputAdapterType ia, const input_format_t format, const bool strict, const bool allow_exceptions, + const error_handler_t error_handler = error_handler_t::keep, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { basic_json result; detail::json_sax_dom_parser sdp(result, allow_exceptions); - binary_reader reader(std::move(ia), format); + binary_reader reader(std::move(ia), format, error_handler); if (!reader.sax_parse(&sdp, strict, tag_handler)) { result = value_t::discarded; @@ -5605,78 +5607,87 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec public: /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static std::vector to_cbor(const basic_json& j) + static std::vector to_cbor(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_cbor(j); + vector_writer(result, error_handler).write_cbor(j); return result; } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static std::vector to_msgpack(const basic_json& j) + static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_msgpack(j); + vector_writer(result, error_handler).write_msgpack(j); return result; } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static std::vector to_ubjson(const basic_json& j, const bool use_size = false, - const bool use_type = false) + const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type); return result; } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a BJData serialization of a given JSON value @@ -5684,11 +5695,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_bjdata(const basic_json& j, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type, true, true, version); return result; } @@ -5696,42 +5708,47 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BJData serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static std::vector to_bson(const basic_json& j) + static std::vector to_bson(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_bson(j); + vector_writer(result, error_handler).write_bson(j); return result; } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BON8 serialization of a given JSON value @@ -5765,9 +5782,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(InputType&& i, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } /// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5778,9 +5796,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(IteratorType first, SentinelType last, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } template @@ -5801,7 +5820,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, error_handler_t::keep, tag_handler); } /// @brief create a JSON value from an input in MessagePack format @@ -5810,9 +5829,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5822,9 +5842,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } template @@ -5852,9 +5873,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5864,9 +5886,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } template @@ -5894,9 +5917,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5906,9 +5930,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BON8 format @@ -5940,9 +5965,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5952,9 +5978,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions, error_handler); } template diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 16e87cc86..511401699 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -104,6 +104,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -140,14 +144,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -156,7 +166,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -6151,6 +6162,59 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// how to treat decoding errors +/// +/// @ref basic_json::dump uses this to decide what to do with ill-formed +/// UTF-8 while escaping a string, and the binary writers (@ref +/// basic_json::to_cbor, @ref basic_json::to_ubjson, @ref +/// basic_json::to_bjdata, @ref basic_json::to_bson) use it the same way for +/// string values and object keys. The binary readers (@ref +/// basic_json::from_cbor, @ref basic_json::from_msgpack, @ref +/// basic_json::from_ubjson, @ref basic_json::from_bjdata, @ref +/// basic_json::from_bson) use it to decide whether to check text strings +/// and object keys for well-formed UTF-8 at all, since none of those +/// formats requires a decoder to do so. +enum class error_handler_t +{ + strict, ///< throw a type_error/parse_error exception in case of invalid UTF-8 + replace, ///< replace invalid UTF-8 sequences with U+FFFD + ignore, ///< ignore invalid UTF-8 sequences + keep ///< keep invalid UTF-8 sequences unchanged +}; + +/// the default error handler of the CBOR, UBJSON, BJData, and BSON writers: +/// error_handler_t::strict if JSON_STRICT_BINARY_UTF8 is enabled, otherwise +/// error_handler_t::keep (the behavior before version 3.13.0) +constexpr error_handler_t binary_writer_default_error_handler() noexcept +{ +#if JSON_STRICT_BINARY_UTF8 + return error_handler_t::strict; +#else + return error_handler_t::keep; +#endif +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -6252,13 +6316,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally written by Björn Hoehrmann. See http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. -The library checks UTF-8 well-formedness (RFC 3629, section 4) in four +The library checks UTF-8 well-formedness (RFC 3629, section 4) in three places, which differ in speed, diagnostics, and how they read the input: -- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in - strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR, - MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in - text strings at decode time). +- decode() below: the serializer, to escape and, in strict mode, reject + ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON, + UBJSON and BJData readers do not use it: none of those specs requires a + decoder to reject ill-formed UTF-8 in text strings, so the readers keep + the bytes as is and leave the check to dump() and the binary writers. - the per-lead-byte switch in lexer::scan_string(): JSON text, with a diagnostic for each kind of error. - validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's @@ -6314,19 +6379,19 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std: } /*! -@brief check whether a string consists solely of valid UTF-8 +@brief check a string for well-formed UTF-8 (RFC 3629, section 4) -Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text -strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the -MessagePack/BSON specifications all require text strings to be UTF-8), so -that malformed input is caught immediately instead of only surfacing later -as a type_error.316 when the resulting value is dumped. +Used by the binary readers (CBOR, MessagePack, UBJSON, BJData, BSON) when an +@ref error_handler_t other than `keep` is requested for a text string value +or object key: none of those formats requires a decoder to reject ill-formed +UTF-8 on its own, so the check is opt-in there, unlike the JSON lexer and the +serializer's @ref decode -based escaping, which always run it. @param[in] s the string to check -@param[in] first index of the first byte to check; the bytes before it are - assumed to have been validated already and to end on a - code point boundary -@return whether @a s (from index @a first on) is valid UTF-8 +@param[in] first the index to start checking at +@return whether `s.substr(first)` is well-formed UTF-8 + +@sa @ref decode */ template inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept @@ -6346,6 +6411,102 @@ inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noex return state == UTF8_ACCEPT; } +/*! +@brief sanitize a string with ill-formed UTF-8 for @ref error_handler_t::replace or @ref error_handler_t::ignore + +Replaces every maximal ill-formed subsequence with U+FFFD (`replace`) or +drops it (`ignore`), using exactly the same boundaries @ref +serializer::dump_escaped_impl uses while escaping a string: a byte that does +not extend the sequence started by the previous byte(s) is reread as the +start of a new one, instead of being swallowed along with them. + +@pre @a error_handler is @ref error_handler_t::replace or @ref error_handler_t::ignore +@note Well-formed input is copied through unchanged, including bytes (e.g. + control characters or quotes) that @ref serializer::dump_escaped_impl + would itself escape; this function only concerns itself with + well-formedness, not with producing valid JSON text. + +@param[in] s the string to sanitize +@param[in] error_handler @ref error_handler_t::replace or @ref error_handler_t::ignore + +@return @a s with every ill-formed subsequence replaced or removed + +@sa @ref decode +*/ +template +inline StringType sanitize_utf8(const StringType& s, const error_handler_t error_handler) +{ + JSON_ASSERT(error_handler == error_handler_t::replace || error_handler == error_handler_t::ignore); + + StringType result; + result.reserve(s.size()); + + std::uint32_t codepoint = 0; + std::uint8_t state = UTF8_ACCEPT; + // length of result after the last accepted code point + std::size_t result_len_after_last_accept = 0; + // whether bytes of an as yet unresolved sequence were already appended + bool pending = false; + + for (std::size_t i = 0; i < s.size(); ++i) + { + switch (decode(state, codepoint, static_cast(s[i]))) + { + case UTF8_ACCEPT: // decode found a well-formed code point + { + result.push_back(s[i]); + result_len_after_last_accept = result.size(); + pending = false; + break; + } + + case UTF8_REJECT: // decode found an ill-formed byte + { + // in case we saw this byte for the first time, read it again, + // because it may be fine for itself, just not for the + // sequence that came before it + if (pending) + { + --i; + } + + // drop the bytes of the ill-formed sequence buffered below + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + result_len_after_last_accept = result.size(); + } + + pending = false; + state = UTF8_ACCEPT; + break; + } + + default: // decode found yet incomplete multibyte code point + { + result.push_back(s[i]); + pending = true; + break; + } + } + } + + // the string ended with an incomplete sequence + if (state != UTF8_ACCEPT) + { + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + } + } + + return result; +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -13452,6 +13613,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -13532,8 +13695,16 @@ class binary_reader @brief create a binary reader @param[in] adapter input adapter to read from + @param[in] format the binary format to parse + @param[in] error_handler how to treat text strings and object keys that + are not well-formed UTF-8; none of the supported formats + requires a decoder to reject those, so the default is to + @ref error_handler_t::keep them unchanged, as every binary + reader did before this parameter existed */ - explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json) noexcept : ia(std::move(adapter)), input_format(format) + explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json, + const error_handler_t error_handler = error_handler_t::keep) noexcept + : ia(std::move(adapter)), input_format(format), error_handler(error_handler) { (void)detail::is_sax_static_asserts {}; } @@ -13852,7 +14023,7 @@ class binary_reader { if (get_bson_cstr_bulk(result, std::integral_constant {})) { - return true; + return check_string_utf8(result, "key"); } auto out = std::back_inserter(result); @@ -13865,7 +14036,7 @@ class binary_reader } if (current == 0x00) { - return true; + return check_string_utf8(result, "key"); } *out++ = static_cast(current); } @@ -13946,7 +14117,7 @@ class binary_reader "string"), nullptr)); } - return true; + return check_string_utf8(result, "string"); } /*! @@ -14573,7 +14744,7 @@ class binary_reader @return whether string creation completed */ - bool get_cbor_string(string_t& result) + bool get_cbor_string(string_t& result, const char* context = "string") { // number of indefinite-length strings that have been opened and not // closed yet. RFC 8949, Section 3.2.3 does not permit nesting them, @@ -14603,7 +14774,7 @@ class binary_reader { if (--open == 0) { - return true; + return check_string_utf8(result, context); } get(); continue; @@ -14616,7 +14787,7 @@ class binary_reader if (open == 0) { - return true; + return check_string_utf8(result, context); } get(); @@ -14640,7 +14811,7 @@ class binary_reader // EOF and major type 3 (text string) are left to get_cbor_string if (current == char_traits::eof() || (static_cast(current) & 0xE0u) == 0x60u) { - return get_cbor_string(result); + return get_cbor_string(result, "key"); } const char* found = nullptr; @@ -15428,7 +15599,7 @@ class binary_reader @return whether string creation completed */ - bool get_msgpack_string(string_t& result) + bool get_msgpack_string(string_t& result, const char* context = "string") { if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "string"))) { @@ -15471,25 +15642,25 @@ class binary_reader case 0xBE: case 0xBF: { - return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result); + return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result) && check_string_utf8(result, context); } case 0xD9: // str 8 { std::uint8_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDA: // str 16 { std::uint16_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDB: // str 32 { std::uint32_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } default: @@ -15567,7 +15738,7 @@ class binary_reader // byte 0xC1 are left to get_msgpack_string if (current == char_traits::eof()) { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } if (current <= 0x7F || current >= 0xE0) { @@ -15583,7 +15754,7 @@ class binary_reader } else { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } break; } @@ -15829,7 +16000,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, true, "key") || !sax->key(key))) { return false; } @@ -15851,7 +16022,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false, "key") || !sax->key(key))) { return false; } @@ -15919,7 +16090,7 @@ class binary_reader @return whether string creation completed */ - bool get_ubjson_string(string_t& result, const bool get_char = true) + bool get_ubjson_string(string_t& result, const bool get_char = true, const char* context = "string") { if (get_char) { @@ -15940,31 +16111,31 @@ class binary_reader case 'U': { std::uint8_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'i': { std::int8_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'I': { std::int16_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'l': { std::int32_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'L': { std::int64_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'u': @@ -15974,7 +16145,7 @@ class binary_reader break; } std::uint16_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'm': @@ -15984,7 +16155,7 @@ class binary_reader break; } std::uint32_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'M': @@ -15994,7 +16165,7 @@ class binary_reader break; } std::uint64_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } default: @@ -17468,27 +17639,50 @@ class binary_reader const NumberType len, string_t& result) { - // get_bytes() appends to result, and CBOR indefinite-length strings - // collect all their chunks in the same result; validating only the - // newly read bytes keeps the check linear in the input size - const std::size_t old_size = result.size(); - if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) + // Strings are taken as is by default: none of CBOR (RFC 8949 §3.1 + // leaves the choice to the decoder), MessagePack (whose spec + // explicitly allows a str object to contain an invalid byte + // sequence), UBJSON, BJData, or BSON requires a decoder to reject + // ill-formed UTF-8. Checking (and, with @ref error_handler_t::strict, + // rejecting, or with `replace`/`ignore`, sanitizing) is opt-in via + // @ref error_handler, applied once the whole string (all chunks of + // an indefinite-length CBOR string included) has been assembled, by + // @ref check_string_utf8 at the call site. + return get_bytes(format, len, "string", result); + } + + /*! + @brief validate a decoded text string (value or object key) against @ref error_handler + + None of the binary formats requires a decoder to reject ill-formed UTF-8 + in a text string (see @ref get_string), so by default + (@ref error_handler_t::keep) this does nothing. A stricter + @ref error_handler opts into the same well-formedness check @ref + serializer::dump_escaped_impl applies when dumping a string: + @ref error_handler_t::strict rejects ill-formed input with + parse_error.113 (honoring `allow_exceptions` via @a sax), while + @ref error_handler_t::replace / @ref error_handler_t::ignore sanitize + @a result in place, using the exact same rules. + + @param[in,out] result the already assembled string to check + @param[in] context further context information (for diagnostics) + @return whether @a result is acceptable (always true for `keep`) + */ + bool check_string_utf8(string_t& result, const char* context) + { + if (error_handler == error_handler_t::keep || is_valid_utf8(result)) { - return false; + return true; } - // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications - // all require text strings to be valid UTF-8; reject anything else - // right here so malformed input is caught at decode time instead of - // only surfacing later as a type_error.316 when the value is dumped - // (which would defeat allow_exceptions=false / strict discarding). - if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) + if (error_handler == error_handler_t::strict) { - return sax->parse_error(chars_read, get_token_string(), - parse_error::create(113, chars_read, - exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)); + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format, "invalid string: ill-formed UTF-8 byte", context), nullptr)); } + result = sanitize_utf8(result, error_handler); return true; } @@ -17665,6 +17859,9 @@ class binary_reader /// input format const input_format_t input_format = input_format_t::json; + /// how to treat text strings/object keys that are not well-formed UTF-8 + const error_handler_t error_handler = error_handler_t::keep; + /// the SAX parser json_sax_t* sax = nullptr; @@ -20753,6 +20950,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // __ _____ _____ _____ // __| | __| | | | JSON for Modern C++ @@ -21120,8 +21319,12 @@ class binary_writer @param[in] sink output sink to write to (a value-type sink such as output_vector_sink, or output_adapter_sink wrapping a type-erased output adapter) + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ - explicit binary_writer(OutputSinkType sink) : oa(std::move(sink)) + explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(std::move(sink)), error_handler(error_handler_) {} /*! @@ -21134,14 +21337,20 @@ class binary_writer from one. @param[in] adapter output adapter to write to + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ template < typename SinkType = OutputSinkType, typename std::enable_if < std::is_constructible>::value, int >::type = 0 > - explicit binary_writer(output_adapter_t adapter) : oa(SinkType(std::move(adapter))) + explicit binary_writer(output_adapter_t adapter, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(SinkType(std::move(adapter))), error_handler(error_handler_) {} /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -21172,6 +21381,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -21238,13 +21449,16 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - write_cbor_head(0x60, j.m_data.m_value.string->size()); + write_cbor_head(0x60, value.size()); // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -21314,6 +21528,17 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // el.first is checked here, against the object as + // diagnostics context, because write_cbor(el.first) + // converts it to a temporary basic_json that would be + // used as the context instead; for error_handler_t::keep + // and ::replace/::ignore the recursive write_cbor(el.first) + // call below handles the key like any other string, so no + // separate check is needed here for those + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_cbor(el.first); write_cbor(el.second); } @@ -21461,8 +21686,11 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); + const auto N = to_msgpack_length(value.size(), j); if (N <= 31) { // fixstr @@ -21489,8 +21717,8 @@ class binary_writer // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -21637,6 +21865,13 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // as in write_cbor, el.first is checked here against the + // object as diagnostics context; the recursive call below + // handles keep/replace/ignore like any other string + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_msgpack(el.first); write_msgpack(el.second); } @@ -21656,6 +21891,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -21705,14 +21942,17 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + if (add_prefix) { oa.write_character(to_char_type('S')); } - write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); + write_number_with_ubjson_prefix(value.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -21867,10 +22107,12 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { - write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); + string_t storage; + const string_t& key = sanitize_utf8_for_write(el.first, j, storage); + write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(el.first.data()), - el.first.size()); + reinterpret_cast(key.data()), + key.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } @@ -21911,8 +22153,12 @@ class binary_writer /*! @return The size of a BSON document entry header, including the id marker and the entry name size (and its null-terminator). + @throw out_of_range.409 if @a name contains U+0000, before anything is + written + @throw type_error.316 if @a name is not valid UTF-8, before anything is + written */ - static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) + std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { const auto it = name.find(static_cast(0)); if (JSON_HEDLEY_UNLIKELY(it != BasicJsonType::string_t::npos)) @@ -21920,8 +22166,10 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - static_cast(j); - return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(name, j, storage); + + return /*id*/ 1ul + sanitized.size() + /*zero-terminator*/1u; } /*! @@ -21941,14 +22189,28 @@ class binary_writer /*! @brief Writes the given @a element_type and @a name to the output adapter + + @a name has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_entry_header_size during the earlier + size pass, so only @ref error_handler_t::replace / @ref + error_handler_t::ignore need to sanitize it again here, to actually write + the bytes that size was computed from. */ void write_bson_entry_header(const string_t& name, const std::uint8_t element_type) { oa.write_character(to_char_type(element_type)); - oa.write_characters( - reinterpret_cast(name.data()), - name.size()); + + if (error_handler == error_handler_t::keep || error_handler == error_handler_t::strict || is_valid_utf8(name)) + { + oa.write_characters(reinterpret_cast(name.data()), name.size()); + } + else + { + const string_t sanitized = sanitize_utf8(name, error_handler); + oa.write_characters(reinterpret_cast(sanitized.data()), sanitized.size()); + } + // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -21976,24 +22238,50 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value + @throw type_error.316 if @a value is not valid UTF-8, before anything is + written + + @note The UTF-8 check is skipped if @a value is already too long for the + 32-bit BSON length field (@ref to_bson_length rejects it later, once + the size of the whole document is known); this also keeps the check + from reading past a StringType that reports a size larger than what + it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value) + std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { + if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, j, storage); + return sizeof(std::int32_t) + sanitized.size() + 1ul; + } return sizeof(std::int32_t) + value.size() + 1ul; } /*! @brief Writes a BSON element with key @a name and string value @a value + + @a value has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_string_size during the earlier size + pass, so only @ref error_handler_t::replace / @ref error_handler_t::ignore + need to sanitize it again here, to actually write the bytes that size was + computed from. */ void write_bson_string(const string_t& name, const string_t& value) { write_bson_entry_header(name, 0x02); - write_number(to_bson_length(value.size() + 1ul), true); + const bool sanitize = error_handler != error_handler_t::keep + && error_handler != error_handler_t::strict + && !is_valid_utf8(value); + const string_t sanitized = sanitize ? sanitize_utf8(value, error_handler) : string_t{}; + const string_t& written = sanitize ? sanitized : value; + + write_number(to_bson_length(written.size() + 1ul), true); oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + reinterpret_cast(written.data()), + written.size()); // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -22107,8 +22395,10 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written + @throw type_error.316 if @a j is a string that is not valid UTF-8, before + anything is written */ - static std::size_t calc_bson_value_size(const BasicJsonType& j) + std::size_t calc_bson_value_size(const BasicJsonType& j) { switch (j.type()) { @@ -22128,7 +22418,7 @@ class binary_writer return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string, j); case value_t::null: return 0ul; @@ -22241,8 +22531,10 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written + @throw type_error.316 if a string value or a key is not valid UTF-8, + before anything is written */ - static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) + std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { // the object or array whose entries are being sized, and the ones it // is in; nothing is allocated unless the document nests @@ -23119,7 +23411,7 @@ class binary_writer */ void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) { - check_bon8_utf8(s, context); + check_utf8(s, context); // a string that follows another string terminates it if (string_open) @@ -23149,7 +23441,7 @@ class binary_writer @throw type_error.316 if @a s is not valid UTF-8; the message names the first byte of the first invalid or incomplete sequence */ - static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + static void check_utf8(const string_t& s, const BasicJsonType& context) { static_cast(context); // only used when exceptions are enabled const auto* data = reinterpret_cast(s.data()); @@ -23160,6 +23452,57 @@ class binary_writer } } + /*! + @brief return @a s as it should be written, honoring @ref error_handler + + Used by @ref write_cbor, @ref write_msgpack, @ref write_ubjson (and so + @ref write_bjdata), and the BSON writing functions for string values and + object keys; never by @ref write_bon8, which always validates, since UTF-8 + lead bytes are structural there. + + - @ref error_handler_t::keep: @a s is returned unchanged, without even + checking it (the behavior of release 3.12.0 and earlier). + - @ref error_handler_t::strict: @ref check_utf8 is called, which throws + type_error.316 if @a s is not valid UTF-8. + - @ref error_handler_t::replace / @ref error_handler_t::ignore: @a s is + sanitized into @a storage with exactly the rules @ref + serializer::dump_escaped_impl uses, so that parsing what @ref + basic_json::dump produces for the same string and the same handler + yields the same result. + + Well-formed input is never copied: this returns a reference to @a s + itself in every case but a sanitized `replace`/`ignore` one, so @a + storage must outlive the returned reference only then. + + @param[in] s the string (value or object key) to write + @param[in] context the value @a s belongs to (for diagnostics) + @param[out] storage backing storage for a sanitized copy + + @return a reference to @a s, or to @a storage once it holds a sanitized copy + */ + const string_t& sanitize_utf8_for_write(const string_t& s, const BasicJsonType& context, string_t& storage) const + { + switch (error_handler) + { + case error_handler_t::keep: + return s; + + case error_handler_t::strict: + check_utf8(s, context); + return s; + + case error_handler_t::replace: + case error_handler_t::ignore: + default: + if (is_valid_utf8(s)) + { + return s; + } + storage = sanitize_utf8(s, error_handler); + return storage; + } + } + /*! @brief write an integer in the shortest encoding @@ -23488,6 +23831,10 @@ class binary_writer /// the output OutputSinkType oa; + + /// how to treat a string value or object key that is not valid UTF-8 + /// (CBOR, MessagePack, UBJSON, BJData, and BSON; not BON8) + const error_handler_t error_handler = binary_writer_default_error_handler(); }; } // namespace detail @@ -24649,6 +24996,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -24668,14 +25017,6 @@ namespace detail // serialization // /////////////////// -/// how to treat decoding errors -enum class error_handler_t -{ - strict, ///< throw a type_error exception in case of invalid UTF-8 - replace, ///< replace invalid UTF-8 sequences with U+FFFD - ignore ///< ignore invalid UTF-8 sequences -}; - template class serializer { @@ -25466,6 +25807,16 @@ class serializer // EnsureAscii parameter is used, non-ASCII characters if ((codepoint <= 0x1F) || (EnsureAscii && (codepoint >= 0x7F))) { + if (EnsureAscii && error_handler == error_handler_t::keep) + { + // this character was buffered as raw bytes + // below in case it turned out to be part of + // an ill-formed sequence (which is kept as + // is); now that it decoded to a well-formed + // code point, undo that and \u-escape it + // like any other character instead + bytes = bytes_after_last_accept; + } if (codepoint <= 0xFFFF) { write_u_escape(bytes, static_cast(codepoint)); @@ -25564,6 +25915,44 @@ class serializer break; } + case error_handler_t::keep: + { + // the bytes of this (now abandoned) ill-formed + // sequence seen so far are already buffered below + // and are kept unchanged in the output + if (undumped_chars > 0) + { + // the byte that ended the sequence may be OK + // for itself (e.g., a quote that must still be + // escaped, or the lead byte of a well-formed + // code point), so read it again + --i; + } + else + { + // a byte that cannot start a sequence (e.g., + // 0xFF or a stray continuation byte) is kept + // as well + string_buffer[bytes++] = s[i]; + } + + // write buffer and reset index; there must be 13 bytes + // left, as this is the maximal number of bytes to be + // written ("\uxxxx\uxxxx\0") for one code point + if (string_buffer.size() - bytes < 13) + { + put_buffer(string_buffer, bytes); + bytes = 0; + } + + bytes_after_last_accept = bytes; + undumped_chars = 0; + + // continue processing the string + state = UTF8_ACCEPT; + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -25572,9 +25961,12 @@ class serializer default: // decode found yet incomplete multibyte code point { - if (!EnsureAscii) + if (!EnsureAscii || error_handler == error_handler_t::keep) { - // code point will not be escaped - copy byte to buffer + // code point will not be escaped (or will be kept as + // is if it turns out to be ill-formed) - copy byte to + // buffer; dropped again above if it decodes to a + // well-formed code point that needs \u-escaping string_buffer[bytes++] = s[i]; } ++undumped_chars; @@ -25625,6 +26017,14 @@ class serializer break; } + case error_handler_t::keep: + { + // write the ill-formed trailing bytes as is; they were + // buffered above regardless of EnsureAscii + put_buffer(string_buffer, bytes); + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -26732,9 +27132,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // used by the vector-returning to_* overloads template using vector_binary_writer = ::nlohmann::detail::binary_writer>; - template static vector_binary_writer vector_writer(std::vector& v) + template static vector_binary_writer vector_writer( + std::vector& v, const ::nlohmann::detail::error_handler_t error_handler = ::nlohmann::detail::binary_writer_default_error_handler()) { - return vector_binary_writer(::nlohmann::detail::output_vector_sink(v)); + return vector_binary_writer(::nlohmann::detail::output_vector_sink(v), error_handler); } JSON_PRIVATE_UNLESS_TESTED: @@ -32120,11 +32521,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template static basic_json from_binary_impl(InputAdapterType ia, const input_format_t format, const bool strict, const bool allow_exceptions, + const error_handler_t error_handler = error_handler_t::keep, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { basic_json result; detail::json_sax_dom_parser sdp(result, allow_exceptions); - binary_reader reader(std::move(ia), format); + binary_reader reader(std::move(ia), format, error_handler); if (!reader.sax_parse(&sdp, strict, tag_handler)) { result = value_t::discarded; @@ -32142,78 +32544,87 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec public: /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static std::vector to_cbor(const basic_json& j) + static std::vector to_cbor(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_cbor(j); + vector_writer(result, error_handler).write_cbor(j); return result; } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static std::vector to_msgpack(const basic_json& j) + static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_msgpack(j); + vector_writer(result, error_handler).write_msgpack(j); return result; } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static std::vector to_ubjson(const basic_json& j, const bool use_size = false, - const bool use_type = false) + const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type); return result; } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a BJData serialization of a given JSON value @@ -32221,11 +32632,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_bjdata(const basic_json& j, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type, true, true, version); return result; } @@ -32233,42 +32645,47 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BJData serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static std::vector to_bson(const basic_json& j) + static std::vector to_bson(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_bson(j); + vector_writer(result, error_handler).write_bson(j); return result; } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BON8 serialization of a given JSON value @@ -32302,9 +32719,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(InputType&& i, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } /// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32315,9 +32733,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(IteratorType first, SentinelType last, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } template @@ -32338,7 +32757,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, error_handler_t::keep, tag_handler); } /// @brief create a JSON value from an input in MessagePack format @@ -32347,9 +32766,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32359,9 +32779,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } template @@ -32389,9 +32810,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32401,9 +32823,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } template @@ -32431,9 +32854,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32443,9 +32867,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BON8 format @@ -32477,9 +32902,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32489,9 +32915,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions, error_handler); } template @@ -33734,6 +34161,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION #undef JSON_STRICT_NUL_HANDLING + #undef JSON_STRICT_BINARY_UTF8 #endif // #include diff --git a/tests/src/unit-binary_utf8_error_handler.cpp b/tests/src/unit-binary_utf8_error_handler.cpp new file mode 100644 index 000000000..85505bffb --- /dev/null +++ b/tests/src/unit-binary_utf8_error_handler.cpp @@ -0,0 +1,362 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +using nlohmann::json; + +#include +#include + +namespace +{ + +struct ill_formed_case +{ + const char* name; + std::string bytes; +}; + +// RFC 3629 ill-formed sequences used throughout this file, plus one +// well-formed sequence for contrast +const std::vector ill_formed_cases = +{ + {"overlong", "\xC0\xAE"}, + {"lone_0xFF", "\xFF"}, + {"truncated", "\xE2\x82"}, + {"surrogate", "\xED\xA0\x80"}, +}; + +const std::string valid_sequence = "\xC3\xA9"; // U+00E9, "é" + +using eh = json::error_handler_t; +const std::vector all_handlers = {eh::strict, eh::replace, eh::ignore, eh::keep}; + +// what dump()+parse() produces for a sanitizing error_handler; this is the +// ground truth every binary writer/reader is checked against +std::string dump_and_parse(const std::string& raw, eh error_handler) +{ + return json::parse(json(raw).dump(-1, ' ', false, error_handler)).get(); +} + +} // namespace + +TEST_CASE("UTF-8 error_handler for the binary readers and writers") +{ + SECTION("writers: string value") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + const json jval = c.bytes; + + CHECK_THROWS_AS(json::to_cbor(jval, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_msgpack(jval, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_ubjson(jval, false, false, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&); + { + json jobj; + jobj["k"] = jval; + CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&); + } + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(json::to_cbor(jval, h)).get() == expected); + CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get() == expected); + CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get() == expected); + CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get() == expected); + { + json jobj; + jobj["k"] = jval; + const auto bytes = json::to_bson(jobj, h); + CHECK(json::from_bson(bytes)["k"].get() == expected); + } + } + + // keep: the writer passes the ill-formed bytes through unchanged, + // exactly as every binary writer did before this parameter existed + CHECK(json::from_cbor(json::to_cbor(jval, eh::keep)).get() == c.bytes); + CHECK(json::from_msgpack(json::to_msgpack(jval, eh::keep)).get() == c.bytes); + CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, eh::keep)).get() == c.bytes); + CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep)).get() == c.bytes); + { + json jobj; + jobj["k"] = jval; + const auto bytes = json::to_bson(jobj, eh::keep); + CHECK(json::from_bson(bytes)["k"].get() == c.bytes); + } + } + } + + SECTION("writers: object key") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + json jobj; + jobj[c.bytes] = 1; + + CHECK_THROWS_AS(json::to_cbor(jobj, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_msgpack(jobj, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_ubjson(jobj, false, false, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&); + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(json::to_cbor(jobj, h)).begin().key() == expected); + CHECK(json::from_msgpack(json::to_msgpack(jobj, h)).begin().key() == expected); + CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, h)).begin().key() == expected); + CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, h)).begin().key() == expected); + CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == expected); + } + + // keep: object keys round-trip unchanged too + CHECK(json::from_cbor(json::to_cbor(jobj, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_msgpack(json::to_msgpack(jobj, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_bson(json::to_bson(jobj, eh::keep)).begin().key() == c.bytes); + } + } + + SECTION("readers: string value") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + + // bytes produced the lenient (keep) way, as any binary reader + // accepted them before this parameter existed + const auto cbor_bytes = json::to_cbor(json(c.bytes), eh::keep); + const auto msgpack_bytes = json::to_msgpack(json(c.bytes)); // to_msgpack has no error_handler; always pass-through + const auto ubjson_bytes = json::to_ubjson(json(c.bytes), false, false, eh::keep); + const auto bjdata_bytes = json::to_bjdata(json(c.bytes), false, false, json::bjdata_version_t::draft2, eh::keep); + const auto bson_bytes = [&c] + { + json jobj; + jobj["k"] = c.bytes; + return json::to_bson(jobj, eh::keep); + }(); + + // keep (the default): bytes are kept unchanged + CHECK(json::from_cbor(cbor_bytes).get() == c.bytes); + CHECK(json::from_msgpack(msgpack_bytes).get() == c.bytes); + CHECK(json::from_ubjson(ubjson_bytes).get() == c.bytes); + CHECK(json::from_bjdata(bjdata_bytes).get() == c.bytes); + CHECK(json::from_bson(bson_bytes)["k"].get() == c.bytes); + + // strict: parse_error.113, discarded (not thrown) when allow_exceptions is false + CHECK_THROWS_AS(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, eh::strict), json::parse_error&); + CHECK(json::from_cbor(cbor_bytes, true, false, json::cbor_tag_handler_t::error, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_msgpack(msgpack_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_msgpack(msgpack_bytes, true, false, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_ubjson(ubjson_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_ubjson(ubjson_bytes, true, false, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_bjdata(bjdata_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_bjdata(bjdata_bytes, true, false, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_bson(bson_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_bson(bson_bytes, true, false, eh::strict).is_discarded()); + + // replace / ignore: match what dump() would have sanitized the same bytes to + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, h).get() == expected); + CHECK(json::from_msgpack(msgpack_bytes, true, true, h).get() == expected); + CHECK(json::from_ubjson(ubjson_bytes, true, true, h).get() == expected); + CHECK(json::from_bjdata(bjdata_bytes, true, true, h).get() == expected); + CHECK(json::from_bson(bson_bytes, true, true, h)["k"].get() == expected); + } + } + } + + SECTION("readers: object key") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + + json jobj; + jobj[c.bytes] = 1; + const auto cbor_bytes = json::to_cbor(jobj, eh::keep); + const auto msgpack_bytes = json::to_msgpack(jobj); + const auto ubjson_bytes = json::to_ubjson(jobj, false, false, eh::keep); + const auto bjdata_bytes = json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep); + const auto bson_bytes = json::to_bson(jobj, eh::keep); + + CHECK(json::from_cbor(cbor_bytes).begin().key() == c.bytes); + CHECK(json::from_msgpack(msgpack_bytes).begin().key() == c.bytes); + CHECK(json::from_ubjson(ubjson_bytes).begin().key() == c.bytes); + CHECK(json::from_bjdata(bjdata_bytes).begin().key() == c.bytes); + CHECK(json::from_bson(bson_bytes).begin().key() == c.bytes); + + CHECK_THROWS_AS(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_msgpack(msgpack_bytes, true, true, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_ubjson(ubjson_bytes, true, true, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_bjdata(bjdata_bytes, true, true, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_bson(bson_bytes, true, true, eh::strict), json::parse_error&); + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, h).begin().key() == expected); + CHECK(json::from_msgpack(msgpack_bytes, true, true, h).begin().key() == expected); + CHECK(json::from_ubjson(ubjson_bytes, true, true, h).begin().key() == expected); + CHECK(json::from_bjdata(bjdata_bytes, true, true, h).begin().key() == expected); + CHECK(json::from_bson(bson_bytes, true, true, h).begin().key() == expected); + } + } + } + + SECTION("well-formed UTF-8 is unaffected by error_handler") + { + const json jval = valid_sequence; + json jobj; + jobj[valid_sequence] = valid_sequence; + + for (const auto h : all_handlers) + { + CAPTURE(static_cast(h)); + + CHECK(json::from_cbor(json::to_cbor(jval, h)).get() == valid_sequence); + CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get() == valid_sequence); + CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get() == valid_sequence); + CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get() == valid_sequence); + CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == valid_sequence); + + CHECK(json::from_cbor(json::to_cbor(jval, eh::keep), true, true, json::cbor_tag_handler_t::error, h).get() == valid_sequence); + CHECK(json::from_msgpack(json::to_msgpack(jval), true, true, h).get() == valid_sequence); + } + } + + SECTION("dump() with error_handler_t::keep writes raw bytes as is") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + + const json jval = c.bytes; + const std::string dumped = jval.dump(-1, ' ', false, eh::keep); + CHECK(dumped.find(c.bytes) != std::string::npos); + + // even with ensure_ascii, the ill-formed bytes are written as is + const std::string dumped_ascii = jval.dump(-1, ' ', true, eh::keep); + CHECK(dumped_ascii.find(c.bytes) != std::string::npos); + } + + // well-formed characters around an ill-formed sequence are still + // escaped as usual under ensure_ascii + const json mixed = valid_sequence + ill_formed_cases[1].bytes; // "é" + lone 0xFF + const std::string dumped_mixed = mixed.dump(-1, ' ', true, eh::keep); + CHECK(dumped_mixed.find("\\u00e9") != std::string::npos); + CHECK(dumped_mixed.find(ill_formed_cases[1].bytes) != std::string::npos); + + // the byte that ends an ill-formed sequence is read again, so a quote, + // a backslash, or a control character after it is still escaped, and + // a well-formed code point after it is escaped under ensure_ascii + for (const bool ensure_ascii : + { + false, true + }) + { + CAPTURE(ensure_ascii); + CHECK(json("\xC3\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\\"\""); + CHECK(json("\xC3\\").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\\\\""); + CHECK(json("\xC3\n").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\n\""); + CHECK(json("\xE2\x82\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xE2\x82\\\"\""); + CHECK(json("\xFF\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xFF\\\"\""); + CHECK(json("a\xE2\x82").dump(-1, ' ', ensure_ascii, eh::keep) == "\"a\xE2\x82\""); + } + CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', false, eh::keep) == "\"\xC3\xC3\xA9\""); + CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', true, eh::keep) == "\"\xC3\\u00e9\""); + } + + SECTION("to_msgpack defaults to keep; to_bon8 is not affected by error_handler") + { + const json jval = ill_formed_cases[1].bytes; // lone 0xFF + + // to_msgpack's error_handler defaults to keep, as MessagePack's spec + // allows any bytes in a str, so the bytes are passed through + CHECK(json::to_msgpack(jval) == json::to_msgpack(jval, eh::keep)); + CHECK(json::from_msgpack(json::to_msgpack(jval)).get() == ill_formed_cases[1].bytes); + + // the diagnostics context of an ill-formed key is the object + json jobj; + jobj["\xFF"] = 1; + CHECK_THROWS_WITH_AS(json::to_msgpack(jobj, eh::strict), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + + // to_bon8 has no error_handler parameter; UTF-8 is structural for + // BON8, so it always rejects ill-formed input + CHECK_THROWS_AS(json::to_bon8(jval), json::type_error&); + } + + SECTION("allow_exceptions=false with error_handler_t::strict discards the value") + { + const auto bytes = json::to_cbor(json(ill_formed_cases[0].bytes), eh::keep); + const json result = json::from_cbor(bytes, true, false, json::cbor_tag_handler_t::error, eh::strict); + CHECK(result.is_discarded()); + } + + SECTION("default parameters are unchanged") + { + const json jval = ill_formed_cases[0].bytes; + + // to_*: the default error_handler is keep, so ill-formed bytes are + // written unchanged, exactly as in release 3.12.0 (it is strict only + // if JSON_STRICT_BINARY_UTF8 is enabled, see + // unit-binary_utf8_strict.cpp) + CHECK(json::to_cbor(jval) == json::to_cbor(jval, eh::keep)); + CHECK(json::to_ubjson(jval) == json::to_ubjson(jval, false, false, eh::keep)); + CHECK(json::to_bjdata(jval) == json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep)); + { + json jobj; + jobj["k"] = jval; + CHECK(json::to_bson(jobj) == json::to_bson(jobj, eh::keep)); + } + + // from_*: the default error_handler is keep, so ill-formed bytes are + // still accepted unchanged, exactly as in release 3.12.0 + const auto cbor_bytes = json::to_cbor(jval, eh::keep); + CHECK(json::from_cbor(cbor_bytes).get() == ill_formed_cases[0].bytes); + const auto ubjson_bytes = json::to_ubjson(jval, false, false, eh::keep); + CHECK(json::from_ubjson(ubjson_bytes).get() == ill_formed_cases[0].bytes); + const auto bjdata_bytes = json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep); + CHECK(json::from_bjdata(bjdata_bytes).get() == ill_formed_cases[0].bytes); + const auto msgpack_bytes = json::to_msgpack(jval); + CHECK(json::from_msgpack(msgpack_bytes).get() == ill_formed_cases[0].bytes); + json bson_obj; + bson_obj["k"] = jval; + const auto bson_bytes = json::to_bson(bson_obj, eh::keep); + CHECK(json::from_bson(bson_bytes)["k"].get() == ill_formed_cases[0].bytes); + } +} diff --git a/tests/src/unit-binary_utf8_strict.cpp b/tests/src/unit-binary_utf8_strict.cpp index c83b5938c..2ac8aabf4 100644 --- a/tests/src/unit-binary_utf8_strict.cpp +++ b/tests/src/unit-binary_utf8_strict.cpp @@ -100,11 +100,23 @@ TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)") CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); } + SECTION("an explicit error_handler overrides the default") + { + // the macro only changes the default of the error_handler parameter + CHECK(json::to_cbor(json("\xFF"), json::error_handler_t::keep) == std::vector({0x61, 0xff})); + CHECK(json::to_ubjson(json("\xFF"), false, false, json::error_handler_t::keep) == std::vector({'S', 'i', 1, 0xff})); + CHECK(json::to_bjdata(json("\xFF"), false, false, json::bjdata_version_t::draft2, json::error_handler_t::keep) == std::vector({'S', 'i', 1, 0xff})); + CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}}, json::error_handler_t::keep)) == json{{"s", "\xFF"}}); + CHECK(json::to_cbor(json("\xFF"), json::error_handler_t::replace) == std::vector({0x63, 0xef, 0xbf, 0xbd})); + } + SECTION("MessagePack and BON8 are unaffected") { - // MessagePack allows any bytes in a str, so to_msgpack() writes them as - // is; BON8 always checks, because the lead bytes mark where strings end + // MessagePack allows any bytes in a str, so to_msgpack() still + // defaults to keep (strict only if passed explicitly); BON8 always + // checks, because the lead bytes mark where strings end CHECK(json::to_msgpack(json("\xFF")) == std::vector({0xa1, 0xff})); + CHECK_THROWS_AS(json::to_msgpack(json("\xFF"), json::error_handler_t::strict), json::type_error&); CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&); } }