From 92c729f6572ab2db319c4cce2bc929f1f14c4aae Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 1 Oct 2026 07:55:17 +0200 Subject: [PATCH] Accept ill-formed UTF-8 in all binary readers again RFC 8949 and the MessagePack/BSON/UBJSON/BJData specs leave UTF-8 well-formedness checking up to the decoder, so following #5529 the binary readers are lenient by default again, as in release 3.12.0 (the reader-side check was added by #5185/#5531, not in any release); reader-side validation becomes opt-in in a follow-up PR. The writers stay strict and throw type_error.316 for ill-formed UTF-8. BON8 is unchanged, since UTF-8 lead bytes are structural there. Signed-off-by: Niels Lohmann --- .../docs/features/binary_formats/bjdata.md | 12 +- .../docs/features/binary_formats/bson.md | 21 +- .../docs/features/binary_formats/cbor.md | 17 +- .../docs/features/binary_formats/ubjson.md | 12 +- docs/mkdocs/docs/home/exceptions.md | 8 +- .../nlohmann/detail/input/binary_reader.hpp | 33 +- include/nlohmann/detail/string_utils.hpp | 44 +- single_include/nlohmann/json.hpp | 379 ++++++++++-------- tests/src/unit-bjdata.cpp | 29 +- tests/src/unit-bson.cpp | 19 +- tests/src/unit-cbor.cpp | 64 ++- tests/src/unit-ubjson.cpp | 29 +- 12 files changed, 366 insertions(+), 301 deletions(-) diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index a0908448b..5220c819b 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -214,12 +214,14 @@ The library maps BJData types to JSON value types as follows: The mapping is **complete** in the sense that any BJData value can be converted to a JSON value. -!!! warning "UTF-8 validation of string values and object keys" +!!! warning "Ill-formed UTF-8 in string values and object keys" - This library validates the bytes of every string value and object key at decode time and rejects ill-formed - UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with - `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value - is dumped. + BJData strings must use UTF-8 encoding, but this is not enforced on read: `from_bjdata()` accepts a string + value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error + handler is passed that replaces or ignores the ill-formed bytes. `to_bjdata()` is strict as well (see above), so + a value read this way cannot be written back to BJData. !!! info "Round trips" diff --git a/docs/mkdocs/docs/features/binary_formats/bson.md b/docs/mkdocs/docs/features/binary_formats/bson.md index ede6c6ca7..acc7ca6d9 100644 --- a/docs/mkdocs/docs/features/binary_formats/bson.md +++ b/docs/mkdocs/docs/features/binary_formats/bson.md @@ -109,18 +109,17 @@ The library maps BSON record types to JSON value types as follows: If BSON input must be validated for strict specification compliance, validate it separately before passing it to `from_bson()`. -!!! warning "UTF-8 validation of string values" +!!! warning "Ill-formed UTF-8 in string values" - The BSON specification requires `string` values (type `0x02`) to be valid UTF-8. This library validates the - bytes of every such string at decode time and rejects ill-formed UTF-8 with a - [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions` - set to `false`, a discarded value), rather than only failing later when the resulting value is dumped. Element - (key) names and `binary` values (type `0x05`) are unaffected and are never validated on read, since they are read - byte-by-byte as a C string, or are not required to hold text, respectively. `to_bson()` validates both string - values and element names and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 in either, so an - object with such a key or value cannot be produced in the first place, even though `from_bson()` would accept it - from another source. + The BSON specification requires `string` values (type `0x02`) to be valid UTF-8, but this is not required of a + decoder. `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back unchanged. + However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error + handler is passed that replaces or ignores the ill-formed bytes. `to_bson()` is strict as well and throws the + same exception for a string value or element (key) name that is not valid UTF-8, so an object with such a key + or value cannot be produced in the first place, even though `from_bson()` would accept it from another source. + Element (key) names are never validated on read, since they are read byte-by-byte as a C string. `binary` + values (type `0x05`) are unaffected, since they are not required to hold text. ??? example diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index c8332c1be..de594d837 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -189,17 +189,16 @@ The library maps CBOR types to JSON value types as follows: ([RFC 8392](https://www.rfc-editor.org/rfc/rfc8392.html)), cannot be read with this library and need a general-purpose CBOR library instead. -!!! warning "UTF-8 validation of text strings" +!!! warning "Ill-formed UTF-8 in text strings" [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings - (major type 3) to be valid UTF-8. This library validates the bytes of every text string (object keys included) at - decode time and rejects ill-formed UTF-8 with a - [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with - `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value is - dumped. Byte strings (major type 2) are unaffected and are never validated, since they are not required to hold - text. `to_cbor()` validates string values and object keys the same way and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, so a value with - such a string cannot be serialized in the first place. + (major type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this. This library does + not: `from_cbor()` accepts a text string (object keys included) whose bytes are not valid UTF-8 and hands them + back unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error + handler is passed that replaces or ignores the ill-formed bytes. `to_cbor()` is strict as well and throws the + same exception for a string value or object key that is not valid UTF-8, so such a value cannot be written back + to CBOR. Byte strings (major type 2) are unaffected, since they are not required to hold text. !!! warning "Tagged items" diff --git a/docs/mkdocs/docs/features/binary_formats/ubjson.md b/docs/mkdocs/docs/features/binary_formats/ubjson.md index a24fedb4f..39da9628e 100644 --- a/docs/mkdocs/docs/features/binary_formats/ubjson.md +++ b/docs/mkdocs/docs/features/binary_formats/ubjson.md @@ -126,12 +126,14 @@ The library maps UBJSON types to JSON value types as follows: The mapping is **complete** in the sense that any UBJSON value can be converted to a JSON value. -!!! warning "UTF-8 validation of string values and object keys" +!!! warning "Ill-formed UTF-8 in string values and object keys" - This library validates the bytes of every string value and object key at decode time and rejects ill-formed - UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with - `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value - is dumped. + UBJSON's required string encoding is UTF-8, but this is not enforced on read: `from_ubjson()` accepts a string + value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error + handler is passed that replaces or ignores the ill-formed bytes. `to_ubjson()` is strict as well (see above), so + a value read this way cannot be written back to UBJSON. ??? example diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 407f3c3f1..1d1bf33d5 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -340,8 +340,9 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde ### json.exception.parse_error.113 A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a -string was read where one was required (for instance as a map key), the string's length specification is invalid, or -the string's bytes are not valid UTF-8. +string was read where one was required (for instance as a map key), or the string's length specification is invalid. +The bytes of a string itself are not checked for valid UTF-8 on read; see the ill-formed UTF-8 notes on the +individual [binary format](../features/binary_formats/index.md) pages for how such a string is handled afterward. CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other type (for instance integers or `null`) are therefore not supported; see the notes on @@ -364,9 +365,6 @@ type (for instance integers or `null`) are therefore not supported; see the note ``` [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData string: string length must not be negative ``` - ``` - [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte - ``` ### json.exception.parse_error.114 diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 195fa6dea..1fce584b0 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -4031,32 +4031,13 @@ class binary_reader const NumberType len, string_t& result) { - // get_bytes() appends to result, and CBOR indefinite-length strings - // collect all their chunks in the same result; validating only the - // newly read bytes keeps the check linear in the input size - const std::size_t old_size = result.size(); - if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) - { - return false; - } - - // RFC 8949 (CBOR) §3.1 and the BSON/UBJSON specifications require - // text strings to be valid UTF-8; reject anything else right here so - // malformed input is caught at decode time instead of only surfacing - // later as a type_error.316 when the value is dumped (which would - // defeat allow_exceptions=false / strict discarding). The MessagePack - // specification explicitly allows a str object to contain an invalid - // byte sequence and expects deserializers to hand back the original - // bytes, so msgpack strings (and map keys, which go through this - // function as well) are exempt. - if (format != input_format_t::msgpack && JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) - { - return sax->parse_error(chars_read, get_token_string(), - parse_error::create(113, chars_read, - exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)); - } - - return true; + // Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the + // choice to the decoder), MessagePack (whose spec explicitly allows + // a str object to contain an invalid byte sequence), UBJSON, BJData, + // or BSON requires a decoder to reject ill-formed UTF-8. The bytes + // are kept unchanged; dump() and the binary writers are the ones + // that check them and report type_error.316 if they are not valid. + return get_bytes(format, len, "string", result); } /*! diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index 2495c5802..fff2f7ff2 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -117,13 +117,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally written by Björn Hoehrmann. See http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. -The library checks UTF-8 well-formedness (RFC 3629, section 4) in four +The library checks UTF-8 well-formedness (RFC 3629, section 4) in three places, which differ in speed, diagnostics, and how they read the input: -- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in - strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR, - MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in - text strings at decode time). +- decode() below: the serializer, to escape and, in strict mode, reject + ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON, + UBJSON and BJData readers do not use it: none of those specs requires a + decoder to reject ill-formed UTF-8 in text strings, so the readers keep + the bytes as is and leave the check to dump() and the binary writers. - the per-lead-byte switch in lexer::scan_string(): JSON text, with a diagnostic for each kind of error. - validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's @@ -178,38 +179,5 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std: return state; } -/*! -@brief check whether a string consists solely of valid UTF-8 - -Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text -strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the -MessagePack/BSON specifications all require text strings to be UTF-8), so -that malformed input is caught immediately instead of only surfacing later -as a type_error.316 when the resulting value is dumped. - -@param[in] s the string to check -@param[in] first index of the first byte to check; the bytes before it are - assumed to have been validated already and to end on a - code point boundary -@return whether @a s (from index @a first on) is valid UTF-8 -*/ -template -inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept -{ - std::uint8_t state = UTF8_ACCEPT; - std::uint32_t codepoint = 0; - - for (std::size_t i = first; i < s.size(); ++i) - { - decode(state, codepoint, static_cast(s[i])); - if (state == UTF8_REJECT) - { - return false; - } - } - - return state == UTF8_ACCEPT; -} - } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index f1de82e0a..19bd2da1e 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -18,16 +18,6 @@ #ifndef INCLUDE_NLOHMANN_JSON_HPP_ #define INCLUDE_NLOHMANN_JSON_HPP_ -// Workaround for GCC template redefinition errors in C++ modules -// When nlohmann/json.hpp is included in a C++20 module preamble after -// other module imports, GCC may report spurious redefinition errors for -// STL templates. These pragmas suppress those false positives. -// See: https://github.com/nlohmann/json/issues/5103 -#if defined(__GNUC__) && !defined(__clang__) && __cplusplus >= 202002L - #pragma GCC diagnostic push - #pragma GCC diagnostic ignored "-Wignored-attributes" -#endif - #include // all_of, find, for_each, none_of #include // isnan #include // nullptr_t, ptrdiff_t, size_t @@ -2617,13 +2607,6 @@ JSON_HEDLEY_DIAGNOSTIC_POP #define JSON_NO_THREAD_LOCAL 1 #endif -// disable documentation warnings on clang -#if defined(__clang__) - #pragma clang diagnostic push - #pragma clang diagnostic ignored "-Wdocumentation" - #pragma clang diagnostic ignored "-Wdocumentation-unknown-command" -#endif - // allow disabling exceptions #if (defined(__cpp_exceptions) || defined(__EXCEPTIONS) || defined(_CPPUNWIND)) && !defined(JSON_NOEXCEPTION) #define JSON_THROW(exception) throw exception @@ -2682,7 +2665,7 @@ JSON_HEDLEY_DIAGNOSTIC_POP { \ /* NOLINTNEXTLINE(modernize-type-traits) we use C++11 */ \ static_assert(std::is_enum::value, #ENUM_TYPE " must be an enum!"); \ - /* NOLINTNEXTLINE(modernize-avoid-c-arrays) we don't want to depend on */ \ + /* NOLINTNEXTLINE(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) we don't want to depend on */ \ static const std::pair m[] = __VA_ARGS__; \ auto it = std::find_if(std::begin(m), std::end(m), \ [e](const std::pair& ej_pair) -> bool \ @@ -2696,7 +2679,7 @@ JSON_HEDLEY_DIAGNOSTIC_POP { \ /* NOLINTNEXTLINE(modernize-type-traits) we use C++11 */ \ static_assert(std::is_enum::value, #ENUM_TYPE " must be an enum!"); \ - /* NOLINTNEXTLINE(modernize-avoid-c-arrays) we don't want to depend on */ \ + /* NOLINTNEXTLINE(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) we don't want to depend on */ \ static const std::pair m[] = __VA_ARGS__; \ auto it = std::find_if(std::begin(m), std::end(m), \ [&j](const std::pair& ej_pair) -> bool \ @@ -2735,7 +2718,7 @@ void templated_json_throw(ExceptionType exception) { \ /* NOLINTNEXTLINE(modernize-type-traits) we use C++11 */ \ static_assert(std::is_enum::value, #ENUM_TYPE " must be an enum!"); \ - /* NOLINTNEXTLINE(modernize-avoid-c-arrays) we don't want to depend on */ \ + /* NOLINTNEXTLINE(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) we don't want to depend on */ \ static const std::pair m[] = __VA_ARGS__; \ auto it = std::find_if(std::begin(m), std::end(m), \ [e](const std::pair& ej_pair) -> bool \ @@ -2750,7 +2733,7 @@ void templated_json_throw(ExceptionType exception) { \ /* NOLINTNEXTLINE(modernize-type-traits) we use C++11 */ \ static_assert(std::is_enum::value, #ENUM_TYPE " must be an enum!"); \ - /* NOLINTNEXTLINE(modernize-avoid-c-arrays) we don't want to depend on */ \ + /* NOLINTNEXTLINE(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) we don't want to depend on */ \ static const std::pair m[] = __VA_ARGS__; \ auto it = std::find_if(std::begin(m), std::end(m), \ [&j](const std::pair& ej_pair) -> bool \ @@ -3578,8 +3561,7 @@ inline StringType escape(const StringType& s) /*! * @brief string unescaping as described in RFC 6901 (Sect. 4) - * @param[in] s string to unescape - * @return unescaped string + * @param[in,out] s string to unescape in place * * Note the order of escaping "~1" to "/" and "~0" to "~" is important. * @@ -5871,22 +5853,22 @@ void()) } template < typename BasicJsonType, typename T, std::size_t... Idx > -std::array from_json_inplace_array_impl(BasicJsonType&& j, +std::array from_json_inplace_array_impl(const BasicJsonType& j, identity_tag> /*unused*/, index_sequence /*unused*/) { - return { { std::forward(j).at(Idx).template get()... } }; + return { { j.at(Idx).template get()... } }; } template < typename BasicJsonType, typename T, std::size_t N > -auto from_json(BasicJsonType&& j, identity_tag> tag) --> decltype(from_json_inplace_array_impl(std::forward(j), tag, make_index_sequence {})) +auto from_json(const BasicJsonType& j, identity_tag> tag) +-> decltype(from_json_inplace_array_impl(j, tag, make_index_sequence {})) { if (JSON_HEDLEY_UNLIKELY(!j.is_array())) { JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); } - return from_json_inplace_array_impl(std::forward(j), tag, make_index_sequence {}); + return from_json_inplace_array_impl(j, tag, make_index_sequence {}); } template @@ -6021,54 +6003,54 @@ template using tuple_type = std::tuple < decltype(from_json_tuple_get_impl(std::declval(), detail::identity_tag {}, detail::priority_tag {}))... >; template -tuple_type from_json_tuple_impl_base(BasicJsonType&& j, index_sequence /*unused*/) +tuple_type from_json_tuple_impl_base(const BasicJsonType& j, index_sequence /*unused*/) { - return tuple_type(from_json_tuple_get_impl(std::forward(j).at(Idx), detail::identity_tag {}, detail::priority_tag {})...); + return tuple_type(from_json_tuple_get_impl(j.at(Idx), detail::identity_tag {}, detail::priority_tag {})...); } template -std::tuple<> from_json_tuple_impl_base(BasicJsonType& /*unused*/, index_sequence<> /*unused*/) +std::tuple<> from_json_tuple_impl_base(const BasicJsonType& /*unused*/, index_sequence<> /*unused*/) { return {}; } template < typename BasicJsonType, class A1, class A2 > -std::pair from_json_tuple_impl(BasicJsonType&& j, identity_tag> /*unused*/, priority_tag<0> /*unused*/) +std::pair from_json_tuple_impl(const BasicJsonType& j, identity_tag> /*unused*/, priority_tag<0> /*unused*/) { - return {std::forward(j).at(0).template get(), - std::forward(j).at(1).template get()}; + return {j.at(0).template get(), + j.at(1).template get()}; } template -inline void from_json_tuple_impl(BasicJsonType&& j, std::pair& p, priority_tag<1> /*unused*/) +inline void from_json_tuple_impl(const BasicJsonType& j, std::pair& p, priority_tag<1> /*unused*/) { - p = from_json_tuple_impl(std::forward(j), identity_tag> {}, priority_tag<0> {}); + p = from_json_tuple_impl(j, identity_tag> {}, priority_tag<0> {}); } template -std::tuple from_json_tuple_impl(BasicJsonType&& j, identity_tag> /*unused*/, priority_tag<2> /*unused*/) +std::tuple from_json_tuple_impl(const BasicJsonType& j, identity_tag> /*unused*/, priority_tag<2> /*unused*/) { - static_assert(cxpr_and>, is_compatible_reference_type>...>::value, + static_assert(cxpr_and>, is_compatible_reference_type>...>::value, "Can not return a tuple containing references to types not contained in a Json, try Json::get_to()"); - return from_json_tuple_impl_base<1, Args...>(std::forward(j), index_sequence_for {}); + return from_json_tuple_impl_base<1, Args...>(j, index_sequence_for {}); } template -inline void from_json_tuple_impl(BasicJsonType&& j, std::tuple& t, priority_tag<3> /*unused*/) +inline void from_json_tuple_impl(const BasicJsonType& j, std::tuple& t, priority_tag<3> /*unused*/) { - t = from_json_tuple_impl_base<2, Args...>(std::forward(j), index_sequence_for {}); + t = from_json_tuple_impl_base<2, Args...>(j, index_sequence_for {}); } template -auto from_json(BasicJsonType&& j, TupleRelated&& t) --> decltype(from_json_tuple_impl(std::forward(j), std::forward(t), priority_tag<3> {})) +auto from_json(const BasicJsonType& j, TupleRelated&& t) +-> decltype(from_json_tuple_impl(j, std::forward(t), priority_tag<3> {})) { if (JSON_HEDLEY_UNLIKELY(!j.is_array())) { JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); } - return from_json_tuple_impl(std::forward(j), std::forward(t), priority_tag<3> {}); + return from_json_tuple_impl(j, std::forward(t), priority_tag<3> {}); } template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator, @@ -6153,7 +6135,7 @@ struct from_json_fn /// namespace to hold default `from_json` function /// to see why this is required: /// http://www.open-std.org/jtc1/sc22/wg21/docs/papers/2015/n4381.html -namespace // NOLINT(cert-dcl59-cpp,fuchsia-header-anon-namespaces,google-build-namespaces) +namespace // NOLINT(cert-dcl59-cpp,fuchsia-header-anon-namespaces,google-build-namespaces,misc-anonymous-namespace-in-header) { #endif JSON_INLINE_VARIABLE constexpr const auto& from_json = // NOLINT(misc-definitions-in-headers) @@ -6339,13 +6321,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally written by Björn Hoehrmann. See http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. -The library checks UTF-8 well-formedness (RFC 3629, section 4) in four +The library checks UTF-8 well-formedness (RFC 3629, section 4) in three places, which differ in speed, diagnostics, and how they read the input: -- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in - strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR, - MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in - text strings at decode time). +- decode() below: the serializer, to escape and, in strict mode, reject + ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON, + UBJSON and BJData readers do not use it: none of those specs requires a + decoder to reject ill-formed UTF-8 in text strings, so the readers keep + the bytes as is and leave the check to dump() and the binary writers. - the per-lead-byte switch in lexer::scan_string(): JSON text, with a diagnostic for each kind of error. - validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's @@ -6400,39 +6383,6 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std: return state; } -/*! -@brief check whether a string consists solely of valid UTF-8 - -Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text -strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the -MessagePack/BSON specifications all require text strings to be UTF-8), so -that malformed input is caught immediately instead of only surfacing later -as a type_error.316 when the resulting value is dumped. - -@param[in] s the string to check -@param[in] first index of the first byte to check; the bytes before it are - assumed to have been validated already and to end on a - code point boundary -@return whether @a s (from index @a first on) is valid UTF-8 -*/ -template -inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept -{ - std::uint8_t state = UTF8_ACCEPT; - std::uint32_t codepoint = 0; - - for (std::size_t i = first; i < s.size(); ++i) - { - decode(state, codepoint, static_cast(s[i])); - if (state == UTF8_REJECT) - { - return false; - } - } - - return state == UTF8_ACCEPT; -} - } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -6629,11 +6579,11 @@ namespace std JSON_HEDLEY_PRAGMA(clang diagnostic ignored "-Wmismatched-tags") #endif template -class tuple_size<::nlohmann::detail::iteration_proxy_value> // NOLINT(cert-dcl58-cpp) +class tuple_size<::nlohmann::detail::iteration_proxy_value> // NOLINT(cert-dcl58-cpp,bugprone-std-namespace-modification) : public std::integral_constant {}; template -class tuple_element> // NOLINT(cert-dcl58-cpp) +class tuple_element> // NOLINT(cert-dcl58-cpp,bugprone-std-namespace-modification) { public: using type = decltype( @@ -7178,7 +7128,7 @@ struct to_json_fn /// namespace to hold default `to_json` function /// to see why this is required: /// http://www.open-std.org/jtc1/sc22/wg21/docs/papers/2015/n4381.html -namespace // NOLINT(cert-dcl59-cpp,fuchsia-header-anon-namespaces,google-build-namespaces) +namespace // NOLINT(cert-dcl59-cpp,fuchsia-header-anon-namespaces,google-build-namespaces,misc-anonymous-namespace-in-header) { #endif JSON_INLINE_VARIABLE constexpr const auto& to_json = // NOLINT(misc-definitions-in-headers) @@ -8399,6 +8349,9 @@ struct container_input_adapter_factory< ContainerType, static adapter_type create(ContainerType&& container) { + // container is forwarded twice on purpose: the resulting begin/end + // iterator types must match adapter_type, computed the same way + // NOLINTNEXTLINE(bugprone-use-after-move) return input_adapter(begin(std::forward(container)), end(std::forward(container))); } }; @@ -10298,9 +10251,9 @@ class lexer : public lexer_base ///////////////////// /*! - @brief get codepoint from 4 hex characters following `\u` + @brief get codepoint from 4 hex characters following `\\u` - For input "\u c1 c2 c3 c4" the codepoint is: + For input "\\u c1 c2 c3 c4" the codepoint is: (c1 * 0x1000) + (c2 * 0x0100) + (c3 * 0x0010) + c4 = (c1 << 12) + (c2 << 8) + (c3 << 4) + (c4 << 0) @@ -13949,7 +13902,7 @@ class binary_reader @brief Parses a C-style string from the BSON input. @param[in,out] result A reference to the string variable where the read string is to be stored. - @return `true` if the \x00-byte indicating the end of the string was + @return `true` if the \\x00-byte indicating the end of the string was encountered before the EOF; false` indicates an unexpected EOF. */ bool get_bson_cstr(string_t& result) @@ -13979,7 +13932,7 @@ class binary_reader @brief read a C-style string from contiguous input in one step @param[in,out] result the string to append to - @return whether the string was read; if the input has no \x00-byte, nothing + @return whether the string was read; if the input has no \\x00-byte, nothing is read, and @ref get_bson_cstr reports the end of the input */ bool get_bson_cstr_bulk(string_t& result, std::true_type /*bulk*/) @@ -17559,28 +17512,13 @@ class binary_reader const NumberType len, string_t& result) { - // get_bytes() appends to result, and CBOR indefinite-length strings - // collect all their chunks in the same result; validating only the - // newly read bytes keeps the check linear in the input size - const std::size_t old_size = result.size(); - if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) - { - return false; - } - - // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications - // all require text strings to be valid UTF-8; reject anything else - // right here so malformed input is caught at decode time instead of - // only surfacing later as a type_error.316 when the value is dumped - // (which would defeat allow_exceptions=false / strict discarding). - if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) - { - return sax->parse_error(chars_read, get_token_string(), - parse_error::create(113, chars_read, - exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)); - } - - return true; + // Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the + // choice to the decoder), MessagePack (whose spec explicitly allows + // a str object to contain an invalid byte sequence), UBJSON, BJData, + // or BSON requires a decoder to reject ill-formed UTF-8. The bytes + // are kept unchanged; dump() and the binary writers are the ones + // that check them and report type_error.316 if they are not valid. + return get_bytes(format, len, "string", result); } /*! @@ -18636,7 +18574,7 @@ This class implements a both iterators (iterator and const_iterator) for the been set (e.g., by a constructor or a copy assignment). If the iterator is default-constructed, it is *uninitialized* and most methods are undefined. **The library uses assertions to detect calls on uninitialized iterators.** -@requirement REQ-JSON-01 The class satisfies the following concept requirements: +This class satisfies the following concept requirements (REQ-JSON-01): - [BidirectionalIterator](https://en.cppreference.com/w/cpp/named_req/BidirectionalIterator): The iterator that can be moved can be moved in both directions (i.e. @@ -19400,7 +19338,7 @@ namespace detail iterator (to create @ref reverse_iterator) and @ref const_iterator (to create @ref const_reverse_iterator). -@requirement REQ-JSON-02 The class satisfies the following concept requirements: +This class satisfies the following concept requirements (REQ-JSON-02): - [BidirectionalIterator](https://en.cppreference.com/w/cpp/named_req/BidirectionalIterator): The iterator that can be moved can be moved in both directions (i.e. @@ -19843,11 +19781,11 @@ class json_pointer JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr)); } - // only triggered on special platforms (like 32bit), see also - // https://github.com/nlohmann/json/pull/2203 + // the index does not fit into size_type; on 64-bit platforms this is + // only SIZE_MAX itself (see #2203 and #5395) if (res >= static_cast((std::numeric_limits::max)())) // NOLINT(runtime/int) { - JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); // LCOV_EXCL_LINE + JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); } return static_cast(res); @@ -19881,7 +19819,7 @@ class json_pointer /*! @brief create and return a reference to the pointed to value - @complexity Linear in the number of reference tokens. + Complexity: Linear in the number of reference tokens. @throw parse_error.106 if an array index begins with '0' @throw parse_error.109 if array index is not a number @@ -19968,7 +19906,7 @@ class json_pointer @return reference to the JSON value pointed to by the JSON pointer - @complexity Linear in the length of the JSON pointer. + Complexity: Linear in the length of the JSON pointer. @throw parse_error.106 if an array index begins with '0' @throw parse_error.109 if an array index was not a number @@ -20880,6 +20818,7 @@ NLOHMANN_JSON_NAMESPACE_END #include // size_t #include // shared_ptr, make_shared #include // basic_string +#include // conditional, integral_constant, is_same #include // move #include // vector @@ -20988,11 +20927,13 @@ class output_stream_adapter : public output_adapter_protocol : stream(s) {} + // NOLINTNEXTLINE(portability-template-virtual-member-function) void write_character(CharType c) override { stream.put(c); } + // NOLINTNEXTLINE(portability-template-virtual-member-function) void write_characters(const CharType* s, std::size_t length) override { stream.write(s, static_cast(length)); @@ -21059,7 +21000,82 @@ class output_adapter_sink output_adapter_t oa; }; -template> +/// @brief whether std::basic_string has a non-deprecated std::char_traits +/// specialization, and is therefore usable as output_adapter's default StringType +/// +/// std::char_traits is only guaranteed (and, on some standard libraries, only +/// implemented without a deprecation warning) for the character types listed +/// below; std::char_traits for any other T (e.g. std::uint8_t, as used by the +/// binary writers) is a non-standard extension some standard libraries deprecate. +/// See https://github.com/nlohmann/json/issues/5725 item 2. +template +struct is_output_adapter_string_char_type : std::integral_constant < bool, + std::is_same::value || + std::is_same::value || + std::is_same::value || + std::is_same::value +#if defined(__cpp_lib_char8_t) && (__cpp_lib_char8_t >= 201907L) + || std::is_same::value +#endif + > {}; + +/// @brief placeholder type for output_adapter's StringType and (with JSON_NO_IO +/// undefined) its std::basic_ostream constructor parameter, for CharType +/// with no non-deprecated std::char_traits specialization +/// +/// Never actually used: the StringType- and std::basic_ostream-based +/// output_adapter constructors are neither documented nor tested for such +/// CharType (only the std::vector-based constructor is used for them, by the +/// binary writers). Naming std::basic_string or +/// std::basic_ostream anywhere such a constructor would otherwise be +/// declared - even as an unused default template argument or an unused, +/// never-called overload - instantiates std::char_traits merely to +/// name the type, which is exactly what triggers the deprecation warning this +/// placeholder avoids. +template +struct output_adapter_no_string_type {}; + +// Select output_adapter's default StringType (and, below, its ostream +// constructor's parameter type) via partial specialization, not +// std::conditional: std::conditional requires both T and F to be named +// as template arguments up front, which would still instantiate (and thus name) +// std::basic_string / std::basic_ostream for every CharType, +// defeating the point. A bool non-type parameter with two specializations only +// ever names the type that is actually selected. +template::value> +struct output_adapter_default_string_type +{ + using type = output_adapter_no_string_type; +}; + +template +struct output_adapter_default_string_type +{ + using type = std::basic_string; +}; + +#ifndef JSON_NO_IO +/// distinct from output_adapter_no_string_type, so the placeholder overloads of +/// output_adapter's constructor (used when CharType is not a character type) +/// stay distinct overloads instead of colliding into a single redeclaration +template +struct output_adapter_no_ostream_type {}; + +template::value> +struct output_adapter_ostream_type +{ + using type = output_adapter_no_ostream_type; +}; + +template +struct output_adapter_ostream_type +{ + using type = std::basic_ostream; +}; +#endif // JSON_NO_IO + +template < typename CharType, typename StringType = + typename output_adapter_default_string_type::type > class output_adapter { public: @@ -21068,7 +21084,7 @@ class output_adapter : oa(std::make_shared>(vec)) {} #ifndef JSON_NO_IO - output_adapter(std::basic_ostream& s) + output_adapter(typename output_adapter_ostream_type::type& s) : oa(std::make_shared>(s)) {} #endif // JSON_NO_IO @@ -21177,6 +21193,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -21207,6 +21225,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -21273,6 +21293,8 @@ class binary_writer case value_t::string: { + check_utf8(*j.m_data.m_value.string, j); + // step 1: write control byte and the string length write_cbor_head(0x60, j.m_data.m_value.string->size()); @@ -21349,6 +21371,11 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // el.first is checked here, against the object as + // diagnostics context, because write_cbor(el.first) + // converts it to a temporary basic_json that would be + // used as the context instead + check_utf8(el.first, j); write_cbor(el.first); write_cbor(el.second); } @@ -21691,6 +21718,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -21740,6 +21769,8 @@ class binary_writer case value_t::string: { + check_utf8(*j.m_data.m_value.string, j); + if (add_prefix) { oa.write_character(to_char_type('S')); @@ -21902,6 +21933,7 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { + check_utf8(el.first, j); write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(el.first.data()), @@ -21946,6 +21978,10 @@ class binary_writer /*! @return The size of a BSON document entry header, including the id marker and the entry name size (and its null-terminator). + @throw out_of_range.409 if @a name contains U+0000, before anything is + written + @throw type_error.316 if @a name is not valid UTF-8, before anything is + written */ static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { @@ -21955,7 +21991,8 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - static_cast(j); + check_utf8(name, j); + return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; } @@ -22011,9 +22048,21 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value + @throw type_error.316 if @a value is not valid UTF-8, before anything is + written + + @note The UTF-8 check is skipped if @a value is already too long for the + 32-bit BSON length field (@ref to_bson_length rejects it later, once + the size of the whole document is known); this also keeps the check + from reading past a StringType that reports a size larger than what + it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value) + static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { + if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) + { + check_utf8(value, j); + } return sizeof(std::int32_t) + value.size() + 1ul; } @@ -22142,6 +22191,8 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written + @throw type_error.316 if @a j is a string that is not valid UTF-8, before + anything is written */ static std::size_t calc_bson_value_size(const BasicJsonType& j) { @@ -22163,7 +22214,7 @@ class binary_writer return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string, j); case value_t::null: return 0ul; @@ -22276,6 +22327,8 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written + @throw type_error.316 if a string value or a key is not valid UTF-8, + before anything is written */ static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { @@ -23154,7 +23207,7 @@ class binary_writer */ void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) { - check_bon8_utf8(s, context); + check_utf8(s, context); // a string that follows another string terminates it if (string_open) @@ -23184,7 +23237,7 @@ class binary_writer @throw type_error.316 if @a s is not valid UTF-8; the message names the first byte of the first invalid or incomplete sequence */ - static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + static void check_utf8(const string_t& s, const BasicJsonType& context) { static_cast(context); // only used when exceptions are enabled const auto* data = reinterpret_cast(s.data()); @@ -24727,7 +24780,7 @@ class serializer @param[in] ichar indentation character to use @param[in] pretty_print_ whether the output shall be pretty-printed @param[in] ensure_ascii_ If @a ensure_ascii_ is true, all non-ASCII - characters in the output are escaped with `\uXXXX` sequences, and the + characters in the output are escaped with `\\uXXXX` sequences, and the result consists of ASCII characters only. @param[in] indent_step_ the indent level @param[in] error_handler_ how to react on decoding errors @@ -25352,7 +25405,7 @@ class serializer @param[in] s the string to escape - @complexity Linear in the length of string @a s. + Complexity: Linear in the length of string @a s. */ void dump_escaped(const string_t& s) { @@ -25856,7 +25909,7 @@ class serializer } /*! - * @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer + * @brief write a lowercase "\\uXXXX" escape sequence into @a string_buffer * * Branch-free replacement for `snprintf(buf, 7, "\\u%04x", codeunit)` in the * string escaping hot path. It writes exactly six characters ('\\', 'u' and @@ -26206,7 +26259,7 @@ class serializer /// whether to pretty-print the output const bool pretty_print; - /// whether to escape non-ASCII characters with \uXXXX sequences + /// whether to escape non-ASCII characters with \\uXXXX sequences const bool ensure_ascii; /// the indent level @@ -29156,12 +29209,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec @throw what @ref json_serializer `from_json()` method throws - @liveexample{The example below shows several conversions from JSON values + The example below shows several conversions from JSON values to other types. There a few things to note: (1) Floating-point numbers can - be converted to integers\, (2) A JSON array can be converted to a standard - `std::vector`\, (3) A JSON object can be converted to C++ - associative containers such as `std::unordered_map`.,get__ValueType_const} + be converted to integers, (2) A JSON array can be converted to a standard + `std::vector`, (3) A JSON object can be converted to C++ + associative containers such as `std::unordered_map`. @since version 2.1.0 */ @@ -29228,7 +29281,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec @return a copy of *this, converted into @a BasicJsonType - @complexity Depending on the implementation of the called `from_json()` + Complexity: Depending on the implementation of the called `from_json()` method. @since version 3.2.0 @@ -29252,7 +29305,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec @return a copy of *this - @complexity Constant. + Complexity: Constant. @since version 2.1.0 */ @@ -29298,7 +29351,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec @tparam ValueTypeCV the provided value type @tparam ValueType the returned value type - @return copy of the JSON value, converted to @tparam ValueType if necessary + @return copy of the JSON value, converted to @a ValueType if necessary @throw what @ref json_serializer `from_json()` method throws if conversion is required @@ -29336,12 +29389,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec @return pointer to the internally stored JSON value if the requested pointer type @a PointerType fits to the JSON value; `nullptr` otherwise - @complexity Constant. + Complexity: Constant. - @liveexample{The example below shows how pointers to internal values of a + The example below shows how pointers to internal values of a JSON value can be requested. Note that no type conversions are made and a `nullptr` is returned if the value and the requested pointer type does not - match.,get__PointerType} + match. @sa see @ref get_ptr() for explicit pointer-member access @@ -29435,14 +29488,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec to the JSON value type (e.g., the JSON value is of type boolean, but a string is requested); see example below - @complexity Linear in the size of the JSON value. + Complexity: Linear in the size of the JSON value. - @liveexample{The example below shows several conversions from JSON values + The example below shows several conversions from JSON values to other types. There a few things to note: (1) Floating-point numbers can - be converted to integers\, (2) A JSON array can be converted to a standard - `std::vector`\, (3) A JSON object can be converted to C++ - associative containers such as `std::unordered_map`.,operator__ValueType} + be converted to integers, (2) A JSON array can be converted to a standard + `std::vector`, (3) A JSON object can be converted to C++ + associative containers such as `std::unordered_map`. @since version 1.0.0 */ @@ -31897,6 +31950,19 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @deprecated This function is deprecated since 3.8.0 and will be removed in /// version 4.0.0 of the library. Please use /// sax_parse(ptr, ptr + len) instead. + // + // Clang reports "declaration is marked with '@deprecated' command but does + // not have a deprecation attribute" for this overload even though + // JSON_HEDLEY_DEPRECATED_FOR below does expand to __attribute__((deprecated)); + // isolated reproductions of this exact declaration shape (doc comment, + // template<>, two stacked __attribute__ lines, an overload set of the same + // name) do not reproduce it, so this looks like a Clang comment/declaration + // association quirk specific to this overload within basic_json, not a + // genuine documentation bug. See #5725 item 2. +#if defined(__clang__) +#pragma clang diagnostic push +#pragma clang diagnostic ignored "-Wdocumentation-deprecated-sync" +#endif template JSON_HEDLEY_DEPRECATED_FOR(3.8.0, sax_parse(ptr, ptr + len, ...)) JSON_HEDLEY_NON_NULL(2) @@ -31913,6 +31979,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } +#if defined(__clang__) +#pragma clang diagnostic pop +#endif #ifndef JSON_NO_IO /// @brief deserialize from stream /// @sa https://json.nlohmann.me/api/basic_json/operator_gtgt/ @@ -33575,7 +33644,7 @@ namespace std // NOLINT(cert-dcl58-cpp) /// @brief hash value for JSON objects /// @sa https://json.nlohmann.me/api/basic_json/std_hash/ NLOHMANN_BASIC_JSON_TPL_DECLARATION -struct hash // NOLINT(cert-dcl58-cpp) +struct hash // NOLINT(cert-dcl58-cpp,bugprone-std-namespace-modification) { std::size_t operator()(const nlohmann::NLOHMANN_BASIC_JSON_TPL& j) const { @@ -33608,7 +33677,7 @@ struct less< ::nlohmann::detail::value_t> // do not remove the space after '<', /// @brief exchanges the values of two JSON objects /// @sa https://json.nlohmann.me/api/basic_json/std_swap/ NLOHMANN_BASIC_JSON_TPL_DECLARATION -inline void swap(nlohmann::NLOHMANN_BASIC_JSON_TPL& j1, nlohmann::NLOHMANN_BASIC_JSON_TPL& j2) noexcept( // NOLINT(readability-inconsistent-declaration-parameter-name, cert-dcl58-cpp) +inline void swap(nlohmann::NLOHMANN_BASIC_JSON_TPL& j1, nlohmann::NLOHMANN_BASIC_JSON_TPL& j2) noexcept( // NOLINT(readability-inconsistent-declaration-parameter-name, cert-dcl58-cpp,bugprone-std-namespace-modification) is_nothrow_move_constructible::value&& // NOLINT(misc-redundant-expression,cppcoreguidelines-noexcept-swap,performance-noexcept-swap) is_nothrow_move_assignable::value) { @@ -33622,7 +33691,7 @@ inline void swap(nlohmann::NLOHMANN_BASIC_JSON_TPL& j1, nlohmann::NLOHMANN_BASIC /// @brief std::formatter specialization for JSON values /// @sa https://json.nlohmann.me/api/basic_json/std_formatter/ NLOHMANN_BASIC_JSON_TPL_DECLARATION -struct formatter // NOLINT(cert-dcl58-cpp) +struct formatter // NOLINT(cert-dcl58-cpp,bugprone-std-namespace-modification) { // -1 means compact output (dump()); any value >= 0 means pretty-printed // output with that many spaces (or indent_char) per level (dump(indent, indent_char)). @@ -33711,11 +33780,6 @@ struct formatter // NOLINT(cert-dcl58-c -// restore clang diagnostic settings -#if defined(__clang__) - #pragma clang diagnostic pop -#endif - // clean up #undef JSON_ASSERT #undef JSON_INTERNAL_CATCH @@ -33916,11 +33980,6 @@ struct formatter // NOLINT(cert-dcl58-c // IWYU pragma: keep -// End of GCC diagnostic pragmas for C++ modules support -#if defined(__GNUC__) && !defined(__clang__) && __cplusplus >= 202002L - #pragma GCC diagnostic pop -#endif - // The user-defined string literals are in a separate header, because their // bodies instantiate the parser in every translation unit that includes them. // Define JSON_NO_AUTOMATIC_UDLS to include only diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 700af1787..09a46ceb2 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -3908,17 +3908,28 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK(json::from_bjdata(v) == j); } - SECTION("ill-formed UTF-8 (see #5651)") + SECTION("ill-formed UTF-8 (see #5529, #5651)") { - // a string value whose bytes are not valid UTF-8 (0xC0 0xAE is an - // overlong encoding of '.') is rejected at decode time, matching - // every other kind of malformed binary input, and to_bjdata() - // rejects it as well, so a value it accepts can always be read - // back + // none of the binary format specs requires a decoder to reject + // ill-formed UTF-8 in a text string, so a value whose bytes are + // not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') + // round-trips byte for byte as a string value; to_bjdata() is + // strict, so such a value cannot be written back const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; - json _; - CHECK_THROWS_WITH_AS(_ = json::from_bjdata(v), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing BJData string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_bjdata(v, true, false).is_discarded()); + json j; + CHECK_NOTHROW(j = json::from_bjdata(v)); + REQUIRE(j.is_string()); + CHECK(j.get_ref() == std::string("\xc0\xae")); + CHECK_THROWS_AS(j.dump(), json::type_error&); + CHECK_THROWS_AS(json::to_bjdata(j), json::type_error&); + + // the same bytes as an object key round-trip as well + const std::vector v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'}; + json j_key; + CHECK_NOTHROW(j_key = json::from_bjdata(v_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK_THROWS_AS(json::to_bjdata(j_key), json::type_error&); CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); // a truncated multi-byte sequence diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index ee2c9c39c..97a437d6b 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -154,11 +154,12 @@ TEST_CASE("BSON") #endif } - SECTION("ill-formed UTF-8 (see #5651)") + SECTION("ill-formed UTF-8 (see #5529, #5651)") { // a BSON document {"s": "\xC0\xAE"} (0xC0 0xAE is an overlong - // encoding of '.'); the reader rejects an ill-formed string value at - // decode time + // encoding of '.'); the BSON spec does not require a decoder to + // reject ill-formed UTF-8 in a string value, so the reader hands the + // bytes back unchanged const std::vector v = { 0x0F, 0x00, 0x00, 0x00, // document length @@ -167,9 +168,15 @@ TEST_CASE("BSON") 0xc0, 0xae, 0x00, // string content and its null terminator 0x00 // document terminator }; - json _; - CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing BSON string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_bson(v, true, false).is_discarded()); + json j; + CHECK_NOTHROW(j = json::from_bson(v)); + REQUIRE(j.is_object()); + REQUIRE(j.contains("s")); + CHECK(j["s"].get_ref() == std::string("\xc0\xae")); + // dump() still requires valid UTF-8 and throws for such a value + CHECK_THROWS_AS(j.dump(), json::type_error&); + // to_bson() is strict as well, so the value cannot be written back + CHECK_THROWS_AS(json::to_bson(j), json::type_error&); // to_bson() rejects the same kind of ill-formed string value, before // any bytes reach the output adapter (the BSON document length diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index 12c3cbd8b..ed2ba98ac 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -1801,19 +1801,40 @@ TEST_CASE("CBOR") CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&); } - SECTION("invalid UTF-8 in string (see #5529)") + SECTION("ill-formed UTF-8 in string (see #5529, #5651)") { + // RFC 8949 §3.1 leaves it up to the decoder whether to reject + // ill-formed UTF-8 in a text string; this library does not, and + // hands the original bytes back unchanged, matching the + // MessagePack reader and the behavior before #5185/#5531 (not in + // any release) + // a two-character text string (major type 3) whose bytes are not - // valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be - // rejected at decode time, matching every other kind of - // malformed binary input, rather than only failing later when - // the resulting value is dumped - json _; - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_cbor(std::vector({0x62, 0xc0, 0xae}), true, false).is_discarded()); + // valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') round-trips + // byte for byte as a string value + const std::vector ill_formed_value = {0x62, 0xc0, 0xae}; + json j_value; + CHECK_NOTHROW(j_value = json::from_cbor(ill_formed_value)); + REQUIRE(j_value.is_string()); + CHECK(j_value.get_ref() == std::string("\xc0\xae")); + // dump() still requires valid UTF-8 and throws for such a value, + // unless an error handler that replaces or ignores the bytes is + // passed + CHECK_THROWS_AS(j_value.dump(), json::type_error&); + // to_cbor() is strict as well, so the value cannot be written back + CHECK_THROWS_AS(json::to_cbor(j_value), json::type_error&); + + // the same bytes as an object key round-trip as well + const std::vector ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01}; + json j_key; + CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK_THROWS_AS(json::to_cbor(j_key), json::type_error&); // a CBOR byte string (major type 2) with the very same bytes is // NOT text and must still be accepted as-is + json _; CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x42, 0xc0, 0xae}))); CHECK(_ == json::binary(std::vector({0xc0, 0xae}))); @@ -1841,17 +1862,27 @@ TEST_CASE("CBOR") CHECK_NOTHROW(json::to_cbor(json::binary(std::vector({0xFF})))); } - SECTION("invalid UTF-8 in indefinite-length string") + SECTION("ill-formed UTF-8 in indefinite-length string") { json _; - // every chunk must be valid UTF-8 on its own (RFC 8949, Section - // 3.2.3), so a code point split across two chunks is rejected - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded()); + // the chunks are concatenated as is, without checking that each + // chunk is valid UTF-8 on its own (RFC 8949, Section 3.2.3), so + // a code point split across two chunks yields a valid string + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}))); + CHECK(_ == "\xc3\xa9"); + CHECK(_.dump() == "\"\xc3\xa9\""); - // an ill-formed later chunk is rejected after valid ones - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + // a truncated code point is kept as is + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0xff}))); + CHECK(_ == "\xc3"); + CHECK_THROWS_AS(_.dump(), json::type_error&); + CHECK_THROWS_AS(json::to_cbor(_), json::type_error&); + + // an ill-formed later chunk is kept after valid ones + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff}))); + CHECK(_ == "\xc3\xa9\xc0\xae"); + CHECK_THROWS_AS(_.dump(), json::type_error&); // valid multi-byte chunks are accepted CHECK(json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6"); @@ -1859,9 +1890,6 @@ TEST_CASE("CBOR") SECTION("many chunks in indefinite-length string") { - // only the newly read chunk is validated, not the whole string - // collected so far; validating the latter made this input take - // quadratic time (about ten seconds for 100000 chunks) constexpr std::size_t chunks = 100000; std::vector v{0x7f}; for (std::size_t i = 0; i < chunks; ++i) diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index 6716a7375..0ca68f384 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -2506,17 +2506,28 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK(json::from_ubjson(v) == j); } - SECTION("ill-formed UTF-8 (see #5651)") + SECTION("ill-formed UTF-8 (see #5529, #5651)") { - // a string value whose bytes are not valid UTF-8 (0xC0 0xAE is an - // overlong encoding of '.') is rejected at decode time, matching - // every other kind of malformed binary input, and to_ubjson() - // rejects it as well, so a value it accepts can always be read - // back + // none of the binary format specs requires a decoder to reject + // ill-formed UTF-8 in a text string, so a value whose bytes are + // not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') + // round-trips byte for byte as a string value; to_ubjson() is + // strict, so such a value cannot be written back const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; - json _; - CHECK_THROWS_WITH_AS(_ = json::from_ubjson(v), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing UBJSON string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_ubjson(v, true, false).is_discarded()); + json j; + CHECK_NOTHROW(j = json::from_ubjson(v)); + REQUIRE(j.is_string()); + CHECK(j.get_ref() == std::string("\xc0\xae")); + CHECK_THROWS_AS(j.dump(), json::type_error&); + CHECK_THROWS_AS(json::to_ubjson(j), json::type_error&); + + // the same bytes as an object key round-trip as well + const std::vector v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'}; + json j_key; + CHECK_NOTHROW(j_key = json::from_ubjson(v_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK_THROWS_AS(json::to_ubjson(j_key), json::type_error&); CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); // a truncated multi-byte sequence