From 73d116547a8233a1ff179fa12dea01ee61da0866 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 22:19:11 +0200 Subject: [PATCH] Reject ill-formed UTF-8 in CBOR/UBJSON/BJData/BSON writers, accept it in MessagePack reader to_cbor(), to_ubjson(), to_bjdata(), and to_bson() wrote a string or object key with ill-formed UTF-8 byte for byte, but from_cbor()/from_ubjson()/ from_bjdata()/from_bson() reject such strings with parse_error.113 since d19f7f5dc (#5185, not yet released): a value that serialized without error could not be read back by the same library (#5651). CBOR (RFC 8949 Section 5), UBJSON, BJData, and BSON all require text strings and object keys/element names to be valid UTF-8, so their writers now validate and throw type_error.316, like to_bon8() already does. For BSON, the check runs in the size-computation pass, before any byte is written, the same way the binary subtype check works. check_bon8_utf8() is renamed to check_utf8() since it is now shared by all of these writers. MessagePack's specification explicitly allows a str object to contain an invalid byte sequence and expects a deserializer to hand the bytes back unchanged, so its writer is unaffected and from_msgpack() (object keys included) no longer validates UTF-8, restoring its pre-#5185 behavior. dump() still rejects ill-formed UTF-8 with type_error.316 unless an error handler is passed. Docs: add the type_error.316 exception to to_cbor/to_ubjson/to_bjdata/to_bson, document the writer-side check on the cbor/bson/ubjson/bjdata format pages, and rewrite the MessagePack UTF-8 warning to describe the round-trip and dump() behavior instead of a validation requirement the spec does not have. Tests: add ill-formed value/key cases (invalid byte, truncated sequence, encoded surrogate, overlong encoding) expecting type_error.316 to unit-cbor.cpp, unit-ubjson.cpp, unit-bjdata.cpp, and unit-bson.cpp (which also checks the output vector stays empty), and turn unit-msgpack.cpp's former parse_error.113 ill-formed-UTF-8 tests into byte-for-byte round-trip tests for both values and keys. This text was written by Claude Code. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/to_bjdata.md | 5 +- docs/mkdocs/docs/api/basic_json/to_bson.md | 4 ++ docs/mkdocs/docs/api/basic_json/to_cbor.md | 6 ++ docs/mkdocs/docs/api/basic_json/to_ubjson.md | 3 + .../docs/features/binary_formats/bjdata.md | 13 ++++ .../docs/features/binary_formats/bson.md | 8 ++- .../docs/features/binary_formats/cbor.md | 4 +- .../features/binary_formats/messagepack.md | 15 ++--- .../docs/features/binary_formats/ubjson.md | 13 ++++ .../nlohmann/detail/input/binary_reader.hpp | 16 +++-- .../nlohmann/detail/output/binary_writer.hpp | 47 ++++++++++++-- single_include/nlohmann/json.hpp | 63 +++++++++++++++---- tests/src/unit-bjdata.cpp | 24 +++++++ tests/src/unit-bson.cpp | 38 +++++++++++ tests/src/unit-cbor.cpp | 19 ++++++ tests/src/unit-msgpack.cpp | 36 ++++++++--- tests/src/unit-ubjson.cpp | 24 +++++++ 17 files changed, 297 insertions(+), 41 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/to_bjdata.md b/docs/mkdocs/docs/api/basic_json/to_bjdata.md index b066e6852..1473e7b70 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/to_bjdata.md @@ -56,6 +56,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false. +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 ## Complexity @@ -89,4 +91,5 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.11.0. -- BJData version parameter (for draft3 binary encoding) added in version 3.12.0. \ No newline at end of file +- BJData version parameter (for draft3 binary encoding) added in version 3.12.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 added in version 3.13.0. \ No newline at end of file diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index 5ba3c8bb4..16b19119a 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -46,6 +46,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the BSON binary subtype; example: `"subtype 70000 is too large for the BSON binary subtype (max 255)"` +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is + not valid UTF-8 ## Complexity @@ -82,3 +84,5 @@ pass before anything is written. - Added in version 3.4.0. - Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0. - `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0. +- Throwing `type_error.316` for a string value or object key that is not valid UTF-8, detected before anything is + written, added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/to_cbor.md b/docs/mkdocs/docs/api/basic_json/to_cbor.md index 3bbd9c7d3..10449a0c1 100644 --- a/docs/mkdocs/docs/api/basic_json/to_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/to_cbor.md @@ -35,6 +35,11 @@ The exact mapping and its limitations are described on a [dedicated page](../../ Strong guarantee: if an exception is thrown, there are no changes in the JSON value. +## Exceptions + +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 + ## Complexity Linear in the size of the JSON value `j`. @@ -68,3 +73,4 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Compact representation of floating-point numbers added in version 3.8.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/to_ubjson.md b/docs/mkdocs/docs/api/basic_json/to_ubjson.md index 6437ba3e5..4fc87dcf9 100644 --- a/docs/mkdocs/docs/api/basic_json/to_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/to_ubjson.md @@ -49,6 +49,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false. +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 ## Complexity @@ -82,3 +84,4 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.1.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 added in version 3.13.0. diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index a0c84edaf..05cb5f196 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -63,6 +63,12 @@ The library uses the following mapping from JSON values types to BJData types ac - strings with more than 18446744073709551615 bytes, i.e., $2^{64}-1$ bytes (theoretical) +!!! warning "UTF-8 validation of string values and object keys" + + BJData strings must use UTF-8 encoding. `to_bjdata()` validates the bytes of every string value and object key + and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, so a + value with such a string cannot be serialized in the first place. + !!! info "Unused BJData markers" The following markers are not used in the conversion: @@ -208,6 +214,13 @@ The library maps BJData types to JSON value types as follows: The mapping is **complete** in the sense that any BJData value can be converted to a JSON value. +!!! warning "UTF-8 validation of string values and object keys" + + This library validates the bytes of every string value and object key at decode time and rejects ill-formed + UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with + `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value + is dumped. + !!! info "Round trips" A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with diff --git a/docs/mkdocs/docs/features/binary_formats/bson.md b/docs/mkdocs/docs/features/binary_formats/bson.md index 6f5603c8c..ede6c6ca7 100644 --- a/docs/mkdocs/docs/features/binary_formats/bson.md +++ b/docs/mkdocs/docs/features/binary_formats/bson.md @@ -115,8 +115,12 @@ The library maps BSON record types to JSON value types as follows: bytes of every such string at decode time and rejects ill-formed UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value is dumped. Element - (key) names and `binary` values (type `0x05`) are unaffected and are never validated, since they are read - byte-by-byte as a C string, or are not required to hold text, respectively. + (key) names and `binary` values (type `0x05`) are unaffected and are never validated on read, since they are read + byte-by-byte as a C string, or are not required to hold text, respectively. `to_bson()` validates both string + values and element names and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 in either, so an + object with such a key or value cannot be produced in the first place, even though `from_bson()` would accept it + from another source. ??? example diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index a488466d4..c8332c1be 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -197,7 +197,9 @@ The library maps CBOR types to JSON value types as follows: [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value is dumped. Byte strings (major type 2) are unaffected and are never validated, since they are not required to hold - text. + text. `to_cbor()` validates string values and object keys the same way and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, so a value with + such a string cannot be serialized in the first place. !!! warning "Tagged items" diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index 3ce5f7620..f3c68c064 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -153,14 +153,15 @@ The library maps MessagePack types to JSON value types as follows: This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed on. Such input needs a general-purpose MessagePack library instead. -!!! warning "UTF-8 validation of string values" +!!! warning "Ill-formed UTF-8 in string values" - The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8. - This library validates the bytes of every such string (object keys included) at decode time and rejects - ill-formed UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, - with `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting - value is dumped. `bin`/`ext`/`fixext` values are unaffected and are never validated, since they are not required - to hold text. + The MessagePack specification explicitly allows a `str` value (`fixstr`, `str 8`, `str 16`, `str 32`) to contain + a byte sequence that is not valid UTF-8, and expects a deserializer to hand the original bytes back unchanged. + This library follows that: `from_msgpack()` reads `str` bytes (object keys included) as-is, without validating + them, and `to_msgpack()` writes them back as-is, so such a value round-trips through `from_msgpack(to_msgpack(j))` + byte for byte. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way, unless an + error handler is passed that replaces or ignores the ill-formed bytes. ??? example diff --git a/docs/mkdocs/docs/features/binary_formats/ubjson.md b/docs/mkdocs/docs/features/binary_formats/ubjson.md index be545b9fe..a24fedb4f 100644 --- a/docs/mkdocs/docs/features/binary_formats/ubjson.md +++ b/docs/mkdocs/docs/features/binary_formats/ubjson.md @@ -47,6 +47,12 @@ The library uses the following mapping from JSON values types to UBJSON types ac - strings with more than 9223372036854775807 bytes (theoretical) +!!! warning "UTF-8 validation of string values and object keys" + + UBJSON's required string encoding is UTF-8. `to_ubjson()` validates the bytes of every string value and object + key and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, so + a value with such a string cannot be serialized in the first place. + !!! info "Unused UBJSON markers" The following markers are not used in the conversion: @@ -120,6 +126,13 @@ The library maps UBJSON types to JSON value types as follows: The mapping is **complete** in the sense that any UBJSON value can be converted to a JSON value. +!!! warning "UTF-8 validation of string values and object keys" + + This library validates the bytes of every string value and object key at decode time and rejects ill-formed + UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with + `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value + is dumped. + ??? example ```cpp diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index b9e6b304b..f09330ddc 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -4111,12 +4111,16 @@ class binary_reader return false; } - // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications - // all require text strings to be valid UTF-8; reject anything else - // right here so malformed input is caught at decode time instead of - // only surfacing later as a type_error.316 when the value is dumped - // (which would defeat allow_exceptions=false / strict discarding). - if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) + // RFC 8949 (CBOR) §3.1 and the BSON/UBJSON specifications require + // text strings to be valid UTF-8; reject anything else right here so + // malformed input is caught at decode time instead of only surfacing + // later as a type_error.316 when the value is dumped (which would + // defeat allow_exceptions=false / strict discarding). The MessagePack + // specification explicitly allows a str object to contain an invalid + // byte sequence and expects deserializers to hand back the original + // bytes, so msgpack strings (and map keys, which go through this + // function as well) are exempt. + if (format != input_format_t::msgpack && JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) { return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 1ebd3c8ab..9204b39ef 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -116,6 +116,8 @@ class binary_writer /*! @param[in] j JSON value to serialize @pre j.type() == value_t::object + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_bson(const BasicJsonType& j) { @@ -145,6 +147,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -211,6 +215,8 @@ class binary_writer case value_t::string: { + check_utf8(*j.m_data.m_value.string, j); + // step 1: write control byte and the string length write_cbor_head(0x60, j.m_data.m_value.string->size()); @@ -280,6 +286,11 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // el.first is checked here, against the object as + // diagnostics context, because write_cbor(el.first) + // converts it to a temporary basic_json that would be + // used as the context instead + check_utf8(el.first, j); write_cbor(el.first); write_cbor(el.second); } @@ -643,6 +654,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -692,6 +705,8 @@ class binary_writer case value_t::string: { + check_utf8(*j.m_data.m_value.string, j); + if (add_prefix) { oa.write_character(to_char_type('S')); @@ -854,6 +869,7 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { + check_utf8(el.first, j); write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(el.first.data()), @@ -898,6 +914,10 @@ class binary_writer /*! @return The size of a BSON document entry header, including the id marker and the entry name size (and its null-terminator). + @throw out_of_range.409 if @a name contains U+0000, before anything is + written + @throw type_error.316 if @a name is not valid UTF-8, before anything is + written */ static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { @@ -907,7 +927,8 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - static_cast(j); + check_utf8(name, j); + return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; } @@ -963,9 +984,21 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value + @throw type_error.316 if @a value is not valid UTF-8, before anything is + written + + @note The UTF-8 check is skipped if @a value is already too long for the + 32-bit BSON length field (@ref to_bson_length rejects it later, once + the size of the whole document is known); this also keeps the check + from reading past a StringType that reports a size larger than what + it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value) + static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { + if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) + { + check_utf8(value, j); + } return sizeof(std::int32_t) + value.size() + 1ul; } @@ -1094,6 +1127,8 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written + @throw type_error.316 if @a j is a string that is not valid UTF-8, before + anything is written */ static std::size_t calc_bson_value_size(const BasicJsonType& j) { @@ -1115,7 +1150,7 @@ class binary_writer return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string, j); case value_t::null: return 0ul; @@ -1228,6 +1263,8 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written + @throw type_error.316 if a string value or a key is not valid UTF-8, + before anything is written */ static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { @@ -2252,7 +2289,7 @@ class binary_writer */ void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) { - check_bon8_utf8(s, context); + check_utf8(s, context); // a string that follows another string terminates it if (string_open) @@ -2282,7 +2319,7 @@ class binary_writer @throw type_error.316 if @a s is not valid UTF-8; the message names the first byte of the first invalid or incomplete sequence */ - static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + static void check_utf8(const string_t& s, const BasicJsonType& context) { static_cast(context); // only used when exceptions are enabled const auto* data = reinterpret_cast(s.data()); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 1fa63f344..9a8ac870a 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -17664,12 +17664,16 @@ class binary_reader return false; } - // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications - // all require text strings to be valid UTF-8; reject anything else - // right here so malformed input is caught at decode time instead of - // only surfacing later as a type_error.316 when the value is dumped - // (which would defeat allow_exceptions=false / strict discarding). - if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) + // RFC 8949 (CBOR) §3.1 and the BSON/UBJSON specifications require + // text strings to be valid UTF-8; reject anything else right here so + // malformed input is caught at decode time instead of only surfacing + // later as a type_error.316 when the value is dumped (which would + // defeat allow_exceptions=false / strict discarding). The MessagePack + // specification explicitly allows a str object to contain an invalid + // byte sequence and expects deserializers to hand back the original + // bytes, so msgpack strings (and map keys, which go through this + // function as well) are exempt. + if (format != input_format_t::msgpack && JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) { return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, @@ -21240,6 +21244,8 @@ class binary_writer /*! @param[in] j JSON value to serialize @pre j.type() == value_t::object + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_bson(const BasicJsonType& j) { @@ -21269,6 +21275,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -21335,6 +21343,8 @@ class binary_writer case value_t::string: { + check_utf8(*j.m_data.m_value.string, j); + // step 1: write control byte and the string length write_cbor_head(0x60, j.m_data.m_value.string->size()); @@ -21404,6 +21414,11 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // el.first is checked here, against the object as + // diagnostics context, because write_cbor(el.first) + // converts it to a temporary basic_json that would be + // used as the context instead + check_utf8(el.first, j); write_cbor(el.first); write_cbor(el.second); } @@ -21767,6 +21782,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -21816,6 +21833,8 @@ class binary_writer case value_t::string: { + check_utf8(*j.m_data.m_value.string, j); + if (add_prefix) { oa.write_character(to_char_type('S')); @@ -21978,6 +21997,7 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { + check_utf8(el.first, j); write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(el.first.data()), @@ -22022,6 +22042,10 @@ class binary_writer /*! @return The size of a BSON document entry header, including the id marker and the entry name size (and its null-terminator). + @throw out_of_range.409 if @a name contains U+0000, before anything is + written + @throw type_error.316 if @a name is not valid UTF-8, before anything is + written */ static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { @@ -22031,7 +22055,8 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - static_cast(j); + check_utf8(name, j); + return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; } @@ -22087,9 +22112,21 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value + @throw type_error.316 if @a value is not valid UTF-8, before anything is + written + + @note The UTF-8 check is skipped if @a value is already too long for the + 32-bit BSON length field (@ref to_bson_length rejects it later, once + the size of the whole document is known); this also keeps the check + from reading past a StringType that reports a size larger than what + it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value) + static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { + if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) + { + check_utf8(value, j); + } return sizeof(std::int32_t) + value.size() + 1ul; } @@ -22218,6 +22255,8 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written + @throw type_error.316 if @a j is a string that is not valid UTF-8, before + anything is written */ static std::size_t calc_bson_value_size(const BasicJsonType& j) { @@ -22239,7 +22278,7 @@ class binary_writer return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string, j); case value_t::null: return 0ul; @@ -22352,6 +22391,8 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written + @throw type_error.316 if a string value or a key is not valid UTF-8, + before anything is written */ static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { @@ -23376,7 +23417,7 @@ class binary_writer */ void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) { - check_bon8_utf8(s, context); + check_utf8(s, context); // a string that follows another string terminates it if (string_open) @@ -23406,7 +23447,7 @@ class binary_writer @throw type_error.316 if @a s is not valid UTF-8; the message names the first byte of the first invalid or incomplete sequence */ - static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + static void check_utf8(const string_t& s, const BasicJsonType& context) { static_cast(context); // only used when exceptions are enabled const auto* data = reinterpret_cast(s.data()); diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index a58507c15..cd1973d02 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -4025,6 +4025,30 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK(json::to_bjdata(j) == v); CHECK(json::from_bjdata(v) == j); } + + SECTION("ill-formed UTF-8 (see #5651)") + { + // a string value whose bytes are not valid UTF-8 (0xC0 0xAE is an + // overlong encoding of '.') is rejected at decode time, matching + // every other kind of malformed binary input, and to_bjdata() + // rejects it as well, so a value it accepts can always be read + // back + const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; + json _; + CHECK_THROWS_WITH_AS(_ = json::from_bjdata(v), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing BJData string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + CHECK(json::from_bjdata(v, true, false).is_discarded()); + + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } } SECTION("Array Type") diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 5bfaac7c2..8d91e23ea 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -151,6 +151,44 @@ TEST_CASE("BSON") #endif } + SECTION("ill-formed UTF-8 (see #5651)") + { + // a BSON document {"s": "\xC0\xAE"} (0xC0 0xAE is an overlong + // encoding of '.'); the reader rejects an ill-formed string value at + // decode time + const std::vector v = + { + 0x0F, 0x00, 0x00, 0x00, // document length + 0x02, 's', 0x00, // type 0x02 (string), key "s" + 0x03, 0x00, 0x00, 0x00, // string length (including null) + 0xc0, 0xae, 0x00, // string content and its null terminator + 0x00 // document terminator + }; + json _; + CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing BSON string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + CHECK(json::from_bson(v, true, false).is_discarded()); + + // to_bson() rejects the same kind of ill-formed string value, before + // any bytes reach the output adapter (the BSON document length + // prefix must be known up front, so nothing is written incrementally) + std::vector out{0x42}; // a sentinel byte the writer must not touch + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + CHECK(out == std::vector {0x42}); + + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected as well; unlike + // the reader (which never validates element names), the writer + // checks both string values and object keys + CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON") { // out_of_range.412 is thrown from a single shared helper diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index fe0fb2644..109539555 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -1896,6 +1896,25 @@ TEST_CASE("CBOR") CHECK(json::from_cbor(json::to_cbor(j)) == j); } + SECTION("to_cbor rejects ill-formed UTF-8 (see #5651)") + { + // to_cbor() must reject the same ill-formed strings from_cbor() + // rejects, so a value it accepts can always be read back + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + + // binary values are not text and are unaffected + CHECK_NOTHROW(json::to_cbor(json::binary(std::vector({0xFF})))); + } + SECTION("invalid UTF-8 in indefinite-length string") { json _; diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index 0876e0f9f..c74a0918f 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -1614,19 +1614,39 @@ TEST_CASE("MessagePack") CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&); } - SECTION("invalid UTF-8 in string (see #5529)") + SECTION("ill-formed UTF-8 in string (see #5529, #5651)") { + // the MessagePack specification explicitly allows a str object to + // contain a byte sequence that is not valid UTF-8 and expects a + // deserializer to hand the original bytes back unchanged; this + // library follows that, unlike CBOR/UBJSON/BJData/BSON, whose + // specifications require text strings to be valid UTF-8 + // a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8 - // (0xC0 0xAE is an overlong encoding of '.') must be rejected at - // decode time, matching every other kind of malformed binary - // input, rather than only failing later when the resulting - // value is dumped - json _; - CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0xa2, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing MessagePack string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_msgpack(std::vector({0xa2, 0xc0, 0xae}), true, false).is_discarded()); + // (0xC0 0xAE is an overlong encoding of '.') round-trips byte for + // byte as a string value + const std::vector ill_formed_value = {0xa2, 0xc0, 0xae}; + json j_value; + CHECK_NOTHROW(j_value = json::from_msgpack(ill_formed_value)); + REQUIRE(j_value.is_string()); + CHECK(j_value.get_ref() == std::string("\xc0\xae")); + CHECK(json::from_msgpack(json::to_msgpack(j_value)) == j_value); + // dump() still requires valid UTF-8 and throws for such a value, + // unless an error handler that replaces or ignores the bytes is + // passed + CHECK_THROWS_AS(j_value.dump(), json::type_error&); + + // the same bytes as an object key round-trip as well + const std::vector ill_formed_key = {0x81, 0xa2, 0xc0, 0xae, 0x01}; + json j_key; + CHECK_NOTHROW(j_key = json::from_msgpack(ill_formed_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_msgpack(json::to_msgpack(j_key)) == j_key); // a MessagePack bin8 blob with the very same bytes is NOT text // and must still be accepted as-is + json _; CHECK_NOTHROW(_ = json::from_msgpack(std::vector({0xc4, 0x02, 0xc0, 0xae}))); CHECK(_ == json::binary(std::vector({0xc0, 0xae}))); diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index c315fac94..78f3d091f 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -2580,6 +2580,30 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK(json::to_ubjson(j) == v); CHECK(json::from_ubjson(v) == j); } + + SECTION("ill-formed UTF-8 (see #5651)") + { + // a string value whose bytes are not valid UTF-8 (0xC0 0xAE is an + // overlong encoding of '.') is rejected at decode time, matching + // every other kind of malformed binary input, and to_ubjson() + // rejects it as well, so a value it accepts can always be read + // back + const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; + json _; + CHECK_THROWS_WITH_AS(_ = json::from_ubjson(v), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing UBJSON string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + CHECK(json::from_ubjson(v, true, false).is_discarded()); + + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } } SECTION("Array Type")