mirror of
https://github.com/nlohmann/json.git
synced 2026-10-03 21:20:30 +00:00
Reject ill-formed UTF-8 in CBOR/UBJSON/BJData/BSON writers, accept it in MessagePack reader
to_cbor(), to_ubjson(), to_bjdata(), and to_bson() wrote a string or object
key with ill-formed UTF-8 byte for byte, but from_cbor()/from_ubjson()/
from_bjdata()/from_bson() reject such strings with parse_error.113 since
d19f7f5dc (#5185, not yet released): a value that serialized without error
could not be read back by the same library (#5651).
CBOR (RFC 8949 Section 5), UBJSON, BJData, and BSON all require text strings
and object keys/element names to be valid UTF-8, so their writers now
validate and throw type_error.316, like to_bon8() already does. For BSON,
the check runs in the size-computation pass, before any byte is written, the
same way the binary subtype check works. check_bon8_utf8() is renamed to
check_utf8() since it is now shared by all of these writers.
MessagePack's specification explicitly allows a str object to contain an
invalid byte sequence and expects a deserializer to hand the bytes back
unchanged, so its writer is unaffected and from_msgpack() (object keys
included) no longer validates UTF-8, restoring its pre-#5185 behavior.
dump() still rejects ill-formed UTF-8 with type_error.316 unless an error
handler is passed.
Docs: add the type_error.316 exception to to_cbor/to_ubjson/to_bjdata/to_bson,
document the writer-side check on the cbor/bson/ubjson/bjdata format pages,
and rewrite the MessagePack UTF-8 warning to describe the round-trip and
dump() behavior instead of a validation requirement the spec does not have.
Tests: add ill-formed value/key cases (invalid byte, truncated sequence,
encoded surrogate, overlong encoding) expecting type_error.316 to
unit-cbor.cpp, unit-ubjson.cpp, unit-bjdata.cpp, and unit-bson.cpp (which
also checks the output vector stays empty), and turn unit-msgpack.cpp's
former parse_error.113 ill-formed-UTF-8 tests into byte-for-byte round-trip
tests for both values and keys.
This text was written by Claude Code.
Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -4111,12 +4111,16 @@ class binary_reader
|
||||
return false;
|
||||
}
|
||||
|
||||
// RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications
|
||||
// all require text strings to be valid UTF-8; reject anything else
|
||||
// right here so malformed input is caught at decode time instead of
|
||||
// only surfacing later as a type_error.316 when the value is dumped
|
||||
// (which would defeat allow_exceptions=false / strict discarding).
|
||||
if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size)))
|
||||
// RFC 8949 (CBOR) §3.1 and the BSON/UBJSON specifications require
|
||||
// text strings to be valid UTF-8; reject anything else right here so
|
||||
// malformed input is caught at decode time instead of only surfacing
|
||||
// later as a type_error.316 when the value is dumped (which would
|
||||
// defeat allow_exceptions=false / strict discarding). The MessagePack
|
||||
// specification explicitly allows a str object to contain an invalid
|
||||
// byte sequence and expects deserializers to hand back the original
|
||||
// bytes, so msgpack strings (and map keys, which go through this
|
||||
// function as well) are exempt.
|
||||
if (format != input_format_t::msgpack && JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
parse_error::create(113, chars_read,
|
||||
|
||||
@@ -116,6 +116,8 @@ class binary_writer
|
||||
/*!
|
||||
@param[in] j JSON value to serialize
|
||||
@pre j.type() == value_t::object
|
||||
@throw type_error.316 if a string value or an object key is not valid
|
||||
UTF-8
|
||||
*/
|
||||
void write_bson(const BasicJsonType& j)
|
||||
{
|
||||
@@ -145,6 +147,8 @@ class binary_writer
|
||||
|
||||
/*!
|
||||
@param[in] j JSON value to serialize
|
||||
@throw type_error.316 if a string value or an object key is not valid
|
||||
UTF-8
|
||||
*/
|
||||
void write_cbor(const BasicJsonType& j)
|
||||
{
|
||||
@@ -211,6 +215,8 @@ class binary_writer
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
check_utf8(*j.m_data.m_value.string, j);
|
||||
|
||||
// step 1: write control byte and the string length
|
||||
write_cbor_head(0x60, j.m_data.m_value.string->size());
|
||||
|
||||
@@ -280,6 +286,11 @@ class binary_writer
|
||||
// step 2: write each element
|
||||
for (const auto& el : *j.m_data.m_value.object)
|
||||
{
|
||||
// el.first is checked here, against the object as
|
||||
// diagnostics context, because write_cbor(el.first)
|
||||
// converts it to a temporary basic_json that would be
|
||||
// used as the context instead
|
||||
check_utf8(el.first, j);
|
||||
write_cbor(el.first);
|
||||
write_cbor(el.second);
|
||||
}
|
||||
@@ -643,6 +654,8 @@ class binary_writer
|
||||
@param[in] add_prefix whether prefixes need to be used for this value
|
||||
@param[in] use_bjdata whether write in BJData format, default is false
|
||||
@param[in] bjdata_version which BJData version to use, default is draft2
|
||||
@throw type_error.316 if a string value or an object key is not valid
|
||||
UTF-8
|
||||
*/
|
||||
void write_ubjson(const BasicJsonType& j, const bool use_count,
|
||||
const bool use_type, const bool add_prefix = true,
|
||||
@@ -692,6 +705,8 @@ class binary_writer
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
check_utf8(*j.m_data.m_value.string, j);
|
||||
|
||||
if (add_prefix)
|
||||
{
|
||||
oa.write_character(to_char_type('S'));
|
||||
@@ -854,6 +869,7 @@ class binary_writer
|
||||
|
||||
for (const auto& el : *j.m_data.m_value.object)
|
||||
{
|
||||
check_utf8(el.first, j);
|
||||
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(el.first.data()),
|
||||
@@ -898,6 +914,10 @@ class binary_writer
|
||||
/*!
|
||||
@return The size of a BSON document entry header, including the id marker
|
||||
and the entry name size (and its null-terminator).
|
||||
@throw out_of_range.409 if @a name contains U+0000, before anything is
|
||||
written
|
||||
@throw type_error.316 if @a name is not valid UTF-8, before anything is
|
||||
written
|
||||
*/
|
||||
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
|
||||
{
|
||||
@@ -907,7 +927,8 @@ class binary_writer
|
||||
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
|
||||
}
|
||||
|
||||
static_cast<void>(j);
|
||||
check_utf8(name, j);
|
||||
|
||||
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
|
||||
}
|
||||
|
||||
@@ -963,9 +984,21 @@ class binary_writer
|
||||
|
||||
/*!
|
||||
@return The size of the BSON-encoded string in @a value
|
||||
@throw type_error.316 if @a value is not valid UTF-8, before anything is
|
||||
written
|
||||
|
||||
@note The UTF-8 check is skipped if @a value is already too long for the
|
||||
32-bit BSON length field (@ref to_bson_length rejects it later, once
|
||||
the size of the whole document is known); this also keeps the check
|
||||
from reading past a StringType that reports a size larger than what
|
||||
it actually holds.
|
||||
*/
|
||||
static std::size_t calc_bson_string_size(const string_t& value)
|
||||
static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<std::int32_t>(value.size())))
|
||||
{
|
||||
check_utf8(value, j);
|
||||
}
|
||||
return sizeof(std::int32_t) + value.size() + 1ul;
|
||||
}
|
||||
|
||||
@@ -1094,6 +1127,8 @@ class binary_writer
|
||||
is neither an object nor an array
|
||||
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
|
||||
into a byte, before anything is written
|
||||
@throw type_error.316 if @a j is a string that is not valid UTF-8, before
|
||||
anything is written
|
||||
*/
|
||||
static std::size_t calc_bson_value_size(const BasicJsonType& j)
|
||||
{
|
||||
@@ -1115,7 +1150,7 @@ class binary_writer
|
||||
return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned);
|
||||
|
||||
case value_t::string:
|
||||
return calc_bson_string_size(*j.m_data.m_value.string);
|
||||
return calc_bson_string_size(*j.m_data.m_value.string, j);
|
||||
|
||||
case value_t::null:
|
||||
return 0ul;
|
||||
@@ -1228,6 +1263,8 @@ class binary_writer
|
||||
written
|
||||
@throw out_of_range.415 if a binary value's subtype does not fit into a
|
||||
byte, before anything is written
|
||||
@throw type_error.316 if a string value or a key is not valid UTF-8,
|
||||
before anything is written
|
||||
*/
|
||||
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
|
||||
{
|
||||
@@ -2252,7 +2289,7 @@ class binary_writer
|
||||
*/
|
||||
void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context)
|
||||
{
|
||||
check_bon8_utf8(s, context);
|
||||
check_utf8(s, context);
|
||||
|
||||
// a string that follows another string terminates it
|
||||
if (string_open)
|
||||
@@ -2282,7 +2319,7 @@ class binary_writer
|
||||
@throw type_error.316 if @a s is not valid UTF-8; the message names the
|
||||
first byte of the first invalid or incomplete sequence
|
||||
*/
|
||||
static void check_bon8_utf8(const string_t& s, const BasicJsonType& context)
|
||||
static void check_utf8(const string_t& s, const BasicJsonType& context)
|
||||
{
|
||||
static_cast<void>(context); // only used when exceptions are enabled
|
||||
const auto* data = reinterpret_cast<const unsigned char*>(s.data());
|
||||
|
||||
Reference in New Issue
Block a user