Follow each binary format's UTF-8 rule: strict writers (CBOR/UBJSON/BJData/BSON), lenient readers (#5741)

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-10-04 12:13:48 +02:00
committed by GitHub
parent 40021f38fb
commit f56b418c56
32 changed files with 651 additions and 128 deletions
@@ -115,6 +115,8 @@ class binary_writer
/*!
@param[in] j JSON value to serialize
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
value or an object key is not valid UTF-8
@throw type_error.317 if @a j is not an object
*/
void write_bson(const BasicJsonType& j)
@@ -145,6 +147,8 @@ class binary_writer
/*!
@param[in] j JSON value to serialize
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
value or an object key is not valid UTF-8
*/
void write_cbor(const BasicJsonType& j)
{
@@ -211,6 +215,8 @@ class binary_writer
case value_t::string:
{
check_text_utf8(*j.m_data.m_value.string, j);
// step 1: write control byte and the string length
write_cbor_head(0x60, j.m_data.m_value.string->size());
@@ -287,6 +293,11 @@ class binary_writer
// step 2: write each element
for (const auto& el : *j.m_data.m_value.object)
{
// el.first is checked here, against the object as
// diagnostics context, because write_cbor(el.first)
// converts it to a temporary basic_json that would be
// used as the context instead
check_text_utf8(el.first, j);
write_cbor(el.first);
write_cbor(el.second);
}
@@ -629,6 +640,8 @@ class binary_writer
@param[in] add_prefix whether prefixes need to be used for this value
@param[in] use_bjdata whether write in BJData format, default is false
@param[in] bjdata_version which BJData version to use, default is draft2
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
value or an object key is not valid UTF-8
*/
void write_ubjson(const BasicJsonType& j, const bool use_count,
const bool use_type, const bool add_prefix = true,
@@ -678,6 +691,8 @@ class binary_writer
case value_t::string:
{
check_text_utf8(*j.m_data.m_value.string, j);
if (add_prefix)
{
oa.write_character(to_char_type('S'));
@@ -840,6 +855,7 @@ class binary_writer
for (const auto& el : *j.m_data.m_value.object)
{
check_text_utf8(el.first, j);
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
oa.write_characters(
reinterpret_cast<const CharType*>(el.first.data()),
@@ -884,6 +900,10 @@ class binary_writer
/*!
@return The size of a BSON document entry header, including the id marker
and the entry name size (and its null-terminator).
@throw out_of_range.409 if @a name contains U+0000, before anything is
written
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a name is
not valid UTF-8, before anything is written
*/
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
{
@@ -893,7 +913,8 @@ class binary_writer
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
}
static_cast<void>(j);
check_text_utf8(name, j);
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
}
@@ -949,9 +970,21 @@ class binary_writer
/*!
@return The size of the BSON-encoded string in @a value
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a value
is not valid UTF-8, before anything is written
@note The UTF-8 check is skipped if @a value is already too long for the
32-bit BSON length field (@ref to_bson_length rejects it later, once
the size of the whole document is known); this also keeps the check
from reading past a StringType that reports a size larger than what
it actually holds.
*/
static std::size_t calc_bson_string_size(const string_t& value)
static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j)
{
if (JSON_HEDLEY_LIKELY(value_in_range_of<std::int32_t>(value.size())))
{
check_text_utf8(value, j);
}
return sizeof(std::int32_t) + value.size() + 1ul;
}
@@ -1080,6 +1113,8 @@ class binary_writer
is neither an object nor an array
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
into a byte, before anything is written
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a j is a
string that is not valid UTF-8, before anything is written
*/
static std::size_t calc_bson_value_size(const BasicJsonType& j)
{
@@ -1101,7 +1136,7 @@ class binary_writer
return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned);
case value_t::string:
return calc_bson_string_size(*j.m_data.m_value.string);
return calc_bson_string_size(*j.m_data.m_value.string, j);
case value_t::null:
return 0ul;
@@ -1214,6 +1249,8 @@ class binary_writer
written
@throw out_of_range.415 if a binary value's subtype does not fit into a
byte, before anything is written
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
value or a key is not valid UTF-8, before anything is written
*/
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
{
@@ -2092,7 +2129,7 @@ class binary_writer
*/
void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context)
{
check_bon8_utf8(s, context);
check_utf8(s, context);
// a string that follows another string terminates it
if (string_open)
@@ -2122,7 +2159,7 @@ class binary_writer
@throw type_error.316 if @a s is not valid UTF-8; the message names the
first byte of the first invalid or incomplete sequence
*/
static void check_bon8_utf8(const string_t& s, const BasicJsonType& context)
static void check_utf8(const string_t& s, const BasicJsonType& context)
{
static_cast<void>(context); // only used when exceptions are enabled
const auto* data = reinterpret_cast<const unsigned char*>(s.data());
@@ -2133,6 +2170,29 @@ class binary_writer
}
}
/*!
@brief check a CBOR, UBJSON, BJData, or BSON text string for valid UTF-8
The check only happens if JSON_STRICT_BINARY_UTF8 is enabled. Otherwise,
the bytes are written unchanged, as before version 3.13.0. MessagePack
always writes the bytes as is, and BON8 always checks them (see
@ref check_utf8).
@param[in] s the string to check
@param[in] context the value that holds @a s (for diagnostics)
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a s is
not valid UTF-8
*/
static void check_text_utf8(const string_t& s, const BasicJsonType& context)
{
#if JSON_STRICT_BINARY_UTF8
check_utf8(s, context);
#else
static_cast<void>(s);
static_cast<void>(context);
#endif
}
/*!
@brief write an integer in the shortest encoding