Add an error_handler parameter for UTF-8 to the binary readers and writers

Adds error_handler_t::keep (invalid UTF-8 sequences are left unchanged) as
a fourth error_handler_t value, and threads an error_handler parameter
through the binary writers and readers:

- to_cbor/to_ubjson/to_bjdata/to_bson gain a trailing error_handler
  parameter (default strict, matching their existing type_error.316
  behavior); keep writes a string value or object key's bytes as is
  instead of throwing, and replace/ignore sanitize it exactly like
  dump() would, including for the BSON length prefix. to_msgpack and
  to_bon8 are unchanged.
- from_cbor/from_msgpack/from_ubjson/from_bjdata/from_bson gain a
  trailing error_handler parameter (default keep, i.e. the lenient
  behavior every binary reader had in 3.12.0 and still has after
  #5741); strict checks every string value and object key and raises
  parse_error.113 for ill-formed UTF-8, honoring allow_exceptions;
  replace/ignore sanitize it like dump() would. from_bon8 is
  unchanged, since UTF-8 lead bytes are structural there.

dump()'s own keep support writes ill-formed bytes as is, even with
ensure_ascii, while still \u-escaping well-formed characters around
them as usual.

The UTF-8 validity check (is_valid_utf8) and the replace/ignore
sanitizing logic (sanitize_utf8) now live in string_utils.hpp, shared
by the serializer and the binary reader/writer; error_handler_t itself
moved to its own header (detail/output/error_handler.hpp) so that
string_utils.hpp does not need to depend on serializer.hpp.

See #5529 and #5741.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-10-01 08:43:15 +02:00
parent 92c729f657
commit fff221c58b
27 changed files with 1601 additions and 308 deletions
+82 -32
View File
@@ -31,6 +31,7 @@
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/is_sax.hpp>
#include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/detail/output/error_handler.hpp>
#include <nlohmann/detail/string_concat.hpp>
#include <nlohmann/detail/string_utils.hpp>
#include <nlohmann/detail/value_t.hpp>
@@ -108,8 +109,16 @@ class binary_reader
@brief create a binary reader
@param[in] adapter input adapter to read from
@param[in] format the binary format to parse
@param[in] error_handler how to treat text strings and object keys that
are not well-formed UTF-8; none of the supported formats
requires a decoder to reject those, so the default is to
@ref error_handler_t::keep them unchanged, as every binary
reader did before this parameter existed
*/
explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json) noexcept : ia(std::move(adapter)), input_format(format)
explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json,
const error_handler_t error_handler = error_handler_t::keep) noexcept
: ia(std::move(adapter)), input_format(format), error_handler(error_handler)
{
(void)detail::is_sax_static_asserts<SAX, BasicJsonType> {};
}
@@ -428,7 +437,7 @@ class binary_reader
{
if (get_bson_cstr_bulk(result, std::integral_constant<bool, bulk_scan> {}))
{
return true;
return check_string_utf8(result, "key");
}
auto out = std::back_inserter(result);
@@ -441,7 +450,7 @@ class binary_reader
}
if (current == 0x00)
{
return true;
return check_string_utf8(result, "key");
}
*out++ = static_cast<typename string_t::value_type>(current);
}
@@ -522,7 +531,7 @@ class binary_reader
"string"), nullptr));
}
return true;
return check_string_utf8(result, "string");
}
/*!
@@ -1149,7 +1158,7 @@ class binary_reader
@return whether string creation completed
*/
bool get_cbor_string(string_t& result)
bool get_cbor_string(string_t& result, const char* context = "string")
{
// number of indefinite-length strings that have been opened and not
// closed yet. RFC 8949, Section 3.2.3 does not permit nesting them,
@@ -1179,7 +1188,7 @@ class binary_reader
{
if (--open == 0)
{
return true;
return check_string_utf8(result, context);
}
get();
continue;
@@ -1192,7 +1201,7 @@ class binary_reader
if (open == 0)
{
return true;
return check_string_utf8(result, context);
}
get();
@@ -1216,7 +1225,7 @@ class binary_reader
// EOF and major type 3 (text string) are left to get_cbor_string
if (current == char_traits<char_type>::eof() || (static_cast<unsigned int>(current) & 0xE0u) == 0x60u)
{
return get_cbor_string(result);
return get_cbor_string(result, "key");
}
const char* found = nullptr;
@@ -2004,7 +2013,7 @@ class binary_reader
@return whether string creation completed
*/
bool get_msgpack_string(string_t& result)
bool get_msgpack_string(string_t& result, const char* context = "string")
{
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "string")))
{
@@ -2047,25 +2056,25 @@ class binary_reader
case 0xBE:
case 0xBF:
{
return get_string(input_format_t::msgpack, static_cast<unsigned int>(current) & 0x1Fu, result);
return get_string(input_format_t::msgpack, static_cast<unsigned int>(current) & 0x1Fu, result) && check_string_utf8(result, context);
}
case 0xD9: // str 8
{
std::uint8_t len{};
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result);
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context);
}
case 0xDA: // str 16
{
std::uint16_t len{};
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result);
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context);
}
case 0xDB: // str 32
{
std::uint32_t len{};
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result);
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context);
}
default:
@@ -2143,7 +2152,7 @@ class binary_reader
// byte 0xC1 are left to get_msgpack_string
if (current == char_traits<char_type>::eof())
{
return get_msgpack_string(result);
return get_msgpack_string(result, "key");
}
if (current <= 0x7F || current >= 0xE0)
{
@@ -2159,7 +2168,7 @@ class binary_reader
}
else
{
return get_msgpack_string(result);
return get_msgpack_string(result, "key");
}
break;
}
@@ -2405,7 +2414,7 @@ class binary_reader
if (top.is_object)
{
key.clear();
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key)))
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, true, "key") || !sax->key(key)))
{
return false;
}
@@ -2427,7 +2436,7 @@ class binary_reader
if (top.is_object)
{
key.clear();
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key)))
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false, "key") || !sax->key(key)))
{
return false;
}
@@ -2495,7 +2504,7 @@ class binary_reader
@return whether string creation completed
*/
bool get_ubjson_string(string_t& result, const bool get_char = true)
bool get_ubjson_string(string_t& result, const bool get_char = true, const char* context = "string")
{
if (get_char)
{
@@ -2516,31 +2525,31 @@ class binary_reader
case 'U':
{
std::uint8_t len{};
return get_number(input_format, len) && get_string(input_format, len, result);
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'i':
{
std::int8_t len{};
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'I':
{
std::int16_t len{};
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'l':
{
std::int32_t len{};
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'L':
{
std::int64_t len{};
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'u':
@@ -2550,7 +2559,7 @@ class binary_reader
break;
}
std::uint16_t len{};
return get_number(input_format, len) && get_string(input_format, len, result);
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'm':
@@ -2560,7 +2569,7 @@ class binary_reader
break;
}
std::uint32_t len{};
return get_number(input_format, len) && get_string(input_format, len, result);
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'M':
@@ -2570,7 +2579,7 @@ class binary_reader
break;
}
std::uint64_t len{};
return get_number(input_format, len) && get_string(input_format, len, result);
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
default:
@@ -4031,15 +4040,53 @@ class binary_reader
const NumberType len,
string_t& result)
{
// Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the
// choice to the decoder), MessagePack (whose spec explicitly allows
// a str object to contain an invalid byte sequence), UBJSON, BJData,
// or BSON requires a decoder to reject ill-formed UTF-8. The bytes
// are kept unchanged; dump() and the binary writers are the ones
// that check them and report type_error.316 if they are not valid.
// Strings are taken as is by default: none of CBOR (RFC 8949 §3.1
// leaves the choice to the decoder), MessagePack (whose spec
// explicitly allows a str object to contain an invalid byte
// sequence), UBJSON, BJData, or BSON requires a decoder to reject
// ill-formed UTF-8. Checking (and, with @ref error_handler_t::strict,
// rejecting, or with `replace`/`ignore`, sanitizing) is opt-in via
// @ref error_handler, applied once the whole string (all chunks of
// an indefinite-length CBOR string included) has been assembled, by
// @ref check_string_utf8 at the call site.
return get_bytes(format, len, "string", result);
}
/*!
@brief validate a decoded text string (value or object key) against @ref error_handler
None of the binary formats requires a decoder to reject ill-formed UTF-8
in a text string (see @ref get_string), so by default
(@ref error_handler_t::keep) this does nothing. A stricter
@ref error_handler opts into the same well-formedness check @ref
serializer::dump_escaped_impl applies when dumping a string:
@ref error_handler_t::strict rejects ill-formed input with
parse_error.113 (honoring `allow_exceptions` via @a sax), while
@ref error_handler_t::replace / @ref error_handler_t::ignore sanitize
@a result in place, using the exact same rules.
@param[in,out] result the already assembled string to check
@param[in] context further context information (for diagnostics)
@return whether @a result is acceptable (always true for `keep`)
*/
bool check_string_utf8(string_t& result, const char* context)
{
if (error_handler == error_handler_t::keep || is_valid_utf8(result))
{
return true;
}
if (error_handler == error_handler_t::strict)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(input_format, "invalid string: ill-formed UTF-8 byte", context), nullptr));
}
result = sanitize_utf8(result, error_handler);
return true;
}
/*!
@brief create a byte array by reading bytes from the input
@@ -4211,6 +4258,9 @@ class binary_reader
/// input format
const input_format_t input_format = input_format_t::json;
/// how to treat text strings/object keys that are not well-formed UTF-8
const error_handler_t error_handler = error_handler_t::keep;
/// the SAX parser
json_sax_t* sax = nullptr;