mirror of
https://github.com/nlohmann/json.git
synced 2026-10-03 21:20:30 +00:00
Accept ill-formed UTF-8 in all binary readers again
RFC 8949 and the MessagePack/BSON/UBJSON/BJData specs leave UTF-8 well-formedness checking up to the decoder, so following #5529 the binary readers are lenient by default again, as in release 3.12.0 (the reader-side check was added by #5185/#5531, not in any release); reader-side validation becomes opt-in in a follow-up PR. The writers stay strict and throw type_error.316 for ill-formed UTF-8. BON8 is unchanged, since UTF-8 lead bytes are structural there. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -4031,32 +4031,13 @@ class binary_reader
|
||||
const NumberType len,
|
||||
string_t& result)
|
||||
{
|
||||
// get_bytes() appends to result, and CBOR indefinite-length strings
|
||||
// collect all their chunks in the same result; validating only the
|
||||
// newly read bytes keeps the check linear in the input size
|
||||
const std::size_t old_size = result.size();
|
||||
if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// RFC 8949 (CBOR) §3.1 and the BSON/UBJSON specifications require
|
||||
// text strings to be valid UTF-8; reject anything else right here so
|
||||
// malformed input is caught at decode time instead of only surfacing
|
||||
// later as a type_error.316 when the value is dumped (which would
|
||||
// defeat allow_exceptions=false / strict discarding). The MessagePack
|
||||
// specification explicitly allows a str object to contain an invalid
|
||||
// byte sequence and expects deserializers to hand back the original
|
||||
// bytes, so msgpack strings (and map keys, which go through this
|
||||
// function as well) are exempt.
|
||||
if (format != input_format_t::msgpack && JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
parse_error::create(113, chars_read,
|
||||
exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr));
|
||||
}
|
||||
|
||||
return true;
|
||||
// Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the
|
||||
// choice to the decoder), MessagePack (whose spec explicitly allows
|
||||
// a str object to contain an invalid byte sequence), UBJSON, BJData,
|
||||
// or BSON requires a decoder to reject ill-formed UTF-8. The bytes
|
||||
// are kept unchanged; dump() and the binary writers are the ones
|
||||
// that check them and report type_error.316 if they are not valid.
|
||||
return get_bytes(format, len, "string", result);
|
||||
}
|
||||
|
||||
/*!
|
||||
|
||||
@@ -117,13 +117,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
|
||||
written by Björn Hoehrmann. See
|
||||
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
|
||||
|
||||
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four
|
||||
The library checks UTF-8 well-formedness (RFC 3629, section 4) in three
|
||||
places, which differ in speed, diagnostics, and how they read the input:
|
||||
|
||||
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in
|
||||
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR,
|
||||
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
|
||||
text strings at decode time).
|
||||
- decode() below: the serializer, to escape and, in strict mode, reject
|
||||
ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON,
|
||||
UBJSON and BJData readers do not use it: none of those specs requires a
|
||||
decoder to reject ill-formed UTF-8 in text strings, so the readers keep
|
||||
the bytes as is and leave the check to dump() and the binary writers.
|
||||
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
|
||||
diagnostic for each kind of error.
|
||||
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
|
||||
@@ -178,38 +179,5 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std:
|
||||
return state;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether a string consists solely of valid UTF-8
|
||||
|
||||
Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text
|
||||
strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the
|
||||
MessagePack/BSON specifications all require text strings to be UTF-8), so
|
||||
that malformed input is caught immediately instead of only surfacing later
|
||||
as a type_error.316 when the resulting value is dumped.
|
||||
|
||||
@param[in] s the string to check
|
||||
@param[in] first index of the first byte to check; the bytes before it are
|
||||
assumed to have been validated already and to end on a
|
||||
code point boundary
|
||||
@return whether @a s (from index @a first on) is valid UTF-8
|
||||
*/
|
||||
template<typename StringType>
|
||||
inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept
|
||||
{
|
||||
std::uint8_t state = UTF8_ACCEPT;
|
||||
std::uint32_t codepoint = 0;
|
||||
|
||||
for (std::size_t i = first; i < s.size(); ++i)
|
||||
{
|
||||
decode(state, codepoint, static_cast<std::uint8_t>(s[i]));
|
||||
if (state == UTF8_REJECT)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return state == UTF8_ACCEPT;
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
Reference in New Issue
Block a user