mirror of
https://github.com/nlohmann/json.git
synced 2026-10-04 13:40:33 +00:00
Follow each binary format's UTF-8 rule: strict writers (CBOR/UBJSON/BJData/BSON), lenient readers (#5741)
Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -46,6 +46,10 @@
|
||||
#define JSON_STRICT_NUL_HANDLING 0
|
||||
#endif
|
||||
|
||||
#ifndef JSON_STRICT_BINARY_UTF8
|
||||
#define JSON_STRICT_BINARY_UTF8 0
|
||||
#endif
|
||||
|
||||
#if JSON_DIAGNOSTICS
|
||||
#define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag
|
||||
#else
|
||||
@@ -82,14 +86,20 @@
|
||||
#define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING
|
||||
#endif
|
||||
|
||||
#if JSON_STRICT_BINARY_UTF8
|
||||
#define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8
|
||||
#else
|
||||
#define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8
|
||||
#endif
|
||||
|
||||
#ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION
|
||||
#define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0
|
||||
#endif
|
||||
|
||||
// Construct the namespace ABI tags component
|
||||
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f
|
||||
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \
|
||||
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f)
|
||||
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g
|
||||
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \
|
||||
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g)
|
||||
|
||||
#define NLOHMANN_JSON_ABI_TAGS \
|
||||
NLOHMANN_JSON_ABI_TAGS_CONCAT( \
|
||||
@@ -98,7 +108,8 @@
|
||||
NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \
|
||||
NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \
|
||||
NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \
|
||||
NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING)
|
||||
NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \
|
||||
NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8)
|
||||
|
||||
// Construct the namespace version component
|
||||
#define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \
|
||||
|
||||
@@ -4044,28 +4044,13 @@ class binary_reader
|
||||
const NumberType len,
|
||||
string_t& result)
|
||||
{
|
||||
// get_bytes() appends to result, and CBOR indefinite-length strings
|
||||
// collect all their chunks in the same result; validating only the
|
||||
// newly read bytes keeps the check linear in the input size
|
||||
const std::size_t old_size = result.size();
|
||||
if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications
|
||||
// all require text strings to be valid UTF-8; reject anything else
|
||||
// right here so malformed input is caught at decode time instead of
|
||||
// only surfacing later as a type_error.316 when the value is dumped
|
||||
// (which would defeat allow_exceptions=false / strict discarding).
|
||||
if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
parse_error::create(113, chars_read,
|
||||
exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr));
|
||||
}
|
||||
|
||||
return true;
|
||||
// Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the
|
||||
// choice to the decoder), MessagePack (whose spec explicitly allows
|
||||
// a str object to contain an invalid byte sequence), UBJSON, BJData,
|
||||
// or BSON requires a decoder to reject ill-formed UTF-8. The bytes
|
||||
// are kept unchanged; dump() and the binary writers are the ones
|
||||
// that check them and report type_error.316 if they are not valid.
|
||||
return get_bytes(format, len, "string", result);
|
||||
}
|
||||
|
||||
/*!
|
||||
|
||||
@@ -42,6 +42,7 @@
|
||||
#undef JSON_BRACE_INIT_COPY_SEMANTICS
|
||||
#undef JSON_PRECISE_STREAM_POSITION
|
||||
#undef JSON_STRICT_NUL_HANDLING
|
||||
#undef JSON_STRICT_BINARY_UTF8
|
||||
#endif
|
||||
|
||||
#include <nlohmann/thirdparty/hedley/hedley_undef.hpp>
|
||||
|
||||
@@ -115,6 +115,8 @@ class binary_writer
|
||||
|
||||
/*!
|
||||
@param[in] j JSON value to serialize
|
||||
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
|
||||
value or an object key is not valid UTF-8
|
||||
@throw type_error.317 if @a j is not an object
|
||||
*/
|
||||
void write_bson(const BasicJsonType& j)
|
||||
@@ -145,6 +147,8 @@ class binary_writer
|
||||
|
||||
/*!
|
||||
@param[in] j JSON value to serialize
|
||||
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
|
||||
value or an object key is not valid UTF-8
|
||||
*/
|
||||
void write_cbor(const BasicJsonType& j)
|
||||
{
|
||||
@@ -211,6 +215,8 @@ class binary_writer
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
check_text_utf8(*j.m_data.m_value.string, j);
|
||||
|
||||
// step 1: write control byte and the string length
|
||||
write_cbor_head(0x60, j.m_data.m_value.string->size());
|
||||
|
||||
@@ -287,6 +293,11 @@ class binary_writer
|
||||
// step 2: write each element
|
||||
for (const auto& el : *j.m_data.m_value.object)
|
||||
{
|
||||
// el.first is checked here, against the object as
|
||||
// diagnostics context, because write_cbor(el.first)
|
||||
// converts it to a temporary basic_json that would be
|
||||
// used as the context instead
|
||||
check_text_utf8(el.first, j);
|
||||
write_cbor(el.first);
|
||||
write_cbor(el.second);
|
||||
}
|
||||
@@ -629,6 +640,8 @@ class binary_writer
|
||||
@param[in] add_prefix whether prefixes need to be used for this value
|
||||
@param[in] use_bjdata whether write in BJData format, default is false
|
||||
@param[in] bjdata_version which BJData version to use, default is draft2
|
||||
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
|
||||
value or an object key is not valid UTF-8
|
||||
*/
|
||||
void write_ubjson(const BasicJsonType& j, const bool use_count,
|
||||
const bool use_type, const bool add_prefix = true,
|
||||
@@ -678,6 +691,8 @@ class binary_writer
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
check_text_utf8(*j.m_data.m_value.string, j);
|
||||
|
||||
if (add_prefix)
|
||||
{
|
||||
oa.write_character(to_char_type('S'));
|
||||
@@ -840,6 +855,7 @@ class binary_writer
|
||||
|
||||
for (const auto& el : *j.m_data.m_value.object)
|
||||
{
|
||||
check_text_utf8(el.first, j);
|
||||
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(el.first.data()),
|
||||
@@ -884,6 +900,10 @@ class binary_writer
|
||||
/*!
|
||||
@return The size of a BSON document entry header, including the id marker
|
||||
and the entry name size (and its null-terminator).
|
||||
@throw out_of_range.409 if @a name contains U+0000, before anything is
|
||||
written
|
||||
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a name is
|
||||
not valid UTF-8, before anything is written
|
||||
*/
|
||||
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
|
||||
{
|
||||
@@ -893,7 +913,8 @@ class binary_writer
|
||||
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
|
||||
}
|
||||
|
||||
static_cast<void>(j);
|
||||
check_text_utf8(name, j);
|
||||
|
||||
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
|
||||
}
|
||||
|
||||
@@ -949,9 +970,21 @@ class binary_writer
|
||||
|
||||
/*!
|
||||
@return The size of the BSON-encoded string in @a value
|
||||
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a value
|
||||
is not valid UTF-8, before anything is written
|
||||
|
||||
@note The UTF-8 check is skipped if @a value is already too long for the
|
||||
32-bit BSON length field (@ref to_bson_length rejects it later, once
|
||||
the size of the whole document is known); this also keeps the check
|
||||
from reading past a StringType that reports a size larger than what
|
||||
it actually holds.
|
||||
*/
|
||||
static std::size_t calc_bson_string_size(const string_t& value)
|
||||
static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<std::int32_t>(value.size())))
|
||||
{
|
||||
check_text_utf8(value, j);
|
||||
}
|
||||
return sizeof(std::int32_t) + value.size() + 1ul;
|
||||
}
|
||||
|
||||
@@ -1080,6 +1113,8 @@ class binary_writer
|
||||
is neither an object nor an array
|
||||
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
|
||||
into a byte, before anything is written
|
||||
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a j is a
|
||||
string that is not valid UTF-8, before anything is written
|
||||
*/
|
||||
static std::size_t calc_bson_value_size(const BasicJsonType& j)
|
||||
{
|
||||
@@ -1101,7 +1136,7 @@ class binary_writer
|
||||
return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned);
|
||||
|
||||
case value_t::string:
|
||||
return calc_bson_string_size(*j.m_data.m_value.string);
|
||||
return calc_bson_string_size(*j.m_data.m_value.string, j);
|
||||
|
||||
case value_t::null:
|
||||
return 0ul;
|
||||
@@ -1214,6 +1249,8 @@ class binary_writer
|
||||
written
|
||||
@throw out_of_range.415 if a binary value's subtype does not fit into a
|
||||
byte, before anything is written
|
||||
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
|
||||
value or a key is not valid UTF-8, before anything is written
|
||||
*/
|
||||
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
|
||||
{
|
||||
@@ -2092,7 +2129,7 @@ class binary_writer
|
||||
*/
|
||||
void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context)
|
||||
{
|
||||
check_bon8_utf8(s, context);
|
||||
check_utf8(s, context);
|
||||
|
||||
// a string that follows another string terminates it
|
||||
if (string_open)
|
||||
@@ -2122,7 +2159,7 @@ class binary_writer
|
||||
@throw type_error.316 if @a s is not valid UTF-8; the message names the
|
||||
first byte of the first invalid or incomplete sequence
|
||||
*/
|
||||
static void check_bon8_utf8(const string_t& s, const BasicJsonType& context)
|
||||
static void check_utf8(const string_t& s, const BasicJsonType& context)
|
||||
{
|
||||
static_cast<void>(context); // only used when exceptions are enabled
|
||||
const auto* data = reinterpret_cast<const unsigned char*>(s.data());
|
||||
@@ -2133,6 +2170,29 @@ class binary_writer
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check a CBOR, UBJSON, BJData, or BSON text string for valid UTF-8
|
||||
|
||||
The check only happens if JSON_STRICT_BINARY_UTF8 is enabled. Otherwise,
|
||||
the bytes are written unchanged, as before version 3.13.0. MessagePack
|
||||
always writes the bytes as is, and BON8 always checks them (see
|
||||
@ref check_utf8).
|
||||
|
||||
@param[in] s the string to check
|
||||
@param[in] context the value that holds @a s (for diagnostics)
|
||||
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a s is
|
||||
not valid UTF-8
|
||||
*/
|
||||
static void check_text_utf8(const string_t& s, const BasicJsonType& context)
|
||||
{
|
||||
#if JSON_STRICT_BINARY_UTF8
|
||||
check_utf8(s, context);
|
||||
#else
|
||||
static_cast<void>(s);
|
||||
static_cast<void>(context);
|
||||
#endif
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief write an integer in the shortest encoding
|
||||
|
||||
|
||||
@@ -117,13 +117,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
|
||||
written by Björn Hoehrmann. See
|
||||
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
|
||||
|
||||
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four
|
||||
The library checks UTF-8 well-formedness (RFC 3629, section 4) in three
|
||||
places, which differ in speed, diagnostics, and how they read the input:
|
||||
|
||||
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in
|
||||
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR,
|
||||
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
|
||||
text strings at decode time).
|
||||
- decode() below: the serializer, to escape and, in strict mode, reject
|
||||
ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON,
|
||||
UBJSON and BJData readers do not use it: none of those specs requires a
|
||||
decoder to reject ill-formed UTF-8 in text strings, so the readers keep
|
||||
the bytes as is and leave the check to dump() and the binary writers.
|
||||
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
|
||||
diagnostic for each kind of error.
|
||||
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
|
||||
@@ -178,38 +179,5 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std:
|
||||
return state;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether a string consists solely of valid UTF-8
|
||||
|
||||
Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text
|
||||
strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the
|
||||
MessagePack/BSON specifications all require text strings to be UTF-8), so
|
||||
that malformed input is caught immediately instead of only surfacing later
|
||||
as a type_error.316 when the resulting value is dumped.
|
||||
|
||||
@param[in] s the string to check
|
||||
@param[in] first index of the first byte to check; the bytes before it are
|
||||
assumed to have been validated already and to end on a
|
||||
code point boundary
|
||||
@return whether @a s (from index @a first on) is valid UTF-8
|
||||
*/
|
||||
template<typename StringType>
|
||||
inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept
|
||||
{
|
||||
std::uint8_t state = UTF8_ACCEPT;
|
||||
std::uint32_t codepoint = 0;
|
||||
|
||||
for (std::size_t i = first; i < s.size(); ++i)
|
||||
{
|
||||
decode(state, codepoint, static_cast<std::uint8_t>(s[i]));
|
||||
if (state == UTF8_REJECT)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return state == UTF8_ACCEPT;
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
Reference in New Issue
Block a user