Keep writing ill-formed UTF-8 unchanged unless JSON_STRICT_BINARY_UTF8 is set

Throwing type_error.316 from to_cbor(), to_ubjson(), to_bjdata(), and
to_bson() by default would break code that works with 3.12.0, for example
code that stores ISO 8859-1 in a string and only writes it to a binary
format. Following the plan for 4.0.0, the check is now opt-in via the new
macro JSON_STRICT_BINARY_UTF8 (default 0), which is planned to become the
default in 4.0.0:

- abi_macros.hpp: default 0 and ABI tag _sbu8; macro_unscope.hpp undefines
  it; CMake option JSON_StrictBinaryUTF8; ABI namespace tests updated.
- binary_writer: the four writers call check_text_utf8(), which checks
  only if the macro is enabled. BON8 always checks, MessagePack never.
- Tests: the strict checks move to unit-binary_utf8_strict.cpp, which
  defines the macro; the per-format tests now check that the default
  writes the bytes unchanged (from_X(to_X(j)) == j).
- Docs: new macro page, macro lists, CMake option, namespace tags, and the
  writer/format pages describe the default and the opt-in.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-10-02 08:45:32 +02:00
parent 92c729f657
commit 809ace26b3
29 changed files with 477 additions and 140 deletions
+15 -4
View File
@@ -46,6 +46,10 @@
#define JSON_STRICT_NUL_HANDLING 0
#endif
#ifndef JSON_STRICT_BINARY_UTF8
#define JSON_STRICT_BINARY_UTF8 0
#endif
#if JSON_DIAGNOSTICS
#define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag
#else
@@ -82,14 +86,20 @@
#define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING
#endif
#if JSON_STRICT_BINARY_UTF8
#define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8
#else
#define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8
#endif
#ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION
#define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0
#endif
// Construct the namespace ABI tags component
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f)
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g)
#define NLOHMANN_JSON_ABI_TAGS \
NLOHMANN_JSON_ABI_TAGS_CONCAT( \
@@ -98,7 +108,8 @@
NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \
NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \
NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \
NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING)
NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \
NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8)
// Construct the namespace version component
#define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \
@@ -41,6 +41,7 @@
#undef JSON_BRACE_INIT_COPY_SEMANTICS
#undef JSON_PRECISE_STREAM_POSITION
#undef JSON_STRICT_NUL_HANDLING
#undef JSON_STRICT_BINARY_UTF8
#endif
#include <nlohmann/thirdparty/hedley/hedley_undef.hpp>
@@ -115,8 +115,8 @@ class binary_writer
/*!
@param[in] j JSON value to serialize
@throw type_error.316 if a string value or an object key is not valid
UTF-8
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
value or an object key is not valid UTF-8
@throw type_error.317 if @a j is not an object
*/
void write_bson(const BasicJsonType& j)
@@ -147,8 +147,8 @@ class binary_writer
/*!
@param[in] j JSON value to serialize
@throw type_error.316 if a string value or an object key is not valid
UTF-8
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
value or an object key is not valid UTF-8
*/
void write_cbor(const BasicJsonType& j)
{
@@ -215,7 +215,7 @@ class binary_writer
case value_t::string:
{
check_utf8(*j.m_data.m_value.string, j);
check_text_utf8(*j.m_data.m_value.string, j);
// step 1: write control byte and the string length
write_cbor_head(0x60, j.m_data.m_value.string->size());
@@ -297,7 +297,7 @@ class binary_writer
// diagnostics context, because write_cbor(el.first)
// converts it to a temporary basic_json that would be
// used as the context instead
check_utf8(el.first, j);
check_text_utf8(el.first, j);
write_cbor(el.first);
write_cbor(el.second);
}
@@ -640,8 +640,8 @@ class binary_writer
@param[in] add_prefix whether prefixes need to be used for this value
@param[in] use_bjdata whether write in BJData format, default is false
@param[in] bjdata_version which BJData version to use, default is draft2
@throw type_error.316 if a string value or an object key is not valid
UTF-8
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
value or an object key is not valid UTF-8
*/
void write_ubjson(const BasicJsonType& j, const bool use_count,
const bool use_type, const bool add_prefix = true,
@@ -691,7 +691,7 @@ class binary_writer
case value_t::string:
{
check_utf8(*j.m_data.m_value.string, j);
check_text_utf8(*j.m_data.m_value.string, j);
if (add_prefix)
{
@@ -855,7 +855,7 @@ class binary_writer
for (const auto& el : *j.m_data.m_value.object)
{
check_utf8(el.first, j);
check_text_utf8(el.first, j);
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
oa.write_characters(
reinterpret_cast<const CharType*>(el.first.data()),
@@ -902,8 +902,8 @@ class binary_writer
and the entry name size (and its null-terminator).
@throw out_of_range.409 if @a name contains U+0000, before anything is
written
@throw type_error.316 if @a name is not valid UTF-8, before anything is
written
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a name is
not valid UTF-8, before anything is written
*/
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
{
@@ -913,7 +913,7 @@ class binary_writer
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
}
check_utf8(name, j);
check_text_utf8(name, j);
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
}
@@ -970,8 +970,8 @@ class binary_writer
/*!
@return The size of the BSON-encoded string in @a value
@throw type_error.316 if @a value is not valid UTF-8, before anything is
written
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a value
is not valid UTF-8, before anything is written
@note The UTF-8 check is skipped if @a value is already too long for the
32-bit BSON length field (@ref to_bson_length rejects it later, once
@@ -983,7 +983,7 @@ class binary_writer
{
if (JSON_HEDLEY_LIKELY(value_in_range_of<std::int32_t>(value.size())))
{
check_utf8(value, j);
check_text_utf8(value, j);
}
return sizeof(std::int32_t) + value.size() + 1ul;
}
@@ -1113,8 +1113,8 @@ class binary_writer
is neither an object nor an array
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
into a byte, before anything is written
@throw type_error.316 if @a j is a string that is not valid UTF-8, before
anything is written
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a j is a
string that is not valid UTF-8, before anything is written
*/
static std::size_t calc_bson_value_size(const BasicJsonType& j)
{
@@ -1249,8 +1249,8 @@ class binary_writer
written
@throw out_of_range.415 if a binary value's subtype does not fit into a
byte, before anything is written
@throw type_error.316 if a string value or a key is not valid UTF-8,
before anything is written
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string
value or a key is not valid UTF-8, before anything is written
*/
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
{
@@ -2170,6 +2170,29 @@ class binary_writer
}
}
/*!
@brief check a CBOR, UBJSON, BJData, or BSON text string for valid UTF-8
The check only happens if JSON_STRICT_BINARY_UTF8 is enabled. Otherwise,
the bytes are written unchanged, as before version 3.13.0. MessagePack
always writes the bytes as is, and BON8 always checks them (see
@ref check_utf8).
@param[in] s the string to check
@param[in] context the value that holds @a s (for diagnostics)
@throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a s is
not valid UTF-8
*/
static void check_text_utf8(const string_t& s, const BasicJsonType& context)
{
#if JSON_STRICT_BINARY_UTF8
check_utf8(s, context);
#else
static_cast<void>(s);
static_cast<void>(context);
#endif
}
/*!
@brief write an integer in the shortest encoding