Keep writing ill-formed UTF-8 unchanged unless JSON_STRICT_BINARY_UTF8 is set

Throwing type_error.316 from to_cbor(), to_ubjson(), to_bjdata(), and
to_bson() by default would break code that works with 3.12.0, for example
code that stores ISO 8859-1 in a string and only writes it to a binary
format. Following the plan for 4.0.0, the check is now opt-in via the new
macro JSON_STRICT_BINARY_UTF8 (default 0), which is planned to become the
default in 4.0.0:

- abi_macros.hpp: default 0 and ABI tag _sbu8; macro_unscope.hpp undefines
  it; CMake option JSON_StrictBinaryUTF8; ABI namespace tests updated.
- binary_writer: the four writers call check_text_utf8(), which checks
  only if the macro is enabled. BON8 always checks, MessagePack never.
- Tests: the strict checks move to unit-binary_utf8_strict.cpp, which
  defines the macro; the per-format tests now check that the default
  writes the bytes unchanged (from_X(to_X(j)) == j).
- Docs: new macro page, macro lists, CMake option, namespace tags, and the
  writer/format pages describe the default and the opt-in.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-10-02 08:45:32 +02:00
parent 92c729f657
commit 809ace26b3
29 changed files with 477 additions and 140 deletions
+4
View File
@@ -44,6 +44,10 @@ TEST_CASE("default namespace")
expected += "_snul";
#endif
#if JSON_STRICT_BINARY_UTF8
expected += "_sbu8";
#endif
expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR);
expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR);
expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json";
+4
View File
@@ -45,6 +45,10 @@ TEST_CASE("default namespace without version component")
expected += "_snul";
#endif
#if JSON_STRICT_BINARY_UTF8
expected += "_sbu8";
#endif
expected += "::basic_json";
// fallback for Clang
+110
View File
@@ -0,0 +1,110 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
// The binary writers check strings and object keys for valid UTF-8 only if
// JSON_STRICT_BINARY_UTF8 is enabled (planned to be the default in 4.0.0).
// Without it, they write the bytes unchanged, as before version 3.13.0; the
// tests for that are next to the other tests of each format.
#ifdef JSON_STRICT_BINARY_UTF8
#undef JSON_STRICT_BINARY_UTF8
#endif
#define JSON_STRICT_BINARY_UTF8 1
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <cstdint>
#include <vector>
TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)")
{
SECTION("CBOR")
{
// a string value with ill-formed UTF-8 is rejected
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
// an object key with ill-formed UTF-8 is rejected the same way
CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// binary values are not text and are unaffected
CHECK_NOTHROW(json::to_cbor(json::binary(std::vector<std::uint8_t>({0xFF}))));
// a value read back from CBOR with ill-formed bytes cannot be written
// back either (the reader is lenient regardless of the macro)
const json j = json::from_cbor(std::vector<std::uint8_t>({0x62, 0xc0, 0xae}));
CHECK_THROWS_WITH_AS(json::to_cbor(j), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
}
SECTION("UBJSON")
{
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
// an object key with ill-formed UTF-8 is rejected the same way
CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
}
SECTION("BJData")
{
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
// an object key with ill-formed UTF-8 is rejected the same way
CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
}
SECTION("BSON")
{
// to_bson() rejects the same kind of ill-formed string value, before
// any bytes reach the output adapter (the BSON document length
// prefix must be known up front, so nothing is written incrementally)
std::vector<std::uint8_t> out{0x42}; // a sentinel byte the writer must not touch
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter<std::uint8_t>(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
CHECK(out == std::vector<std::uint8_t> {0x42});
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
// an object key with ill-formed UTF-8 is rejected as well; unlike
// the reader (which never validates element names), the writer
// checks both string values and object keys
CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
}
SECTION("MessagePack and BON8 are unaffected")
{
// MessagePack allows any bytes in a str, so to_msgpack() writes them as
// is; BON8 always checks, because the lead bytes mark where strings end
CHECK(json::to_msgpack(json("\xFF")) == std::vector<std::uint8_t>({0xa1, 0xff}));
CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&);
}
}
+12 -10
View File
@@ -3913,15 +3913,17 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
// none of the binary format specs requires a decoder to reject
// ill-formed UTF-8 in a text string, so a value whose bytes are
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
// round-trips byte for byte as a string value; to_bjdata() is
// strict, so such a value cannot be written back
// round-trips byte for byte as a string value; to_bjdata() writes
// the bytes unchanged, as before 3.13.0, unless
// JSON_STRICT_BINARY_UTF8 is enabled (see
// unit-binary_utf8_strict.cpp)
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
json j;
CHECK_NOTHROW(j = json::from_bjdata(v));
REQUIRE(j.is_string());
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
CHECK_THROWS_AS(j.dump(), json::type_error&);
CHECK_THROWS_AS(json::to_bjdata(j), json::type_error&);
CHECK(json::from_bjdata(json::to_bjdata(j)) == j);
// the same bytes as an object key round-trip as well
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
@@ -3929,18 +3931,18 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
CHECK_NOTHROW(j_key = json::from_bjdata(v_key));
REQUIRE(j_key.is_object());
CHECK(j_key.contains(std::string("\xc0\xae")));
CHECK_THROWS_AS(json::to_bjdata(j_key), json::type_error&);
CHECK(json::from_bjdata(json::to_bjdata(j_key)) == j_key);
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
CHECK(json::from_bjdata(json::to_bjdata(json("\xFF"))) == json("\xFF"));
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
CHECK(json::from_bjdata(json::to_bjdata(json("\xC3"))) == json("\xC3"));
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
CHECK(json::from_bjdata(json::to_bjdata(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
CHECK(json::from_bjdata(json::to_bjdata(json("\xC0\xAF"))) == json("\xC0\xAF"));
// an object key with ill-formed UTF-8 is rejected the same way
CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// an object key with ill-formed UTF-8 is kept the same way
CHECK(json::from_bjdata(json::to_bjdata(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
}
}
+9 -17
View File
@@ -175,28 +175,20 @@ TEST_CASE("BSON")
CHECK(j["s"].get_ref<const json::string_t&>() == std::string("\xc0\xae"));
// dump() still requires valid UTF-8 and throws for such a value
CHECK_THROWS_AS(j.dump(), json::type_error&);
// to_bson() is strict as well, so the value cannot be written back
CHECK_THROWS_AS(json::to_bson(j), json::type_error&);
// to_bson() writes the bytes back unchanged, as before 3.13.0,
// unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp)
CHECK(json::from_bson(json::to_bson(j)) == j);
// to_bson() rejects the same kind of ill-formed string value, before
// any bytes reach the output adapter (the BSON document length
// prefix must be known up front, so nothing is written incrementally)
std::vector<std::uint8_t> out{0x42}; // a sentinel byte the writer must not touch
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter<std::uint8_t>(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
CHECK(out == std::vector<std::uint8_t> {0x42});
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}})) == json{{"s", "\xFF"}});
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
CHECK(json::from_bson(json::to_bson(json{{"s", "\xC3"}})) == json{{"s", "\xC3"}});
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
CHECK(json::from_bson(json::to_bson(json{{"s", "\xED\xA0\x80"}})) == json{{"s", "\xED\xA0\x80"}});
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
CHECK(json::from_bson(json::to_bson(json{{"s", "\xC0\xAF"}})) == json{{"s", "\xC0\xAF"}});
// an object key with ill-formed UTF-8 is rejected as well; unlike
// the reader (which never validates element names), the writer
// checks both string values and object keys
CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// an object key with ill-formed UTF-8 is kept as well
CHECK(json::from_bson(json::to_bson(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
}
SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON")
+15 -13
View File
@@ -1821,8 +1821,9 @@ TEST_CASE("CBOR")
// unless an error handler that replaces or ignores the bytes is
// passed
CHECK_THROWS_AS(j_value.dump(), json::type_error&);
// to_cbor() is strict as well, so the value cannot be written back
CHECK_THROWS_AS(json::to_cbor(j_value), json::type_error&);
// to_cbor() writes the bytes back unchanged, as before 3.13.0,
// unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp)
CHECK(json::from_cbor(json::to_cbor(j_value)) == j_value);
// the same bytes as an object key round-trip as well
const std::vector<uint8_t> ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01};
@@ -1830,7 +1831,7 @@ TEST_CASE("CBOR")
CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key));
REQUIRE(j_key.is_object());
CHECK(j_key.contains(std::string("\xc0\xae")));
CHECK_THROWS_AS(json::to_cbor(j_key), json::type_error&);
CHECK(json::from_cbor(json::to_cbor(j_key)) == j_key);
// a CBOR byte string (major type 2) with the very same bytes is
// NOT text and must still be accepted as-is
@@ -1843,20 +1844,21 @@ TEST_CASE("CBOR")
CHECK(json::from_cbor(json::to_cbor(j)) == j);
}
SECTION("to_cbor rejects ill-formed UTF-8 (see #5651)")
SECTION("to_cbor keeps ill-formed UTF-8 (see #5651)")
{
// to_cbor() must reject the same ill-formed strings from_cbor()
// rejects, so a value it accepts can always be read back
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// to_cbor() writes the bytes unchanged, as before 3.13.0, unless
// JSON_STRICT_BINARY_UTF8 is enabled (see
// unit-binary_utf8_strict.cpp); from_cbor() reads them back as is
CHECK(json::from_cbor(json::to_cbor(json("\xFF"))) == json("\xFF"));
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
CHECK(json::from_cbor(json::to_cbor(json("\xC3"))) == json("\xC3"));
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
CHECK(json::from_cbor(json::to_cbor(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
CHECK(json::from_cbor(json::to_cbor(json("\xC0\xAF"))) == json("\xC0\xAF"));
// an object key with ill-formed UTF-8 is rejected the same way
CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// an object key with ill-formed UTF-8 is kept the same way
CHECK(json::from_cbor(json::to_cbor(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
// binary values are not text and are unaffected
CHECK_NOTHROW(json::to_cbor(json::binary(std::vector<std::uint8_t>({0xFF}))));
@@ -1877,7 +1879,7 @@ TEST_CASE("CBOR")
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0xff})));
CHECK(_ == "\xc3");
CHECK_THROWS_AS(_.dump(), json::type_error&);
CHECK_THROWS_AS(json::to_cbor(_), json::type_error&);
CHECK(json::from_cbor(json::to_cbor(_)) == _);
// an ill-formed later chunk is kept after valid ones
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})));
+12 -10
View File
@@ -2511,15 +2511,17 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
// none of the binary format specs requires a decoder to reject
// ill-formed UTF-8 in a text string, so a value whose bytes are
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
// round-trips byte for byte as a string value; to_ubjson() is
// strict, so such a value cannot be written back
// round-trips byte for byte as a string value; to_ubjson() writes
// the bytes unchanged, as before 3.13.0, unless
// JSON_STRICT_BINARY_UTF8 is enabled (see
// unit-binary_utf8_strict.cpp)
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
json j;
CHECK_NOTHROW(j = json::from_ubjson(v));
REQUIRE(j.is_string());
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
CHECK_THROWS_AS(j.dump(), json::type_error&);
CHECK_THROWS_AS(json::to_ubjson(j), json::type_error&);
CHECK(json::from_ubjson(json::to_ubjson(j)) == j);
// the same bytes as an object key round-trip as well
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
@@ -2527,18 +2529,18 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
CHECK_NOTHROW(j_key = json::from_ubjson(v_key));
REQUIRE(j_key.is_object());
CHECK(j_key.contains(std::string("\xc0\xae")));
CHECK_THROWS_AS(json::to_ubjson(j_key), json::type_error&);
CHECK(json::from_ubjson(json::to_ubjson(j_key)) == j_key);
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
CHECK(json::from_ubjson(json::to_ubjson(json("\xFF"))) == json("\xFF"));
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
CHECK(json::from_ubjson(json::to_ubjson(json("\xC3"))) == json("\xC3"));
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
CHECK(json::from_ubjson(json::to_ubjson(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
CHECK(json::from_ubjson(json::to_ubjson(json("\xC0\xAF"))) == json("\xC0\xAF"));
// an object key with ill-formed UTF-8 is rejected the same way
CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// an object key with ill-formed UTF-8 is kept the same way
CHECK(json::from_ubjson(json::to_ubjson(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
}
}