mirror of
https://github.com/nlohmann/json.git
synced 2026-10-03 13:10:33 +00:00
Accept ill-formed UTF-8 in all binary readers again
RFC 8949 and the MessagePack/BSON/UBJSON/BJData specs leave UTF-8 well-formedness checking up to the decoder, so following #5529 the binary readers are lenient by default again, as in release 3.12.0 (the reader-side check was added by #5185/#5531, not in any release); reader-side validation becomes opt-in in a follow-up PR. The writers stay strict and throw type_error.316 for ill-formed UTF-8. BON8 is unchanged, since UTF-8 lead bytes are structural there. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -3908,17 +3908,28 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||
CHECK(json::from_bjdata(v) == j);
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 (see #5651)")
|
||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
||||
{
|
||||
// a string value whose bytes are not valid UTF-8 (0xC0 0xAE is an
|
||||
// overlong encoding of '.') is rejected at decode time, matching
|
||||
// every other kind of malformed binary input, and to_bjdata()
|
||||
// rejects it as well, so a value it accepts can always be read
|
||||
// back
|
||||
// none of the binary format specs requires a decoder to reject
|
||||
// ill-formed UTF-8 in a text string, so a value whose bytes are
|
||||
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
|
||||
// round-trips byte for byte as a string value; to_bjdata() is
|
||||
// strict, so such a value cannot be written back
|
||||
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(v), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing BJData string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_bjdata(v, true, false).is_discarded());
|
||||
json j;
|
||||
CHECK_NOTHROW(j = json::from_bjdata(v));
|
||||
REQUIRE(j.is_string());
|
||||
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_bjdata(j), json::type_error&);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_bjdata(v_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK_THROWS_AS(json::to_bjdata(j_key), json::type_error&);
|
||||
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
// a truncated multi-byte sequence
|
||||
|
||||
+13
-6
@@ -154,11 +154,12 @@ TEST_CASE("BSON")
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 (see #5651)")
|
||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
||||
{
|
||||
// a BSON document {"s": "\xC0\xAE"} (0xC0 0xAE is an overlong
|
||||
// encoding of '.'); the reader rejects an ill-formed string value at
|
||||
// decode time
|
||||
// encoding of '.'); the BSON spec does not require a decoder to
|
||||
// reject ill-formed UTF-8 in a string value, so the reader hands the
|
||||
// bytes back unchanged
|
||||
const std::vector<uint8_t> v =
|
||||
{
|
||||
0x0F, 0x00, 0x00, 0x00, // document length
|
||||
@@ -167,9 +168,15 @@ TEST_CASE("BSON")
|
||||
0xc0, 0xae, 0x00, // string content and its null terminator
|
||||
0x00 // document terminator
|
||||
};
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing BSON string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_bson(v, true, false).is_discarded());
|
||||
json j;
|
||||
CHECK_NOTHROW(j = json::from_bson(v));
|
||||
REQUIRE(j.is_object());
|
||||
REQUIRE(j.contains("s"));
|
||||
CHECK(j["s"].get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
// dump() still requires valid UTF-8 and throws for such a value
|
||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
||||
// to_bson() is strict as well, so the value cannot be written back
|
||||
CHECK_THROWS_AS(json::to_bson(j), json::type_error&);
|
||||
|
||||
// to_bson() rejects the same kind of ill-formed string value, before
|
||||
// any bytes reach the output adapter (the BSON document length
|
||||
|
||||
+46
-18
@@ -1801,19 +1801,40 @@ TEST_CASE("CBOR")
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&);
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in string (see #5529)")
|
||||
SECTION("ill-formed UTF-8 in string (see #5529, #5651)")
|
||||
{
|
||||
// RFC 8949 §3.1 leaves it up to the decoder whether to reject
|
||||
// ill-formed UTF-8 in a text string; this library does not, and
|
||||
// hands the original bytes back unchanged, matching the
|
||||
// MessagePack reader and the behavior before #5185/#5531 (not in
|
||||
// any release)
|
||||
|
||||
// a two-character text string (major type 3) whose bytes are not
|
||||
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be
|
||||
// rejected at decode time, matching every other kind of
|
||||
// malformed binary input, rather than only failing later when
|
||||
// the resulting value is dumped
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae}), true, false).is_discarded());
|
||||
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') round-trips
|
||||
// byte for byte as a string value
|
||||
const std::vector<uint8_t> ill_formed_value = {0x62, 0xc0, 0xae};
|
||||
json j_value;
|
||||
CHECK_NOTHROW(j_value = json::from_cbor(ill_formed_value));
|
||||
REQUIRE(j_value.is_string());
|
||||
CHECK(j_value.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
// dump() still requires valid UTF-8 and throws for such a value,
|
||||
// unless an error handler that replaces or ignores the bytes is
|
||||
// passed
|
||||
CHECK_THROWS_AS(j_value.dump(), json::type_error&);
|
||||
// to_cbor() is strict as well, so the value cannot be written back
|
||||
CHECK_THROWS_AS(json::to_cbor(j_value), json::type_error&);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK_THROWS_AS(json::to_cbor(j_key), json::type_error&);
|
||||
|
||||
// a CBOR byte string (major type 2) with the very same bytes is
|
||||
// NOT text and must still be accepted as-is
|
||||
json _;
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x42, 0xc0, 0xae})));
|
||||
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
||||
|
||||
@@ -1841,17 +1862,27 @@ TEST_CASE("CBOR")
|
||||
CHECK_NOTHROW(json::to_cbor(json::binary(std::vector<std::uint8_t>({0xFF}))));
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in indefinite-length string")
|
||||
SECTION("ill-formed UTF-8 in indefinite-length string")
|
||||
{
|
||||
json _;
|
||||
|
||||
// every chunk must be valid UTF-8 on its own (RFC 8949, Section
|
||||
// 3.2.3), so a code point split across two chunks is rejected
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded());
|
||||
// the chunks are concatenated as is, without checking that each
|
||||
// chunk is valid UTF-8 on its own (RFC 8949, Section 3.2.3), so
|
||||
// a code point split across two chunks yields a valid string
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})));
|
||||
CHECK(_ == "\xc3\xa9");
|
||||
CHECK(_.dump() == "\"\xc3\xa9\"");
|
||||
|
||||
// an ill-formed later chunk is rejected after valid ones
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
// a truncated code point is kept as is
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0xff})));
|
||||
CHECK(_ == "\xc3");
|
||||
CHECK_THROWS_AS(_.dump(), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_cbor(_), json::type_error&);
|
||||
|
||||
// an ill-formed later chunk is kept after valid ones
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})));
|
||||
CHECK(_ == "\xc3\xa9\xc0\xae");
|
||||
CHECK_THROWS_AS(_.dump(), json::type_error&);
|
||||
|
||||
// valid multi-byte chunks are accepted
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6");
|
||||
@@ -1859,9 +1890,6 @@ TEST_CASE("CBOR")
|
||||
|
||||
SECTION("many chunks in indefinite-length string")
|
||||
{
|
||||
// only the newly read chunk is validated, not the whole string
|
||||
// collected so far; validating the latter made this input take
|
||||
// quadratic time (about ten seconds for 100000 chunks)
|
||||
constexpr std::size_t chunks = 100000;
|
||||
std::vector<uint8_t> v{0x7f};
|
||||
for (std::size_t i = 0; i < chunks; ++i)
|
||||
|
||||
@@ -2506,17 +2506,28 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||
CHECK(json::from_ubjson(v) == j);
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 (see #5651)")
|
||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
||||
{
|
||||
// a string value whose bytes are not valid UTF-8 (0xC0 0xAE is an
|
||||
// overlong encoding of '.') is rejected at decode time, matching
|
||||
// every other kind of malformed binary input, and to_ubjson()
|
||||
// rejects it as well, so a value it accepts can always be read
|
||||
// back
|
||||
// none of the binary format specs requires a decoder to reject
|
||||
// ill-formed UTF-8 in a text string, so a value whose bytes are
|
||||
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
|
||||
// round-trips byte for byte as a string value; to_ubjson() is
|
||||
// strict, so such a value cannot be written back
|
||||
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(v), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing UBJSON string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_ubjson(v, true, false).is_discarded());
|
||||
json j;
|
||||
CHECK_NOTHROW(j = json::from_ubjson(v));
|
||||
REQUIRE(j.is_string());
|
||||
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_ubjson(j), json::type_error&);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_ubjson(v_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK_THROWS_AS(json::to_ubjson(j_key), json::type_error&);
|
||||
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
// a truncated multi-byte sequence
|
||||
|
||||
Reference in New Issue
Block a user