mirror of
https://github.com/nlohmann/json.git
synced 2026-10-04 21:50:33 +00:00
Follow each binary format's UTF-8 rule: strict writers (CBOR/UBJSON/BJData/BSON), lenient readers (#5741)
Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -44,6 +44,10 @@ TEST_CASE("default namespace")
|
||||
expected += "_snul";
|
||||
#endif
|
||||
|
||||
#if JSON_STRICT_BINARY_UTF8
|
||||
expected += "_sbu8";
|
||||
#endif
|
||||
|
||||
expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR);
|
||||
expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR);
|
||||
expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json";
|
||||
|
||||
@@ -45,6 +45,10 @@ TEST_CASE("default namespace without version component")
|
||||
expected += "_snul";
|
||||
#endif
|
||||
|
||||
#if JSON_STRICT_BINARY_UTF8
|
||||
expected += "_sbu8";
|
||||
#endif
|
||||
|
||||
expected += "::basic_json";
|
||||
|
||||
// fallback for Clang
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
// The binary writers check strings and object keys for valid UTF-8 only if
|
||||
// JSON_STRICT_BINARY_UTF8 is enabled (planned to be the default in 4.0.0).
|
||||
// Without it, they write the bytes unchanged, as before version 3.13.0; the
|
||||
// tests for that are next to the other tests of each format.
|
||||
#ifdef JSON_STRICT_BINARY_UTF8
|
||||
#undef JSON_STRICT_BINARY_UTF8
|
||||
#endif
|
||||
|
||||
#define JSON_STRICT_BINARY_UTF8 1
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)")
|
||||
{
|
||||
SECTION("CBOR")
|
||||
{
|
||||
// a string value with ill-formed UTF-8 is rejected
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
// a truncated multi-byte sequence
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
||||
// an overlong encoding of '.'
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
||||
|
||||
// an object key with ill-formed UTF-8 is rejected the same way
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
|
||||
// binary values are not text and are unaffected
|
||||
CHECK_NOTHROW(json::to_cbor(json::binary(std::vector<std::uint8_t>({0xFF}))));
|
||||
|
||||
// a value read back from CBOR with ill-formed bytes cannot be written
|
||||
// back either (the reader is lenient regardless of the macro)
|
||||
const json j = json::from_cbor(std::vector<std::uint8_t>({0x62, 0xc0, 0xae}));
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(j), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
||||
}
|
||||
|
||||
SECTION("UBJSON")
|
||||
{
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
// a truncated multi-byte sequence
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
||||
// an overlong encoding of '.'
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
||||
|
||||
// an object key with ill-formed UTF-8 is rejected the same way
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
}
|
||||
|
||||
SECTION("BJData")
|
||||
{
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
// a truncated multi-byte sequence
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
||||
// an overlong encoding of '.'
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
||||
|
||||
// an object key with ill-formed UTF-8 is rejected the same way
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
}
|
||||
|
||||
SECTION("BSON")
|
||||
{
|
||||
// to_bson() rejects the same kind of ill-formed string value, before
|
||||
// any bytes reach the output adapter (the BSON document length
|
||||
// prefix must be known up front, so nothing is written incrementally)
|
||||
std::vector<std::uint8_t> out{0x42}; // a sentinel byte the writer must not touch
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter<std::uint8_t>(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
CHECK(out == std::vector<std::uint8_t> {0x42});
|
||||
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
// a truncated multi-byte sequence
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
||||
// an overlong encoding of '.'
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
||||
|
||||
// an object key with ill-formed UTF-8 is rejected as well; unlike
|
||||
// the reader (which never validates element names), the writer
|
||||
// checks both string values and object keys
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
}
|
||||
|
||||
SECTION("MessagePack and BON8 are unaffected")
|
||||
{
|
||||
// MessagePack allows any bytes in a str, so to_msgpack() writes them as
|
||||
// is; BON8 always checks, because the lead bytes mark where strings end
|
||||
CHECK(json::to_msgpack(json("\xFF")) == std::vector<std::uint8_t>({0xa1, 0xff}));
|
||||
CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&);
|
||||
}
|
||||
}
|
||||
@@ -3906,6 +3906,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||
CHECK(json::to_bjdata(j) == v);
|
||||
CHECK(json::from_bjdata(v) == j);
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
||||
{
|
||||
// none of the binary format specs requires a decoder to reject
|
||||
// ill-formed UTF-8 in a text string, so a value whose bytes are
|
||||
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
|
||||
// round-trips byte for byte as a string value; to_bjdata() writes
|
||||
// the bytes unchanged, as before 3.13.0, unless
|
||||
// JSON_STRICT_BINARY_UTF8 is enabled (see
|
||||
// unit-binary_utf8_strict.cpp)
|
||||
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
|
||||
json j;
|
||||
CHECK_NOTHROW(j = json::from_bjdata(v));
|
||||
REQUIRE(j.is_string());
|
||||
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
||||
CHECK(json::from_bjdata(json::to_bjdata(j)) == j);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_bjdata(v_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK(json::from_bjdata(json::to_bjdata(j_key)) == j_key);
|
||||
|
||||
CHECK(json::from_bjdata(json::to_bjdata(json("\xFF"))) == json("\xFF"));
|
||||
// a truncated multi-byte sequence
|
||||
CHECK(json::from_bjdata(json::to_bjdata(json("\xC3"))) == json("\xC3"));
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK(json::from_bjdata(json::to_bjdata(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
|
||||
// an overlong encoding of '.'
|
||||
CHECK(json::from_bjdata(json::to_bjdata(json("\xC0\xAF"))) == json("\xC0\xAF"));
|
||||
|
||||
// an object key with ill-formed UTF-8 is kept the same way
|
||||
CHECK(json::from_bjdata(json::to_bjdata(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("Array Type")
|
||||
|
||||
@@ -154,6 +154,43 @@ TEST_CASE("BSON")
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
||||
{
|
||||
// a BSON document {"s": "\xC0\xAE"} (0xC0 0xAE is an overlong
|
||||
// encoding of '.'); the BSON spec does not require a decoder to
|
||||
// reject ill-formed UTF-8 in a string value, so the reader hands the
|
||||
// bytes back unchanged
|
||||
const std::vector<uint8_t> v =
|
||||
{
|
||||
0x0F, 0x00, 0x00, 0x00, // document length
|
||||
0x02, 's', 0x00, // type 0x02 (string), key "s"
|
||||
0x03, 0x00, 0x00, 0x00, // string length (including null)
|
||||
0xc0, 0xae, 0x00, // string content and its null terminator
|
||||
0x00 // document terminator
|
||||
};
|
||||
json j;
|
||||
CHECK_NOTHROW(j = json::from_bson(v));
|
||||
REQUIRE(j.is_object());
|
||||
REQUIRE(j.contains("s"));
|
||||
CHECK(j["s"].get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
// dump() still requires valid UTF-8 and throws for such a value
|
||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
||||
// to_bson() writes the bytes back unchanged, as before 3.13.0,
|
||||
// unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp)
|
||||
CHECK(json::from_bson(json::to_bson(j)) == j);
|
||||
|
||||
CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}})) == json{{"s", "\xFF"}});
|
||||
// a truncated multi-byte sequence
|
||||
CHECK(json::from_bson(json::to_bson(json{{"s", "\xC3"}})) == json{{"s", "\xC3"}});
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK(json::from_bson(json::to_bson(json{{"s", "\xED\xA0\x80"}})) == json{{"s", "\xED\xA0\x80"}});
|
||||
// an overlong encoding of '.'
|
||||
CHECK(json::from_bson(json::to_bson(json{{"s", "\xC0\xAF"}})) == json{{"s", "\xC0\xAF"}});
|
||||
|
||||
// an object key with ill-formed UTF-8 is kept as well
|
||||
CHECK(json::from_bson(json::to_bson(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
|
||||
}
|
||||
|
||||
SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON")
|
||||
{
|
||||
// out_of_range.412 is thrown from a single shared helper
|
||||
|
||||
+67
-18
@@ -1801,19 +1801,41 @@ TEST_CASE("CBOR")
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&);
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in string (see #5529)")
|
||||
SECTION("ill-formed UTF-8 in string (see #5529, #5651)")
|
||||
{
|
||||
// RFC 8949 §3.1 leaves it up to the decoder whether to reject
|
||||
// ill-formed UTF-8 in a text string; this library does not, and
|
||||
// hands the original bytes back unchanged, matching the
|
||||
// MessagePack reader and the behavior before #5185/#5531 (not in
|
||||
// any release)
|
||||
|
||||
// a two-character text string (major type 3) whose bytes are not
|
||||
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be
|
||||
// rejected at decode time, matching every other kind of
|
||||
// malformed binary input, rather than only failing later when
|
||||
// the resulting value is dumped
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae}), true, false).is_discarded());
|
||||
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') round-trips
|
||||
// byte for byte as a string value
|
||||
const std::vector<uint8_t> ill_formed_value = {0x62, 0xc0, 0xae};
|
||||
json j_value;
|
||||
CHECK_NOTHROW(j_value = json::from_cbor(ill_formed_value));
|
||||
REQUIRE(j_value.is_string());
|
||||
CHECK(j_value.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
// dump() still requires valid UTF-8 and throws for such a value,
|
||||
// unless an error handler that replaces or ignores the bytes is
|
||||
// passed
|
||||
CHECK_THROWS_AS(j_value.dump(), json::type_error&);
|
||||
// to_cbor() writes the bytes back unchanged, as before 3.13.0,
|
||||
// unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp)
|
||||
CHECK(json::from_cbor(json::to_cbor(j_value)) == j_value);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK(json::from_cbor(json::to_cbor(j_key)) == j_key);
|
||||
|
||||
// a CBOR byte string (major type 2) with the very same bytes is
|
||||
// NOT text and must still be accepted as-is
|
||||
json _;
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x42, 0xc0, 0xae})));
|
||||
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
||||
|
||||
@@ -1822,17 +1844,47 @@ TEST_CASE("CBOR")
|
||||
CHECK(json::from_cbor(json::to_cbor(j)) == j);
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in indefinite-length string")
|
||||
SECTION("to_cbor keeps ill-formed UTF-8 (see #5651)")
|
||||
{
|
||||
// to_cbor() writes the bytes unchanged, as before 3.13.0, unless
|
||||
// JSON_STRICT_BINARY_UTF8 is enabled (see
|
||||
// unit-binary_utf8_strict.cpp); from_cbor() reads them back as is
|
||||
CHECK(json::from_cbor(json::to_cbor(json("\xFF"))) == json("\xFF"));
|
||||
// a truncated multi-byte sequence
|
||||
CHECK(json::from_cbor(json::to_cbor(json("\xC3"))) == json("\xC3"));
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK(json::from_cbor(json::to_cbor(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
|
||||
// an overlong encoding of '.'
|
||||
CHECK(json::from_cbor(json::to_cbor(json("\xC0\xAF"))) == json("\xC0\xAF"));
|
||||
|
||||
// an object key with ill-formed UTF-8 is kept the same way
|
||||
CHECK(json::from_cbor(json::to_cbor(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
|
||||
|
||||
// binary values are not text and are unaffected
|
||||
CHECK_NOTHROW(json::to_cbor(json::binary(std::vector<std::uint8_t>({0xFF}))));
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 in indefinite-length string")
|
||||
{
|
||||
json _;
|
||||
|
||||
// every chunk must be valid UTF-8 on its own (RFC 8949, Section
|
||||
// 3.2.3), so a code point split across two chunks is rejected
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded());
|
||||
// the chunks are concatenated as is, without checking that each
|
||||
// chunk is valid UTF-8 on its own (RFC 8949, Section 3.2.3), so
|
||||
// a code point split across two chunks yields a valid string
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})));
|
||||
CHECK(_ == "\xc3\xa9");
|
||||
CHECK(_.dump() == "\"\xc3\xa9\"");
|
||||
|
||||
// an ill-formed later chunk is rejected after valid ones
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
// a truncated code point is kept as is
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0xff})));
|
||||
CHECK(_ == "\xc3");
|
||||
CHECK_THROWS_AS(_.dump(), json::type_error&);
|
||||
CHECK(json::from_cbor(json::to_cbor(_)) == _);
|
||||
|
||||
// an ill-formed later chunk is kept after valid ones
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})));
|
||||
CHECK(_ == "\xc3\xa9\xc0\xae");
|
||||
CHECK_THROWS_AS(_.dump(), json::type_error&);
|
||||
|
||||
// valid multi-byte chunks are accepted
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6");
|
||||
@@ -1840,9 +1892,6 @@ TEST_CASE("CBOR")
|
||||
|
||||
SECTION("many chunks in indefinite-length string")
|
||||
{
|
||||
// only the newly read chunk is validated, not the whole string
|
||||
// collected so far; validating the latter made this input take
|
||||
// quadratic time (about ten seconds for 100000 chunks)
|
||||
constexpr std::size_t chunks = 100000;
|
||||
std::vector<uint8_t> v{0x7f};
|
||||
for (std::size_t i = 0; i < chunks; ++i)
|
||||
|
||||
@@ -1540,19 +1540,39 @@ TEST_CASE("MessagePack")
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&);
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in string (see #5529)")
|
||||
SECTION("ill-formed UTF-8 in string (see #5529, #5651)")
|
||||
{
|
||||
// the MessagePack specification explicitly allows a str object to
|
||||
// contain a byte sequence that is not valid UTF-8 and expects a
|
||||
// deserializer to hand the original bytes back unchanged; this
|
||||
// library follows that, unlike CBOR/UBJSON/BJData/BSON, whose
|
||||
// specifications require text strings to be valid UTF-8
|
||||
|
||||
// a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8
|
||||
// (0xC0 0xAE is an overlong encoding of '.') must be rejected at
|
||||
// decode time, matching every other kind of malformed binary
|
||||
// input, rather than only failing later when the resulting
|
||||
// value is dumped
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0xa2, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing MessagePack string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_msgpack(std::vector<uint8_t>({0xa2, 0xc0, 0xae}), true, false).is_discarded());
|
||||
// (0xC0 0xAE is an overlong encoding of '.') round-trips byte for
|
||||
// byte as a string value
|
||||
const std::vector<uint8_t> ill_formed_value = {0xa2, 0xc0, 0xae};
|
||||
json j_value;
|
||||
CHECK_NOTHROW(j_value = json::from_msgpack(ill_formed_value));
|
||||
REQUIRE(j_value.is_string());
|
||||
CHECK(j_value.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
CHECK(json::from_msgpack(json::to_msgpack(j_value)) == j_value);
|
||||
// dump() still requires valid UTF-8 and throws for such a value,
|
||||
// unless an error handler that replaces or ignores the bytes is
|
||||
// passed
|
||||
CHECK_THROWS_AS(j_value.dump(), json::type_error&);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> ill_formed_key = {0x81, 0xa2, 0xc0, 0xae, 0x01};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_msgpack(ill_formed_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK(json::from_msgpack(json::to_msgpack(j_key)) == j_key);
|
||||
|
||||
// a MessagePack bin8 blob with the very same bytes is NOT text
|
||||
// and must still be accepted as-is
|
||||
json _;
|
||||
CHECK_NOTHROW(_ = json::from_msgpack(std::vector<uint8_t>({0xc4, 0x02, 0xc0, 0xae})));
|
||||
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
||||
|
||||
|
||||
@@ -2505,6 +2505,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||
CHECK(json::to_ubjson(j) == v);
|
||||
CHECK(json::from_ubjson(v) == j);
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
||||
{
|
||||
// none of the binary format specs requires a decoder to reject
|
||||
// ill-formed UTF-8 in a text string, so a value whose bytes are
|
||||
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
|
||||
// round-trips byte for byte as a string value; to_ubjson() writes
|
||||
// the bytes unchanged, as before 3.13.0, unless
|
||||
// JSON_STRICT_BINARY_UTF8 is enabled (see
|
||||
// unit-binary_utf8_strict.cpp)
|
||||
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
|
||||
json j;
|
||||
CHECK_NOTHROW(j = json::from_ubjson(v));
|
||||
REQUIRE(j.is_string());
|
||||
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
||||
CHECK(json::from_ubjson(json::to_ubjson(j)) == j);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_ubjson(v_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK(json::from_ubjson(json::to_ubjson(j_key)) == j_key);
|
||||
|
||||
CHECK(json::from_ubjson(json::to_ubjson(json("\xFF"))) == json("\xFF"));
|
||||
// a truncated multi-byte sequence
|
||||
CHECK(json::from_ubjson(json::to_ubjson(json("\xC3"))) == json("\xC3"));
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK(json::from_ubjson(json::to_ubjson(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
|
||||
// an overlong encoding of '.'
|
||||
CHECK(json::from_ubjson(json::to_ubjson(json("\xC0\xAF"))) == json("\xC0\xAF"));
|
||||
|
||||
// an object key with ill-formed UTF-8 is kept the same way
|
||||
CHECK(json::from_ubjson(json::to_ubjson(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("Array Type")
|
||||
|
||||
Reference in New Issue
Block a user