mirror of
https://github.com/nlohmann/json.git
synced 2026-10-04 05:30:31 +00:00
Validate UTF-8 in CBOR/MessagePack/BSON/UBJSON/BJData text strings on develop
PR #5531 fixed the UTF-8-validation gap described in #5529, but it was merged onto the still-unmerged bson-sizes branch rather than develop, so develop was left with the original bug for all affected formats. A follow-up comment on #5529 reproduced this on develop and additionally found that UBJSON (and, by the same code path, BJData) has the identical gap, undocumented. Port the same fix directly onto develop: extract the UTF-8 DFA decoder out of serializer<>::decode() into a shared detail::decode()/is_valid_utf8() in string_utils.hpp, and call it from binary_reader::get_string() - the single choke point shared by all five binary readers - so malformed text strings are rejected at decode time (parse_error.113) instead of only failing later on dump() (type_error.316). Byte/binary payloads are unaffected. Add matching decode-time tests for CBOR, MessagePack, BSON, UBJSON, and BJData, and document the new behavior on all five binary format pages (the two UBJSON/BJData pages didn't get this note in #5531). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017sdieJCn6BHxzRMaXP49sP
This commit is contained in:
@@ -2892,6 +2892,23 @@ TEST_CASE("BJData")
|
||||
CHECK(json::from_bjdata(vl, true, false).is_discarded());
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in string (see #5529)")
|
||||
{
|
||||
// a BJData string of length 2 whose bytes are not valid
|
||||
// UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be
|
||||
// rejected at decode time, matching every other kind of
|
||||
// malformed binary input, rather than only failing later
|
||||
// when the resulting value is dumped
|
||||
std::vector<uint8_t> const v = {'S', 'i', 0x02, 0xc0, 0xae};
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(v), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing BJData string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_bjdata(v, true, false).is_discarded());
|
||||
|
||||
// valid UTF-8 must still round-trip
|
||||
const json j = "h\xc3\xa9llo, w\xc3\xb6rld! \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e"; // héllo, wörld! 日本語
|
||||
CHECK(json::from_bjdata(json::to_bjdata(j)) == j);
|
||||
}
|
||||
|
||||
SECTION("parse bjdata markers in ubjson")
|
||||
{
|
||||
// create a single-character string for all number types
|
||||
|
||||
@@ -199,6 +199,32 @@ TEST_CASE("BSON")
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 10: syntax error while parsing BSON string: string length must be at least 1, is -2147483648", json::parse_error&);
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in string (see #5529)")
|
||||
{
|
||||
// a BSON document with a string field "k" whose value bytes are not
|
||||
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be
|
||||
// rejected at decode time, matching every other kind of malformed
|
||||
// binary input, rather than only failing later when the resulting
|
||||
// value is dumped
|
||||
std::vector<std::uint8_t> const v =
|
||||
{
|
||||
0x0F, 0x00, 0x00, 0x00, // size (little endian)
|
||||
0x02, /// entry: string (UTF-8)
|
||||
'k', 0x00, // key "k"
|
||||
0x03, 0x00, 0x00, 0x00, // string length (including trailing zero byte)
|
||||
0xc0, 0xae, // ill-formed UTF-8
|
||||
0x00, // string terminator
|
||||
0x00 // end marker
|
||||
};
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing BSON string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_bson(v, true, false).is_discarded());
|
||||
|
||||
// valid UTF-8 must still round-trip
|
||||
const json j = {{"k", "h\xc3\xa9llo, w\xc3\xb6rld! \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e"}}; // héllo, wörld! 日本語
|
||||
CHECK(json::from_bson(json::to_bson(j)) == j);
|
||||
}
|
||||
|
||||
SECTION("objects")
|
||||
{
|
||||
SECTION("empty object")
|
||||
|
||||
@@ -1833,6 +1833,27 @@ TEST_CASE("CBOR")
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0xa1, 0xff, 0x01}), true, false).is_discarded());
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in string (see #5529)")
|
||||
{
|
||||
// a two-character text string (major type 3) whose bytes are not
|
||||
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be
|
||||
// rejected at decode time, matching every other kind of
|
||||
// malformed binary input, rather than only failing later when
|
||||
// the resulting value is dumped
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae}), true, false).is_discarded());
|
||||
|
||||
// a CBOR byte string (major type 2) with the very same bytes is
|
||||
// NOT text and must still be accepted as-is
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x42, 0xc0, 0xae})));
|
||||
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
||||
|
||||
// valid UTF-8 must still round-trip
|
||||
const json j = "h\xc3\xa9llo, w\xc3\xb6rld! \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e"; // héllo, wörld! 日本語
|
||||
CHECK(json::from_cbor(json::to_cbor(j)) == j);
|
||||
}
|
||||
|
||||
SECTION("strict mode")
|
||||
{
|
||||
std::vector<uint8_t> const vec = {0xf6, 0xf6};
|
||||
|
||||
@@ -1554,6 +1554,27 @@ TEST_CASE("MessagePack")
|
||||
CHECK(json::from_msgpack(std::vector<uint8_t>({0x81, 0xff, 0x01}), true, false).is_discarded());
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in string (see #5529)")
|
||||
{
|
||||
// a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8
|
||||
// (0xC0 0xAE is an overlong encoding of '.') must be rejected at
|
||||
// decode time, matching every other kind of malformed binary
|
||||
// input, rather than only failing later when the resulting
|
||||
// value is dumped
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0xa2, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing MessagePack string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_msgpack(std::vector<uint8_t>({0xa2, 0xc0, 0xae}), true, false).is_discarded());
|
||||
|
||||
// a MessagePack bin8 blob with the very same bytes is NOT text
|
||||
// and must still be accepted as-is
|
||||
CHECK_NOTHROW(_ = json::from_msgpack(std::vector<uint8_t>({0xc4, 0x02, 0xc0, 0xae})));
|
||||
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
||||
|
||||
// valid UTF-8 must still round-trip
|
||||
const json j = "h\xc3\xa9llo, w\xc3\xb6rld! \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e"; // héllo, wörld! 日本語
|
||||
CHECK(json::from_msgpack(json::to_msgpack(j)) == j);
|
||||
}
|
||||
|
||||
SECTION("strict mode")
|
||||
{
|
||||
std::vector<uint8_t> const vec = {0xc0, 0xc0};
|
||||
|
||||
@@ -1927,6 +1927,23 @@ TEST_CASE("UBJSON")
|
||||
std::vector<uint8_t> const v0 = {'S', 'i', 0};
|
||||
CHECK(json::from_ubjson(v0) == json(""));
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in string (see #5529)")
|
||||
{
|
||||
// a UBJSON string of length 2 whose bytes are not valid
|
||||
// UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be
|
||||
// rejected at decode time, matching every other kind of
|
||||
// malformed binary input, rather than only failing later
|
||||
// when the resulting value is dumped
|
||||
std::vector<uint8_t> const v = {'S', 'i', 0x02, 0xc0, 0xae};
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(v), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing UBJSON string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_ubjson(v, true, false).is_discarded());
|
||||
|
||||
// valid UTF-8 must still round-trip
|
||||
const json j = "h\xc3\xa9llo, w\xc3\xb6rld! \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e"; // héllo, wörld! 日本語
|
||||
CHECK(json::from_ubjson(json::to_ubjson(j)) == j);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("array")
|
||||
|
||||
Reference in New Issue
Block a user