Compare commits

..
Author SHA1 Message Date
Niels Lohmann af25bf67f3 Report BON8 input that ends after a UTF-8 lead byte as truncated
A lead byte (0xC2..0xF7) inside a string begins either another character
(if a continuation byte follows) or an integer (otherwise). When the input
ended right after the lead byte, the reader took the missing byte as "not
a continuation byte", ended the string before the lead byte, and treated
the lead byte as the start of the next value. With strict=false, a message
cut off there was therefore read as a shorter value: the 11 bytes of
"😀😀é" cut after 9 bytes gave "😀😀", and ["aé"] cut after 3 of its 5
bytes gave ["a"]. With strict=true, the input was rejected with a
misleading message ("expected end of input"), or, for a key, with
parse_error.112 instead of 110.

Either reading of the lead byte leaves the message incomplete: a string at
the end of a message must be terminated by 0xFF, so the lead byte cannot
belong to a following message. Report parse_error.110 (unexpected end of
input) for strings and keys, as the comment on get_bon8_string() already
requires and as the reference decoder (HikoGUI) does.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 20:25:03 +02:00
5 changed files with 65 additions and 150 deletions
@@ -3821,6 +3821,11 @@ class binary_reader
if (0xC2 <= byte && byte <= 0xF7)
{
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer
return unexpect_eof(input_format_t::bon8, "key");
}
unget_bon8(second);
if (is_bon8_continuation(second))
{
@@ -3919,6 +3924,12 @@ class binary_reader
// a lead byte ends the string if no continuation byte follows: it
// is then the first byte of an integer
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer: either
// way, the message is incomplete
return unexpect_eof(input_format_t::bon8, "string");
}
if (!is_bon8_continuation(second))
{
unget_bon8(second);
+1 -51
View File
@@ -1307,65 +1307,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
*/
template<bool Ordered>
static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept
{
return compare_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
}
/// @brief compare two leaves that are only being checked for equality
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::false_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::false_type {});
return order_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
}
#if JSON_HAS_THREE_WAY_COMPARISON
/*!
@brief compare two leaves that are being ordered, for operator<=>
Reached only from operator<=>, so the leaves must be classified exactly
as operator<=> classifies them - which is not the same as asking
== and then order_leaves(), the way the other overload does it. The two
disagree on a binary value: == also compares the subtype, but <=> compares
only the bytes, through std::vector<std::uint8_t>::operator<=>. Using <=>
itself here keeps a leaf pair classified the same way regardless of how
deep it is nested - == first would again call operator<=> a level down
through order_leaves(), but call it after a mismatching == already ended
the comparison for a pair that <=> alone would still call equivalent.
*/
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
const std::partial_ordering order = lhs <=> rhs; // *NOPAD*
if (order == 0)
{
return compare_result::equal;
}
if (order < 0)
{
return compare_result::less;
}
if (order > 0)
{
return compare_result::greater;
}
return compare_result::unordered;
}
#else
/// @brief compare two leaves that are being ordered, for operator<
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::true_type {});
}
#endif
/*!
@brief compare two object keys
+12 -51
View File
@@ -16588,6 +16588,11 @@ class binary_reader
if (0xC2 <= byte && byte <= 0xF7)
{
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer
return unexpect_eof(input_format_t::bon8, "key");
}
unget_bon8(second);
if (is_bon8_continuation(second))
{
@@ -16686,6 +16691,12 @@ class binary_reader
// a lead byte ends the string if no continuation byte follows: it
// is then the first byte of an integer
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer: either
// way, the message is incomplete
return unexpect_eof(input_format_t::bon8, "string");
}
if (!is_bon8_continuation(second))
{
unget_bon8(second);
@@ -27388,65 +27399,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
*/
template<bool Ordered>
static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept
{
return compare_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
}
/// @brief compare two leaves that are only being checked for equality
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::false_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::false_type {});
return order_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
}
#if JSON_HAS_THREE_WAY_COMPARISON
/*!
@brief compare two leaves that are being ordered, for operator<=>
Reached only from operator<=>, so the leaves must be classified exactly
as operator<=> classifies them - which is not the same as asking
== and then order_leaves(), the way the other overload does it. The two
disagree on a binary value: == also compares the subtype, but <=> compares
only the bytes, through std::vector<std::uint8_t>::operator<=>. Using <=>
itself here keeps a leaf pair classified the same way regardless of how
deep it is nested - == first would again call operator<=> a level down
through order_leaves(), but call it after a mismatching == already ended
the comparison for a pair that <=> alone would still call equivalent.
*/
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
const std::partial_ordering order = lhs <=> rhs; // *NOPAD*
if (order == 0)
{
return compare_result::equal;
}
if (order < 0)
{
return compare_result::less;
}
if (order > 0)
{
return compare_result::greater;
}
return compare_result::unordered;
}
#else
/// @brief compare two leaves that are being ordered, for operator<
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::true_type {});
}
#endif
/*!
@brief compare two object keys
+41
View File
@@ -531,6 +531,47 @@ TEST_CASE("BON8")
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a'}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
}
SECTION("input that ends after a UTF-8 lead byte")
{
// the lead byte begins either a character or an integer; both are
// incomplete, so the lead byte must not end the string before it
for (const bool strict :
{
true, false
})
{
CAPTURE(strict)
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xE2}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xF0}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 0xC3, 0xA9, 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 key: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x88, 'a', 0x91, 0xE2}, strict), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 key: unexpected end of input", json::parse_error&);
}
}
SECTION("a message that is cut off is not read as a shorter value")
{
const json values = {"\xC3\xA9", "a\xE2\x82\xAC", "\xF0\x9F\x98\x80\xC3\xA9", {"a\xC3\xA9"}, {{"\xC3\xA9", "\xE2\x82\xAC"}}, {{"a", {"b\xC3\xA9", 1}}}};
for (const auto& j : values)
{
const bytes message = json::to_bon8(j);
for (std::size_t length = 0; length < message.size(); ++length)
{
CAPTURE(j)
CAPTURE(length)
bytes prefix = message;
prefix.resize(length);
CHECK(json::from_bon8(prefix, false, false).is_discarded());
// a stream is read byte by byte rather than in bulk
std::istringstream stream(str(prefix));
CHECK(json::from_bon8(stream, false, false).is_discarded());
}
}
}
SECTION("invalid UTF-8")
{
// overlong
-48
View File
@@ -952,51 +952,3 @@ TEST_CASE("containers are compared element by element")
}
}
}
#if JSON_HAS_THREE_WAY_COMPARISON
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
TEST_CASE("operator<=> of binary values with a different subtype does not depend on nesting depth")
{
// #5654: std::vector<std::uint8_t>::operator<=>, which the binary type's
// own operator<=> uses, ignores the subtype that operator== checks. So a
// pair of binary values with the same bytes but a different subtype is
// unequal, yet <=>-equivalent - the same inconsistency between == and <=>
// that a NaN has. Within the nesting bound, an array compares itself
// with std::vector's own operator<=>, which treats an equivalent pair as
// undecided and lets the next element decide, same as
// std::lexicographical_compare_three_way does. Past the bound,
// compare_iteratively<true>() takes over and must classify the pair the
// same way, or the result of operator<=> - and of <, which C++20 derives
// from it - depends on how deeply the values are nested.
const json a = json::array({json::binary({1}, 1), 1});
const json b = json::array({json::binary({1}, 2), 2});
// the root inconsistency: unequal, yet <=>-equivalent
CHECK_FALSE(a[0] == b[0]);
CHECK((a[0] <=> b[0]) == std::partial_ordering::equivalent); // *NOPAD*
const auto deep = [](const json & j, const std::size_t depth)
{
json result = j;
for (std::size_t i = 0; i < depth; ++i)
{
result = json::array({std::move(result)});
}
return result;
};
// 127 levels stay within nesting_depth_limit() (128); 128 and 200 do not,
// and must still agree with the levels that do
for (const std::size_t depth : std::vector<std::size_t> {0, 127, 128, 200})
{
CAPTURE(depth);
const json x = deep(a, depth);
const json y = deep(b, depth);
CHECK((x <=> y) == std::partial_ordering::less); // *NOPAD*
CHECK((y <=> x) == std::partial_ordering::greater); // *NOPAD*
CHECK(x < y);
CHECK(y > x);
CHECK_FALSE(y < x);
}
}
#endif