Compare commits

..
Author SHA1 Message Date
Niels Lohmann af25bf67f3 Report BON8 input that ends after a UTF-8 lead byte as truncated
A lead byte (0xC2..0xF7) inside a string begins either another character
(if a continuation byte follows) or an integer (otherwise). When the input
ended right after the lead byte, the reader took the missing byte as "not
a continuation byte", ended the string before the lead byte, and treated
the lead byte as the start of the next value. With strict=false, a message
cut off there was therefore read as a shorter value: the 11 bytes of
"😀😀é" cut after 9 bytes gave "😀😀", and ["aé"] cut after 3 of its 5
bytes gave ["a"]. With strict=true, the input was rejected with a
misleading message ("expected end of input"), or, for a key, with
parse_error.112 instead of 110.

Either reading of the lead byte leaves the message incomplete: a string at
the end of a message must be terminated by 0xFF, so the lead byte cannot
belong to a following message. Report parse_error.110 (unexpected end of
input) for strings and keys, as the comment on get_bon8_string() already
requires and as the reference decoder (HikoGUI) does.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 20:25:03 +02:00
7 changed files with 66 additions and 65 deletions
+1 -3
View File
@@ -195,7 +195,5 @@ Strong exception safety: if an exception occurs, the original value stays intact
1. Added in version 1.0.0.
2. Added in version 1.0.0.
3. Added in version 1.0.0.
4. Added in version 1.0.0. Fixed in version 3.13.0 to copy the values before inserting; before, an `ilist` that
referred to elements of the array being inserted into could insert wrong values, because the range insert could
move from or shift an element before it was copied.
4. Added in version 1.0.0.
5. Added in version 3.0.0.
@@ -3821,6 +3821,11 @@ class binary_reader
if (0xC2 <= byte && byte <= 0xF7)
{
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer
return unexpect_eof(input_format_t::bon8, "key");
}
unget_bon8(second);
if (is_bon8_continuation(second))
{
@@ -3919,6 +3924,12 @@ class binary_reader
// a lead byte ends the string if no continuation byte follows: it
// is then the first byte of an integer
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer: either
// way, the message is incomplete
return unexpect_eof(input_format_t::bon8, "string");
}
if (!is_bon8_continuation(second))
{
unget_bon8(second);
+1 -9
View File
@@ -4152,16 +4152,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
JSON_THROW(invalid_iterator::create(202, "iterator does not fit current value", this));
}
// copy the values first: ilist may refer to elements of this array
array_t values;
detail::reserve_array(values, ilist.size(), detail::priority_tag<1> {});
for (const auto& element : ilist)
{
values.push_back(element.moved_or_copied());
}
// insert to array and return iterator
return insert_iterator(pos, std::make_move_iterator(values.begin()), std::make_move_iterator(values.end()));
return insert_iterator(pos, ilist.begin(), ilist.end());
}
/// @brief inserts range of elements into object
+12 -9
View File
@@ -16588,6 +16588,11 @@ class binary_reader
if (0xC2 <= byte && byte <= 0xF7)
{
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer
return unexpect_eof(input_format_t::bon8, "key");
}
unget_bon8(second);
if (is_bon8_continuation(second))
{
@@ -16686,6 +16691,12 @@ class binary_reader
// a lead byte ends the string if no continuation byte follows: it
// is then the first byte of an integer
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer: either
// way, the message is incomplete
return unexpect_eof(input_format_t::bon8, "string");
}
if (!is_bon8_continuation(second))
{
unget_bon8(second);
@@ -30233,16 +30244,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
JSON_THROW(invalid_iterator::create(202, "iterator does not fit current value", this));
}
// copy the values first: ilist may refer to elements of this array
array_t values;
detail::reserve_array(values, ilist.size(), detail::priority_tag<1> {});
for (const auto& element : ilist)
{
values.push_back(element.moved_or_copied());
}
// insert to array and return iterator
return insert_iterator(pos, std::make_move_iterator(values.begin()), std::make_move_iterator(values.end()));
return insert_iterator(pos, ilist.begin(), ilist.end());
}
/// @brief inserts range of elements into object
+41
View File
@@ -531,6 +531,47 @@ TEST_CASE("BON8")
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a'}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
}
SECTION("input that ends after a UTF-8 lead byte")
{
// the lead byte begins either a character or an integer; both are
// incomplete, so the lead byte must not end the string before it
for (const bool strict :
{
true, false
})
{
CAPTURE(strict)
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xE2}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xF0}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 0xC3, 0xA9, 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 key: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x88, 'a', 0x91, 0xE2}, strict), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 key: unexpected end of input", json::parse_error&);
}
}
SECTION("a message that is cut off is not read as a shorter value")
{
const json values = {"\xC3\xA9", "a\xE2\x82\xAC", "\xF0\x9F\x98\x80\xC3\xA9", {"a\xC3\xA9"}, {{"\xC3\xA9", "\xE2\x82\xAC"}}, {{"a", {"b\xC3\xA9", 1}}}};
for (const auto& j : values)
{
const bytes message = json::to_bon8(j);
for (std::size_t length = 0; length < message.size(); ++length)
{
CAPTURE(j)
CAPTURE(length)
bytes prefix = message;
prefix.resize(length);
CHECK(json::from_bon8(prefix, false, false).is_discarded());
// a stream is read byte by byte rather than in bulk
std::istringstream stream(str(prefix));
CHECK(json::from_bon8(stream, false, false).is_discarded());
}
}
}
SECTION("invalid UTF-8")
{
// overlong
-16
View File
@@ -117,22 +117,6 @@ TEST_CASE("array type without capacity()")
CHECK(nested.flatten().unflatten() == nested);
}
SECTION("insert(pos, initializer_list) compiles and works without reserve()")
{
// std::deque has no reserve() either; insert(pos, ilist) must not
// require it (regression test for #5656, which also covers an ilist
// that refers to elements of the array being inserted into)
deque_json j = deque_json::array();
j.push_back("a");
j.push_back("b");
j.push_back("c");
const deque_json& cj = j;
auto it = j.insert(j.begin(), {cj[0], cj[1]});
CHECK(*it == deque_json("a"));
CHECK(j == deque_json({"a", "b", "a", "b", "c"}));
}
SECTION("references stay valid while the array grows")
{
deque_json j = deque_json::array();
-28
View File
@@ -761,34 +761,6 @@ TEST_CASE("modifiers")
}
}
SECTION("initializer list referring to the array's own elements (#5656)")
{
SECTION("sufficient capacity (no reallocation)")
{
json j_own = json::array();
j_own.get_ref<json::array_t&>().reserve(8);
j_own.push_back("a");
j_own.push_back("b");
j_own.push_back("c");
const json& j_own_cref = j_own;
auto it = j_own.insert(j_own.begin(), {j_own_cref[0], j_own_cref[1]});
CHECK(*it == json("a"));
CHECK(j_own == json({"a", "b", "a", "b", "c"}));
}
SECTION("insufficient capacity (reallocation)")
{
json j_own = {"a", "b", "c"};
j_own.get_ref<json::array_t&>().shrink_to_fit();
const json& j_own_cref = j_own;
auto it = j_own.insert(j_own.begin(), {j_own_cref[2]});
CHECK(*it == json("c"));
CHECK(j_own == json({"c", "a", "b", "c"}));
}
}
SECTION("invalid iterator")
{
// pass iterator to a different array