From af25bf67f3933fcfe3c61a6e5a90b74f7cb5e5f2 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 29 Sep 2026 20:25:03 +0200 Subject: [PATCH] Report BON8 input that ends after a UTF-8 lead byte as truncated MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A lead byte (0xC2..0xF7) inside a string begins either another character (if a continuation byte follows) or an integer (otherwise). When the input ended right after the lead byte, the reader took the missing byte as "not a continuation byte", ended the string before the lead byte, and treated the lead byte as the start of the next value. With strict=false, a message cut off there was therefore read as a shorter value: the 11 bytes of "šŸ˜€šŸ˜€Ć©" cut after 9 bytes gave "šŸ˜€šŸ˜€", and ["aĆ©"] cut after 3 of its 5 bytes gave ["a"]. With strict=true, the input was rejected with a misleading message ("expected end of input"), or, for a key, with parse_error.112 instead of 110. Either reading of the lead byte leaves the message incomplete: a string at the end of a message must be terminated by 0xFF, so the lead byte cannot belong to a following message. Report parse_error.110 (unexpected end of input) for strings and keys, as the comment on get_bon8_string() already requires and as the reference decoder (HikoGUI) does. Signed-off-by: Niels Lohmann --- .../nlohmann/detail/input/binary_reader.hpp | 11 +++++ single_include/nlohmann/json.hpp | 11 +++++ tests/src/unit-bon8.cpp | 41 +++++++++++++++++++ 3 files changed, 63 insertions(+) diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index b9e6b304b..5a1da639c 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -3821,6 +3821,11 @@ class binary_reader if (0xC2 <= byte && byte <= 0xF7) { const auto second = get_bon8(); + if (second == char_traits::eof()) + { + // the input ends inside a character or an integer + return unexpect_eof(input_format_t::bon8, "key"); + } unget_bon8(second); if (is_bon8_continuation(second)) { @@ -3919,6 +3924,12 @@ class binary_reader // a lead byte ends the string if no continuation byte follows: it // is then the first byte of an integer const auto second = get_bon8(); + if (second == char_traits::eof()) + { + // the input ends inside a character or an integer: either + // way, the message is incomplete + return unexpect_eof(input_format_t::bon8, "string"); + } if (!is_bon8_continuation(second)) { unget_bon8(second); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 576498738..6db31291a 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -16588,6 +16588,11 @@ class binary_reader if (0xC2 <= byte && byte <= 0xF7) { const auto second = get_bon8(); + if (second == char_traits::eof()) + { + // the input ends inside a character or an integer + return unexpect_eof(input_format_t::bon8, "key"); + } unget_bon8(second); if (is_bon8_continuation(second)) { @@ -16686,6 +16691,12 @@ class binary_reader // a lead byte ends the string if no continuation byte follows: it // is then the first byte of an integer const auto second = get_bon8(); + if (second == char_traits::eof()) + { + // the input ends inside a character or an integer: either + // way, the message is incomplete + return unexpect_eof(input_format_t::bon8, "string"); + } if (!is_bon8_continuation(second)) { unget_bon8(second); diff --git a/tests/src/unit-bon8.cpp b/tests/src/unit-bon8.cpp index 030e1d5c3..b2a92b006 100644 --- a/tests/src/unit-bon8.cpp +++ b/tests/src/unit-bon8.cpp @@ -531,6 +531,47 @@ TEST_CASE("BON8") CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a'}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); } + SECTION("input that ends after a UTF-8 lead byte") + { + // the lead byte begins either a character or an integer; both are + // incomplete, so the lead byte must not end the string before it + for (const bool strict : + { + true, false + }) + { + CAPTURE(strict) + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xE2}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xF0}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 0xC3, 0xA9, 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 key: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x88, 'a', 0x91, 0xE2}, strict), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 key: unexpected end of input", json::parse_error&); + } + } + + SECTION("a message that is cut off is not read as a shorter value") + { + const json values = {"\xC3\xA9", "a\xE2\x82\xAC", "\xF0\x9F\x98\x80\xC3\xA9", {"a\xC3\xA9"}, {{"\xC3\xA9", "\xE2\x82\xAC"}}, {{"a", {"b\xC3\xA9", 1}}}}; + for (const auto& j : values) + { + const bytes message = json::to_bon8(j); + for (std::size_t length = 0; length < message.size(); ++length) + { + CAPTURE(j) + CAPTURE(length) + bytes prefix = message; + prefix.resize(length); + CHECK(json::from_bon8(prefix, false, false).is_discarded()); + // a stream is read byte by byte rather than in bulk + std::istringstream stream(str(prefix)); + CHECK(json::from_bon8(stream, false, false).is_discarded()); + } + } + } + SECTION("invalid UTF-8") { // overlong