Compare commits

..
Author SHA1 Message Date
Niels Lohmann af25bf67f3 Report BON8 input that ends after a UTF-8 lead byte as truncated
A lead byte (0xC2..0xF7) inside a string begins either another character
(if a continuation byte follows) or an integer (otherwise). When the input
ended right after the lead byte, the reader took the missing byte as "not
a continuation byte", ended the string before the lead byte, and treated
the lead byte as the start of the next value. With strict=false, a message
cut off there was therefore read as a shorter value: the 11 bytes of
"😀😀é" cut after 9 bytes gave "😀😀", and ["aé"] cut after 3 of its 5
bytes gave ["a"]. With strict=true, the input was rejected with a
misleading message ("expected end of input"), or, for a key, with
parse_error.112 instead of 110.

Either reading of the lead byte leaves the message incomplete: a string at
the end of a message must be terminated by 0xFF, so the lead byte cannot
belong to a following message. Report parse_error.110 (unexpected end of
input) for strings and keys, as the comment on get_bon8_string() already
requires and as the reference decoder (HikoGUI) does.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 20:25:03 +02:00
7 changed files with 68 additions and 40 deletions
@@ -3821,6 +3821,11 @@ class binary_reader
if (0xC2 <= byte && byte <= 0xF7)
{
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer
return unexpect_eof(input_format_t::bon8, "key");
}
unget_bon8(second);
if (is_bon8_continuation(second))
{
@@ -3919,6 +3924,12 @@ class binary_reader
// a lead byte ends the string if no continuation byte follows: it
// is then the first byte of an integer
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer: either
// way, the message is incomplete
return unexpect_eof(input_format_t::bon8, "string");
}
if (!is_bon8_continuation(second))
{
unget_bon8(second);
+11
View File
@@ -16588,6 +16588,11 @@ class binary_reader
if (0xC2 <= byte && byte <= 0xF7)
{
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer
return unexpect_eof(input_format_t::bon8, "key");
}
unget_bon8(second);
if (is_bon8_continuation(second))
{
@@ -16686,6 +16691,12 @@ class binary_reader
// a lead byte ends the string if no continuation byte follows: it
// is then the first byte of an integer
const auto second = get_bon8();
if (second == char_traits<char_type>::eof())
{
// the input ends inside a character or an integer: either
// way, the message is incomplete
return unexpect_eof(input_format_t::bon8, "string");
}
if (!is_bon8_continuation(second))
{
unget_bon8(second);
+41
View File
@@ -531,6 +531,47 @@ TEST_CASE("BON8")
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a'}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
}
SECTION("input that ends after a UTF-8 lead byte")
{
// the lead byte begins either a character or an integer; both are
// incomplete, so the lead byte must not end the string before it
for (const bool strict :
{
true, false
})
{
CAPTURE(strict)
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xE2}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xF0}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 0xC3, 0xA9, 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a', 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 0xC3}, strict), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 key: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x88, 'a', 0x91, 0xE2}, strict), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 key: unexpected end of input", json::parse_error&);
}
}
SECTION("a message that is cut off is not read as a shorter value")
{
const json values = {"\xC3\xA9", "a\xE2\x82\xAC", "\xF0\x9F\x98\x80\xC3\xA9", {"a\xC3\xA9"}, {{"\xC3\xA9", "\xE2\x82\xAC"}}, {{"a", {"b\xC3\xA9", 1}}}};
for (const auto& j : values)
{
const bytes message = json::to_bon8(j);
for (std::size_t length = 0; length < message.size(); ++length)
{
CAPTURE(j)
CAPTURE(length)
bytes prefix = message;
prefix.resize(length);
CHECK(json::from_bon8(prefix, false, false).is_discarded());
// a stream is read byte by byte rather than in bulk
std::istringstream stream(str(prefix));
CHECK(json::from_bon8(stream, false, false).is_discarded());
}
}
}
SECTION("invalid UTF-8")
{
// overlong
-3
View File
@@ -8,6 +8,3 @@ The following changes have been made to the code with respect to <https://github
- membership check
- made function from `_is_within`
- removed unused variable `actual_path`
- Added the optional config key `external`: include paths listed there are kept as
`#include` directives instead of being inlined (the first directive per path; the
repeated ones are commented out).
-5
View File
@@ -57,11 +57,6 @@ Python v.2.7.0 or higher is required.
amalgamation. Have a look at `test/source.c.json` and `test/include.h.json`
to see two examples.
The optional `external` list names include paths that are kept as `#include`
directives instead of being inlined, e.g. `["nlohmann/json.hpp"]` for a header
that includes another amalgamated header. Only the first directive for each
of these paths is kept; the repeated ones are commented out.
* The `-s, --source` option should specify the path to the source directory.
This is useful for supporting separate source and build directories.
+5 -23
View File
@@ -62,10 +62,6 @@ class Amalgamation(object):
return None
def __init__(self, args):
# include paths that are kept as #include directives instead of
# being inlined (e.g. a header amalgamated on its own)
self.external = []
self.included_external = []
with open(args.config, 'r') as f:
config = json.loads(f.read())
for key in config:
@@ -224,14 +220,11 @@ class TranslationUnit(object):
while include_match:
if not _is_within(include_match, skippable_contexts):
include_path = include_match.group("path")
if include_path in self.amalgamation.external:
includes.append((include_match, None))
else:
search_same_dir = include_match.group(1) == '"'
found_included_path = self.amalgamation.find_included_file(
include_path, self.file_dir if search_same_dir else None)
if found_included_path:
includes.append((include_match, found_included_path))
search_same_dir = include_match.group(1) == '"'
found_included_path = self.amalgamation.find_included_file(
include_path, self.file_dir if search_same_dir else None)
if found_included_path:
includes.append((include_match, found_included_path))
include_match = self.include_pattern.search(self.content,
include_match.end())
@@ -242,17 +235,6 @@ class TranslationUnit(object):
for include in includes:
include_match, found_included_path = include
tmp_content += self.content[prev_end:include_match.start()]
if found_included_path is None:
# an external header: keep the first directive and comment
# out the repeated ones
include_path = include_match.group("path")
if include_path in self.amalgamation.included_external:
tmp_content += "// {0}".format(include_match.group(0))
else:
self.amalgamation.included_external.append(include_path)
tmp_content += include_match.group(0)
prev_end = include_match.end()
continue
tmp_content += "// {0}\n".format(include_match.group(0))
if found_included_path not in self.amalgamation.included_files:
t = TranslationUnit(found_included_path, self.amalgamation, False)
-9
View File
@@ -1,9 +0,0 @@
{
"project": "JSON for Modern C++",
"target": "single_include/nlohmann/json_view.hpp",
"sources": [
"include/nlohmann/json_view.hpp"
],
"include_paths": ["include"],
"external": ["nlohmann/json.hpp"]
}