Read BON8 strings in bulk from contiguous input

- copy the valid UTF-8 of a string in one step when the input is
  contiguous (twitter.json is read in 1.68 instead of 2.52 ms,
  jeopardy.json in 196 instead of 297 ms, close to CBOR and MessagePack)
- share the new valid_utf8_prefix() with the writer's UTF-8 check, which
  now skips ASCII 8 bytes at a time
- let the fuzzer check that contiguous and stream input give the same
  value or error, and test both paths in the unit tests
- clarify that a second 0xFF after a string is an empty string

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-25 22:06:54 +02:00
parent 31db00f2b4
commit 43a346cf99
7 changed files with 269 additions and 35 deletions
@@ -2178,7 +2178,7 @@ class binary_writer
if (N > 4)
{
oa.write_character(to_char_type(0xFE));
write_bon8_marker(0xFE, string_open);
}
break;
}
@@ -2248,20 +2248,10 @@ class binary_writer
{
static_cast<void>(context); // only used when exceptions are enabled
const auto* data = reinterpret_cast<const unsigned char*>(s.data());
for (std::size_t i = 0; i < s.size();)
const std::size_t valid = valid_utf8_prefix(data, s.size());
if (JSON_HEDLEY_UNLIKELY(valid != s.size()))
{
if (data[i] < 0x80)
{
++i;
continue;
}
const std::size_t length = validate_one_utf8(data + i, s.size() - i);
if (JSON_HEDLEY_UNLIKELY(length == 0))
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", hex_byte(data[i])), &context));
}
i += length;
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context));
}
}