Decode \u escapes with a table in the lexer

get_codepoint() read the four hex digits of a \u escape with four calls
to get(), each classified by a chain of range comparisons. For contiguous
input, get_codepoint_bulk() now decodes them with one lookup per byte
(hex_codepoint() in string_scan.hpp, after yyjson's read_hex_u16): a
256-entry table maps a byte to its value, or 0xFF for anything else, and
an invalid digit shows in the OR of the four values. It then skips the
four bytes and updates the position counters as four get() calls would.
If a digit is invalid or fewer than four bytes are left, it changes
nothing and the existing loop runs, so errors are reported with the same
message and position as before.

json::parse, best of 5 runs in separate processes (M1 Max): the escaped
twitter.json (every non-ASCII character as \u) -13.6%, all other files
within 0.3%.

Tests compare the contiguous and the streaming path (value or exception
message) for valid escapes, surrogate pairs, truncated and invalid digits
at every position, and 3,000 seeded random escapes.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-29 03:49:27 +02:00
parent 31789f51cf
commit a08f5501a7
4 changed files with 365 additions and 2 deletions
+46
View File
@@ -219,6 +219,44 @@ class lexer : public lexer_base<BasicJsonType>
// scan functions
/////////////////////
/// contiguous input: try to decode the 4 hex digits following `\u`
/// directly from the input buffer via hex_codepoint(), instead of 4 calls
/// to get(). On success, advances the adapter and the position counters
/// exactly as those 4 get() calls would (a hex digit is never '\n', so
/// only the flat counters move) and leaves @a current holding the last of
/// the 4 digits, just as the last such get() would; the codepoint is
/// written to @a out. Makes no state change and returns false - for a
/// pending unget, fewer than 4 remaining bytes, or any of the 4 bytes not
/// being a hex digit - so the caller falls back unchanged to the
/// per-character loop, which then reports the same diagnostic (stopping
/// at the first invalid digit) as before this optimization.
bool get_codepoint_bulk(std::true_type /*bulk*/, int& out)
{
if (next_unget || ia.bulk_remaining() < 4)
{
return false;
}
const char_type* const raw = ia.bulk_data();
const int codepoint = hex_codepoint(reinterpret_cast<const unsigned char*>(raw));
if (codepoint < 0)
{
return false;
}
ia.bulk_skip(4);
// a hex digit is never a newline, so only the flat counters advance
position.chars_read_total += 4;
position.chars_read_current_line += 4;
current = char_traits<char_type>::to_int_type(raw[3]);
out = codepoint;
return true;
}
/// streaming input: no bulk fast path
bool get_codepoint_bulk(std::false_type /*bulk*/, int& /*out*/) const noexcept
{
return false;
}
/*!
@brief get codepoint from 4 hex characters following `\u`
@@ -238,6 +276,14 @@ class lexer : public lexer_base<BasicJsonType>
{
// this function only makes sense after reading `\u`
JSON_ASSERT(current == 'u');
// contiguous input: decode all 4 hex digits directly from the buffer
int fast_codepoint = 0;
if (get_codepoint_bulk(std::integral_constant<bool, bulk_scan> {}, fast_codepoint))
{
return fast_codepoint;
}
int codepoint = 0;
const auto factors = { 12u, 8u, 4u, 0u };
+47 -1
View File
@@ -8,8 +8,9 @@
#pragma once
#include <array> // array
#include <cstddef> // size_t
#include <cstdint> // uint64_t
#include <cstdint> // uint64_t, uint8_t
#include <cstring> // memcpy
#include <nlohmann/detail/bit_ops.hpp>
@@ -315,5 +316,50 @@ inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noe
return scalar_string_bulk_run(data, n);
}
// Decode the 4 hex digits at [data, data+4) - the digits following a `\u`
// escape - into a codepoint 0x0000..0xFFFF via one table lookup per byte
// (after yyjson's read_hex_u16), or return -1 if any of the 4 bytes is not a
// hex digit ('0'..'9', 'A'..'F', 'a'..'f'). The caller must already have
// checked that 4 bytes are available; used by lexer::get_codepoint()'s
// contiguous fast path. On -1 it falls back to the byte-at-a-time loop, which
// stops at the first invalid digit, so the reported error and position are
// unaffected by this fast path.
inline int hex_codepoint(const unsigned char* data) noexcept
{
static const std::array<std::uint8_t, 256> hex_digit_table = // NOLINT(cppcoreguidelines-avoid-non-const-global-variables)
{
{
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 00..0F
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 10..1F
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 20..2F
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 30..3F ('0'..'9')
0xFF, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 40..4F ('A'..'F')
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 50..5F
0xFF, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 60..6F ('a'..'f')
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 70..7F
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 80..8F
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 90..9F
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // A0..AF
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // B0..BF
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // C0..CF
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // D0..DF
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // E0..EF
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF // F0..FF
}
};
const std::uint8_t d0 = hex_digit_table[data[0]];
const std::uint8_t d1 = hex_digit_table[data[1]];
const std::uint8_t d2 = hex_digit_table[data[2]];
const std::uint8_t d3 = hex_digit_table[data[3]];
// every valid digit is <= 0xF; the combined OR only exceeds it if at
// least one of the four bytes was not a hex digit (looked up as 0xFF)
if ((d0 | d1 | d2 | d3) > 0x0F)
{
return -1;
}
return (d0 << 12) | (d1 << 8) | (d2 << 4) | d3;
}
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END