mirror of
https://github.com/nlohmann/json.git
synced 2026-09-30 11:40:30 +00:00
Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1372829f4c | ||
|
|
046ad7ebd8 | ||
|
|
b7b2397e8f | ||
|
|
a08f5501a7 | ||
|
|
31789f51cf |
@@ -39,6 +39,25 @@ inline int count_leading_zeros(std::uint64_t x) noexcept
|
||||
#endif
|
||||
}
|
||||
|
||||
/// number of trailing zero bits of x (x != 0)
|
||||
inline int count_trailing_zeros(std::uint64_t x) noexcept
|
||||
{
|
||||
#if defined(__GNUC__) || defined(__clang__)
|
||||
return __builtin_ctzll(x);
|
||||
#else
|
||||
int n = 0;
|
||||
for (int shift = 32; shift != 0; shift >>= 1)
|
||||
{
|
||||
if ((x << (64 - shift)) == 0)
|
||||
{
|
||||
n += shift;
|
||||
x >>= shift;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
#endif
|
||||
}
|
||||
|
||||
/// the 128-bit product of two 64-bit numbers
|
||||
struct uint128_parts
|
||||
{
|
||||
@@ -68,14 +87,19 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc
|
||||
|
||||
/// eight bytes as a little-endian word (compilers fold this into one load on
|
||||
/// little-endian targets)
|
||||
inline std::uint64_t read_eight_bytes(const char* p) noexcept
|
||||
inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
|
||||
{
|
||||
const auto* b = reinterpret_cast<const unsigned char*>(p); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
return static_cast<std::uint64_t>(b[0]) | (static_cast<std::uint64_t>(b[1]) << 8u)
|
||||
| (static_cast<std::uint64_t>(b[2]) << 16u) | (static_cast<std::uint64_t>(b[3]) << 24u)
|
||||
| (static_cast<std::uint64_t>(b[4]) << 32u) | (static_cast<std::uint64_t>(b[5]) << 40u)
|
||||
| (static_cast<std::uint64_t>(b[6]) << 48u) | (static_cast<std::uint64_t>(b[7]) << 56u);
|
||||
}
|
||||
|
||||
/// eight bytes as a little-endian word
|
||||
inline std::uint64_t read_eight_bytes(const char* p) noexcept
|
||||
{
|
||||
return read_eight_bytes(reinterpret_cast<const unsigned char*>(p)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -219,6 +219,44 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
// scan functions
|
||||
/////////////////////
|
||||
|
||||
/// contiguous input: try to decode the 4 hex digits following `\u`
|
||||
/// directly from the input buffer via hex_codepoint(), instead of 4 calls
|
||||
/// to get(). On success, advances the adapter and the position counters
|
||||
/// exactly as those 4 get() calls would (a hex digit is never '\n', so
|
||||
/// only the flat counters move) and leaves @a current holding the last of
|
||||
/// the 4 digits, just as the last such get() would; the codepoint is
|
||||
/// written to @a out. Makes no state change and returns false - for a
|
||||
/// pending unget, fewer than 4 remaining bytes, or any of the 4 bytes not
|
||||
/// being a hex digit - so the caller falls back unchanged to the
|
||||
/// per-character loop, which then reports the same diagnostic (stopping
|
||||
/// at the first invalid digit) as before this optimization.
|
||||
bool get_codepoint_bulk(std::true_type /*bulk*/, int& out)
|
||||
{
|
||||
if (next_unget || ia.bulk_remaining() < 4)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
const char_type* const raw = ia.bulk_data();
|
||||
const int codepoint = hex_codepoint(reinterpret_cast<const unsigned char*>(raw));
|
||||
if (codepoint < 0)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
ia.bulk_skip(4);
|
||||
// a hex digit is never a newline, so only the flat counters advance
|
||||
position.chars_read_total += 4;
|
||||
position.chars_read_current_line += 4;
|
||||
current = char_traits<char_type>::to_int_type(raw[3]);
|
||||
out = codepoint;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// streaming input: no bulk fast path
|
||||
bool get_codepoint_bulk(std::false_type /*bulk*/, int& /*out*/) const noexcept
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief get codepoint from 4 hex characters following `\u`
|
||||
|
||||
@@ -238,6 +276,14 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
{
|
||||
// this function only makes sense after reading `\u`
|
||||
JSON_ASSERT(current == 'u');
|
||||
|
||||
// contiguous input: decode all 4 hex digits directly from the buffer
|
||||
int fast_codepoint = 0;
|
||||
if (get_codepoint_bulk(std::integral_constant<bool, bulk_scan> {}, fast_codepoint))
|
||||
{
|
||||
return fast_codepoint;
|
||||
}
|
||||
|
||||
int codepoint = 0;
|
||||
|
||||
const auto factors = { 12u, 8u, 4u, 0u };
|
||||
|
||||
@@ -8,10 +8,12 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint64_t
|
||||
#include <cstdint> // uint64_t, uint8_t
|
||||
#include <cstring> // memcpy
|
||||
|
||||
#include <nlohmann/detail/bit_ops.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in external
|
||||
@@ -69,18 +71,12 @@ inline std::size_t find_string_special(const unsigned char* data, std::size_t n)
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t word = 0;
|
||||
std::memcpy(&word, data + i, sizeof(word));
|
||||
if (swar_string_special(word) != 0)
|
||||
const std::uint64_t special = swar_string_special(read_eight_bytes(data + i));
|
||||
if (special != 0)
|
||||
{
|
||||
// a special byte is in this word; locate it (endian-agnostic)
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
if (is_string_special(data[i + j]))
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
// the lowest flagged byte is the first special one: the borrows of
|
||||
// the subtractions can only flag bytes above a true hit
|
||||
return i + (static_cast<std::size_t>(count_trailing_zeros(special)) / 8);
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
@@ -114,8 +110,7 @@ inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, data + i, sizeof(v));
|
||||
const std::uint64_t v = read_eight_bytes(data + i);
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
||||
const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F)
|
||||
@@ -126,7 +121,9 @@ inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_
|
||||
| (v & high); // >= 0x80
|
||||
if (stop != 0)
|
||||
{
|
||||
break;
|
||||
// the lowest flagged byte is the first one to stop at (see
|
||||
// find_string_special())
|
||||
return i + (static_cast<std::size_t>(count_trailing_zeros(stop)) / 8);
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
@@ -253,12 +250,18 @@ inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t
|
||||
{
|
||||
break; // end of buffer, or a quote/escape/control byte
|
||||
}
|
||||
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
||||
if (seq == 0)
|
||||
// a run of multi-byte sequences (e.g. CJK text) is validated sequence
|
||||
// by sequence without searching for the next special byte in between
|
||||
do
|
||||
{
|
||||
break; // ill-formed or truncated: let the byte path diagnose it
|
||||
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
||||
if (seq == 0)
|
||||
{
|
||||
return pos; // ill-formed or truncated: let the byte path diagnose it
|
||||
}
|
||||
pos += seq;
|
||||
}
|
||||
pos += seq;
|
||||
while (pos < n && data[pos] >= 0x80u);
|
||||
}
|
||||
return pos;
|
||||
}
|
||||
@@ -273,8 +276,7 @@ inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, data + i, sizeof(v));
|
||||
const std::uint64_t v = read_eight_bytes(data + i);
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull;
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull;
|
||||
const std::uint64_t hit = ((q - ones) & ~q & high)
|
||||
@@ -282,14 +284,8 @@ inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t
|
||||
| ((v - 0x2020202020202020ull) & ~v & high);
|
||||
if (hit != 0)
|
||||
{
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
const unsigned char c = data[i + j];
|
||||
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
// the lowest flagged byte is the first delimiter (see find_string_special())
|
||||
return i + (static_cast<std::size_t>(count_trailing_zeros(hit)) / 8);
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
@@ -320,5 +316,50 @@ inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noe
|
||||
return scalar_string_bulk_run(data, n);
|
||||
}
|
||||
|
||||
// Decode the 4 hex digits at [data, data+4) - the digits following a `\u`
|
||||
// escape - into a codepoint 0x0000..0xFFFF via one table lookup per byte
|
||||
// (after yyjson's read_hex_u16), or return -1 if any of the 4 bytes is not a
|
||||
// hex digit ('0'..'9', 'A'..'F', 'a'..'f'). The caller must already have
|
||||
// checked that 4 bytes are available; used by lexer::get_codepoint()'s
|
||||
// contiguous fast path. On -1 it falls back to the byte-at-a-time loop, which
|
||||
// stops at the first invalid digit, so the reported error and position are
|
||||
// unaffected by this fast path.
|
||||
inline int hex_codepoint(const unsigned char* data) noexcept
|
||||
{
|
||||
static const std::array<std::uint8_t, 256> hex_digit_table = // NOLINT(cppcoreguidelines-avoid-non-const-global-variables)
|
||||
{
|
||||
{
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 00..0F
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 10..1F
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 20..2F
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 30..3F ('0'..'9')
|
||||
0xFF, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 40..4F ('A'..'F')
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 50..5F
|
||||
0xFF, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 60..6F ('a'..'f')
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 70..7F
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 80..8F
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 90..9F
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // A0..AF
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // B0..BF
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // C0..CF
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // D0..DF
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // E0..EF
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF // F0..FF
|
||||
}
|
||||
};
|
||||
|
||||
const std::uint8_t d0 = hex_digit_table[data[0]];
|
||||
const std::uint8_t d1 = hex_digit_table[data[1]];
|
||||
const std::uint8_t d2 = hex_digit_table[data[2]];
|
||||
const std::uint8_t d3 = hex_digit_table[data[3]];
|
||||
// every valid digit is <= 0xF; the combined OR only exceeds it if at
|
||||
// least one of the four bytes was not a hex digit (looked up as 0xFF)
|
||||
if ((d0 | d1 | d2 | d3) > 0x0F)
|
||||
{
|
||||
return -1;
|
||||
}
|
||||
return (d0 << 12) | (d1 << 8) | (d2 << 4) | d3;
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -8546,6 +8546,25 @@ inline int count_leading_zeros(std::uint64_t x) noexcept
|
||||
#endif
|
||||
}
|
||||
|
||||
/// number of trailing zero bits of x (x != 0)
|
||||
inline int count_trailing_zeros(std::uint64_t x) noexcept
|
||||
{
|
||||
#if defined(__GNUC__) || defined(__clang__)
|
||||
return __builtin_ctzll(x);
|
||||
#else
|
||||
int n = 0;
|
||||
for (int shift = 32; shift != 0; shift >>= 1)
|
||||
{
|
||||
if ((x << (64 - shift)) == 0)
|
||||
{
|
||||
n += shift;
|
||||
x >>= shift;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
#endif
|
||||
}
|
||||
|
||||
/// the 128-bit product of two 64-bit numbers
|
||||
struct uint128_parts
|
||||
{
|
||||
@@ -8575,15 +8594,20 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc
|
||||
|
||||
/// eight bytes as a little-endian word (compilers fold this into one load on
|
||||
/// little-endian targets)
|
||||
inline std::uint64_t read_eight_bytes(const char* p) noexcept
|
||||
inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
|
||||
{
|
||||
const auto* b = reinterpret_cast<const unsigned char*>(p); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
return static_cast<std::uint64_t>(b[0]) | (static_cast<std::uint64_t>(b[1]) << 8u)
|
||||
| (static_cast<std::uint64_t>(b[2]) << 16u) | (static_cast<std::uint64_t>(b[3]) << 24u)
|
||||
| (static_cast<std::uint64_t>(b[4]) << 32u) | (static_cast<std::uint64_t>(b[5]) << 40u)
|
||||
| (static_cast<std::uint64_t>(b[6]) << 48u) | (static_cast<std::uint64_t>(b[7]) << 56u);
|
||||
}
|
||||
|
||||
/// eight bytes as a little-endian word
|
||||
inline std::uint64_t read_eight_bytes(const char* p) noexcept
|
||||
{
|
||||
return read_eight_bytes(reinterpret_cast<const unsigned char*>(p)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -9668,10 +9692,13 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint64_t
|
||||
#include <cstdint> // uint64_t, uint8_t
|
||||
#include <cstring> // memcpy
|
||||
|
||||
// #include <nlohmann/detail/bit_ops.hpp>
|
||||
|
||||
// #include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
|
||||
@@ -9730,18 +9757,12 @@ inline std::size_t find_string_special(const unsigned char* data, std::size_t n)
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t word = 0;
|
||||
std::memcpy(&word, data + i, sizeof(word));
|
||||
if (swar_string_special(word) != 0)
|
||||
const std::uint64_t special = swar_string_special(read_eight_bytes(data + i));
|
||||
if (special != 0)
|
||||
{
|
||||
// a special byte is in this word; locate it (endian-agnostic)
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
if (is_string_special(data[i + j]))
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
// the lowest flagged byte is the first special one: the borrows of
|
||||
// the subtractions can only flag bytes above a true hit
|
||||
return i + (static_cast<std::size_t>(count_trailing_zeros(special)) / 8);
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
@@ -9775,8 +9796,7 @@ inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, data + i, sizeof(v));
|
||||
const std::uint64_t v = read_eight_bytes(data + i);
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
||||
const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F)
|
||||
@@ -9787,7 +9807,9 @@ inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_
|
||||
| (v & high); // >= 0x80
|
||||
if (stop != 0)
|
||||
{
|
||||
break;
|
||||
// the lowest flagged byte is the first one to stop at (see
|
||||
// find_string_special())
|
||||
return i + (static_cast<std::size_t>(count_trailing_zeros(stop)) / 8);
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
@@ -9914,12 +9936,18 @@ inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t
|
||||
{
|
||||
break; // end of buffer, or a quote/escape/control byte
|
||||
}
|
||||
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
||||
if (seq == 0)
|
||||
// a run of multi-byte sequences (e.g. CJK text) is validated sequence
|
||||
// by sequence without searching for the next special byte in between
|
||||
do
|
||||
{
|
||||
break; // ill-formed or truncated: let the byte path diagnose it
|
||||
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
||||
if (seq == 0)
|
||||
{
|
||||
return pos; // ill-formed or truncated: let the byte path diagnose it
|
||||
}
|
||||
pos += seq;
|
||||
}
|
||||
pos += seq;
|
||||
while (pos < n && data[pos] >= 0x80u);
|
||||
}
|
||||
return pos;
|
||||
}
|
||||
@@ -9934,8 +9962,7 @@ inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, data + i, sizeof(v));
|
||||
const std::uint64_t v = read_eight_bytes(data + i);
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull;
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull;
|
||||
const std::uint64_t hit = ((q - ones) & ~q & high)
|
||||
@@ -9943,14 +9970,8 @@ inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t
|
||||
| ((v - 0x2020202020202020ull) & ~v & high);
|
||||
if (hit != 0)
|
||||
{
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
const unsigned char c = data[i + j];
|
||||
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
// the lowest flagged byte is the first delimiter (see find_string_special())
|
||||
return i + (static_cast<std::size_t>(count_trailing_zeros(hit)) / 8);
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
@@ -9981,6 +10002,51 @@ inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noe
|
||||
return scalar_string_bulk_run(data, n);
|
||||
}
|
||||
|
||||
// Decode the 4 hex digits at [data, data+4) - the digits following a `\u`
|
||||
// escape - into a codepoint 0x0000..0xFFFF via one table lookup per byte
|
||||
// (after yyjson's read_hex_u16), or return -1 if any of the 4 bytes is not a
|
||||
// hex digit ('0'..'9', 'A'..'F', 'a'..'f'). The caller must already have
|
||||
// checked that 4 bytes are available; used by lexer::get_codepoint()'s
|
||||
// contiguous fast path. On -1 it falls back to the byte-at-a-time loop, which
|
||||
// stops at the first invalid digit, so the reported error and position are
|
||||
// unaffected by this fast path.
|
||||
inline int hex_codepoint(const unsigned char* data) noexcept
|
||||
{
|
||||
static const std::array<std::uint8_t, 256> hex_digit_table = // NOLINT(cppcoreguidelines-avoid-non-const-global-variables)
|
||||
{
|
||||
{
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 00..0F
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 10..1F
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 20..2F
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 30..3F ('0'..'9')
|
||||
0xFF, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 40..4F ('A'..'F')
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 50..5F
|
||||
0xFF, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 60..6F ('a'..'f')
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 70..7F
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 80..8F
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 90..9F
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // A0..AF
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // B0..BF
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // C0..CF
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // D0..DF
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // E0..EF
|
||||
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF // F0..FF
|
||||
}
|
||||
};
|
||||
|
||||
const std::uint8_t d0 = hex_digit_table[data[0]];
|
||||
const std::uint8_t d1 = hex_digit_table[data[1]];
|
||||
const std::uint8_t d2 = hex_digit_table[data[2]];
|
||||
const std::uint8_t d3 = hex_digit_table[data[3]];
|
||||
// every valid digit is <= 0xF; the combined OR only exceeds it if at
|
||||
// least one of the four bytes was not a hex digit (looked up as 0xFF)
|
||||
if ((d0 | d1 | d2 | d3) > 0x0F)
|
||||
{
|
||||
return -1;
|
||||
}
|
||||
return (d0 << 12) | (d1 << 8) | (d2 << 4) | d3;
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -10185,6 +10251,44 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
// scan functions
|
||||
/////////////////////
|
||||
|
||||
/// contiguous input: try to decode the 4 hex digits following `\u`
|
||||
/// directly from the input buffer via hex_codepoint(), instead of 4 calls
|
||||
/// to get(). On success, advances the adapter and the position counters
|
||||
/// exactly as those 4 get() calls would (a hex digit is never '\n', so
|
||||
/// only the flat counters move) and leaves @a current holding the last of
|
||||
/// the 4 digits, just as the last such get() would; the codepoint is
|
||||
/// written to @a out. Makes no state change and returns false - for a
|
||||
/// pending unget, fewer than 4 remaining bytes, or any of the 4 bytes not
|
||||
/// being a hex digit - so the caller falls back unchanged to the
|
||||
/// per-character loop, which then reports the same diagnostic (stopping
|
||||
/// at the first invalid digit) as before this optimization.
|
||||
bool get_codepoint_bulk(std::true_type /*bulk*/, int& out)
|
||||
{
|
||||
if (next_unget || ia.bulk_remaining() < 4)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
const char_type* const raw = ia.bulk_data();
|
||||
const int codepoint = hex_codepoint(reinterpret_cast<const unsigned char*>(raw));
|
||||
if (codepoint < 0)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
ia.bulk_skip(4);
|
||||
// a hex digit is never a newline, so only the flat counters advance
|
||||
position.chars_read_total += 4;
|
||||
position.chars_read_current_line += 4;
|
||||
current = char_traits<char_type>::to_int_type(raw[3]);
|
||||
out = codepoint;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// streaming input: no bulk fast path
|
||||
bool get_codepoint_bulk(std::false_type /*bulk*/, int& /*out*/) const noexcept
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief get codepoint from 4 hex characters following `\u`
|
||||
|
||||
@@ -10204,6 +10308,14 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
{
|
||||
// this function only makes sense after reading `\u`
|
||||
JSON_ASSERT(current == 'u');
|
||||
|
||||
// contiguous input: decode all 4 hex digits directly from the buffer
|
||||
int fast_codepoint = 0;
|
||||
if (get_codepoint_bulk(std::integral_constant<bool, bulk_scan> {}, fast_codepoint))
|
||||
{
|
||||
return fast_codepoint;
|
||||
}
|
||||
|
||||
int codepoint = 0;
|
||||
|
||||
const auto factors = { 12u, 8u, 4u, 0u };
|
||||
|
||||
@@ -17,6 +17,7 @@ using nlohmann::json;
|
||||
#include <cstdint> // uint32_t, uint64_t
|
||||
#include <cstdlib> // strtod
|
||||
#include <cstring> // memcpy
|
||||
#include <random> // mt19937
|
||||
#include <sstream> // stringstream
|
||||
#include <string> // string
|
||||
#include <utility> // pair
|
||||
@@ -663,6 +664,184 @@ TEST_CASE("lexer string fast path")
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("lexer escape fast path")
|
||||
{
|
||||
// json::accept() never throws, so this section stays covered without
|
||||
// exceptions; it pins which of the cases below are valid/invalid and
|
||||
// checks the contiguous and streaming paths agree on that classification.
|
||||
SECTION("accept() parity")
|
||||
{
|
||||
const std::vector<std::pair<std::string, bool>> cases =
|
||||
{
|
||||
{"\\u0041", true}, {"\\u00e4", true}, {"\\u00E4", true},
|
||||
{"\\uD83D\\uDE00", true},
|
||||
{"\\u12", false}, {"\\u12G4", false}, {"\\uXYZW", false},
|
||||
{"\\uD800", false}, {"\\uD800A", false}, {"\\uD800\\u0041", false},
|
||||
{"\\uDC00", false}, {"\\u", false}
|
||||
};
|
||||
|
||||
for (const auto& c : cases)
|
||||
{
|
||||
for (const std::size_t offset :
|
||||
{
|
||||
std::size_t{0}, std::size_t{9}
|
||||
})
|
||||
{
|
||||
const std::string doc = "[\"" + std::string(offset, 'a') + c.first + "\"]";
|
||||
CAPTURE(doc);
|
||||
CHECK(json::accept(doc) == c.second);
|
||||
std::stringstream ss(doc);
|
||||
CHECK(json::accept(ss) == c.second);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
// the full outcome of parsing @a doc: the parsed value, or the exact
|
||||
// error message, so a mismatch in either is caught
|
||||
const auto outcome = [](const std::string & doc, bool streaming) -> std::string
|
||||
{
|
||||
try
|
||||
{
|
||||
if (streaming)
|
||||
{
|
||||
std::stringstream ss(doc);
|
||||
const json j = json::parse(ss);
|
||||
return j.dump();
|
||||
}
|
||||
const json j = json::parse(doc);
|
||||
return j.dump();
|
||||
}
|
||||
catch (const json::exception& e)
|
||||
{
|
||||
return {e.what()};
|
||||
}
|
||||
};
|
||||
|
||||
SECTION("contiguous vs streaming parity")
|
||||
{
|
||||
const std::vector<std::string> escapes =
|
||||
{
|
||||
"\\u0041", // "A"
|
||||
"\\u00e4", // "ä" (lowercase hex)
|
||||
"\\u00E4", // "ä" (uppercase hex)
|
||||
"\\uD83D\\uDE00", // valid surrogate pair (an emoji)
|
||||
"\\u12", // truncated: only 2 hex digits before the closing quote
|
||||
"\\u12G4", // invalid hex digit at the 3rd position
|
||||
"\\uXYZW", // all 4 bytes invalid
|
||||
"\\uD800", // lone high surrogate, string ends right after
|
||||
"\\uD800A", // high surrogate not followed by another \u escape
|
||||
"\\uD800\\u0041", // high surrogate followed by \u, but not a low surrogate
|
||||
"\\uDC00", // lone low surrogate
|
||||
"\\u", // '\u' with nothing after (closing quote right away)
|
||||
};
|
||||
|
||||
// once at the start of the string and once past the first 8-byte SWAR
|
||||
// word of the outer string_bulk_run, so the escape is reached both
|
||||
// right after the opening quote and mid-run
|
||||
for (const auto& escape : escapes)
|
||||
{
|
||||
for (const std::size_t offset :
|
||||
{
|
||||
std::size_t{0}, std::size_t{9}
|
||||
})
|
||||
{
|
||||
const std::string doc = "[\"" + std::string(offset, 'a') + escape + "\"]";
|
||||
CAPTURE(doc);
|
||||
CHECK(outcome(doc, false) == outcome(doc, true));
|
||||
}
|
||||
|
||||
// the escape is the last thing before end of input: no closing
|
||||
// quote at all
|
||||
const std::string truncated_doc = "[\"" + escape;
|
||||
CAPTURE(truncated_doc);
|
||||
CHECK(outcome(truncated_doc, false) == outcome(truncated_doc, true));
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("truncated \\u escape at every distance from the end of input")
|
||||
{
|
||||
// ia.bulk_remaining() must correctly report fewer than 4 bytes for
|
||||
// every possible count of trailing hex-looking bytes (0, 1, 2, or 3)
|
||||
// before end of input, so the fast path declines and the byte path
|
||||
// alone reports the "must be followed by 4 hex digits" error, at the
|
||||
// same position, in every case
|
||||
for (const std::string& tail :
|
||||
{
|
||||
std::string{}, std::string("1"), std::string("12"), std::string("123")
|
||||
})
|
||||
{
|
||||
const std::string doc = "[\"\\u" + tail;
|
||||
CAPTURE(doc);
|
||||
CHECK(outcome(doc, false) == outcome(doc, true));
|
||||
CHECK(outcome(doc, false).find("must be followed by 4 hex digits") != std::string::npos);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("invalid hex digit at every position of the 4")
|
||||
{
|
||||
// the fast path must decline for *any* invalid byte among the 4, not
|
||||
// just the first, and the byte path must then stop at exactly that
|
||||
// position - same as it always has
|
||||
for (std::size_t bad_pos = 0; bad_pos < 4; ++bad_pos)
|
||||
{
|
||||
std::string digits = "1234";
|
||||
digits[bad_pos] = 'g'; // not a hex digit
|
||||
const std::string doc = "[\"\\u" + digits + "\"]";
|
||||
CAPTURE(doc);
|
||||
CHECK(outcome(doc, false) == outcome(doc, true));
|
||||
CHECK(outcome(doc, false).find("must be followed by 4 hex digits") != std::string::npos);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("random escapes")
|
||||
{
|
||||
// A seeded PRNG builds the 4 bytes following `\u` from a mix of hex
|
||||
// digits and non-hex bytes, at varying distances from the start of
|
||||
// the string, to compare the two scanners on many more shapes than
|
||||
// are practical to enumerate by hand.
|
||||
std::mt19937 gen(7654321); // NOLINT(cert-msc32-c,cert-msc51-cpp)
|
||||
const std::string hex_alphabet = "0123456789AaBbCcDdEeFf";
|
||||
std::uniform_int_distribution<std::size_t> pick_hex(0, hex_alphabet.size() - 1);
|
||||
std::uniform_int_distribution<int> pick_byte(1, 255); // never NUL
|
||||
std::uniform_int_distribution<int> pick_is_hex(0, 4); // 4-in-5 chance of a hex digit
|
||||
std::uniform_int_distribution<std::size_t> pick_offset(0, 12);
|
||||
|
||||
std::vector<std::string> mismatches;
|
||||
for (int iter = 0; iter < 3000; ++iter)
|
||||
{
|
||||
std::string digits;
|
||||
for (int i = 0; i < 4; ++i)
|
||||
{
|
||||
if (pick_is_hex(gen) != 0)
|
||||
{
|
||||
digits += hex_alphabet[pick_hex(gen)];
|
||||
}
|
||||
else
|
||||
{
|
||||
char c = static_cast<char>(pick_byte(gen));
|
||||
if (c == '"' || c == '\\')
|
||||
{
|
||||
// keep the string well-formed apart from the escape
|
||||
// itself, so any mismatch is attributable to the \u
|
||||
// handling and not to an unrelated quote/escape
|
||||
c = 'z';
|
||||
}
|
||||
digits += c;
|
||||
}
|
||||
}
|
||||
const std::string doc = "[\"" + std::string(pick_offset(gen), 'a') + "\\u" + digits + "\"]";
|
||||
if (outcome(doc, false) != outcome(doc, true))
|
||||
{
|
||||
mismatches.push_back(doc);
|
||||
}
|
||||
}
|
||||
CAPTURE(mismatches);
|
||||
CHECK(mismatches.empty());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
TEST_CASE("parse_float_fast declines what it cannot convert exactly")
|
||||
{
|
||||
// The lexer only hands well-formed numbers to parse_float_fast, so the
|
||||
@@ -1315,3 +1494,104 @@ TEST_CASE("Eisel-Lemire float conversion")
|
||||
"[json.exception.out_of_range.406] number overflow parsing '1.7976931348623159e308'", json::out_of_range&);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("string scanning kernels")
|
||||
{
|
||||
// the word-at-a-time kernels must stop exactly where a byte-by-byte scan
|
||||
// stops, for any content, length, and alignment
|
||||
const auto reference_special = [](const unsigned char* data, std::size_t n)
|
||||
{
|
||||
std::size_t i = 0;
|
||||
while (i < n && !nlohmann::detail::is_string_special(data[i]))
|
||||
{
|
||||
++i;
|
||||
}
|
||||
return i;
|
||||
};
|
||||
const auto reference_copyable = [](const unsigned char* data, std::size_t n)
|
||||
{
|
||||
std::size_t i = 0;
|
||||
while (i < n && nlohmann::detail::is_ascii_copyable(data[i]))
|
||||
{
|
||||
++i;
|
||||
}
|
||||
return i;
|
||||
};
|
||||
const auto reference_bulk_run = [](const unsigned char* data, std::size_t n)
|
||||
{
|
||||
std::size_t i = 0;
|
||||
while (i < n)
|
||||
{
|
||||
if (data[i] < 0x80u)
|
||||
{
|
||||
if (nlohmann::detail::is_string_special(data[i]))
|
||||
{
|
||||
break;
|
||||
}
|
||||
++i;
|
||||
continue;
|
||||
}
|
||||
const std::size_t seq = nlohmann::detail::validate_one_utf8(data + i, n - i);
|
||||
if (seq == 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
i += seq;
|
||||
}
|
||||
return i;
|
||||
};
|
||||
|
||||
// pieces: ordinary ASCII, stops, DEL, well-formed sequences of every
|
||||
// length, and ill-formed or truncated ones
|
||||
const std::vector<std::string> pieces =
|
||||
{
|
||||
"a", "Z", " ", "~", "0123456789", "\"", "\\", std::string(1, '\0'), "\n", "\x1F", "\x7F",
|
||||
"\xC3\xA4", "\xE2\x82\xAC", "\xE6\x97\xA5\xE6\x9C\xAC", "\xF0\x9F\x98\x80", "\xED\x9F\xBF",
|
||||
"\x80", "\xC0\x80", "\xC3", "\xE2\x82", "\xED\xA0\x80", "\xF4\x90\x80\x80", "\xFF",
|
||||
};
|
||||
std::uint64_t state = 5295;
|
||||
const auto next = [&state]()
|
||||
{
|
||||
state ^= state << 13u;
|
||||
state ^= state >> 7u;
|
||||
state ^= state << 17u;
|
||||
return state;
|
||||
};
|
||||
// the upper half as a 32-bit value: converts to std::size_t implicitly on
|
||||
// every platform (a cast of std::uint64_t is useless where both are the
|
||||
// same type, and required where std::size_t is 32 bits wide)
|
||||
const auto next_small = [&next]()
|
||||
{
|
||||
return static_cast<std::uint32_t>(next() >> 32u);
|
||||
};
|
||||
for (int round = 0; round < 100000; ++round)
|
||||
{
|
||||
// mostly ordinary text, so that runs span several words
|
||||
std::string text(next_small() % 8u, '.');
|
||||
const std::size_t count = next_small() % 12u;
|
||||
for (std::size_t k = 0; k < count; ++k)
|
||||
{
|
||||
const std::size_t p = (next() % 4 == 0) ? next_small() % pieces.size() : 0;
|
||||
text += pieces[p];
|
||||
text += std::string(next_small() % 10u, 'x');
|
||||
}
|
||||
const auto* data = reinterpret_cast<const unsigned char*>(text.data()); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
for (std::size_t offset = 0; offset < 3 && offset <= text.size(); ++offset)
|
||||
{
|
||||
const std::size_t n = text.size() - offset;
|
||||
CAPTURE(text);
|
||||
CAPTURE(offset);
|
||||
CHECK(nlohmann::detail::find_string_special(data + offset, n) == reference_special(data + offset, n));
|
||||
CHECK(nlohmann::detail::find_ascii_copyable_run(data + offset, n) == reference_copyable(data + offset, n));
|
||||
CHECK(nlohmann::detail::scalar_string_bulk_run(data + offset, n) == reference_bulk_run(data + offset, n));
|
||||
}
|
||||
}
|
||||
|
||||
// the trailing-zero count, whichever implementation the compiler gets
|
||||
for (int k = 0; k < 64; ++k)
|
||||
{
|
||||
const std::uint64_t bit = std::uint64_t{1} << k;
|
||||
CHECK(nlohmann::detail::count_trailing_zeros(bit) == k);
|
||||
CHECK(nlohmann::detail::count_trailing_zeros(bit | (bit << 1u) | 0x8000000000000000u) == k);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user