mirror of
https://github.com/nlohmann/json.git
synced 2026-10-06 22:47:13 +00:00
Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
66e72cc129 | ||
|
|
cf2af0396d | ||
|
|
44fba02ec2 | ||
|
|
42f8043413 | ||
|
|
606c1e710b |
No files matched your search
@@ -49,6 +49,11 @@ whether or not the new parse succeeds; take fresh views from [`root()`](root.md)
|
||||
`input` is borrowed or owned by the same rules as [`parse()`](parse.md#notes); a document can borrow on one call and
|
||||
own on the next, since ownership is decided freshly each time.
|
||||
|
||||
Reusing a document matters most for large inputs: the operating system provides the memory of a fresh node index one
|
||||
page at a time, and every page costs a page fault the first time it is written. On x86-64 Linux (4 KiB pages), parsing
|
||||
a 55 MB document into a reused document took about 40 % less time than parsing it into a fresh one. Programs that parse
|
||||
many documents of similar size should therefore keep one document and call `read()`.
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
|
||||
@@ -5,13 +5,17 @@
|
||||
```
|
||||
|
||||
When defined on x86-64, the parser of [`basic_json_document`](../basic_json_document/index.md)
|
||||
(`<nlohmann/json_view.hpp>`) validates non-ASCII text in strings with SSSE3, 16 bytes at a time, using the "lookup4"
|
||||
algorithm of [simdjson](https://github.com/simdjson/simdjson). Without it, non-ASCII text is validated one UTF-8
|
||||
sequence at a time on x86-64; on AArch64, the vector check uses NEON and is always on.
|
||||
(`<nlohmann/json_view.hpp>`) validates non-ASCII text in strings with SSSE3 without asking the CPU first.
|
||||
|
||||
SSSE3 is not part of the x86-64 baseline, so the code must be compiled for it: define the macro only together with a
|
||||
compiler option that enables SSSE3 (e.g. `-mssse3`, or `-march=` with a CPU that has it), and only for programs that
|
||||
run on such CPUs. The same input is accepted or rejected either way; only the speed of non-ASCII text differs.
|
||||
By default, the parser checks once at run time whether the CPU has SSSE3 (all x86-64 CPUs since about 2011 have it)
|
||||
and then validates non-ASCII text 16 bytes at a time, using the "lookup4" algorithm of
|
||||
[simdjson](https://github.com/simdjson/simdjson); on CPUs without SSSE3, it validates one UTF-8 sequence at a time.
|
||||
The vector check is compiled for SSSE3 with a function attribute (GCC 4.9 and later, Clang), so this needs no compiler
|
||||
option. With MSVC, the check uses `__cpuid`. On AArch64, the vector check uses NEON and is always on.
|
||||
|
||||
Define the macro only together with a compiler option that enables SSSE3 (e.g. `-mssse3`, or `-march=` with a CPU that
|
||||
has it), and only for programs that run on such CPUs. It saves the check of the CPU, which costs little. The same
|
||||
input is accepted or rejected either way; only the speed of non-ASCII text differs.
|
||||
|
||||
!!! warning "Define consistently"
|
||||
|
||||
|
||||
@@ -86,8 +86,9 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc
|
||||
}
|
||||
|
||||
/// eight bytes as a little-endian word (compilers fold this into one load on
|
||||
/// little-endian targets)
|
||||
inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
|
||||
/// little-endian targets; always inlined, as GCC otherwise calls it in the
|
||||
/// number loops)
|
||||
JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
|
||||
{
|
||||
return static_cast<std::uint64_t>(b[0]) | (static_cast<std::uint64_t>(b[1]) << 8u)
|
||||
| (static_cast<std::uint64_t>(b[2]) << 16u) | (static_cast<std::uint64_t>(b[3]) << 24u)
|
||||
@@ -96,7 +97,7 @@ inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
|
||||
}
|
||||
|
||||
/// eight bytes as a little-endian word
|
||||
inline std::uint64_t read_eight_bytes(const char* p) noexcept
|
||||
JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const char* p) noexcept
|
||||
{
|
||||
return read_eight_bytes(reinterpret_cast<const unsigned char*>(p)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
}
|
||||
|
||||
@@ -21,6 +21,21 @@
|
||||
#include <cstdlib> // _byteswap_uint64
|
||||
#endif
|
||||
|
||||
// SSE2 (every x86-64 CPU) and NEON (every 64-bit Arm CPU) convert the 16
|
||||
// digits of a double at once
|
||||
#if defined(__x86_64__) || (defined(_M_X64) && !defined(_M_ARM64EC))
|
||||
#include <emmintrin.h>
|
||||
#define JSON_DTOA_SSE2 1
|
||||
#define JSON_DTOA_NEON 0
|
||||
#elif (defined(__aarch64__) || defined(_M_ARM64)) && !defined(_M_ARM64EC) && !defined(__ARM_BIG_ENDIAN)
|
||||
#include <arm_neon.h>
|
||||
#define JSON_DTOA_SSE2 0
|
||||
#define JSON_DTOA_NEON 1
|
||||
#else
|
||||
#define JSON_DTOA_SSE2 0
|
||||
#define JSON_DTOA_NEON 0
|
||||
#endif
|
||||
|
||||
#include <nlohmann/detail/conversions/zmij.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
@@ -1253,6 +1268,203 @@ inline char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept
|
||||
return end + (three ? 5 : 4);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the shortest decimal of a positive double (Zmij), as write_decimal()
|
||||
writes it
|
||||
|
||||
For a normal double, the shorter candidate has 15 or 16 digits: they are
|
||||
converted at once (two halves of eight digits) and followed by the digit
|
||||
after them, if there is one, without the multiplication and division by 10
|
||||
that counting the digits of one number would take. The fixed layouts move
|
||||
the digits after the point by one byte.
|
||||
|
||||
@return a pointer past the text; up to 41 bytes at @a first are written
|
||||
(some beyond the returned end)
|
||||
*/
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcept
|
||||
{
|
||||
const std::uint64_t sig = d.integral;
|
||||
if (JSON_HEDLEY_UNLIKELY(sig < 100000000000000u || sig >= 10000000000000000u))
|
||||
{
|
||||
// (subnormals)
|
||||
return d.has_digit ? write_decimal(first, (sig * 10) + d.digit, d.exponent) : write_decimal(first, sig, d.exponent + 1);
|
||||
}
|
||||
const bool sixteen = sig >= 1000000000000000u; // (else 15 digits)
|
||||
const int last = d.has_digit ? d.digit : 0;
|
||||
const std::uint64_t upper = sig / 100000000u;
|
||||
#if JSON_DTOA_SSE2
|
||||
// NOLINTBEGIN(portability-simd-intrinsics)
|
||||
// the two halves in the 64-bit lanes, each as abcd * 2^32 + efgh, then as
|
||||
// bytes (as eight_digit_bytes(), one lane each)
|
||||
const __m128i x = _mm_set_epi64x(static_cast<long long>(sig - (upper * 100000000u)), static_cast<long long>(upper));
|
||||
const __m128i abcd = _mm_srli_epi64(_mm_mul_epu32(x, _mm_set1_epi64x(109951163)), 40); // 2^40 / 10000 + 1
|
||||
const __m128i abcd_efgh = _mm_add_epi64(x, _mm_mul_epu32(abcd, _mm_set1_epi64x(4294957296))); // 2^32 - 10000
|
||||
// 32-bit lanes in the order of the text: abcd, efgh of both halves
|
||||
const __m128i fours = _mm_shuffle_epi32(abcd_efgh, _MM_SHUFFLE(2, 3, 0, 1));
|
||||
const __m128i ab = _mm_srli_epi16(_mm_mulhi_epu16(fours, _mm_set1_epi32(5243)), 3);
|
||||
const __m128i ab_cd = _mm_or_si128(_mm_slli_epi32(_mm_sub_epi16(fours, _mm_mullo_epi16(ab, _mm_set1_epi32(100))), 16), ab);
|
||||
// 16-bit lanes ab (< 100) -> bytes a, b: 256 * ab - 2559 * (ab / 10)
|
||||
const __m128i bytes = _mm_sub_epi16(_mm_slli_epi16(ab_cd, 8), _mm_mullo_epi16(_mm_set1_epi16(2559), _mm_mulhi_epu16(ab_cd, _mm_set1_epi16(6554))));
|
||||
// the last digit that is not 0 (sig is not 0)
|
||||
const auto nonzero = static_cast<std::uint64_t>(_mm_movemask_epi8(_mm_cmpgt_epi8(bytes, _mm_setzero_si128())));
|
||||
const int digits = 63 - count_leading_zeros(nonzero) + (sixteen ? 1 : 0); // without trailing zeros
|
||||
const __m128i chars = _mm_add_epi8(bytes, _mm_set1_epi8('0'));
|
||||
// the 16 characters from the first digit
|
||||
const __m128i s = sixteen ? chars : _mm_or_si128(_mm_srli_si128(chars, 1), _mm_slli_si128(_mm_cvtsi32_si128('0' + last), 15));
|
||||
const char s16 = static_cast<char>(sixteen ? '0' + last : '0'); // the 17th
|
||||
const auto store_16 = [&s](char* p) noexcept
|
||||
{
|
||||
std::memcpy(p, &s, 16);
|
||||
};
|
||||
const char first_digit = static_cast<char>(_mm_cvtsi128_si32(s));
|
||||
// NOLINTEND(portability-simd-intrinsics)
|
||||
#elif JSON_DTOA_NEON
|
||||
// as with SSE2: the halves in 32-bit lanes, then abcd, efgh of both
|
||||
const uint32x2_t halves = vcreate_u32(upper | ((sig - (upper * 100000000u)) << 32u));
|
||||
const uint32x2_t abcd = vmovn_u64(vshrq_n_u64(vmull_n_u32(halves, static_cast<std::uint32_t>(((std::uint64_t{1} << 40u) / 10000u) + 1u)), 40));
|
||||
const uint32x2_t efgh = vmls_n_u32(halves, abcd, 10000u);
|
||||
const uint32x4_t fours = vcombine_u32(vzip1_u32(abcd, efgh), vzip2_u32(abcd, efgh));
|
||||
const uint32x4_t ab = vshrq_n_u32(vmulq_n_u32(fours, 5243u), 19);
|
||||
const uint16x8_t ab_cd = vreinterpretq_u16_u32(vorrq_u32(ab, vshlq_n_u32(vmlsq_n_u32(fours, ab, 100u), 16)));
|
||||
const uint16x8_t tens = vshrq_n_u16(vmulq_n_u16(ab_cd, 103u), 10);
|
||||
const uint8x16_t bytes = vreinterpretq_u8_u16(vorrq_u16(tens, vshlq_n_u16(vmlsq_n_u16(ab_cd, tens, 10u), 8)));
|
||||
// the last digit that is not 0 (sig is not 0): a nibble per byte
|
||||
const std::uint64_t nonzero = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(bytes, bytes)), 4)), 0);
|
||||
const int digits = ((63 - count_leading_zeros(nonzero)) / 4) + (sixteen ? 1 : 0); // without trailing zeros
|
||||
const uint8x16_t chars = vaddq_u8(bytes, vdupq_n_u8('0'));
|
||||
// the 16 characters from the first digit
|
||||
const uint8x16_t s = sixteen ? chars : vextq_u8(chars, vdupq_n_u8(static_cast<std::uint8_t>('0' + last)), 1);
|
||||
const char s16 = static_cast<char>(sixteen ? '0' + last : '0'); // the 17th
|
||||
const auto store_16 = [&s](char* p) noexcept
|
||||
{
|
||||
vst1q_u8(reinterpret_cast<std::uint8_t*>(p), s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
};
|
||||
const auto first_digit = static_cast<char>(vgetq_lane_u8(s, 0));
|
||||
#else
|
||||
const std::uint64_t hi = eight_digit_bytes(upper);
|
||||
const std::uint64_t lo = eight_digit_bytes(sig - (upper * 100000000u));
|
||||
// trailing zero digits: zero bytes (sig is not 0)
|
||||
const int zeros = lo != 0 ? count_trailing_zeros(lo) / 8 : 8 + (count_trailing_zeros(hi) / 8);
|
||||
const int digits = 15 - zeros + (sixteen ? 1 : 0); // without trailing zeros
|
||||
// the 16 characters from the first digit
|
||||
const std::uint64_t s_hi = (sixteen ? hi : (hi << 8u) | (lo >> 56u)) + 0x3030303030303030u;
|
||||
const std::uint64_t s_lo = (sixteen ? lo : (lo << 8u) | static_cast<std::uint64_t>(last)) + 0x3030303030303030u;
|
||||
const char s16 = static_cast<char>(sixteen ? '0' + last : '0'); // the 17th
|
||||
const auto store_16 = [s_hi, s_lo](char* p) noexcept
|
||||
{
|
||||
store_msb_first(p, s_hi);
|
||||
store_msb_first(p + 8, s_lo);
|
||||
};
|
||||
const auto first_digit = static_cast<char>(s_hi >> 56u);
|
||||
#endif
|
||||
const int len = d.has_digit ? 16 + (sixteen ? 1 : 0) : digits; // significant digits
|
||||
const int n = 16 + (sixteen ? 1 : 0) + d.exponent; // digits before the point
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(n >= 1 && n <= 15))
|
||||
{
|
||||
// "dig.its" and "digits[000].0": the digits after the point move by
|
||||
// one byte ('0's follow the digits)
|
||||
#if JSON_DTOA_SSE2
|
||||
// NOLINTBEGIN(portability-simd-intrinsics)
|
||||
// (in the register: reading the digits back from memory right after
|
||||
// storing them waits until the stores are done)
|
||||
const __m128i at = _mm_set1_epi8(static_cast<char>(n));
|
||||
const __m128i index = _mm_setr_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15);
|
||||
const __m128i before = _mm_cmpgt_epi8(at, index);
|
||||
const __m128i after = _mm_cmpgt_epi8(index, at);
|
||||
const __m128i text = _mm_or_si128(_mm_or_si128(_mm_and_si128(s, before), _mm_and_si128(_mm_slli_si128(s, 1), after)),
|
||||
_mm_andnot_si128(_mm_or_si128(before, after), _mm_set1_epi8('.')));
|
||||
std::memcpy(first, &text, 16);
|
||||
first[16] = static_cast<char>(_mm_extract_epi16(s, 7) >> 8);
|
||||
first[17] = s16;
|
||||
// NOLINTEND(portability-simd-intrinsics)
|
||||
#elif JSON_DTOA_NEON
|
||||
const uint8x16_t index = vcombine_u8(vcreate_u8(0x0706050403020100u), vcreate_u8(0x0F0E0D0C0B0A0908u));
|
||||
const uint8x16_t at = vdupq_n_u8(static_cast<std::uint8_t>(n));
|
||||
const uint8x16_t after_point = vbslq_u8(vcgtq_u8(index, at), vextq_u8(vdupq_n_u8(0), s, 15), vdupq_n_u8('.'));
|
||||
vst1q_u8(reinterpret_cast<std::uint8_t*>(first), vbslq_u8(vcltq_u8(index, at), s, after_point)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
first[16] = static_cast<char>(vgetq_lane_u8(s, 15));
|
||||
first[17] = s16;
|
||||
#else
|
||||
store_16(first);
|
||||
first[16] = s16;
|
||||
std::uint64_t after_point[2]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
|
||||
std::memcpy(after_point, first + n, 16);
|
||||
std::memcpy(first + n + 1, after_point, 16);
|
||||
first[n] = '.';
|
||||
#endif
|
||||
return first + (n >= len ? n + 2 : len + 1);
|
||||
}
|
||||
if (n <= 0 && n > -4)
|
||||
{
|
||||
// "0.[000]digits"
|
||||
std::memset(first, '0', 8);
|
||||
first[1] = '.';
|
||||
store_16(first + 2 - n);
|
||||
first[18 - n] = s16;
|
||||
return first + 2 - n + len;
|
||||
}
|
||||
// d.igitse+XX, with at least two exponent digits (as append_exponent())
|
||||
store_16(first + 1);
|
||||
first[17] = s16;
|
||||
first[0] = first_digit;
|
||||
first[1] = '.';
|
||||
char* const end = first + (len == 1 ? 1 : len + 1);
|
||||
const int e = n - 1;
|
||||
const auto ea = static_cast<unsigned>(e < 0 ? -e : e);
|
||||
const bool three = ea >= 100;
|
||||
end[0] = 'e';
|
||||
end[1] = e < 0 ? '-' : '+';
|
||||
end[2] = static_cast<char>('0' + (three ? ea / 100 : (ea / 10) % 10));
|
||||
end[3] = static_cast<char>('0' + (three ? (ea / 10) % 10 : ea % 10));
|
||||
end[4] = static_cast<char>('0' + (ea % 10));
|
||||
return end + (three ? 5 : 4);
|
||||
}
|
||||
|
||||
/// the powers of ten up to 10^16
|
||||
inline const std::array<std::uint64_t, 17>& powers_of_ten_16() noexcept
|
||||
{
|
||||
static const std::array<std::uint64_t, 17> powers =
|
||||
{
|
||||
{
|
||||
1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u,
|
||||
100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u
|
||||
}
|
||||
};
|
||||
return powers;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief digits * 10^exp, as write_decimal() writes it, for the digits of a
|
||||
double that need no conversion (count digits, at most 15, the first not 0;
|
||||
trailing zeros allowed): extended to 16 digits and written by write_shortest()
|
||||
|
||||
@return a pointer past the text; up to 41 bytes at @a first are written
|
||||
(some beyond the returned end)
|
||||
*/
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept
|
||||
{
|
||||
JSON_ASSERT(digits >= powers_of_ten_16()[static_cast<std::size_t>(count - 1)] && count <= 15);
|
||||
const int scale = 16 - count;
|
||||
return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast<std::size_t>(scale)], exp - scale - 1, 0, false});
|
||||
}
|
||||
|
||||
/// as write_short_decimal(), counting the digits (not 0, less than 10^15)
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept
|
||||
{
|
||||
JSON_ASSERT(digits != 0 && digits < 1000000000000000u);
|
||||
// floor(log10(2^bits)) + 1 digits, or one less
|
||||
const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12;
|
||||
const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast<std::size_t>(log2_bound)] ? 1 : 0);
|
||||
return write_short_decimal(first, digits, count, exp);
|
||||
}
|
||||
|
||||
/// a positive finite float (other than double): Grisu2 and format_buffer()
|
||||
template<typename FloatType>
|
||||
JSON_HEDLEY_NON_NULL(1, 2)
|
||||
@@ -1284,7 +1496,7 @@ char* write_positive(char* first, const char* last, FloatType value)
|
||||
}
|
||||
|
||||
/// a positive finite double: the shortest digits (Zmij), laid out by
|
||||
/// write_decimal() (through a local buffer if [first, last) is shorter than
|
||||
/// write_shortest() (through a local buffer if [first, last) is shorter than
|
||||
/// the 41 bytes it may write)
|
||||
JSON_HEDLEY_NON_NULL(1, 2)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
@@ -1294,13 +1506,13 @@ inline char* write_positive(char* first, const char* last, double value)
|
||||
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
|
||||
std::uint64_t bits = 0;
|
||||
std::memcpy(&bits, &value, sizeof(bits));
|
||||
const zmij::decimal d = zmij::to_decimal(bits);
|
||||
const zmij::shortest_decimal d = zmij::to_shortest(bits);
|
||||
if (JSON_HEDLEY_LIKELY(last - first >= 41))
|
||||
{
|
||||
return write_decimal(first, d.significand, d.exponent);
|
||||
return write_shortest(first, d);
|
||||
}
|
||||
std::array<char, 64> buf; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
|
||||
const auto len = static_cast<std::size_t>(write_decimal(buf.data(), d.significand, d.exponent) - buf.data());
|
||||
const auto len = static_cast<std::size_t>(write_shortest(buf.data(), d) - buf.data());
|
||||
JSON_ASSERT(static_cast<std::size_t>(last - first) >= len);
|
||||
std::memcpy(first, buf.data(), len);
|
||||
return first + len;
|
||||
|
||||
@@ -157,10 +157,22 @@ inline std::uint64_t umul128_add_hi64(std::uint64_t x, std::uint64_t y, std::uin
|
||||
return p.high + (p.low + c < p.low ? 1u : 0u);
|
||||
}
|
||||
|
||||
/// the result of Zmij: the shorter candidate and, if that is outside the
|
||||
/// rounding interval, the digit after it (16 bytes: returned in registers)
|
||||
struct shortest_decimal
|
||||
{
|
||||
std::uint64_t integral; ///< the shorter candidate (15 or 16 digits for normal doubles)
|
||||
int exponent; ///< the decimal exponent of the digit after it
|
||||
unsigned char digit; ///< the digit after it (if has_digit)
|
||||
bool has_digit; ///< whether the shortest decimal is integral * 10 + digit
|
||||
};
|
||||
|
||||
/// The shortest decimal in the rounding interval of a positive finite double
|
||||
/// given by its bits, the closest one if there are several (to_decimal of
|
||||
/// Zmij). The significand can end in zeros.
|
||||
inline decimal to_decimal(std::uint64_t bits) noexcept
|
||||
/// Zmij, which keeps the last digit apart: the 15 or 16 digits before it can be
|
||||
/// converted without a multiplication by 10 first). Always inlined: GCC
|
||||
/// otherwise calls it, and its result goes through memory.
|
||||
JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexcept
|
||||
{
|
||||
constexpr int extra_shift = 9;
|
||||
const auto raw_exp = static_cast<int>((bits >> 52u) & 0x7FFu);
|
||||
@@ -205,12 +217,20 @@ inline decimal to_decimal(std::uint64_t bits) noexcept
|
||||
digit = digit < lowest ? lowest : digit;
|
||||
}
|
||||
integral += round_up ? 1u : 0u;
|
||||
if (!round_up && !round_down)
|
||||
// if the shorter candidate is outside the rounding interval: one digit more
|
||||
return shortest_decimal{integral, dec_exp, static_cast<unsigned char>(digit), !round_up && !round_down};
|
||||
}
|
||||
|
||||
/// The shortest decimal in the rounding interval of a positive finite double
|
||||
/// given by its bits, as one number. The significand can end in zeros.
|
||||
inline decimal to_decimal(std::uint64_t bits) noexcept
|
||||
{
|
||||
const shortest_decimal d = to_shortest(bits);
|
||||
if (d.has_digit)
|
||||
{
|
||||
// the shorter candidate is outside the rounding interval: one digit more
|
||||
return decimal{(integral * 10) + digit, dec_exp};
|
||||
return decimal{(d.integral * 10) + d.digit, d.exponent};
|
||||
}
|
||||
return decimal{integral, dec_exp + 1};
|
||||
return decimal{d.integral, d.exponent + 1};
|
||||
}
|
||||
|
||||
} // namespace zmij
|
||||
|
||||
@@ -249,8 +249,9 @@ template<typename FloatType>
|
||||
using native_float_t = typename std::conditional<std::numeric_limits<FloatType>::digits == 24, float, double>::type;
|
||||
|
||||
/// the value of the eight ASCII digits in @a v (see read_eight_bytes()), three
|
||||
/// multiplications instead of eight (after simdjson and fast_float)
|
||||
inline std::uint32_t parse_eight_digits(std::uint64_t v) noexcept
|
||||
/// multiplications instead of eight (after simdjson and fast_float); always
|
||||
/// inlined, as GCC otherwise calls it in the number loops
|
||||
JSON_HEDLEY_ALWAYS_INLINE std::uint32_t parse_eight_digits(std::uint64_t v) noexcept
|
||||
{
|
||||
v = ((v & 0x0F0F0F0F0F0F0F0Fu) * 2561u) >> 8u;
|
||||
v = ((v & 0x00FF00FF00FF00FFu) * 6553601u) >> 16u;
|
||||
|
||||
@@ -21,6 +21,8 @@
|
||||
#undef JSON_NO_UNIQUE_ADDRESS
|
||||
#undef JSON_DISABLE_ENUM_SERIALIZATION
|
||||
#undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION
|
||||
#undef JSON_DTOA_SSE2
|
||||
#undef JSON_DTOA_NEON
|
||||
|
||||
#ifndef JSON_TEST_KEEP_MACROS
|
||||
#undef JSON_CATCH
|
||||
|
||||
@@ -1366,8 +1366,9 @@ class serializer
|
||||
/*!
|
||||
@brief dump an integer
|
||||
|
||||
Dump a given integer, appending it to @ref write_buffer. Works internally with
|
||||
@a number_buffer.
|
||||
Dump a given integer, appending it to @ref write_buffer (directly: copying
|
||||
the digits from another buffer right after writing them waits until the
|
||||
stores are done).
|
||||
|
||||
@param[in] x integer number (signed or unsigned) to dump
|
||||
@tparam NumberType either @a number_integer_t or @a number_unsigned_t
|
||||
@@ -1402,33 +1403,57 @@ class serializer
|
||||
return;
|
||||
}
|
||||
|
||||
// use a pointer to fill the buffer
|
||||
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto)
|
||||
// use a pointer to fill the buffer (room for as much as number_buffer holds)
|
||||
if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + number_buffer.size() > write_buffer.size()))
|
||||
{
|
||||
flush();
|
||||
}
|
||||
auto* buffer_ptr = write_buffer.data() + write_buffer_pos;
|
||||
|
||||
number_unsigned_t abs_value;
|
||||
|
||||
unsigned int n_chars{};
|
||||
// one byte for the minus sign
|
||||
unsigned int n_chars = 0;
|
||||
|
||||
if (is_negative_number(x))
|
||||
{
|
||||
*buffer_ptr = '-';
|
||||
abs_value = remove_sign(static_cast<number_integer_t>(x));
|
||||
|
||||
// account one more byte for the minus sign
|
||||
n_chars = 1 + count_digits(abs_value);
|
||||
n_chars = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
abs_value = static_cast<number_unsigned_t>(x);
|
||||
n_chars = count_digits(abs_value);
|
||||
}
|
||||
|
||||
// up to 16 digits: eight at a time (as the digits of floats), written
|
||||
// without leading zeros
|
||||
if (abs_value < 10000000000000000u)
|
||||
{
|
||||
const std::uint64_t value = abs_value;
|
||||
const std::uint64_t upper = value / 100000000u;
|
||||
const std::uint64_t first = dtoa_impl::eight_digit_bytes(upper != 0 ? upper : value);
|
||||
const auto leading = static_cast<unsigned>(count_leading_zeros(first) / 8); // (first is not 0)
|
||||
char* const p = buffer_ptr + n_chars;
|
||||
dtoa_impl::store_msb_first(p, (first << (8 * leading)) + 0x3030303030303030u);
|
||||
n_chars += 8 - leading;
|
||||
if (upper != 0)
|
||||
{
|
||||
dtoa_impl::store_msb_first(p + 8 - leading, dtoa_impl::eight_digit_bytes(value - (upper * 100000000u)) + 0x3030303030303030u);
|
||||
n_chars += 8;
|
||||
}
|
||||
write_buffer_pos += n_chars;
|
||||
return;
|
||||
}
|
||||
|
||||
n_chars += count_digits(abs_value);
|
||||
|
||||
// spare 1 byte for '\0'
|
||||
JSON_ASSERT(n_chars < number_buffer.size() - 1);
|
||||
|
||||
// jump to the end to generate the string from backward,
|
||||
// so we later avoid reversing the result
|
||||
buffer_ptr += static_cast<typename decltype(number_buffer)::difference_type>(n_chars);
|
||||
buffer_ptr += n_chars;
|
||||
|
||||
// Fast int2ascii implementation inspired by "Fastware" talk by Andrei Alexandrescu
|
||||
// See: https://www.youtube.com/watch?v=o4-CwDo2zpg
|
||||
@@ -1451,14 +1476,13 @@ class serializer
|
||||
*(--buffer_ptr) = static_cast<char>('0' + abs_value);
|
||||
}
|
||||
|
||||
put_buffer(number_buffer, n_chars);
|
||||
write_buffer_pos += n_chars;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief dump a floating-point number
|
||||
|
||||
Dump a given floating-point number, appending it to @ref write_buffer. Works internally
|
||||
with @a number_buffer.
|
||||
Dump a given floating-point number, appending it to @ref write_buffer.
|
||||
|
||||
@param[in] x floating-point number to dump
|
||||
*/
|
||||
@@ -1485,10 +1509,15 @@ class serializer
|
||||
|
||||
void dump_float(number_float_t x, std::true_type /*is_ieee_single_or_double*/)
|
||||
{
|
||||
auto* begin = number_buffer.data();
|
||||
// directly into the write buffer: copying the text from number_buffer
|
||||
// right after to_chars() wrote it waits until its stores are done
|
||||
if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + number_buffer.size() > write_buffer.size()))
|
||||
{
|
||||
flush();
|
||||
}
|
||||
auto* begin = write_buffer.data() + write_buffer_pos;
|
||||
auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x);
|
||||
|
||||
put_buffer(number_buffer, static_cast<std::size_t>(end - begin));
|
||||
write_buffer_pos += static_cast<std::size_t>(end - begin);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
|
||||
@@ -828,14 +828,19 @@ indent_done:
|
||||
const auto idx = static_cast<std::uint32_t>(emit(k, 0, 0, static_cast<std::size_t>(p - b), 0) - base);
|
||||
if (depth != 0)
|
||||
{
|
||||
const frame f = {cur_idx, cur_count, cur_is_object};
|
||||
if (NLOHMANN_VIEW_LIKELY(depth <= 64))
|
||||
{
|
||||
cold.shallow[depth - 1] = f;
|
||||
// field by field: a frame put together on the stack and
|
||||
// copied would be read back wider than it was written,
|
||||
// and that load waits until the stores are done
|
||||
frame& f = cold.shallow[depth - 1];
|
||||
f.idx = cur_idx;
|
||||
f.count = cur_count;
|
||||
f.is_object = cur_is_object;
|
||||
}
|
||||
else
|
||||
{
|
||||
cold.deep.push_back(f);
|
||||
cold.deep.push_back(frame{cur_idx, cur_count, cur_is_object});
|
||||
}
|
||||
}
|
||||
++depth;
|
||||
@@ -851,19 +856,21 @@ indent_done:
|
||||
n.next = static_cast<std::uint32_t>(out - base) - cur_idx;
|
||||
if (--depth != 0)
|
||||
{
|
||||
frame f{};
|
||||
if (NLOHMANN_VIEW_LIKELY(depth <= 64))
|
||||
{
|
||||
f = cold.shallow[depth - 1];
|
||||
const frame& f = cold.shallow[depth - 1];
|
||||
cur_idx = f.idx;
|
||||
cur_count = f.count;
|
||||
cur_is_object = f.is_object;
|
||||
}
|
||||
else
|
||||
{
|
||||
f = cold.deep.back();
|
||||
const frame f = cold.deep.back();
|
||||
cold.deep.pop_back();
|
||||
cur_idx = f.idx;
|
||||
cur_count = f.count;
|
||||
cur_is_object = f.is_object;
|
||||
}
|
||||
cur_idx = f.idx;
|
||||
cur_count = f.count;
|
||||
cur_is_object = f.is_object;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -22,5 +22,7 @@
|
||||
#undef NLOHMANN_VIEW_NEON
|
||||
#undef NLOHMANN_VIEW_SSE2
|
||||
#undef NLOHMANN_VIEW_SSSE3
|
||||
#undef NLOHMANN_VIEW_SSSE3_DISPATCH
|
||||
#undef NLOHMANN_VIEW_SSSE3_TARGET
|
||||
#undef NLOHMANN_VIEW_VECTOR
|
||||
#undef NLOHMANN_VIEW_VECTOR_UTF8
|
||||
@@ -67,13 +67,19 @@ NLOHMANN_VIEW_ALWAYS_INLINE std::uint16_t load16(const unsigned char* p) noexcep
|
||||
/// in predicted branches: 16 for keys, whose lengths repeat from record to
|
||||
/// record, and 8 for string values (Value) where a vector loop follows, as
|
||||
/// their lengths vary more. Longer runs continue 16 bytes at a time with NEON
|
||||
/// or SSE2, else eight bytes at a time.
|
||||
/// or SSE2, else eight bytes at a time. With SSE2, the run is checked 16 bytes
|
||||
/// at a time from its first byte instead: on x86-64, one compare that finds
|
||||
/// the end of most keys and short values is faster than a branch per byte (on
|
||||
/// AArch64, where a NEON mask costs more and branches predict well, slower).
|
||||
template<bool Value = false>
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned char* p, const unsigned char* e) noexcept
|
||||
{
|
||||
const std::uint8_t* plain = string_plain();
|
||||
for (;;)
|
||||
{
|
||||
#if NLOHMANN_VIEW_SSE2
|
||||
p = vector_plain_run(p, e);
|
||||
#else
|
||||
if (e - p >= 16)
|
||||
{
|
||||
#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(plain[p[i]] != 0)) {} else { p += (i); goto stop; }
|
||||
@@ -107,6 +113,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
#endif
|
||||
continue;
|
||||
}
|
||||
#endif
|
||||
while (p != e && plain[*p] != 0)
|
||||
{
|
||||
++p;
|
||||
@@ -115,15 +122,23 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
{
|
||||
return p;
|
||||
}
|
||||
#if !NLOHMANN_VIEW_SSE2
|
||||
stop:
|
||||
#endif
|
||||
if (*p < 0x80)
|
||||
{
|
||||
return p; // quote, backslash, or control character
|
||||
}
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
// non-ASCII: the vector check, out of line
|
||||
return scan_string_vector(p, e, plain);
|
||||
#else
|
||||
#if NLOHMANN_VIEW_SSSE3_DISPATCH
|
||||
if (NLOHMANN_VIEW_LIKELY(cpu_has_ssse3()))
|
||||
#endif
|
||||
{
|
||||
// non-ASCII: the vector check, out of line
|
||||
return scan_string_vector(p, e, plain);
|
||||
}
|
||||
#endif
|
||||
#if !NLOHMANN_VIEW_VECTOR_UTF8 || NLOHMANN_VIEW_SSSE3_DISPATCH
|
||||
// non-ASCII: a run of well-formed sequences (the library's check, so
|
||||
// that exactly what json::parse accepts is accepted)
|
||||
do
|
||||
|
||||
@@ -23,7 +23,6 @@
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/node.hpp>
|
||||
#include <nlohmann/detail/view/number.hpp>
|
||||
#include <nlohmann/detail/view/simd.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -32,7 +31,11 @@ namespace view
|
||||
{
|
||||
|
||||
/// append-only output buffer: writes through a raw pointer into a string that
|
||||
/// is resized ahead, and trimmed by finish()
|
||||
/// is resized ahead, and trimmed by finish(). The estimate is reserved, and the
|
||||
/// string grows in steps of 64 KiB within it: resize() fills the new bytes
|
||||
/// with zeros (before C++23, a string cannot grow without), and a small step
|
||||
/// is filled while the writer is about to use it, in the cache, instead of
|
||||
/// filling the whole estimate in memory first.
|
||||
template<typename StringType>
|
||||
class output_buffer
|
||||
{
|
||||
@@ -94,16 +97,31 @@ class output_buffer
|
||||
}
|
||||
|
||||
private:
|
||||
/// the size of a growth step (a function: std::min() takes a reference,
|
||||
/// which a static constexpr member does not have before C++17)
|
||||
static constexpr std::size_t step() noexcept
|
||||
{
|
||||
return 65536;
|
||||
}
|
||||
|
||||
static StringType& sized(StringType& out, std::size_t estimate)
|
||||
{
|
||||
out.resize((std::max)(estimate, static_cast<std::size_t>(64)));
|
||||
out.reserve(estimate);
|
||||
out.resize((std::min)((std::max)(estimate, static_cast<std::size_t>(64)), step()));
|
||||
return out;
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_NOINLINE void grow(std::size_t n)
|
||||
{
|
||||
const auto used = static_cast<std::size_t>(m_pos - m_out.data());
|
||||
m_out.resize((std::max)(m_out.size() * 2, used + n + 256));
|
||||
// (a step does not go beyond the reserved estimate, so that a good
|
||||
// estimate is never copied to a larger allocation)
|
||||
const std::size_t size = (std::max)((std::min)(m_out.size() + step(), m_out.capacity()), used + n + 256);
|
||||
if (size > m_out.capacity())
|
||||
{
|
||||
m_out.reserve((std::max)(m_out.capacity() * 2, size));
|
||||
}
|
||||
m_out.resize(size);
|
||||
m_pos = &m_out[0] + used;
|
||||
m_end = &m_out[0] + m_out.size();
|
||||
}
|
||||
@@ -222,69 +240,6 @@ struct dump_style
|
||||
bool source_numbers = false; ///< copy number tokens from the source
|
||||
};
|
||||
|
||||
/*!
|
||||
@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0,
|
||||
at most 17 digits; up to 41 bytes are written at first)
|
||||
|
||||
With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in
|
||||
vector registers: output byte i is byte s + i of the digits (after '0's)
|
||||
before the point, and byte s + i - 1 after it. The portable code writes the
|
||||
digits to a buffer and copies them from there at another offset, and a load
|
||||
that spans several recent stores waits until they reach the cache.
|
||||
*/
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept
|
||||
{
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
namespace dtoa = ::nlohmann::detail::dtoa_impl;
|
||||
const std::uint64_t upper = digits / 100000000u;
|
||||
const std::uint64_t b0 = upper / 100000000u; // one digit
|
||||
const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u);
|
||||
const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u);
|
||||
// leading and trailing zero digits (as dtoa_impl::write_decimal())
|
||||
int leading = 7;
|
||||
if (b0 == 0)
|
||||
{
|
||||
leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8);
|
||||
}
|
||||
int zeros = 16;
|
||||
if (b2 != 0)
|
||||
{
|
||||
zeros = count_trailing_zeros(b2) / 8;
|
||||
}
|
||||
else if (b1 != 0)
|
||||
{
|
||||
zeros = 8 + (count_trailing_zeros(b1) / 8);
|
||||
}
|
||||
const int k = 24 - leading - zeros; // significant digits
|
||||
const int n = k + exp + zeros; // position of the point after the first digit
|
||||
if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15))
|
||||
{
|
||||
static const std::array<std::uint8_t, 32> iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}};
|
||||
const int pad = n <= 0 ? 1 - n : 0;
|
||||
const int len = k + pad;
|
||||
const int point = n + pad;
|
||||
// the 24 digit bytes in memory order, then '0's
|
||||
const std::uint64_t zero_chars = 0x3030303030303030u;
|
||||
const uint8x16x2_t table = {{
|
||||
vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))),
|
||||
vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0'))
|
||||
}
|
||||
};
|
||||
const uint8x16_t s = vdupq_n_u8(static_cast<std::uint8_t>(leading - pad));
|
||||
const uint8x16_t at_point = vdupq_n_u8(static_cast<std::uint8_t>(point));
|
||||
for (std::size_t half = 0; half < 2; ++half)
|
||||
{
|
||||
const uint8x16_t i = vld1q_u8(iota.data() + (16 * half));
|
||||
// (+ 0xFF is - 1 after the point; indexes past the digits read a '0')
|
||||
const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31));
|
||||
vst1q_u8(reinterpret_cast<std::uint8_t*>(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
}
|
||||
return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0"
|
||||
}
|
||||
#endif
|
||||
return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief write a view's subtree as basic_json::dump() writes the value
|
||||
|
||||
@@ -786,7 +741,11 @@ class view_serializer
|
||||
if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290)
|
||||
{
|
||||
*w = '-';
|
||||
return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast<int>(d.exponent));
|
||||
w += d.negative ? 1 : 0;
|
||||
// (without leading zeros, all digits of the token count)
|
||||
const unsigned char lead = first[d.negative ? 1 : 0];
|
||||
return lead != '0' ? ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast<int>(int_digits + frac_digits), static_cast<int>(d.exponent))
|
||||
: ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast<int>(d.exponent));
|
||||
}
|
||||
return write_double_value_at(w, decimal_to_float<double>(d)); // (without reading the token again)
|
||||
}
|
||||
@@ -803,13 +762,13 @@ class view_serializer
|
||||
/// a double as dump() writes it, at w (64 bytes of room)
|
||||
static char* write_double_value_at(char* w, double x)
|
||||
{
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x)))
|
||||
// (from the bits: without the checks of to_chars())
|
||||
std::uint64_t bits = 0;
|
||||
std::memcpy(&bits, &x, sizeof(bits));
|
||||
if (NLOHMANN_VIEW_UNLIKELY((bits & 0x7FF0000000000000u) == 0x7FF0000000000000u))
|
||||
{
|
||||
return write_text_at(w, "null", 4);
|
||||
}
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
std::uint64_t bits = 0;
|
||||
std::memcpy(&bits, &x, sizeof(bits));
|
||||
*w = '-';
|
||||
w += bits >> 63u;
|
||||
bits &= ~(std::uint64_t{1} << 63u);
|
||||
@@ -817,11 +776,7 @@ class view_serializer
|
||||
{
|
||||
return write_text_at(w, "0.0", 3);
|
||||
}
|
||||
const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits);
|
||||
return write_decimal(w, d.significand, d.exponent);
|
||||
#else
|
||||
return ::nlohmann::detail::to_chars(w, w + 64, x);
|
||||
#endif
|
||||
return ::nlohmann::detail::dtoa_impl::write_shortest(w, ::nlohmann::detail::zmij::to_shortest(bits));
|
||||
}
|
||||
|
||||
/// as serializer::dump_float()
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <atomic> // atomic
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t, uint64_t
|
||||
|
||||
@@ -18,11 +19,13 @@
|
||||
|
||||
// Vector code for long runs of string bytes. NEON (AArch64) and SSE2 (x86-64)
|
||||
// belong to the baseline instruction sets and are used by default. The vector
|
||||
// UTF-8 check needs NEON, or SSSE3 if JSON_VIEW_USE_SSSE3 is defined: SSSE3 is
|
||||
// not part of x86-64, so it must not depend on the flags of a translation unit
|
||||
// (two translation units with different flags would have different
|
||||
// definitions of the same inline functions). JSON_VIEW_NO_SIMD selects the
|
||||
// portable code.
|
||||
// UTF-8 check needs NEON or SSSE3. SSSE3 is not part of x86-64, and the code
|
||||
// must not depend on the flags of a translation unit (two translation units
|
||||
// with different flags would have different definitions of the same inline
|
||||
// functions): the check is compiled for SSSE3 with a function attribute and
|
||||
// used where the CPU has SSSE3 (all x86-64 CPUs since about 2011), else the
|
||||
// portable check. JSON_VIEW_USE_SSSE3 skips the CPU check (for code compiled
|
||||
// for SSSE3 anyway); JSON_VIEW_NO_SIMD selects the portable code.
|
||||
#if !defined(JSON_VIEW_NO_SIMD) && defined(__aarch64__) && (defined(__GNUC__) || defined(__clang__)) && NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
#include <arm_neon.h>
|
||||
#define NLOHMANN_VIEW_NEON 1
|
||||
@@ -41,8 +44,24 @@
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSSE3 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#endif
|
||||
#if NLOHMANN_VIEW_SSE2 && !NLOHMANN_VIEW_SSSE3 && ((defined(__clang__) && __clang_major__ >= 4) || (defined(__GNUC__) && !defined(__clang__) && (__GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 9))))
|
||||
// (GCC before 4.9 has no SSSE3 intrinsics without -mssse3)
|
||||
#include <cpuid.h>
|
||||
#include <tmmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSSE3_DISPATCH 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#define NLOHMANN_VIEW_SSSE3_TARGET __attribute__((target("ssse3")))
|
||||
#elif NLOHMANN_VIEW_SSE2 && !NLOHMANN_VIEW_SSSE3 && defined(_MSC_VER)
|
||||
// (MSVC compiles intrinsics of any instruction set)
|
||||
#include <intrin.h>
|
||||
#include <tmmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSSE3_DISPATCH 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#define NLOHMANN_VIEW_SSSE3_TARGET
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSSE3_DISPATCH 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#define NLOHMANN_VIEW_SSSE3_TARGET
|
||||
#endif
|
||||
#define NLOHMANN_VIEW_VECTOR (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSE2)
|
||||
#define NLOHMANN_VIEW_VECTOR_UTF8 (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSSE3)
|
||||
#define NLOHMANN_VIEW_VECTOR_UTF8 (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSSE3 || NLOHMANN_VIEW_SSSE3_DISPATCH)
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -90,6 +109,40 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* vector_plain_run(const unsigned
|
||||
}
|
||||
#endif
|
||||
|
||||
#if NLOHMANN_VIEW_SSSE3_DISPATCH
|
||||
/// whether the CPU has SSSE3 (CPUID leaf 1, ECX bit 9)
|
||||
inline bool cpu_ssse3() noexcept
|
||||
{
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
std::array<int, 4> regs {{}};
|
||||
__cpuid(regs.data(), 1);
|
||||
return (static_cast<unsigned>(regs[2]) & (1u << 9u)) != 0;
|
||||
#else
|
||||
unsigned eax = 0;
|
||||
unsigned ebx = 0;
|
||||
unsigned ecx = 0;
|
||||
unsigned edx = 0;
|
||||
return __get_cpuid(1, &eax, &ebx, &ecx, &edx) != 0 && (ecx & (1u << 9u)) != 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
/// whether the CPU has SSSE3, asked once: the answer is kept in an atomic
|
||||
/// that is initialized at compile time, so that neither a guard of a local
|
||||
/// static nor a global constructor is needed (threads that ask at the same
|
||||
/// time all store the same answer)
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE bool cpu_has_ssse3() noexcept
|
||||
{
|
||||
static std::atomic<int> known{0}; // 0: not asked yet, 1: no, 2: yes
|
||||
int state = known.load(std::memory_order_relaxed);
|
||||
if (NLOHMANN_VIEW_UNLIKELY(state == 0))
|
||||
{
|
||||
state = cpu_ssse3() ? 2 : 1;
|
||||
known.store(state, std::memory_order_relaxed);
|
||||
}
|
||||
return state == 2;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
/// Tables of the UTF-8 check of J. Keiser and D. Lemire, "Validating UTF-8 In
|
||||
/// Less Than One Instruction Per Byte" (2021), as in simdjson ("lookup4"): each
|
||||
@@ -192,8 +245,9 @@ compares, and the UTF-8 check covers the bytes up to it. Returns where the
|
||||
string scan stops, like scan_string_run: before ill-formed UTF-8 and for the
|
||||
last bytes of the input, the bytes are checked one sequence at a time. Out of
|
||||
line, so that no constants of the check occupy registers in the parse loop.
|
||||
On x86-64, it is compiled for SSSE3 (see cpu_has_ssse3()).
|
||||
*/
|
||||
NLOHMANN_VIEW_NOINLINE inline const unsigned char* scan_string_vector(const unsigned char* p, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
NLOHMANN_VIEW_SSSE3_TARGET NLOHMANN_VIEW_NOINLINE inline const unsigned char* scan_string_vector(const unsigned char* p, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
{
|
||||
using lookup = utf8_lookup4<>;
|
||||
const unsigned char* block = p;
|
||||
|
||||
@@ -8860,8 +8860,9 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc
|
||||
}
|
||||
|
||||
/// eight bytes as a little-endian word (compilers fold this into one load on
|
||||
/// little-endian targets)
|
||||
inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
|
||||
/// little-endian targets; always inlined, as GCC otherwise calls it in the
|
||||
/// number loops)
|
||||
JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
|
||||
{
|
||||
return static_cast<std::uint64_t>(b[0]) | (static_cast<std::uint64_t>(b[1]) << 8u)
|
||||
| (static_cast<std::uint64_t>(b[2]) << 16u) | (static_cast<std::uint64_t>(b[3]) << 24u)
|
||||
@@ -8870,7 +8871,7 @@ inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
|
||||
}
|
||||
|
||||
/// eight bytes as a little-endian word
|
||||
inline std::uint64_t read_eight_bytes(const char* p) noexcept
|
||||
JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const char* p) noexcept
|
||||
{
|
||||
return read_eight_bytes(reinterpret_cast<const unsigned char*>(p)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
}
|
||||
@@ -9479,8 +9480,9 @@ template<typename FloatType>
|
||||
using native_float_t = typename std::conditional<std::numeric_limits<FloatType>::digits == 24, float, double>::type;
|
||||
|
||||
/// the value of the eight ASCII digits in @a v (see read_eight_bytes()), three
|
||||
/// multiplications instead of eight (after simdjson and fast_float)
|
||||
inline std::uint32_t parse_eight_digits(std::uint64_t v) noexcept
|
||||
/// multiplications instead of eight (after simdjson and fast_float); always
|
||||
/// inlined, as GCC otherwise calls it in the number loops
|
||||
JSON_HEDLEY_ALWAYS_INLINE std::uint32_t parse_eight_digits(std::uint64_t v) noexcept
|
||||
{
|
||||
v = ((v & 0x0F0F0F0F0F0F0F0Fu) * 2561u) >> 8u;
|
||||
v = ((v & 0x00FF00FF00FF00FFu) * 6553601u) >> 16u;
|
||||
@@ -24426,6 +24428,21 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
#include <cstdlib> // _byteswap_uint64
|
||||
#endif
|
||||
|
||||
// SSE2 (every x86-64 CPU) and NEON (every 64-bit Arm CPU) convert the 16
|
||||
// digits of a double at once
|
||||
#if defined(__x86_64__) || (defined(_M_X64) && !defined(_M_ARM64EC))
|
||||
#include <emmintrin.h>
|
||||
#define JSON_DTOA_SSE2 1
|
||||
#define JSON_DTOA_NEON 0
|
||||
#elif (defined(__aarch64__) || defined(_M_ARM64)) && !defined(_M_ARM64EC) && !defined(__ARM_BIG_ENDIAN)
|
||||
#include <arm_neon.h>
|
||||
#define JSON_DTOA_SSE2 0
|
||||
#define JSON_DTOA_NEON 1
|
||||
#else
|
||||
#define JSON_DTOA_SSE2 0
|
||||
#define JSON_DTOA_NEON 0
|
||||
#endif
|
||||
|
||||
// #include <nlohmann/detail/conversions/zmij.hpp>
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
@@ -24590,10 +24607,22 @@ inline std::uint64_t umul128_add_hi64(std::uint64_t x, std::uint64_t y, std::uin
|
||||
return p.high + (p.low + c < p.low ? 1u : 0u);
|
||||
}
|
||||
|
||||
/// the result of Zmij: the shorter candidate and, if that is outside the
|
||||
/// rounding interval, the digit after it (16 bytes: returned in registers)
|
||||
struct shortest_decimal
|
||||
{
|
||||
std::uint64_t integral; ///< the shorter candidate (15 or 16 digits for normal doubles)
|
||||
int exponent; ///< the decimal exponent of the digit after it
|
||||
unsigned char digit; ///< the digit after it (if has_digit)
|
||||
bool has_digit; ///< whether the shortest decimal is integral * 10 + digit
|
||||
};
|
||||
|
||||
/// The shortest decimal in the rounding interval of a positive finite double
|
||||
/// given by its bits, the closest one if there are several (to_decimal of
|
||||
/// Zmij). The significand can end in zeros.
|
||||
inline decimal to_decimal(std::uint64_t bits) noexcept
|
||||
/// Zmij, which keeps the last digit apart: the 15 or 16 digits before it can be
|
||||
/// converted without a multiplication by 10 first). Always inlined: GCC
|
||||
/// otherwise calls it, and its result goes through memory.
|
||||
JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexcept
|
||||
{
|
||||
constexpr int extra_shift = 9;
|
||||
const auto raw_exp = static_cast<int>((bits >> 52u) & 0x7FFu);
|
||||
@@ -24638,12 +24667,20 @@ inline decimal to_decimal(std::uint64_t bits) noexcept
|
||||
digit = digit < lowest ? lowest : digit;
|
||||
}
|
||||
integral += round_up ? 1u : 0u;
|
||||
if (!round_up && !round_down)
|
||||
// if the shorter candidate is outside the rounding interval: one digit more
|
||||
return shortest_decimal{integral, dec_exp, static_cast<unsigned char>(digit), !round_up && !round_down};
|
||||
}
|
||||
|
||||
/// The shortest decimal in the rounding interval of a positive finite double
|
||||
/// given by its bits, as one number. The significand can end in zeros.
|
||||
inline decimal to_decimal(std::uint64_t bits) noexcept
|
||||
{
|
||||
const shortest_decimal d = to_shortest(bits);
|
||||
if (d.has_digit)
|
||||
{
|
||||
// the shorter candidate is outside the rounding interval: one digit more
|
||||
return decimal{(integral * 10) + digit, dec_exp};
|
||||
return decimal{(d.integral * 10) + d.digit, d.exponent};
|
||||
}
|
||||
return decimal{integral, dec_exp + 1};
|
||||
return decimal{d.integral, d.exponent + 1};
|
||||
}
|
||||
|
||||
} // namespace zmij
|
||||
@@ -25882,6 +25919,203 @@ inline char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept
|
||||
return end + (three ? 5 : 4);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the shortest decimal of a positive double (Zmij), as write_decimal()
|
||||
writes it
|
||||
|
||||
For a normal double, the shorter candidate has 15 or 16 digits: they are
|
||||
converted at once (two halves of eight digits) and followed by the digit
|
||||
after them, if there is one, without the multiplication and division by 10
|
||||
that counting the digits of one number would take. The fixed layouts move
|
||||
the digits after the point by one byte.
|
||||
|
||||
@return a pointer past the text; up to 41 bytes at @a first are written
|
||||
(some beyond the returned end)
|
||||
*/
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcept
|
||||
{
|
||||
const std::uint64_t sig = d.integral;
|
||||
if (JSON_HEDLEY_UNLIKELY(sig < 100000000000000u || sig >= 10000000000000000u))
|
||||
{
|
||||
// (subnormals)
|
||||
return d.has_digit ? write_decimal(first, (sig * 10) + d.digit, d.exponent) : write_decimal(first, sig, d.exponent + 1);
|
||||
}
|
||||
const bool sixteen = sig >= 1000000000000000u; // (else 15 digits)
|
||||
const int last = d.has_digit ? d.digit : 0;
|
||||
const std::uint64_t upper = sig / 100000000u;
|
||||
#if JSON_DTOA_SSE2
|
||||
// NOLINTBEGIN(portability-simd-intrinsics)
|
||||
// the two halves in the 64-bit lanes, each as abcd * 2^32 + efgh, then as
|
||||
// bytes (as eight_digit_bytes(), one lane each)
|
||||
const __m128i x = _mm_set_epi64x(static_cast<long long>(sig - (upper * 100000000u)), static_cast<long long>(upper));
|
||||
const __m128i abcd = _mm_srli_epi64(_mm_mul_epu32(x, _mm_set1_epi64x(109951163)), 40); // 2^40 / 10000 + 1
|
||||
const __m128i abcd_efgh = _mm_add_epi64(x, _mm_mul_epu32(abcd, _mm_set1_epi64x(4294957296))); // 2^32 - 10000
|
||||
// 32-bit lanes in the order of the text: abcd, efgh of both halves
|
||||
const __m128i fours = _mm_shuffle_epi32(abcd_efgh, _MM_SHUFFLE(2, 3, 0, 1));
|
||||
const __m128i ab = _mm_srli_epi16(_mm_mulhi_epu16(fours, _mm_set1_epi32(5243)), 3);
|
||||
const __m128i ab_cd = _mm_or_si128(_mm_slli_epi32(_mm_sub_epi16(fours, _mm_mullo_epi16(ab, _mm_set1_epi32(100))), 16), ab);
|
||||
// 16-bit lanes ab (< 100) -> bytes a, b: 256 * ab - 2559 * (ab / 10)
|
||||
const __m128i bytes = _mm_sub_epi16(_mm_slli_epi16(ab_cd, 8), _mm_mullo_epi16(_mm_set1_epi16(2559), _mm_mulhi_epu16(ab_cd, _mm_set1_epi16(6554))));
|
||||
// the last digit that is not 0 (sig is not 0)
|
||||
const auto nonzero = static_cast<std::uint64_t>(_mm_movemask_epi8(_mm_cmpgt_epi8(bytes, _mm_setzero_si128())));
|
||||
const int digits = 63 - count_leading_zeros(nonzero) + (sixteen ? 1 : 0); // without trailing zeros
|
||||
const __m128i chars = _mm_add_epi8(bytes, _mm_set1_epi8('0'));
|
||||
// the 16 characters from the first digit
|
||||
const __m128i s = sixteen ? chars : _mm_or_si128(_mm_srli_si128(chars, 1), _mm_slli_si128(_mm_cvtsi32_si128('0' + last), 15));
|
||||
const char s16 = static_cast<char>(sixteen ? '0' + last : '0'); // the 17th
|
||||
const auto store_16 = [&s](char* p) noexcept
|
||||
{
|
||||
std::memcpy(p, &s, 16);
|
||||
};
|
||||
const char first_digit = static_cast<char>(_mm_cvtsi128_si32(s));
|
||||
// NOLINTEND(portability-simd-intrinsics)
|
||||
#elif JSON_DTOA_NEON
|
||||
// as with SSE2: the halves in 32-bit lanes, then abcd, efgh of both
|
||||
const uint32x2_t halves = vcreate_u32(upper | ((sig - (upper * 100000000u)) << 32u));
|
||||
const uint32x2_t abcd = vmovn_u64(vshrq_n_u64(vmull_n_u32(halves, static_cast<std::uint32_t>(((std::uint64_t{1} << 40u) / 10000u) + 1u)), 40));
|
||||
const uint32x2_t efgh = vmls_n_u32(halves, abcd, 10000u);
|
||||
const uint32x4_t fours = vcombine_u32(vzip1_u32(abcd, efgh), vzip2_u32(abcd, efgh));
|
||||
const uint32x4_t ab = vshrq_n_u32(vmulq_n_u32(fours, 5243u), 19);
|
||||
const uint16x8_t ab_cd = vreinterpretq_u16_u32(vorrq_u32(ab, vshlq_n_u32(vmlsq_n_u32(fours, ab, 100u), 16)));
|
||||
const uint16x8_t tens = vshrq_n_u16(vmulq_n_u16(ab_cd, 103u), 10);
|
||||
const uint8x16_t bytes = vreinterpretq_u8_u16(vorrq_u16(tens, vshlq_n_u16(vmlsq_n_u16(ab_cd, tens, 10u), 8)));
|
||||
// the last digit that is not 0 (sig is not 0): a nibble per byte
|
||||
const std::uint64_t nonzero = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(bytes, bytes)), 4)), 0);
|
||||
const int digits = ((63 - count_leading_zeros(nonzero)) / 4) + (sixteen ? 1 : 0); // without trailing zeros
|
||||
const uint8x16_t chars = vaddq_u8(bytes, vdupq_n_u8('0'));
|
||||
// the 16 characters from the first digit
|
||||
const uint8x16_t s = sixteen ? chars : vextq_u8(chars, vdupq_n_u8(static_cast<std::uint8_t>('0' + last)), 1);
|
||||
const char s16 = static_cast<char>(sixteen ? '0' + last : '0'); // the 17th
|
||||
const auto store_16 = [&s](char* p) noexcept
|
||||
{
|
||||
vst1q_u8(reinterpret_cast<std::uint8_t*>(p), s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
};
|
||||
const auto first_digit = static_cast<char>(vgetq_lane_u8(s, 0));
|
||||
#else
|
||||
const std::uint64_t hi = eight_digit_bytes(upper);
|
||||
const std::uint64_t lo = eight_digit_bytes(sig - (upper * 100000000u));
|
||||
// trailing zero digits: zero bytes (sig is not 0)
|
||||
const int zeros = lo != 0 ? count_trailing_zeros(lo) / 8 : 8 + (count_trailing_zeros(hi) / 8);
|
||||
const int digits = 15 - zeros + (sixteen ? 1 : 0); // without trailing zeros
|
||||
// the 16 characters from the first digit
|
||||
const std::uint64_t s_hi = (sixteen ? hi : (hi << 8u) | (lo >> 56u)) + 0x3030303030303030u;
|
||||
const std::uint64_t s_lo = (sixteen ? lo : (lo << 8u) | static_cast<std::uint64_t>(last)) + 0x3030303030303030u;
|
||||
const char s16 = static_cast<char>(sixteen ? '0' + last : '0'); // the 17th
|
||||
const auto store_16 = [s_hi, s_lo](char* p) noexcept
|
||||
{
|
||||
store_msb_first(p, s_hi);
|
||||
store_msb_first(p + 8, s_lo);
|
||||
};
|
||||
const auto first_digit = static_cast<char>(s_hi >> 56u);
|
||||
#endif
|
||||
const int len = d.has_digit ? 16 + (sixteen ? 1 : 0) : digits; // significant digits
|
||||
const int n = 16 + (sixteen ? 1 : 0) + d.exponent; // digits before the point
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(n >= 1 && n <= 15))
|
||||
{
|
||||
// "dig.its" and "digits[000].0": the digits after the point move by
|
||||
// one byte ('0's follow the digits)
|
||||
#if JSON_DTOA_SSE2
|
||||
// NOLINTBEGIN(portability-simd-intrinsics)
|
||||
// (in the register: reading the digits back from memory right after
|
||||
// storing them waits until the stores are done)
|
||||
const __m128i at = _mm_set1_epi8(static_cast<char>(n));
|
||||
const __m128i index = _mm_setr_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15);
|
||||
const __m128i before = _mm_cmpgt_epi8(at, index);
|
||||
const __m128i after = _mm_cmpgt_epi8(index, at);
|
||||
const __m128i text = _mm_or_si128(_mm_or_si128(_mm_and_si128(s, before), _mm_and_si128(_mm_slli_si128(s, 1), after)),
|
||||
_mm_andnot_si128(_mm_or_si128(before, after), _mm_set1_epi8('.')));
|
||||
std::memcpy(first, &text, 16);
|
||||
first[16] = static_cast<char>(_mm_extract_epi16(s, 7) >> 8);
|
||||
first[17] = s16;
|
||||
// NOLINTEND(portability-simd-intrinsics)
|
||||
#elif JSON_DTOA_NEON
|
||||
const uint8x16_t index = vcombine_u8(vcreate_u8(0x0706050403020100u), vcreate_u8(0x0F0E0D0C0B0A0908u));
|
||||
const uint8x16_t at = vdupq_n_u8(static_cast<std::uint8_t>(n));
|
||||
const uint8x16_t after_point = vbslq_u8(vcgtq_u8(index, at), vextq_u8(vdupq_n_u8(0), s, 15), vdupq_n_u8('.'));
|
||||
vst1q_u8(reinterpret_cast<std::uint8_t*>(first), vbslq_u8(vcltq_u8(index, at), s, after_point)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
first[16] = static_cast<char>(vgetq_lane_u8(s, 15));
|
||||
first[17] = s16;
|
||||
#else
|
||||
store_16(first);
|
||||
first[16] = s16;
|
||||
std::uint64_t after_point[2]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
|
||||
std::memcpy(after_point, first + n, 16);
|
||||
std::memcpy(first + n + 1, after_point, 16);
|
||||
first[n] = '.';
|
||||
#endif
|
||||
return first + (n >= len ? n + 2 : len + 1);
|
||||
}
|
||||
if (n <= 0 && n > -4)
|
||||
{
|
||||
// "0.[000]digits"
|
||||
std::memset(first, '0', 8);
|
||||
first[1] = '.';
|
||||
store_16(first + 2 - n);
|
||||
first[18 - n] = s16;
|
||||
return first + 2 - n + len;
|
||||
}
|
||||
// d.igitse+XX, with at least two exponent digits (as append_exponent())
|
||||
store_16(first + 1);
|
||||
first[17] = s16;
|
||||
first[0] = first_digit;
|
||||
first[1] = '.';
|
||||
char* const end = first + (len == 1 ? 1 : len + 1);
|
||||
const int e = n - 1;
|
||||
const auto ea = static_cast<unsigned>(e < 0 ? -e : e);
|
||||
const bool three = ea >= 100;
|
||||
end[0] = 'e';
|
||||
end[1] = e < 0 ? '-' : '+';
|
||||
end[2] = static_cast<char>('0' + (three ? ea / 100 : (ea / 10) % 10));
|
||||
end[3] = static_cast<char>('0' + (three ? (ea / 10) % 10 : ea % 10));
|
||||
end[4] = static_cast<char>('0' + (ea % 10));
|
||||
return end + (three ? 5 : 4);
|
||||
}
|
||||
|
||||
/// the powers of ten up to 10^16
|
||||
inline const std::array<std::uint64_t, 17>& powers_of_ten_16() noexcept
|
||||
{
|
||||
static const std::array<std::uint64_t, 17> powers =
|
||||
{
|
||||
{
|
||||
1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u,
|
||||
100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u
|
||||
}
|
||||
};
|
||||
return powers;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief digits * 10^exp, as write_decimal() writes it, for the digits of a
|
||||
double that need no conversion (count digits, at most 15, the first not 0;
|
||||
trailing zeros allowed): extended to 16 digits and written by write_shortest()
|
||||
|
||||
@return a pointer past the text; up to 41 bytes at @a first are written
|
||||
(some beyond the returned end)
|
||||
*/
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept
|
||||
{
|
||||
JSON_ASSERT(digits >= powers_of_ten_16()[static_cast<std::size_t>(count - 1)] && count <= 15);
|
||||
const int scale = 16 - count;
|
||||
return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast<std::size_t>(scale)], exp - scale - 1, 0, false});
|
||||
}
|
||||
|
||||
/// as write_short_decimal(), counting the digits (not 0, less than 10^15)
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept
|
||||
{
|
||||
JSON_ASSERT(digits != 0 && digits < 1000000000000000u);
|
||||
// floor(log10(2^bits)) + 1 digits, or one less
|
||||
const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12;
|
||||
const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast<std::size_t>(log2_bound)] ? 1 : 0);
|
||||
return write_short_decimal(first, digits, count, exp);
|
||||
}
|
||||
|
||||
/// a positive finite float (other than double): Grisu2 and format_buffer()
|
||||
template<typename FloatType>
|
||||
JSON_HEDLEY_NON_NULL(1, 2)
|
||||
@@ -25913,7 +26147,7 @@ char* write_positive(char* first, const char* last, FloatType value)
|
||||
}
|
||||
|
||||
/// a positive finite double: the shortest digits (Zmij), laid out by
|
||||
/// write_decimal() (through a local buffer if [first, last) is shorter than
|
||||
/// write_shortest() (through a local buffer if [first, last) is shorter than
|
||||
/// the 41 bytes it may write)
|
||||
JSON_HEDLEY_NON_NULL(1, 2)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
@@ -25923,13 +26157,13 @@ inline char* write_positive(char* first, const char* last, double value)
|
||||
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
|
||||
std::uint64_t bits = 0;
|
||||
std::memcpy(&bits, &value, sizeof(bits));
|
||||
const zmij::decimal d = zmij::to_decimal(bits);
|
||||
const zmij::shortest_decimal d = zmij::to_shortest(bits);
|
||||
if (JSON_HEDLEY_LIKELY(last - first >= 41))
|
||||
{
|
||||
return write_decimal(first, d.significand, d.exponent);
|
||||
return write_shortest(first, d);
|
||||
}
|
||||
std::array<char, 64> buf; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
|
||||
const auto len = static_cast<std::size_t>(write_decimal(buf.data(), d.significand, d.exponent) - buf.data());
|
||||
const auto len = static_cast<std::size_t>(write_shortest(buf.data(), d) - buf.data());
|
||||
JSON_ASSERT(static_cast<std::size_t>(last - first) >= len);
|
||||
std::memcpy(first, buf.data(), len);
|
||||
return first + len;
|
||||
@@ -27336,8 +27570,9 @@ class serializer
|
||||
/*!
|
||||
@brief dump an integer
|
||||
|
||||
Dump a given integer, appending it to @ref write_buffer. Works internally with
|
||||
@a number_buffer.
|
||||
Dump a given integer, appending it to @ref write_buffer (directly: copying
|
||||
the digits from another buffer right after writing them waits until the
|
||||
stores are done).
|
||||
|
||||
@param[in] x integer number (signed or unsigned) to dump
|
||||
@tparam NumberType either @a number_integer_t or @a number_unsigned_t
|
||||
@@ -27372,33 +27607,57 @@ class serializer
|
||||
return;
|
||||
}
|
||||
|
||||
// use a pointer to fill the buffer
|
||||
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto)
|
||||
// use a pointer to fill the buffer (room for as much as number_buffer holds)
|
||||
if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + number_buffer.size() > write_buffer.size()))
|
||||
{
|
||||
flush();
|
||||
}
|
||||
auto* buffer_ptr = write_buffer.data() + write_buffer_pos;
|
||||
|
||||
number_unsigned_t abs_value;
|
||||
|
||||
unsigned int n_chars{};
|
||||
// one byte for the minus sign
|
||||
unsigned int n_chars = 0;
|
||||
|
||||
if (is_negative_number(x))
|
||||
{
|
||||
*buffer_ptr = '-';
|
||||
abs_value = remove_sign(static_cast<number_integer_t>(x));
|
||||
|
||||
// account one more byte for the minus sign
|
||||
n_chars = 1 + count_digits(abs_value);
|
||||
n_chars = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
abs_value = static_cast<number_unsigned_t>(x);
|
||||
n_chars = count_digits(abs_value);
|
||||
}
|
||||
|
||||
// up to 16 digits: eight at a time (as the digits of floats), written
|
||||
// without leading zeros
|
||||
if (abs_value < 10000000000000000u)
|
||||
{
|
||||
const std::uint64_t value = abs_value;
|
||||
const std::uint64_t upper = value / 100000000u;
|
||||
const std::uint64_t first = dtoa_impl::eight_digit_bytes(upper != 0 ? upper : value);
|
||||
const auto leading = static_cast<unsigned>(count_leading_zeros(first) / 8); // (first is not 0)
|
||||
char* const p = buffer_ptr + n_chars;
|
||||
dtoa_impl::store_msb_first(p, (first << (8 * leading)) + 0x3030303030303030u);
|
||||
n_chars += 8 - leading;
|
||||
if (upper != 0)
|
||||
{
|
||||
dtoa_impl::store_msb_first(p + 8 - leading, dtoa_impl::eight_digit_bytes(value - (upper * 100000000u)) + 0x3030303030303030u);
|
||||
n_chars += 8;
|
||||
}
|
||||
write_buffer_pos += n_chars;
|
||||
return;
|
||||
}
|
||||
|
||||
n_chars += count_digits(abs_value);
|
||||
|
||||
// spare 1 byte for '\0'
|
||||
JSON_ASSERT(n_chars < number_buffer.size() - 1);
|
||||
|
||||
// jump to the end to generate the string from backward,
|
||||
// so we later avoid reversing the result
|
||||
buffer_ptr += static_cast<typename decltype(number_buffer)::difference_type>(n_chars);
|
||||
buffer_ptr += n_chars;
|
||||
|
||||
// Fast int2ascii implementation inspired by "Fastware" talk by Andrei Alexandrescu
|
||||
// See: https://www.youtube.com/watch?v=o4-CwDo2zpg
|
||||
@@ -27421,14 +27680,13 @@ class serializer
|
||||
*(--buffer_ptr) = static_cast<char>('0' + abs_value);
|
||||
}
|
||||
|
||||
put_buffer(number_buffer, n_chars);
|
||||
write_buffer_pos += n_chars;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief dump a floating-point number
|
||||
|
||||
Dump a given floating-point number, appending it to @ref write_buffer. Works internally
|
||||
with @a number_buffer.
|
||||
Dump a given floating-point number, appending it to @ref write_buffer.
|
||||
|
||||
@param[in] x floating-point number to dump
|
||||
*/
|
||||
@@ -27455,10 +27713,15 @@ class serializer
|
||||
|
||||
void dump_float(number_float_t x, std::true_type /*is_ieee_single_or_double*/)
|
||||
{
|
||||
auto* begin = number_buffer.data();
|
||||
// directly into the write buffer: copying the text from number_buffer
|
||||
// right after to_chars() wrote it waits until its stores are done
|
||||
if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + number_buffer.size() > write_buffer.size()))
|
||||
{
|
||||
flush();
|
||||
}
|
||||
auto* begin = write_buffer.data() + write_buffer_pos;
|
||||
auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x);
|
||||
|
||||
put_buffer(number_buffer, static_cast<std::size_t>(end - begin));
|
||||
write_buffer_pos += static_cast<std::size_t>(end - begin);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
@@ -35141,6 +35404,8 @@ struct formatter<nlohmann::NLOHMANN_BASIC_JSON_TPL, char> // NOLINT(cert-dcl58-c
|
||||
#undef JSON_NO_UNIQUE_ADDRESS
|
||||
#undef JSON_DISABLE_ENUM_SERIALIZATION
|
||||
#undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION
|
||||
#undef JSON_DTOA_SSE2
|
||||
#undef JSON_DTOA_NEON
|
||||
|
||||
#ifndef JSON_TEST_KEEP_MACROS
|
||||
#undef JSON_CATCH
|
||||
|
||||
@@ -542,6 +542,7 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
|
||||
#include <array> // array
|
||||
#include <atomic> // atomic
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t, uint64_t
|
||||
|
||||
@@ -551,11 +552,13 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// Vector code for long runs of string bytes. NEON (AArch64) and SSE2 (x86-64)
|
||||
// belong to the baseline instruction sets and are used by default. The vector
|
||||
// UTF-8 check needs NEON, or SSSE3 if JSON_VIEW_USE_SSSE3 is defined: SSSE3 is
|
||||
// not part of x86-64, so it must not depend on the flags of a translation unit
|
||||
// (two translation units with different flags would have different
|
||||
// definitions of the same inline functions). JSON_VIEW_NO_SIMD selects the
|
||||
// portable code.
|
||||
// UTF-8 check needs NEON or SSSE3. SSSE3 is not part of x86-64, and the code
|
||||
// must not depend on the flags of a translation unit (two translation units
|
||||
// with different flags would have different definitions of the same inline
|
||||
// functions): the check is compiled for SSSE3 with a function attribute and
|
||||
// used where the CPU has SSSE3 (all x86-64 CPUs since about 2011), else the
|
||||
// portable check. JSON_VIEW_USE_SSSE3 skips the CPU check (for code compiled
|
||||
// for SSSE3 anyway); JSON_VIEW_NO_SIMD selects the portable code.
|
||||
#if !defined(JSON_VIEW_NO_SIMD) && defined(__aarch64__) && (defined(__GNUC__) || defined(__clang__)) && NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
#include <arm_neon.h>
|
||||
#define NLOHMANN_VIEW_NEON 1
|
||||
@@ -574,8 +577,24 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSSE3 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#endif
|
||||
#if NLOHMANN_VIEW_SSE2 && !NLOHMANN_VIEW_SSSE3 && ((defined(__clang__) && __clang_major__ >= 4) || (defined(__GNUC__) && !defined(__clang__) && (__GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 9))))
|
||||
// (GCC before 4.9 has no SSSE3 intrinsics without -mssse3)
|
||||
#include <cpuid.h>
|
||||
#include <tmmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSSE3_DISPATCH 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#define NLOHMANN_VIEW_SSSE3_TARGET __attribute__((target("ssse3")))
|
||||
#elif NLOHMANN_VIEW_SSE2 && !NLOHMANN_VIEW_SSSE3 && defined(_MSC_VER)
|
||||
// (MSVC compiles intrinsics of any instruction set)
|
||||
#include <intrin.h>
|
||||
#include <tmmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSSE3_DISPATCH 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#define NLOHMANN_VIEW_SSSE3_TARGET
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSSE3_DISPATCH 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#define NLOHMANN_VIEW_SSSE3_TARGET
|
||||
#endif
|
||||
#define NLOHMANN_VIEW_VECTOR (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSE2)
|
||||
#define NLOHMANN_VIEW_VECTOR_UTF8 (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSSE3)
|
||||
#define NLOHMANN_VIEW_VECTOR_UTF8 (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSSE3 || NLOHMANN_VIEW_SSSE3_DISPATCH)
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -623,6 +642,40 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* vector_plain_run(const unsigned
|
||||
}
|
||||
#endif
|
||||
|
||||
#if NLOHMANN_VIEW_SSSE3_DISPATCH
|
||||
/// whether the CPU has SSSE3 (CPUID leaf 1, ECX bit 9)
|
||||
inline bool cpu_ssse3() noexcept
|
||||
{
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
std::array<int, 4> regs {{}};
|
||||
__cpuid(regs.data(), 1);
|
||||
return (static_cast<unsigned>(regs[2]) & (1u << 9u)) != 0;
|
||||
#else
|
||||
unsigned eax = 0;
|
||||
unsigned ebx = 0;
|
||||
unsigned ecx = 0;
|
||||
unsigned edx = 0;
|
||||
return __get_cpuid(1, &eax, &ebx, &ecx, &edx) != 0 && (ecx & (1u << 9u)) != 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
/// whether the CPU has SSSE3, asked once: the answer is kept in an atomic
|
||||
/// that is initialized at compile time, so that neither a guard of a local
|
||||
/// static nor a global constructor is needed (threads that ask at the same
|
||||
/// time all store the same answer)
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE bool cpu_has_ssse3() noexcept
|
||||
{
|
||||
static std::atomic<int> known{0}; // 0: not asked yet, 1: no, 2: yes
|
||||
int state = known.load(std::memory_order_relaxed);
|
||||
if (NLOHMANN_VIEW_UNLIKELY(state == 0))
|
||||
{
|
||||
state = cpu_ssse3() ? 2 : 1;
|
||||
known.store(state, std::memory_order_relaxed);
|
||||
}
|
||||
return state == 2;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
/// Tables of the UTF-8 check of J. Keiser and D. Lemire, "Validating UTF-8 In
|
||||
/// Less Than One Instruction Per Byte" (2021), as in simdjson ("lookup4"): each
|
||||
@@ -725,8 +778,9 @@ compares, and the UTF-8 check covers the bytes up to it. Returns where the
|
||||
string scan stops, like scan_string_run: before ill-formed UTF-8 and for the
|
||||
last bytes of the input, the bytes are checked one sequence at a time. Out of
|
||||
line, so that no constants of the check occupy registers in the parse loop.
|
||||
On x86-64, it is compiled for SSSE3 (see cpu_has_ssse3()).
|
||||
*/
|
||||
NLOHMANN_VIEW_NOINLINE inline const unsigned char* scan_string_vector(const unsigned char* p, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
NLOHMANN_VIEW_SSSE3_TARGET NLOHMANN_VIEW_NOINLINE inline const unsigned char* scan_string_vector(const unsigned char* p, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
{
|
||||
using lookup = utf8_lookup4<>;
|
||||
const unsigned char* block = p;
|
||||
@@ -863,13 +917,19 @@ NLOHMANN_VIEW_ALWAYS_INLINE std::uint16_t load16(const unsigned char* p) noexcep
|
||||
/// in predicted branches: 16 for keys, whose lengths repeat from record to
|
||||
/// record, and 8 for string values (Value) where a vector loop follows, as
|
||||
/// their lengths vary more. Longer runs continue 16 bytes at a time with NEON
|
||||
/// or SSE2, else eight bytes at a time.
|
||||
/// or SSE2, else eight bytes at a time. With SSE2, the run is checked 16 bytes
|
||||
/// at a time from its first byte instead: on x86-64, one compare that finds
|
||||
/// the end of most keys and short values is faster than a branch per byte (on
|
||||
/// AArch64, where a NEON mask costs more and branches predict well, slower).
|
||||
template<bool Value = false>
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned char* p, const unsigned char* e) noexcept
|
||||
{
|
||||
const std::uint8_t* plain = string_plain();
|
||||
for (;;)
|
||||
{
|
||||
#if NLOHMANN_VIEW_SSE2
|
||||
p = vector_plain_run(p, e);
|
||||
#else
|
||||
if (e - p >= 16)
|
||||
{
|
||||
#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(plain[p[i]] != 0)) {} else { p += (i); goto stop; }
|
||||
@@ -903,6 +963,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
#endif
|
||||
continue;
|
||||
}
|
||||
#endif
|
||||
while (p != e && plain[*p] != 0)
|
||||
{
|
||||
++p;
|
||||
@@ -911,15 +972,23 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
{
|
||||
return p;
|
||||
}
|
||||
#if !NLOHMANN_VIEW_SSE2
|
||||
stop:
|
||||
#endif
|
||||
if (*p < 0x80)
|
||||
{
|
||||
return p; // quote, backslash, or control character
|
||||
}
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
// non-ASCII: the vector check, out of line
|
||||
return scan_string_vector(p, e, plain);
|
||||
#else
|
||||
#if NLOHMANN_VIEW_SSSE3_DISPATCH
|
||||
if (NLOHMANN_VIEW_LIKELY(cpu_has_ssse3()))
|
||||
#endif
|
||||
{
|
||||
// non-ASCII: the vector check, out of line
|
||||
return scan_string_vector(p, e, plain);
|
||||
}
|
||||
#endif
|
||||
#if !NLOHMANN_VIEW_VECTOR_UTF8 || NLOHMANN_VIEW_SSSE3_DISPATCH
|
||||
// non-ASCII: a run of well-formed sequences (the library's check, so
|
||||
// that exactly what json::parse accepts is accepted)
|
||||
do
|
||||
@@ -1813,14 +1882,19 @@ indent_done:
|
||||
const auto idx = static_cast<std::uint32_t>(emit(k, 0, 0, static_cast<std::size_t>(p - b), 0) - base);
|
||||
if (depth != 0)
|
||||
{
|
||||
const frame f = {cur_idx, cur_count, cur_is_object};
|
||||
if (NLOHMANN_VIEW_LIKELY(depth <= 64))
|
||||
{
|
||||
cold.shallow[depth - 1] = f;
|
||||
// field by field: a frame put together on the stack and
|
||||
// copied would be read back wider than it was written,
|
||||
// and that load waits until the stores are done
|
||||
frame& f = cold.shallow[depth - 1];
|
||||
f.idx = cur_idx;
|
||||
f.count = cur_count;
|
||||
f.is_object = cur_is_object;
|
||||
}
|
||||
else
|
||||
{
|
||||
cold.deep.push_back(f);
|
||||
cold.deep.push_back(frame{cur_idx, cur_count, cur_is_object});
|
||||
}
|
||||
}
|
||||
++depth;
|
||||
@@ -1836,19 +1910,21 @@ indent_done:
|
||||
n.next = static_cast<std::uint32_t>(out - base) - cur_idx;
|
||||
if (--depth != 0)
|
||||
{
|
||||
frame f{};
|
||||
if (NLOHMANN_VIEW_LIKELY(depth <= 64))
|
||||
{
|
||||
f = cold.shallow[depth - 1];
|
||||
const frame& f = cold.shallow[depth - 1];
|
||||
cur_idx = f.idx;
|
||||
cur_count = f.count;
|
||||
cur_is_object = f.is_object;
|
||||
}
|
||||
else
|
||||
{
|
||||
f = cold.deep.back();
|
||||
const frame f = cold.deep.back();
|
||||
cold.deep.pop_back();
|
||||
cur_idx = f.idx;
|
||||
cur_count = f.count;
|
||||
cur_is_object = f.is_object;
|
||||
}
|
||||
cur_idx = f.idx;
|
||||
cur_count = f.count;
|
||||
cur_is_object = f.is_object;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5325,8 +5401,6 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// #include <nlohmann/detail/view/number.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/simd.hpp>
|
||||
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -5335,7 +5409,11 @@ namespace view
|
||||
{
|
||||
|
||||
/// append-only output buffer: writes through a raw pointer into a string that
|
||||
/// is resized ahead, and trimmed by finish()
|
||||
/// is resized ahead, and trimmed by finish(). The estimate is reserved, and the
|
||||
/// string grows in steps of 64 KiB within it: resize() fills the new bytes
|
||||
/// with zeros (before C++23, a string cannot grow without), and a small step
|
||||
/// is filled while the writer is about to use it, in the cache, instead of
|
||||
/// filling the whole estimate in memory first.
|
||||
template<typename StringType>
|
||||
class output_buffer
|
||||
{
|
||||
@@ -5397,16 +5475,31 @@ class output_buffer
|
||||
}
|
||||
|
||||
private:
|
||||
/// the size of a growth step (a function: std::min() takes a reference,
|
||||
/// which a static constexpr member does not have before C++17)
|
||||
static constexpr std::size_t step() noexcept
|
||||
{
|
||||
return 65536;
|
||||
}
|
||||
|
||||
static StringType& sized(StringType& out, std::size_t estimate)
|
||||
{
|
||||
out.resize((std::max)(estimate, static_cast<std::size_t>(64)));
|
||||
out.reserve(estimate);
|
||||
out.resize((std::min)((std::max)(estimate, static_cast<std::size_t>(64)), step()));
|
||||
return out;
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_NOINLINE void grow(std::size_t n)
|
||||
{
|
||||
const auto used = static_cast<std::size_t>(m_pos - m_out.data());
|
||||
m_out.resize((std::max)(m_out.size() * 2, used + n + 256));
|
||||
// (a step does not go beyond the reserved estimate, so that a good
|
||||
// estimate is never copied to a larger allocation)
|
||||
const std::size_t size = (std::max)((std::min)(m_out.size() + step(), m_out.capacity()), used + n + 256);
|
||||
if (size > m_out.capacity())
|
||||
{
|
||||
m_out.reserve((std::max)(m_out.capacity() * 2, size));
|
||||
}
|
||||
m_out.resize(size);
|
||||
m_pos = &m_out[0] + used;
|
||||
m_end = &m_out[0] + m_out.size();
|
||||
}
|
||||
@@ -5525,69 +5618,6 @@ struct dump_style
|
||||
bool source_numbers = false; ///< copy number tokens from the source
|
||||
};
|
||||
|
||||
/*!
|
||||
@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0,
|
||||
at most 17 digits; up to 41 bytes are written at first)
|
||||
|
||||
With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in
|
||||
vector registers: output byte i is byte s + i of the digits (after '0's)
|
||||
before the point, and byte s + i - 1 after it. The portable code writes the
|
||||
digits to a buffer and copies them from there at another offset, and a load
|
||||
that spans several recent stores waits until they reach the cache.
|
||||
*/
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept
|
||||
{
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
namespace dtoa = ::nlohmann::detail::dtoa_impl;
|
||||
const std::uint64_t upper = digits / 100000000u;
|
||||
const std::uint64_t b0 = upper / 100000000u; // one digit
|
||||
const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u);
|
||||
const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u);
|
||||
// leading and trailing zero digits (as dtoa_impl::write_decimal())
|
||||
int leading = 7;
|
||||
if (b0 == 0)
|
||||
{
|
||||
leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8);
|
||||
}
|
||||
int zeros = 16;
|
||||
if (b2 != 0)
|
||||
{
|
||||
zeros = count_trailing_zeros(b2) / 8;
|
||||
}
|
||||
else if (b1 != 0)
|
||||
{
|
||||
zeros = 8 + (count_trailing_zeros(b1) / 8);
|
||||
}
|
||||
const int k = 24 - leading - zeros; // significant digits
|
||||
const int n = k + exp + zeros; // position of the point after the first digit
|
||||
if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15))
|
||||
{
|
||||
static const std::array<std::uint8_t, 32> iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}};
|
||||
const int pad = n <= 0 ? 1 - n : 0;
|
||||
const int len = k + pad;
|
||||
const int point = n + pad;
|
||||
// the 24 digit bytes in memory order, then '0's
|
||||
const std::uint64_t zero_chars = 0x3030303030303030u;
|
||||
const uint8x16x2_t table = {{
|
||||
vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))),
|
||||
vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0'))
|
||||
}
|
||||
};
|
||||
const uint8x16_t s = vdupq_n_u8(static_cast<std::uint8_t>(leading - pad));
|
||||
const uint8x16_t at_point = vdupq_n_u8(static_cast<std::uint8_t>(point));
|
||||
for (std::size_t half = 0; half < 2; ++half)
|
||||
{
|
||||
const uint8x16_t i = vld1q_u8(iota.data() + (16 * half));
|
||||
// (+ 0xFF is - 1 after the point; indexes past the digits read a '0')
|
||||
const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31));
|
||||
vst1q_u8(reinterpret_cast<std::uint8_t*>(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
}
|
||||
return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0"
|
||||
}
|
||||
#endif
|
||||
return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief write a view's subtree as basic_json::dump() writes the value
|
||||
|
||||
@@ -6089,7 +6119,11 @@ class view_serializer
|
||||
if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290)
|
||||
{
|
||||
*w = '-';
|
||||
return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast<int>(d.exponent));
|
||||
w += d.negative ? 1 : 0;
|
||||
// (without leading zeros, all digits of the token count)
|
||||
const unsigned char lead = first[d.negative ? 1 : 0];
|
||||
return lead != '0' ? ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast<int>(int_digits + frac_digits), static_cast<int>(d.exponent))
|
||||
: ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast<int>(d.exponent));
|
||||
}
|
||||
return write_double_value_at(w, decimal_to_float<double>(d)); // (without reading the token again)
|
||||
}
|
||||
@@ -6106,13 +6140,13 @@ class view_serializer
|
||||
/// a double as dump() writes it, at w (64 bytes of room)
|
||||
static char* write_double_value_at(char* w, double x)
|
||||
{
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x)))
|
||||
// (from the bits: without the checks of to_chars())
|
||||
std::uint64_t bits = 0;
|
||||
std::memcpy(&bits, &x, sizeof(bits));
|
||||
if (NLOHMANN_VIEW_UNLIKELY((bits & 0x7FF0000000000000u) == 0x7FF0000000000000u))
|
||||
{
|
||||
return write_text_at(w, "null", 4);
|
||||
}
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
std::uint64_t bits = 0;
|
||||
std::memcpy(&bits, &x, sizeof(bits));
|
||||
*w = '-';
|
||||
w += bits >> 63u;
|
||||
bits &= ~(std::uint64_t{1} << 63u);
|
||||
@@ -6120,11 +6154,7 @@ class view_serializer
|
||||
{
|
||||
return write_text_at(w, "0.0", 3);
|
||||
}
|
||||
const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits);
|
||||
return write_decimal(w, d.significand, d.exponent);
|
||||
#else
|
||||
return ::nlohmann::detail::to_chars(w, w + 64, x);
|
||||
#endif
|
||||
return ::nlohmann::detail::dtoa_impl::write_shortest(w, ::nlohmann::detail::zmij::to_shortest(bits));
|
||||
}
|
||||
|
||||
/// as serializer::dump_float()
|
||||
@@ -7891,6 +7921,8 @@ class tuple_element<N, ::nlohmann::detail::view::view_item<View>> // NOLINT(cert
|
||||
#undef NLOHMANN_VIEW_NEON
|
||||
#undef NLOHMANN_VIEW_SSE2
|
||||
#undef NLOHMANN_VIEW_SSSE3
|
||||
#undef NLOHMANN_VIEW_SSSE3_DISPATCH
|
||||
#undef NLOHMANN_VIEW_SSSE3_TARGET
|
||||
#undef NLOHMANN_VIEW_VECTOR
|
||||
#undef NLOHMANN_VIEW_VECTOR_UTF8
|
||||
|
||||
|
||||
@@ -50,7 +50,8 @@ JSON-RPC request (`rpc`):
|
||||
| dump | serialize a parsed document (compact) |
|
||||
|
||||
`bench_corpus.cpp` runs parse, traverse, and dump on any list of files, so that no library is tuned to a handful of
|
||||
documents.
|
||||
documents. Its dump also writes the numbers as they are in the input: `json_view` with `number_format::source`, and
|
||||
yyjson with numbers read as raw text (`YYJSON_READ_NUMBER_AS_RAW`), without converting them.
|
||||
|
||||
`bench_edit.cpp` measures read-modify-write: parse, apply the same logical edits with each library's own API, and
|
||||
serialize (compact). Workloads: `patch` (a handful of edits at fixed places) and `update` (edits in every record).
|
||||
@@ -60,7 +61,10 @@ outputs are checked to describe the same value.
|
||||
|
||||
Before anything is timed, all engines must accept each document and agree on the traversal: the number of values, the
|
||||
bytes of all strings and keys, and the sum of all numbers. All engines run interleaved in every round, and the best
|
||||
round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`).
|
||||
round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`). Each timed
|
||||
call follows an untimed call of the same engine: otherwise the engine after `json::parse` pays for the allocator
|
||||
cleaning up the tens of thousands of nodes `json::parse` just freed (with glibc, this made `json_view` look 1.7 times
|
||||
slower on citm_catalog traverse).
|
||||
|
||||
The engines do not all offer the same features, which the numbers should be read with:
|
||||
|
||||
@@ -68,11 +72,18 @@ The engines do not all offer the same features, which the numbers should be read
|
||||
|---|---|---|---|---|
|
||||
| `json_view` | immutable index into the text | yes | no | a fresh document per parse; "reused" parses into the same document |
|
||||
| yyjson | immutable (`yyjson_read`) | yes | via a mutable copy | |
|
||||
| simdjson DOM | immutable, parser reused | yes | no | |
|
||||
| simdjson DOM | immutable, parser reused | yes | no | "fresh" uses a new parser per parse |
|
||||
| simdjson On-Demand | none: forward-only, lazy | no | no | only traverse and select |
|
||||
| Boost.JSON | owning, mutable DOM | yes | yes | monotonic resource |
|
||||
| `json::parse` | owning, mutable DOM | yes | yes | |
|
||||
|
||||
Reusing memory matters as much as the parser. simdjson DOM reuses its parser, so it writes into memory it already
|
||||
touched; a fresh `json_view` document or yyjson document gets new memory for every parse. On Linux, glibc returns large
|
||||
blocks to the system when they are freed, so every fresh parse of a large document pays a page fault per 4 KiB page:
|
||||
on x86-64 Linux, a fresh `json_view` parse of jeopardy took about twice as long as a reused one. On macOS on Apple
|
||||
silicon, with 16 KiB pages, the difference is much smaller. Compare "json_view (reused)" with "simdjson DOM", and the
|
||||
fresh `json_view` with "simdjson DOM (fresh)" and yyjson.
|
||||
|
||||
## Published results
|
||||
|
||||
Results are only published with the file `compare.py` wrote, which names the machine and the versions; see
|
||||
|
||||
@@ -15,7 +15,8 @@
|
||||
// count, string bytes, sum of numbers) before anything is timed. Workloads:
|
||||
// parse (build and free a document), traverse (visit every value, convert
|
||||
// every number), dump (compact), and for json_view also dump with the source
|
||||
// number text. Results go to bench_corpus.csv.
|
||||
// number text, compared with yyjson writing numbers read as raw text
|
||||
// (YYJSON_READ_NUMBER_AS_RAW). Results go to bench_corpus.csv.
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
@@ -29,6 +30,7 @@
|
||||
#include <chrono>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
@@ -195,6 +197,11 @@ static void walk(const boost::json::value& v, stats& st)
|
||||
static std::string slurp(const std::string& p)
|
||||
{
|
||||
std::ifstream f(p, std::ios::binary);
|
||||
if (!f)
|
||||
{
|
||||
std::fprintf(stderr, "cannot open %s\n", p.c_str());
|
||||
std::exit(1);
|
||||
}
|
||||
std::stringstream ss;
|
||||
ss << f.rdbuf();
|
||||
return ss.str();
|
||||
@@ -222,6 +229,7 @@ int main(int argc, char** argv)
|
||||
}
|
||||
std::FILE* csv = std::fopen("bench_corpus.csv", "w");
|
||||
std::fprintf(csv, "file,bytes,workload,engine,ns\n");
|
||||
json_document reused;
|
||||
simdjson::dom::parser sj;
|
||||
for (const auto& path : files)
|
||||
{
|
||||
@@ -265,6 +273,7 @@ int main(int argc, char** argv)
|
||||
};
|
||||
json_document vd = json_document::parse(s);
|
||||
yyjson_doc* yd = yyjson_read(s.data(), s.size(), 0);
|
||||
yyjson_doc* yd_raw = yyjson_read(s.data(), s.size(), YYJSON_READ_NUMBER_AS_RAW);
|
||||
simdjson::dom::parser sjd;
|
||||
const simdjson::dom::element se = sjd.parse(ps).value_unsafe();
|
||||
const std::vector<std::pair<std::string, std::vector<engine>>> workloads =
|
||||
@@ -272,8 +281,10 @@ int main(int argc, char** argv)
|
||||
{
|
||||
"parse", {
|
||||
{"json_view", [&] { auto x = json_document::parse(s); g_sink = static_cast<double>(x.node_count()); }},
|
||||
{"json_view (reused)", [&] { reused.read(s); g_sink = static_cast<double>(reused.node_count()); }},
|
||||
{"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast<double>(yyjson_doc_get_val_count(x)); yyjson_doc_free(x); }},
|
||||
{"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
|
||||
{"simdjson DOM (fresh)", [&] { simdjson::dom::parser p; auto e = p.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }},
|
||||
#endif
|
||||
@@ -295,6 +306,7 @@ int main(int argc, char** argv)
|
||||
{"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast<double>(n); std::free(o); }},
|
||||
{"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast<double>(o.size()); }},
|
||||
{"json_view (source numbers)", [&] { std::string o = vd.root().dump(-1, ' ', false, json_view::number_format::source); g_sink = static_cast<double>(o.size()); }},
|
||||
{"yyjson (raw numbers)", [&] { std::size_t n = 0; char* o = yyjson_write(yd_raw, 0, &n); g_sink = static_cast<double>(n); std::free(o); }},
|
||||
}
|
||||
},
|
||||
};
|
||||
@@ -306,6 +318,9 @@ int main(int argc, char** argv)
|
||||
{
|
||||
for (std::size_t k = 0; k < wl.second.size(); ++k)
|
||||
{
|
||||
// an untimed call first: whatever the previous engine left to the allocator
|
||||
// (e.g. thousands of freed json nodes) is cleaned up here, not in the timing
|
||||
wl.second[k].fn();
|
||||
const auto t0 = std::chrono::steady_clock::now();
|
||||
wl.second[k].fn();
|
||||
best[k] = std::min(best[k], std::chrono::duration<double, std::nano>(std::chrono::steady_clock::now() - t0).count());
|
||||
@@ -321,6 +336,7 @@ int main(int argc, char** argv)
|
||||
std::fflush(stdout);
|
||||
}
|
||||
yyjson_doc_free(yd);
|
||||
yyjson_doc_free(yd_raw);
|
||||
}
|
||||
std::fclose(csv);
|
||||
}
|
||||
@@ -479,6 +479,11 @@ static std::string edit_boost(const std::string& name, const std::string& s, boo
|
||||
static std::string slurp(const std::string& p)
|
||||
{
|
||||
std::ifstream f(p, std::ios::binary);
|
||||
if (!f)
|
||||
{
|
||||
std::fprintf(stderr, "cannot open %s\n", p.c_str());
|
||||
std::exit(1);
|
||||
}
|
||||
std::stringstream ss;
|
||||
ss << f.rdbuf();
|
||||
return ss.str();
|
||||
@@ -550,6 +555,9 @@ int main(int argc, char** argv)
|
||||
{
|
||||
for (std::size_t k = 0; k < engines.size(); ++k)
|
||||
{
|
||||
// an untimed call first: whatever the previous engine left to the allocator
|
||||
// (e.g. thousands of freed json nodes) is cleaned up here, not in the timing
|
||||
g_sink = engines[k].second(dc.name, dc.text, update).size();
|
||||
const auto t0 = std::chrono::steady_clock::now();
|
||||
for (int b = 0; b < dc.batch; ++b)
|
||||
{
|
||||
|
||||
@@ -10,7 +10,8 @@
|
||||
//
|
||||
// json_view nlohmann/json_view.hpp (fresh document per parse / reused)
|
||||
// yyjson yyjson_read(): immutable document, random access
|
||||
// simdjson DOM dom::parser (reused, as recommended): immutable, random access
|
||||
// simdjson DOM dom::parser (reused, as recommended; "fresh": a new parser
|
||||
// per parse): immutable, random access
|
||||
// references (different feature sets):
|
||||
// simdjson OD On-Demand: forward-only, lazy
|
||||
// Boost.JSON owning, mutable DOM (monotonic resource)
|
||||
@@ -19,7 +20,8 @@
|
||||
// Workloads: parse (build + free), traverse (visit everything, convert every
|
||||
// number, touch every string and key), select (a few fields per document),
|
||||
// dump (compact serialization of the parsed document).
|
||||
// All engines run interleaved in every round; the best round is reported.
|
||||
// All engines run interleaved in every round, each timed call after an untimed
|
||||
// one of the same engine; the best round is reported.
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
@@ -33,6 +35,7 @@
|
||||
#include <chrono>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
#include <map>
|
||||
@@ -581,6 +584,11 @@ static double pick_od(const std::string& name, simdjson::ondemand::document& d)
|
||||
static std::string slurp(const std::string& p)
|
||||
{
|
||||
std::ifstream f(p, std::ios::binary);
|
||||
if (!f)
|
||||
{
|
||||
std::fprintf(stderr, "cannot open %s\n", p.c_str());
|
||||
std::exit(1);
|
||||
}
|
||||
std::stringstream ss;
|
||||
ss << f.rdbuf();
|
||||
return ss.str();
|
||||
@@ -657,6 +665,7 @@ int main(int argc, char** argv)
|
||||
{"json_view (reused)", [&] { reused.read(s); g_sink = static_cast<double>(reused.node_count()); }},
|
||||
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast<double>(yyjson_doc_get_val_count(d)); yyjson_doc_free(d); }},
|
||||
{"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
|
||||
{"simdjson DOM (fresh)", [&] { simdjson::dom::parser p; auto e = p.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }},
|
||||
#endif
|
||||
@@ -664,6 +673,7 @@ int main(int argc, char** argv)
|
||||
}});
|
||||
workloads.push_back({"traverse", {
|
||||
{"json_view", [&] { auto d = json_document::parse(s); stats st; walk(d.root(), st); g_sink = st.num; }},
|
||||
{"json_view (reused)", [&] { reused.read(s); stats st; walk(reused.root(), st); g_sink = st.num; }},
|
||||
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(d), st); g_sink = st.num; yyjson_doc_free(d); }},
|
||||
{"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }},
|
||||
{"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); stats st; walk_od(d.get_value().value_unsafe(), st); g_sink = st.num; }},
|
||||
@@ -674,6 +684,7 @@ int main(int argc, char** argv)
|
||||
}});
|
||||
workloads.push_back({"select", {
|
||||
{"json_view", [&] { auto d = json_document::parse(s); g_sink = pick(name, d.root()); }},
|
||||
{"json_view (reused)", [&] { reused.read(s); g_sink = pick(name, reused.root()); }},
|
||||
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = pick(name, yyjson_doc_get_root(d)); yyjson_doc_free(d); }},
|
||||
{"simdjson DOM", [&] { g_sink = pick(name, sj.parse(ps).value_unsafe()); }},
|
||||
{"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); g_sink = pick_od(name, d); }},
|
||||
@@ -710,6 +721,9 @@ int main(int argc, char** argv)
|
||||
{
|
||||
for (std::size_t k = 0; k < wl.second.size(); ++k)
|
||||
{
|
||||
// an untimed call first: whatever the previous engine left to the allocator
|
||||
// (e.g. thousands of freed json nodes) is cleaned up here, not in the timing
|
||||
wl.second[k].fn();
|
||||
const auto t0 = std::chrono::steady_clock::now();
|
||||
for (int b = 0; b < dc.batch; ++b)
|
||||
{
|
||||
|
||||
@@ -154,11 +154,14 @@ def download_library(name, work):
|
||||
if not os.path.isfile(archive):
|
||||
print(f'downloading {pin["url"]}', flush=True)
|
||||
# the URLs are the https constants in PINNED, and the SHA-256 is checked below
|
||||
urllib.request.urlretrieve(pin['url'], archive) # nosec B310
|
||||
# (into a .part file first, so that an interrupted download is not kept)
|
||||
urllib.request.urlretrieve(pin['url'], archive + '.part') # nosec B310
|
||||
os.replace(archive + '.part', archive)
|
||||
with open(archive, 'rb') as f:
|
||||
digest = hashlib.sha256(f.read()).hexdigest()
|
||||
if digest != pin['sha256']:
|
||||
sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]}')
|
||||
os.remove(archive) # downloaded again by the next run
|
||||
sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]} (removed)')
|
||||
src = os.path.join(work, 'download', pin['dir'])
|
||||
if not os.path.isdir(src):
|
||||
with tarfile.open(archive) as t:
|
||||
@@ -214,6 +217,10 @@ def main():
|
||||
ap.add_argument('--corpus', nargs='*', default=[], help='more files for bench_corpus')
|
||||
ap.add_argument('--build-dir', default=os.path.join(HERE, 'build'), help='where to build (default: build/ next to this script)')
|
||||
args = ap.parse_args()
|
||||
# the benchmarks run in the build directory: make the paths absolute
|
||||
args.data = os.path.abspath(args.data)
|
||||
args.corpus = [os.path.abspath(f) for f in args.corpus]
|
||||
args.build_dir = os.path.abspath(args.build_dir)
|
||||
|
||||
cxx = os.environ.get('CXX', 'c++')
|
||||
cc = os.environ.get('CC', 'cc')
|
||||
|
||||
Reference in new issue
Block a user