Format the digits of numbers in vector registers and without reloads

- zmij::to_shortest() keeps the last digit apart from the 15 or 16 digits
  before it, and dtoa_impl::write_shortest() converts those at once: with
  SSE2 on x86-64 and NEON on 64-bit Arm (no CPU check needed), else eight
  digits at a time. The decimal point is inserted in the register; reading
  the digits back right after storing them stalled store forwarding.
- json::dump() writes floats and integers straight into its write buffer
  instead of copying them from number_buffer, and integers below 10^16
  eight digits at a time.
- json_view's dump() writes doubles from their bits and tokens of up to 15
  digits through the same code; its own NEON writer is removed.

to_chars() on canada.json: 35 -> 29 ns per double (x86-64), 18.8 -> 14.2 ns
(Apple M1); json::dump() of canada.json 16-29% faster, of citm_catalog.json
14-19%.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann committed 2026-10-05 22:28:58 +02:00
1 parent 42f8043413
commit 44fba02ec2
7 files changed
+598 -201

No files matched your search

+289 -26
View File
@@ -24428,6 +24428,21 @@ NLOHMANN_JSON_NAMESPACE_END
#include <cstdlib> // _byteswap_uint64
#endif
// SSE2 (every x86-64 CPU) and NEON (every 64-bit Arm CPU) convert the 16
// digits of a double at once
#if defined(__x86_64__) || (defined(_M_X64) && !defined(_M_ARM64EC))
#include <emmintrin.h>
#define JSON_DTOA_SSE2 1
#define JSON_DTOA_NEON 0
#elif (defined(__aarch64__) || defined(_M_ARM64)) && !defined(_M_ARM64EC) && !defined(__ARM_BIG_ENDIAN)
#include <arm_neon.h>
#define JSON_DTOA_SSE2 0
#define JSON_DTOA_NEON 1
#else
#define JSON_DTOA_SSE2 0
#define JSON_DTOA_NEON 0
#endif
// #include <nlohmann/detail/conversions/zmij.hpp>
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
@@ -24592,10 +24607,22 @@ inline std::uint64_t umul128_add_hi64(std::uint64_t x, std::uint64_t y, std::uin
return p.high + (p.low + c < p.low ? 1u : 0u);
}
/// the result of Zmij: the shorter candidate and, if that is outside the
/// rounding interval, the digit after it (16 bytes: returned in registers)
struct shortest_decimal
{
std::uint64_t integral; ///< the shorter candidate (15 or 16 digits for normal doubles)
int exponent; ///< the decimal exponent of the digit after it
unsigned char digit; ///< the digit after it (if has_digit)
bool has_digit; ///< whether the shortest decimal is integral * 10 + digit
};
/// The shortest decimal in the rounding interval of a positive finite double
/// given by its bits, the closest one if there are several (to_decimal of
/// Zmij). The significand can end in zeros.
inline decimal to_decimal(std::uint64_t bits) noexcept
/// Zmij, which keeps the last digit apart: the 15 or 16 digits before it can be
/// converted without a multiplication by 10 first). Always inlined: GCC
/// otherwise calls it, and its result goes through memory.
JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexcept
{
constexpr int extra_shift = 9;
const auto raw_exp = static_cast<int>((bits >> 52u) & 0x7FFu);
@@ -24640,12 +24667,20 @@ inline decimal to_decimal(std::uint64_t bits) noexcept
digit = digit < lowest ? lowest : digit;
}
integral += round_up ? 1u : 0u;
if (!round_up && !round_down)
// if the shorter candidate is outside the rounding interval: one digit more
return shortest_decimal{integral, dec_exp, static_cast<unsigned char>(digit), !round_up && !round_down};
}
/// The shortest decimal in the rounding interval of a positive finite double
/// given by its bits, as one number. The significand can end in zeros.
inline decimal to_decimal(std::uint64_t bits) noexcept
{
const shortest_decimal d = to_shortest(bits);
if (d.has_digit)
{
// the shorter candidate is outside the rounding interval: one digit more
return decimal{(integral * 10) + digit, dec_exp};
return decimal{(d.integral * 10) + d.digit, d.exponent};
}
return decimal{integral, dec_exp + 1};
return decimal{d.integral, d.exponent + 1};
}
} // namespace zmij
@@ -25884,6 +25919,203 @@ inline char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept
return end + (three ? 5 : 4);
}
/*!
@brief the shortest decimal of a positive double (Zmij), as write_decimal()
writes it
For a normal double, the shorter candidate has 15 or 16 digits: they are
converted at once (two halves of eight digits) and followed by the digit
after them, if there is one, without the multiplication and division by 10
that counting the digits of one number would take. The fixed layouts move
the digits after the point by one byte.
@return a pointer past the text; up to 41 bytes at @a first are written
(some beyond the returned end)
*/
JSON_HEDLEY_NON_NULL(1)
JSON_HEDLEY_RETURNS_NON_NULL
inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcept
{
const std::uint64_t sig = d.integral;
if (JSON_HEDLEY_UNLIKELY(sig < 100000000000000u || sig >= 10000000000000000u))
{
// (subnormals)
return d.has_digit ? write_decimal(first, (sig * 10) + d.digit, d.exponent) : write_decimal(first, sig, d.exponent + 1);
}
const bool sixteen = sig >= 1000000000000000u; // (else 15 digits)
const int last = d.has_digit ? d.digit : 0;
const std::uint64_t upper = sig / 100000000u;
#if JSON_DTOA_SSE2
// NOLINTBEGIN(portability-simd-intrinsics)
// the two halves in the 64-bit lanes, each as abcd * 2^32 + efgh, then as
// bytes (as eight_digit_bytes(), one lane each)
const __m128i x = _mm_set_epi64x(static_cast<long long>(sig - (upper * 100000000u)), static_cast<long long>(upper));
const __m128i abcd = _mm_srli_epi64(_mm_mul_epu32(x, _mm_set1_epi64x(109951163)), 40); // 2^40 / 10000 + 1
const __m128i abcd_efgh = _mm_add_epi64(x, _mm_mul_epu32(abcd, _mm_set1_epi64x(4294957296))); // 2^32 - 10000
// 32-bit lanes in the order of the text: abcd, efgh of both halves
const __m128i fours = _mm_shuffle_epi32(abcd_efgh, _MM_SHUFFLE(2, 3, 0, 1));
const __m128i ab = _mm_srli_epi16(_mm_mulhi_epu16(fours, _mm_set1_epi32(5243)), 3);
const __m128i ab_cd = _mm_or_si128(_mm_slli_epi32(_mm_sub_epi16(fours, _mm_mullo_epi16(ab, _mm_set1_epi32(100))), 16), ab);
// 16-bit lanes ab (< 100) -> bytes a, b: 256 * ab - 2559 * (ab / 10)
const __m128i bytes = _mm_sub_epi16(_mm_slli_epi16(ab_cd, 8), _mm_mullo_epi16(_mm_set1_epi16(2559), _mm_mulhi_epu16(ab_cd, _mm_set1_epi16(6554))));
// the last digit that is not 0 (sig is not 0)
const auto nonzero = static_cast<std::uint64_t>(_mm_movemask_epi8(_mm_cmpgt_epi8(bytes, _mm_setzero_si128())));
const int digits = 63 - count_leading_zeros(nonzero) + (sixteen ? 1 : 0); // without trailing zeros
const __m128i chars = _mm_add_epi8(bytes, _mm_set1_epi8('0'));
// the 16 characters from the first digit
const __m128i s = sixteen ? chars : _mm_or_si128(_mm_srli_si128(chars, 1), _mm_slli_si128(_mm_cvtsi32_si128('0' + last), 15));
const char s16 = static_cast<char>(sixteen ? '0' + last : '0'); // the 17th
const auto store_16 = [&s](char* p) noexcept
{
std::memcpy(p, &s, 16);
};
const char first_digit = static_cast<char>(_mm_cvtsi128_si32(s));
// NOLINTEND(portability-simd-intrinsics)
#elif JSON_DTOA_NEON
// as with SSE2: the halves in 32-bit lanes, then abcd, efgh of both
const uint32x2_t halves = vcreate_u32(upper | ((sig - (upper * 100000000u)) << 32u));
const uint32x2_t abcd = vmovn_u64(vshrq_n_u64(vmull_n_u32(halves, static_cast<std::uint32_t>(((std::uint64_t{1} << 40u) / 10000u) + 1u)), 40));
const uint32x2_t efgh = vmls_n_u32(halves, abcd, 10000u);
const uint32x4_t fours = vcombine_u32(vzip1_u32(abcd, efgh), vzip2_u32(abcd, efgh));
const uint32x4_t ab = vshrq_n_u32(vmulq_n_u32(fours, 5243u), 19);
const uint16x8_t ab_cd = vreinterpretq_u16_u32(vorrq_u32(ab, vshlq_n_u32(vmlsq_n_u32(fours, ab, 100u), 16)));
const uint16x8_t tens = vshrq_n_u16(vmulq_n_u16(ab_cd, 103u), 10);
const uint8x16_t bytes = vreinterpretq_u8_u16(vorrq_u16(tens, vshlq_n_u16(vmlsq_n_u16(ab_cd, tens, 10u), 8)));
// the last digit that is not 0 (sig is not 0): a nibble per byte
const std::uint64_t nonzero = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(bytes, bytes)), 4)), 0);
const int digits = ((63 - count_leading_zeros(nonzero)) / 4) + (sixteen ? 1 : 0); // without trailing zeros
const uint8x16_t chars = vaddq_u8(bytes, vdupq_n_u8('0'));
// the 16 characters from the first digit
const uint8x16_t s = sixteen ? chars : vextq_u8(chars, vdupq_n_u8(static_cast<std::uint8_t>('0' + last)), 1);
const char s16 = static_cast<char>(sixteen ? '0' + last : '0'); // the 17th
const auto store_16 = [&s](char* p) noexcept
{
vst1q_u8(reinterpret_cast<std::uint8_t*>(p), s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
};
const auto first_digit = static_cast<char>(vgetq_lane_u8(s, 0));
#else
const std::uint64_t hi = eight_digit_bytes(upper);
const std::uint64_t lo = eight_digit_bytes(sig - (upper * 100000000u));
// trailing zero digits: zero bytes (sig is not 0)
const int zeros = lo != 0 ? count_trailing_zeros(lo) / 8 : 8 + (count_trailing_zeros(hi) / 8);
const int digits = 15 - zeros + (sixteen ? 1 : 0); // without trailing zeros
// the 16 characters from the first digit
const std::uint64_t s_hi = (sixteen ? hi : (hi << 8u) | (lo >> 56u)) + 0x3030303030303030u;
const std::uint64_t s_lo = (sixteen ? lo : (lo << 8u) | static_cast<std::uint64_t>(last)) + 0x3030303030303030u;
const char s16 = static_cast<char>(sixteen ? '0' + last : '0'); // the 17th
const auto store_16 = [s_hi, s_lo](char* p) noexcept
{
store_msb_first(p, s_hi);
store_msb_first(p + 8, s_lo);
};
const auto first_digit = static_cast<char>(s_hi >> 56u);
#endif
const int len = d.has_digit ? 16 + (sixteen ? 1 : 0) : digits; // significant digits
const int n = 16 + (sixteen ? 1 : 0) + d.exponent; // digits before the point
if (JSON_HEDLEY_LIKELY(n >= 1 && n <= 15))
{
// "dig.its" and "digits[000].0": the digits after the point move by
// one byte ('0's follow the digits)
#if JSON_DTOA_SSE2
// NOLINTBEGIN(portability-simd-intrinsics)
// (in the register: reading the digits back from memory right after
// storing them waits until the stores are done)
const __m128i at = _mm_set1_epi8(static_cast<char>(n));
const __m128i index = _mm_setr_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15);
const __m128i before = _mm_cmpgt_epi8(at, index);
const __m128i after = _mm_cmpgt_epi8(index, at);
const __m128i text = _mm_or_si128(_mm_or_si128(_mm_and_si128(s, before), _mm_and_si128(_mm_slli_si128(s, 1), after)),
_mm_andnot_si128(_mm_or_si128(before, after), _mm_set1_epi8('.')));
std::memcpy(first, &text, 16);
first[16] = static_cast<char>(_mm_extract_epi16(s, 7) >> 8);
first[17] = s16;
// NOLINTEND(portability-simd-intrinsics)
#elif JSON_DTOA_NEON
const uint8x16_t index = vcombine_u8(vcreate_u8(0x0706050403020100u), vcreate_u8(0x0F0E0D0C0B0A0908u));
const uint8x16_t at = vdupq_n_u8(static_cast<std::uint8_t>(n));
const uint8x16_t after_point = vbslq_u8(vcgtq_u8(index, at), vextq_u8(vdupq_n_u8(0), s, 15), vdupq_n_u8('.'));
vst1q_u8(reinterpret_cast<std::uint8_t*>(first), vbslq_u8(vcltq_u8(index, at), s, after_point)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
first[16] = static_cast<char>(vgetq_lane_u8(s, 15));
first[17] = s16;
#else
store_16(first);
first[16] = s16;
std::uint64_t after_point[2]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
std::memcpy(after_point, first + n, 16);
std::memcpy(first + n + 1, after_point, 16);
first[n] = '.';
#endif
return first + (n >= len ? n + 2 : len + 1);
}
if (n <= 0 && n > -4)
{
// "0.[000]digits"
std::memset(first, '0', 8);
first[1] = '.';
store_16(first + 2 - n);
first[18 - n] = s16;
return first + 2 - n + len;
}
// d.igitse+XX, with at least two exponent digits (as append_exponent())
store_16(first + 1);
first[17] = s16;
first[0] = first_digit;
first[1] = '.';
char* const end = first + (len == 1 ? 1 : len + 1);
const int e = n - 1;
const auto ea = static_cast<unsigned>(e < 0 ? -e : e);
const bool three = ea >= 100;
end[0] = 'e';
end[1] = e < 0 ? '-' : '+';
end[2] = static_cast<char>('0' + (three ? ea / 100 : (ea / 10) % 10));
end[3] = static_cast<char>('0' + (three ? (ea / 10) % 10 : ea % 10));
end[4] = static_cast<char>('0' + (ea % 10));
return end + (three ? 5 : 4);
}
/// the powers of ten up to 10^16
inline const std::array<std::uint64_t, 17>& powers_of_ten_16() noexcept
{
static const std::array<std::uint64_t, 17> powers =
{
{
1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u,
100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u
}
};
return powers;
}
/*!
@brief digits * 10^exp, as write_decimal() writes it, for the digits of a
double that need no conversion (count digits, at most 15, the first not 0;
trailing zeros allowed): extended to 16 digits and written by write_shortest()
@return a pointer past the text; up to 41 bytes at @a first are written
(some beyond the returned end)
*/
JSON_HEDLEY_NON_NULL(1)
JSON_HEDLEY_RETURNS_NON_NULL
inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept
{
JSON_ASSERT(digits >= powers_of_ten_16()[static_cast<std::size_t>(count - 1)] && count <= 15);
const int scale = 16 - count;
return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast<std::size_t>(scale)], exp - scale - 1, 0, false});
}
/// as write_short_decimal(), counting the digits (not 0, less than 10^15)
JSON_HEDLEY_NON_NULL(1)
JSON_HEDLEY_RETURNS_NON_NULL
inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept
{
JSON_ASSERT(digits != 0 && digits < 1000000000000000u);
// floor(log10(2^bits)) + 1 digits, or one less
const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12;
const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast<std::size_t>(log2_bound)] ? 1 : 0);
return write_short_decimal(first, digits, count, exp);
}
/// a positive finite float (other than double): Grisu2 and format_buffer()
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
@@ -25915,7 +26147,7 @@ char* write_positive(char* first, const char* last, FloatType value)
}
/// a positive finite double: the shortest digits (Zmij), laid out by
/// write_decimal() (through a local buffer if [first, last) is shorter than
/// write_shortest() (through a local buffer if [first, last) is shorter than
/// the 41 bytes it may write)
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
@@ -25925,13 +26157,13 @@ inline char* write_positive(char* first, const char* last, double value)
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
std::uint64_t bits = 0;
std::memcpy(&bits, &value, sizeof(bits));
const zmij::decimal d = zmij::to_decimal(bits);
const zmij::shortest_decimal d = zmij::to_shortest(bits);
if (JSON_HEDLEY_LIKELY(last - first >= 41))
{
return write_decimal(first, d.significand, d.exponent);
return write_shortest(first, d);
}
std::array<char, 64> buf; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
const auto len = static_cast<std::size_t>(write_decimal(buf.data(), d.significand, d.exponent) - buf.data());
const auto len = static_cast<std::size_t>(write_shortest(buf.data(), d) - buf.data());
JSON_ASSERT(static_cast<std::size_t>(last - first) >= len);
std::memcpy(first, buf.data(), len);
return first + len;
@@ -27338,8 +27570,9 @@ class serializer
/*!
@brief dump an integer
Dump a given integer, appending it to @ref write_buffer. Works internally with
@a number_buffer.
Dump a given integer, appending it to @ref write_buffer (directly: copying
the digits from another buffer right after writing them waits until the
stores are done).
@param[in] x integer number (signed or unsigned) to dump
@tparam NumberType either @a number_integer_t or @a number_unsigned_t
@@ -27374,33 +27607,57 @@ class serializer
return;
}
// use a pointer to fill the buffer
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto)
// use a pointer to fill the buffer (room for as much as number_buffer holds)
if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + number_buffer.size() > write_buffer.size()))
{
flush();
}
auto* buffer_ptr = write_buffer.data() + write_buffer_pos;
number_unsigned_t abs_value;
unsigned int n_chars{};
// one byte for the minus sign
unsigned int n_chars = 0;
if (is_negative_number(x))
{
*buffer_ptr = '-';
abs_value = remove_sign(static_cast<number_integer_t>(x));
// account one more byte for the minus sign
n_chars = 1 + count_digits(abs_value);
n_chars = 1;
}
else
{
abs_value = static_cast<number_unsigned_t>(x);
n_chars = count_digits(abs_value);
}
// up to 16 digits: eight at a time (as the digits of floats), written
// without leading zeros
if (abs_value < 10000000000000000u)
{
const std::uint64_t value = abs_value;
const std::uint64_t upper = value / 100000000u;
const std::uint64_t first = dtoa_impl::eight_digit_bytes(upper != 0 ? upper : value);
const auto leading = static_cast<unsigned>(count_leading_zeros(first) / 8); // (first is not 0)
char* const p = buffer_ptr + n_chars;
dtoa_impl::store_msb_first(p, (first << (8 * leading)) + 0x3030303030303030u);
n_chars += 8 - leading;
if (upper != 0)
{
dtoa_impl::store_msb_first(p + 8 - leading, dtoa_impl::eight_digit_bytes(value - (upper * 100000000u)) + 0x3030303030303030u);
n_chars += 8;
}
write_buffer_pos += n_chars;
return;
}
n_chars += count_digits(abs_value);
// spare 1 byte for '\0'
JSON_ASSERT(n_chars < number_buffer.size() - 1);
// jump to the end to generate the string from backward,
// so we later avoid reversing the result
buffer_ptr += static_cast<typename decltype(number_buffer)::difference_type>(n_chars);
buffer_ptr += n_chars;
// Fast int2ascii implementation inspired by "Fastware" talk by Andrei Alexandrescu
// See: https://www.youtube.com/watch?v=o4-CwDo2zpg
@@ -27423,14 +27680,13 @@ class serializer
*(--buffer_ptr) = static_cast<char>('0' + abs_value);
}
put_buffer(number_buffer, n_chars);
write_buffer_pos += n_chars;
}
/*!
@brief dump a floating-point number
Dump a given floating-point number, appending it to @ref write_buffer. Works internally
with @a number_buffer.
Dump a given floating-point number, appending it to @ref write_buffer.
@param[in] x floating-point number to dump
*/
@@ -27457,10 +27713,15 @@ class serializer
void dump_float(number_float_t x, std::true_type /*is_ieee_single_or_double*/)
{
auto* begin = number_buffer.data();
// directly into the write buffer: copying the text from number_buffer
// right after to_chars() wrote it waits until its stores are done
if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + number_buffer.size() > write_buffer.size()))
{
flush();
}
auto* begin = write_buffer.data() + write_buffer_pos;
auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x);
put_buffer(number_buffer, static_cast<std::size_t>(end - begin));
write_buffer_pos += static_cast<std::size_t>(end - begin);
}
JSON_HEDLEY_NON_NULL(1)
@@ -35143,6 +35404,8 @@ struct formatter<nlohmann::NLOHMANN_BASIC_JSON_TPL, char> // NOLINT(cert-dcl58-c
#undef JSON_NO_UNIQUE_ADDRESS
#undef JSON_DISABLE_ENUM_SERIALIZATION
#undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION
#undef JSON_DTOA_SSE2
#undef JSON_DTOA_NEON
#ifndef JSON_TEST_KEEP_MACROS
#undef JSON_CATCH
+10 -75
View File
@@ -5385,8 +5385,6 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/view/number.hpp>
// #include <nlohmann/detail/view/simd.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -5585,69 +5583,6 @@ struct dump_style
bool source_numbers = false; ///< copy number tokens from the source
};
/*!
@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0,
at most 17 digits; up to 41 bytes are written at first)
With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in
vector registers: output byte i is byte s + i of the digits (after '0's)
before the point, and byte s + i - 1 after it. The portable code writes the
digits to a buffer and copies them from there at another offset, and a load
that spans several recent stores waits until they reach the cache.
*/
NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept
{
#if NLOHMANN_VIEW_NEON
namespace dtoa = ::nlohmann::detail::dtoa_impl;
const std::uint64_t upper = digits / 100000000u;
const std::uint64_t b0 = upper / 100000000u; // one digit
const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u);
const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u);
// leading and trailing zero digits (as dtoa_impl::write_decimal())
int leading = 7;
if (b0 == 0)
{
leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8);
}
int zeros = 16;
if (b2 != 0)
{
zeros = count_trailing_zeros(b2) / 8;
}
else if (b1 != 0)
{
zeros = 8 + (count_trailing_zeros(b1) / 8);
}
const int k = 24 - leading - zeros; // significant digits
const int n = k + exp + zeros; // position of the point after the first digit
if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15))
{
static const std::array<std::uint8_t, 32> iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}};
const int pad = n <= 0 ? 1 - n : 0;
const int len = k + pad;
const int point = n + pad;
// the 24 digit bytes in memory order, then '0's
const std::uint64_t zero_chars = 0x3030303030303030u;
const uint8x16x2_t table = {{
vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))),
vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0'))
}
};
const uint8x16_t s = vdupq_n_u8(static_cast<std::uint8_t>(leading - pad));
const uint8x16_t at_point = vdupq_n_u8(static_cast<std::uint8_t>(point));
for (std::size_t half = 0; half < 2; ++half)
{
const uint8x16_t i = vld1q_u8(iota.data() + (16 * half));
// (+ 0xFF is - 1 after the point; indexes past the digits read a '0')
const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31));
vst1q_u8(reinterpret_cast<std::uint8_t*>(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
}
return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0"
}
#endif
return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp);
}
/*!
@brief write a view's subtree as basic_json::dump() writes the value
@@ -6149,7 +6084,11 @@ class view_serializer
if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290)
{
*w = '-';
return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast<int>(d.exponent));
w += d.negative ? 1 : 0;
// (without leading zeros, all digits of the token count)
const unsigned char lead = first[d.negative ? 1 : 0];
return lead != '0' ? ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast<int>(int_digits + frac_digits), static_cast<int>(d.exponent))
: ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast<int>(d.exponent));
}
return write_double_value_at(w, decimal_to_float<double>(d)); // (without reading the token again)
}
@@ -6166,13 +6105,13 @@ class view_serializer
/// a double as dump() writes it, at w (64 bytes of room)
static char* write_double_value_at(char* w, double x)
{
if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x)))
// (from the bits: without the checks of to_chars())
std::uint64_t bits = 0;
std::memcpy(&bits, &x, sizeof(bits));
if (NLOHMANN_VIEW_UNLIKELY((bits & 0x7FF0000000000000u) == 0x7FF0000000000000u))
{
return write_text_at(w, "null", 4);
}
#if NLOHMANN_VIEW_NEON
std::uint64_t bits = 0;
std::memcpy(&bits, &x, sizeof(bits));
*w = '-';
w += bits >> 63u;
bits &= ~(std::uint64_t{1} << 63u);
@@ -6180,11 +6119,7 @@ class view_serializer
{
return write_text_at(w, "0.0", 3);
}
const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits);
return write_decimal(w, d.significand, d.exponent);
#else
return ::nlohmann::detail::to_chars(w, w + 64, x);
#endif
return ::nlohmann::detail::dtoa_impl::write_shortest(w, ::nlohmann::detail::zmij::to_shortest(bits));
}
/// as serializer::dump_float()