From 44fba02ec272b687cddfe82fda7252f892454fe7 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Mon, 5 Oct 2026 22:28:58 +0200 Subject: [PATCH] Format the digits of numbers in vector registers and without reloads - zmij::to_shortest() keeps the last digit apart from the 15 or 16 digits before it, and dtoa_impl::write_shortest() converts those at once: with SSE2 on x86-64 and NEON on 64-bit Arm (no CPU check needed), else eight digits at a time. The decimal point is inserted in the register; reading the digits back right after storing them stalled store forwarding. - json::dump() writes floats and integers straight into its write buffer instead of copying them from number_buffer, and integers below 10^16 eight digits at a time. - json_view's dump() writes doubles from their bits and tokens of up to 15 digits through the same code; its own NEON writer is removed. to_chars() on canada.json: 35 -> 29 ns per double (x86-64), 18.8 -> 14.2 ns (Apple M1); json::dump() of canada.json 16-29% faster, of citm_catalog.json 14-19%. Signed-off-by: Niels Lohmann --- .../nlohmann/detail/conversions/to_chars.hpp | 220 +++++++++++- include/nlohmann/detail/conversions/zmij.hpp | 32 +- include/nlohmann/detail/macro_unscope.hpp | 2 + include/nlohmann/detail/output/serializer.hpp | 61 +++- include/nlohmann/detail/view/serializer.hpp | 84 +---- single_include/nlohmann/json.hpp | 315 ++++++++++++++++-- single_include/nlohmann/json_view.hpp | 85 +---- 7 files changed, 598 insertions(+), 201 deletions(-) diff --git a/include/nlohmann/detail/conversions/to_chars.hpp b/include/nlohmann/detail/conversions/to_chars.hpp index 93c27c162..f06cd549f 100644 --- a/include/nlohmann/detail/conversions/to_chars.hpp +++ b/include/nlohmann/detail/conversions/to_chars.hpp @@ -21,6 +21,21 @@ #include // _byteswap_uint64 #endif +// SSE2 (every x86-64 CPU) and NEON (every 64-bit Arm CPU) convert the 16 +// digits of a double at once +#if defined(__x86_64__) || (defined(_M_X64) && !defined(_M_ARM64EC)) + #include + #define JSON_DTOA_SSE2 1 + #define JSON_DTOA_NEON 0 +#elif (defined(__aarch64__) || defined(_M_ARM64)) && !defined(_M_ARM64EC) && !defined(__ARM_BIG_ENDIAN) + #include + #define JSON_DTOA_SSE2 0 + #define JSON_DTOA_NEON 1 +#else + #define JSON_DTOA_SSE2 0 + #define JSON_DTOA_NEON 0 +#endif + #include #include @@ -1253,6 +1268,203 @@ inline char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept return end + (three ? 5 : 4); } +/*! +@brief the shortest decimal of a positive double (Zmij), as write_decimal() +writes it + +For a normal double, the shorter candidate has 15 or 16 digits: they are +converted at once (two halves of eight digits) and followed by the digit +after them, if there is one, without the multiplication and division by 10 +that counting the digits of one number would take. The fixed layouts move +the digits after the point by one byte. + +@return a pointer past the text; up to 41 bytes at @a first are written + (some beyond the returned end) +*/ +JSON_HEDLEY_NON_NULL(1) +JSON_HEDLEY_RETURNS_NON_NULL +inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcept +{ + const std::uint64_t sig = d.integral; + if (JSON_HEDLEY_UNLIKELY(sig < 100000000000000u || sig >= 10000000000000000u)) + { + // (subnormals) + return d.has_digit ? write_decimal(first, (sig * 10) + d.digit, d.exponent) : write_decimal(first, sig, d.exponent + 1); + } + const bool sixteen = sig >= 1000000000000000u; // (else 15 digits) + const int last = d.has_digit ? d.digit : 0; + const std::uint64_t upper = sig / 100000000u; +#if JSON_DTOA_SSE2 + // NOLINTBEGIN(portability-simd-intrinsics) + // the two halves in the 64-bit lanes, each as abcd * 2^32 + efgh, then as + // bytes (as eight_digit_bytes(), one lane each) + const __m128i x = _mm_set_epi64x(static_cast(sig - (upper * 100000000u)), static_cast(upper)); + const __m128i abcd = _mm_srli_epi64(_mm_mul_epu32(x, _mm_set1_epi64x(109951163)), 40); // 2^40 / 10000 + 1 + const __m128i abcd_efgh = _mm_add_epi64(x, _mm_mul_epu32(abcd, _mm_set1_epi64x(4294957296))); // 2^32 - 10000 + // 32-bit lanes in the order of the text: abcd, efgh of both halves + const __m128i fours = _mm_shuffle_epi32(abcd_efgh, _MM_SHUFFLE(2, 3, 0, 1)); + const __m128i ab = _mm_srli_epi16(_mm_mulhi_epu16(fours, _mm_set1_epi32(5243)), 3); + const __m128i ab_cd = _mm_or_si128(_mm_slli_epi32(_mm_sub_epi16(fours, _mm_mullo_epi16(ab, _mm_set1_epi32(100))), 16), ab); + // 16-bit lanes ab (< 100) -> bytes a, b: 256 * ab - 2559 * (ab / 10) + const __m128i bytes = _mm_sub_epi16(_mm_slli_epi16(ab_cd, 8), _mm_mullo_epi16(_mm_set1_epi16(2559), _mm_mulhi_epu16(ab_cd, _mm_set1_epi16(6554)))); + // the last digit that is not 0 (sig is not 0) + const auto nonzero = static_cast(_mm_movemask_epi8(_mm_cmpgt_epi8(bytes, _mm_setzero_si128()))); + const int digits = 63 - count_leading_zeros(nonzero) + (sixteen ? 1 : 0); // without trailing zeros + const __m128i chars = _mm_add_epi8(bytes, _mm_set1_epi8('0')); + // the 16 characters from the first digit + const __m128i s = sixteen ? chars : _mm_or_si128(_mm_srli_si128(chars, 1), _mm_slli_si128(_mm_cvtsi32_si128('0' + last), 15)); + const char s16 = static_cast(sixteen ? '0' + last : '0'); // the 17th + const auto store_16 = [&s](char* p) noexcept + { + std::memcpy(p, &s, 16); + }; + const char first_digit = static_cast(_mm_cvtsi128_si32(s)); + // NOLINTEND(portability-simd-intrinsics) +#elif JSON_DTOA_NEON + // as with SSE2: the halves in 32-bit lanes, then abcd, efgh of both + const uint32x2_t halves = vcreate_u32(upper | ((sig - (upper * 100000000u)) << 32u)); + const uint32x2_t abcd = vmovn_u64(vshrq_n_u64(vmull_n_u32(halves, static_cast(((std::uint64_t{1} << 40u) / 10000u) + 1u)), 40)); + const uint32x2_t efgh = vmls_n_u32(halves, abcd, 10000u); + const uint32x4_t fours = vcombine_u32(vzip1_u32(abcd, efgh), vzip2_u32(abcd, efgh)); + const uint32x4_t ab = vshrq_n_u32(vmulq_n_u32(fours, 5243u), 19); + const uint16x8_t ab_cd = vreinterpretq_u16_u32(vorrq_u32(ab, vshlq_n_u32(vmlsq_n_u32(fours, ab, 100u), 16))); + const uint16x8_t tens = vshrq_n_u16(vmulq_n_u16(ab_cd, 103u), 10); + const uint8x16_t bytes = vreinterpretq_u8_u16(vorrq_u16(tens, vshlq_n_u16(vmlsq_n_u16(ab_cd, tens, 10u), 8))); + // the last digit that is not 0 (sig is not 0): a nibble per byte + const std::uint64_t nonzero = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(bytes, bytes)), 4)), 0); + const int digits = ((63 - count_leading_zeros(nonzero)) / 4) + (sixteen ? 1 : 0); // without trailing zeros + const uint8x16_t chars = vaddq_u8(bytes, vdupq_n_u8('0')); + // the 16 characters from the first digit + const uint8x16_t s = sixteen ? chars : vextq_u8(chars, vdupq_n_u8(static_cast('0' + last)), 1); + const char s16 = static_cast(sixteen ? '0' + last : '0'); // the 17th + const auto store_16 = [&s](char* p) noexcept + { + vst1q_u8(reinterpret_cast(p), s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + }; + const auto first_digit = static_cast(vgetq_lane_u8(s, 0)); +#else + const std::uint64_t hi = eight_digit_bytes(upper); + const std::uint64_t lo = eight_digit_bytes(sig - (upper * 100000000u)); + // trailing zero digits: zero bytes (sig is not 0) + const int zeros = lo != 0 ? count_trailing_zeros(lo) / 8 : 8 + (count_trailing_zeros(hi) / 8); + const int digits = 15 - zeros + (sixteen ? 1 : 0); // without trailing zeros + // the 16 characters from the first digit + const std::uint64_t s_hi = (sixteen ? hi : (hi << 8u) | (lo >> 56u)) + 0x3030303030303030u; + const std::uint64_t s_lo = (sixteen ? lo : (lo << 8u) | static_cast(last)) + 0x3030303030303030u; + const char s16 = static_cast(sixteen ? '0' + last : '0'); // the 17th + const auto store_16 = [s_hi, s_lo](char* p) noexcept + { + store_msb_first(p, s_hi); + store_msb_first(p + 8, s_lo); + }; + const auto first_digit = static_cast(s_hi >> 56u); +#endif + const int len = d.has_digit ? 16 + (sixteen ? 1 : 0) : digits; // significant digits + const int n = 16 + (sixteen ? 1 : 0) + d.exponent; // digits before the point + + if (JSON_HEDLEY_LIKELY(n >= 1 && n <= 15)) + { + // "dig.its" and "digits[000].0": the digits after the point move by + // one byte ('0's follow the digits) +#if JSON_DTOA_SSE2 + // NOLINTBEGIN(portability-simd-intrinsics) + // (in the register: reading the digits back from memory right after + // storing them waits until the stores are done) + const __m128i at = _mm_set1_epi8(static_cast(n)); + const __m128i index = _mm_setr_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15); + const __m128i before = _mm_cmpgt_epi8(at, index); + const __m128i after = _mm_cmpgt_epi8(index, at); + const __m128i text = _mm_or_si128(_mm_or_si128(_mm_and_si128(s, before), _mm_and_si128(_mm_slli_si128(s, 1), after)), + _mm_andnot_si128(_mm_or_si128(before, after), _mm_set1_epi8('.'))); + std::memcpy(first, &text, 16); + first[16] = static_cast(_mm_extract_epi16(s, 7) >> 8); + first[17] = s16; + // NOLINTEND(portability-simd-intrinsics) +#elif JSON_DTOA_NEON + const uint8x16_t index = vcombine_u8(vcreate_u8(0x0706050403020100u), vcreate_u8(0x0F0E0D0C0B0A0908u)); + const uint8x16_t at = vdupq_n_u8(static_cast(n)); + const uint8x16_t after_point = vbslq_u8(vcgtq_u8(index, at), vextq_u8(vdupq_n_u8(0), s, 15), vdupq_n_u8('.')); + vst1q_u8(reinterpret_cast(first), vbslq_u8(vcltq_u8(index, at), s, after_point)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + first[16] = static_cast(vgetq_lane_u8(s, 15)); + first[17] = s16; +#else + store_16(first); + first[16] = s16; + std::uint64_t after_point[2]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read + std::memcpy(after_point, first + n, 16); + std::memcpy(first + n + 1, after_point, 16); + first[n] = '.'; +#endif + return first + (n >= len ? n + 2 : len + 1); + } + if (n <= 0 && n > -4) + { + // "0.[000]digits" + std::memset(first, '0', 8); + first[1] = '.'; + store_16(first + 2 - n); + first[18 - n] = s16; + return first + 2 - n + len; + } + // d.igitse+XX, with at least two exponent digits (as append_exponent()) + store_16(first + 1); + first[17] = s16; + first[0] = first_digit; + first[1] = '.'; + char* const end = first + (len == 1 ? 1 : len + 1); + const int e = n - 1; + const auto ea = static_cast(e < 0 ? -e : e); + const bool three = ea >= 100; + end[0] = 'e'; + end[1] = e < 0 ? '-' : '+'; + end[2] = static_cast('0' + (three ? ea / 100 : (ea / 10) % 10)); + end[3] = static_cast('0' + (three ? (ea / 10) % 10 : ea % 10)); + end[4] = static_cast('0' + (ea % 10)); + return end + (three ? 5 : 4); +} + +/// the powers of ten up to 10^16 +inline const std::array& powers_of_ten_16() noexcept +{ + static const std::array powers = + { + { + 1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u, + 100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u + } + }; + return powers; +} + +/*! +@brief digits * 10^exp, as write_decimal() writes it, for the digits of a +double that need no conversion (count digits, at most 15, the first not 0; +trailing zeros allowed): extended to 16 digits and written by write_shortest() + +@return a pointer past the text; up to 41 bytes at @a first are written + (some beyond the returned end) +*/ +JSON_HEDLEY_NON_NULL(1) +JSON_HEDLEY_RETURNS_NON_NULL +inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept +{ + JSON_ASSERT(digits >= powers_of_ten_16()[static_cast(count - 1)] && count <= 15); + const int scale = 16 - count; + return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast(scale)], exp - scale - 1, 0, false}); +} + +/// as write_short_decimal(), counting the digits (not 0, less than 10^15) +JSON_HEDLEY_NON_NULL(1) +JSON_HEDLEY_RETURNS_NON_NULL +inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept +{ + JSON_ASSERT(digits != 0 && digits < 1000000000000000u); + // floor(log10(2^bits)) + 1 digits, or one less + const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12; + const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast(log2_bound)] ? 1 : 0); + return write_short_decimal(first, digits, count, exp); +} + /// a positive finite float (other than double): Grisu2 and format_buffer() template JSON_HEDLEY_NON_NULL(1, 2) @@ -1284,7 +1496,7 @@ char* write_positive(char* first, const char* last, FloatType value) } /// a positive finite double: the shortest digits (Zmij), laid out by -/// write_decimal() (through a local buffer if [first, last) is shorter than +/// write_shortest() (through a local buffer if [first, last) is shorter than /// the 41 bytes it may write) JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL @@ -1294,13 +1506,13 @@ inline char* write_positive(char* first, const char* last, double value) "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); std::uint64_t bits = 0; std::memcpy(&bits, &value, sizeof(bits)); - const zmij::decimal d = zmij::to_decimal(bits); + const zmij::shortest_decimal d = zmij::to_shortest(bits); if (JSON_HEDLEY_LIKELY(last - first >= 41)) { - return write_decimal(first, d.significand, d.exponent); + return write_shortest(first, d); } std::array buf; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read - const auto len = static_cast(write_decimal(buf.data(), d.significand, d.exponent) - buf.data()); + const auto len = static_cast(write_shortest(buf.data(), d) - buf.data()); JSON_ASSERT(static_cast(last - first) >= len); std::memcpy(first, buf.data(), len); return first + len; diff --git a/include/nlohmann/detail/conversions/zmij.hpp b/include/nlohmann/detail/conversions/zmij.hpp index 6022d9338..ffff8380e 100644 --- a/include/nlohmann/detail/conversions/zmij.hpp +++ b/include/nlohmann/detail/conversions/zmij.hpp @@ -157,10 +157,22 @@ inline std::uint64_t umul128_add_hi64(std::uint64_t x, std::uint64_t y, std::uin return p.high + (p.low + c < p.low ? 1u : 0u); } +/// the result of Zmij: the shorter candidate and, if that is outside the +/// rounding interval, the digit after it (16 bytes: returned in registers) +struct shortest_decimal +{ + std::uint64_t integral; ///< the shorter candidate (15 or 16 digits for normal doubles) + int exponent; ///< the decimal exponent of the digit after it + unsigned char digit; ///< the digit after it (if has_digit) + bool has_digit; ///< whether the shortest decimal is integral * 10 + digit +}; + /// The shortest decimal in the rounding interval of a positive finite double /// given by its bits, the closest one if there are several (to_decimal of -/// Zmij). The significand can end in zeros. -inline decimal to_decimal(std::uint64_t bits) noexcept +/// Zmij, which keeps the last digit apart: the 15 or 16 digits before it can be +/// converted without a multiplication by 10 first). Always inlined: GCC +/// otherwise calls it, and its result goes through memory. +JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexcept { constexpr int extra_shift = 9; const auto raw_exp = static_cast((bits >> 52u) & 0x7FFu); @@ -205,12 +217,20 @@ inline decimal to_decimal(std::uint64_t bits) noexcept digit = digit < lowest ? lowest : digit; } integral += round_up ? 1u : 0u; - if (!round_up && !round_down) + // if the shorter candidate is outside the rounding interval: one digit more + return shortest_decimal{integral, dec_exp, static_cast(digit), !round_up && !round_down}; +} + +/// The shortest decimal in the rounding interval of a positive finite double +/// given by its bits, as one number. The significand can end in zeros. +inline decimal to_decimal(std::uint64_t bits) noexcept +{ + const shortest_decimal d = to_shortest(bits); + if (d.has_digit) { - // the shorter candidate is outside the rounding interval: one digit more - return decimal{(integral * 10) + digit, dec_exp}; + return decimal{(d.integral * 10) + d.digit, d.exponent}; } - return decimal{integral, dec_exp + 1}; + return decimal{d.integral, d.exponent + 1}; } } // namespace zmij diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index 55b2ac99e..bf31ac7ae 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -21,6 +21,8 @@ #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION +#undef JSON_DTOA_SSE2 +#undef JSON_DTOA_NEON #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index 8306440ef..769f92de6 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -1366,8 +1366,9 @@ class serializer /*! @brief dump an integer - Dump a given integer, appending it to @ref write_buffer. Works internally with - @a number_buffer. + Dump a given integer, appending it to @ref write_buffer (directly: copying + the digits from another buffer right after writing them waits until the + stores are done). @param[in] x integer number (signed or unsigned) to dump @tparam NumberType either @a number_integer_t or @a number_unsigned_t @@ -1402,33 +1403,57 @@ class serializer return; } - // use a pointer to fill the buffer - auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto) + // use a pointer to fill the buffer (room for as much as number_buffer holds) + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + number_buffer.size() > write_buffer.size())) + { + flush(); + } + auto* buffer_ptr = write_buffer.data() + write_buffer_pos; number_unsigned_t abs_value; - unsigned int n_chars{}; + // one byte for the minus sign + unsigned int n_chars = 0; if (is_negative_number(x)) { *buffer_ptr = '-'; abs_value = remove_sign(static_cast(x)); - - // account one more byte for the minus sign - n_chars = 1 + count_digits(abs_value); + n_chars = 1; } else { abs_value = static_cast(x); - n_chars = count_digits(abs_value); } + // up to 16 digits: eight at a time (as the digits of floats), written + // without leading zeros + if (abs_value < 10000000000000000u) + { + const std::uint64_t value = abs_value; + const std::uint64_t upper = value / 100000000u; + const std::uint64_t first = dtoa_impl::eight_digit_bytes(upper != 0 ? upper : value); + const auto leading = static_cast(count_leading_zeros(first) / 8); // (first is not 0) + char* const p = buffer_ptr + n_chars; + dtoa_impl::store_msb_first(p, (first << (8 * leading)) + 0x3030303030303030u); + n_chars += 8 - leading; + if (upper != 0) + { + dtoa_impl::store_msb_first(p + 8 - leading, dtoa_impl::eight_digit_bytes(value - (upper * 100000000u)) + 0x3030303030303030u); + n_chars += 8; + } + write_buffer_pos += n_chars; + return; + } + + n_chars += count_digits(abs_value); + // spare 1 byte for '\0' JSON_ASSERT(n_chars < number_buffer.size() - 1); // jump to the end to generate the string from backward, // so we later avoid reversing the result - buffer_ptr += static_cast(n_chars); + buffer_ptr += n_chars; // Fast int2ascii implementation inspired by "Fastware" talk by Andrei Alexandrescu // See: https://www.youtube.com/watch?v=o4-CwDo2zpg @@ -1451,14 +1476,13 @@ class serializer *(--buffer_ptr) = static_cast('0' + abs_value); } - put_buffer(number_buffer, n_chars); + write_buffer_pos += n_chars; } /*! @brief dump a floating-point number - Dump a given floating-point number, appending it to @ref write_buffer. Works internally - with @a number_buffer. + Dump a given floating-point number, appending it to @ref write_buffer. @param[in] x floating-point number to dump */ @@ -1485,10 +1509,15 @@ class serializer void dump_float(number_float_t x, std::true_type /*is_ieee_single_or_double*/) { - auto* begin = number_buffer.data(); + // directly into the write buffer: copying the text from number_buffer + // right after to_chars() wrote it waits until its stores are done + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + number_buffer.size() > write_buffer.size())) + { + flush(); + } + auto* begin = write_buffer.data() + write_buffer_pos; auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x); - - put_buffer(number_buffer, static_cast(end - begin)); + write_buffer_pos += static_cast(end - begin); } JSON_HEDLEY_NON_NULL(1) diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp index f471f1642..f15ff410c 100644 --- a/include/nlohmann/detail/view/serializer.hpp +++ b/include/nlohmann/detail/view/serializer.hpp @@ -23,7 +23,6 @@ #include #include #include -#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -222,69 +221,6 @@ struct dump_style bool source_numbers = false; ///< copy number tokens from the source }; -/*! -@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0, -at most 17 digits; up to 41 bytes are written at first) - -With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in -vector registers: output byte i is byte s + i of the digits (after '0's) -before the point, and byte s + i - 1 after it. The portable code writes the -digits to a buffer and copies them from there at another offset, and a load -that spans several recent stores waits until they reach the cache. -*/ -NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept -{ -#if NLOHMANN_VIEW_NEON - namespace dtoa = ::nlohmann::detail::dtoa_impl; - const std::uint64_t upper = digits / 100000000u; - const std::uint64_t b0 = upper / 100000000u; // one digit - const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u); - const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u); - // leading and trailing zero digits (as dtoa_impl::write_decimal()) - int leading = 7; - if (b0 == 0) - { - leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8); - } - int zeros = 16; - if (b2 != 0) - { - zeros = count_trailing_zeros(b2) / 8; - } - else if (b1 != 0) - { - zeros = 8 + (count_trailing_zeros(b1) / 8); - } - const int k = 24 - leading - zeros; // significant digits - const int n = k + exp + zeros; // position of the point after the first digit - if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15)) - { - static const std::array iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}}; - const int pad = n <= 0 ? 1 - n : 0; - const int len = k + pad; - const int point = n + pad; - // the 24 digit bytes in memory order, then '0's - const std::uint64_t zero_chars = 0x3030303030303030u; - const uint8x16x2_t table = {{ - vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))), - vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0')) - } - }; - const uint8x16_t s = vdupq_n_u8(static_cast(leading - pad)); - const uint8x16_t at_point = vdupq_n_u8(static_cast(point)); - for (std::size_t half = 0; half < 2; ++half) - { - const uint8x16_t i = vld1q_u8(iota.data() + (16 * half)); - // (+ 0xFF is - 1 after the point; indexes past the digits read a '0') - const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31)); - vst1q_u8(reinterpret_cast(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) - } - return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0" - } -#endif - return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp); -} - /*! @brief write a view's subtree as basic_json::dump() writes the value @@ -786,7 +722,11 @@ class view_serializer if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290) { *w = '-'; - return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast(d.exponent)); + w += d.negative ? 1 : 0; + // (without leading zeros, all digits of the token count) + const unsigned char lead = first[d.negative ? 1 : 0]; + return lead != '0' ? ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast(int_digits + frac_digits), static_cast(d.exponent)) + : ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast(d.exponent)); } return write_double_value_at(w, decimal_to_float(d)); // (without reading the token again) } @@ -803,13 +743,13 @@ class view_serializer /// a double as dump() writes it, at w (64 bytes of room) static char* write_double_value_at(char* w, double x) { - if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x))) + // (from the bits: without the checks of to_chars()) + std::uint64_t bits = 0; + std::memcpy(&bits, &x, sizeof(bits)); + if (NLOHMANN_VIEW_UNLIKELY((bits & 0x7FF0000000000000u) == 0x7FF0000000000000u)) { return write_text_at(w, "null", 4); } -#if NLOHMANN_VIEW_NEON - std::uint64_t bits = 0; - std::memcpy(&bits, &x, sizeof(bits)); *w = '-'; w += bits >> 63u; bits &= ~(std::uint64_t{1} << 63u); @@ -817,11 +757,7 @@ class view_serializer { return write_text_at(w, "0.0", 3); } - const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits); - return write_decimal(w, d.significand, d.exponent); -#else - return ::nlohmann::detail::to_chars(w, w + 64, x); -#endif + return ::nlohmann::detail::dtoa_impl::write_shortest(w, ::nlohmann::detail::zmij::to_shortest(bits)); } /// as serializer::dump_float() diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index e4a2ff312..ac2125b2d 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -24428,6 +24428,21 @@ NLOHMANN_JSON_NAMESPACE_END #include // _byteswap_uint64 #endif +// SSE2 (every x86-64 CPU) and NEON (every 64-bit Arm CPU) convert the 16 +// digits of a double at once +#if defined(__x86_64__) || (defined(_M_X64) && !defined(_M_ARM64EC)) + #include + #define JSON_DTOA_SSE2 1 + #define JSON_DTOA_NEON 0 +#elif (defined(__aarch64__) || defined(_M_ARM64)) && !defined(_M_ARM64EC) && !defined(__ARM_BIG_ENDIAN) + #include + #define JSON_DTOA_SSE2 0 + #define JSON_DTOA_NEON 1 +#else + #define JSON_DTOA_SSE2 0 + #define JSON_DTOA_NEON 0 +#endif + // #include // __ _____ _____ _____ // __| | __| | | | JSON for Modern C++ @@ -24592,10 +24607,22 @@ inline std::uint64_t umul128_add_hi64(std::uint64_t x, std::uint64_t y, std::uin return p.high + (p.low + c < p.low ? 1u : 0u); } +/// the result of Zmij: the shorter candidate and, if that is outside the +/// rounding interval, the digit after it (16 bytes: returned in registers) +struct shortest_decimal +{ + std::uint64_t integral; ///< the shorter candidate (15 or 16 digits for normal doubles) + int exponent; ///< the decimal exponent of the digit after it + unsigned char digit; ///< the digit after it (if has_digit) + bool has_digit; ///< whether the shortest decimal is integral * 10 + digit +}; + /// The shortest decimal in the rounding interval of a positive finite double /// given by its bits, the closest one if there are several (to_decimal of -/// Zmij). The significand can end in zeros. -inline decimal to_decimal(std::uint64_t bits) noexcept +/// Zmij, which keeps the last digit apart: the 15 or 16 digits before it can be +/// converted without a multiplication by 10 first). Always inlined: GCC +/// otherwise calls it, and its result goes through memory. +JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexcept { constexpr int extra_shift = 9; const auto raw_exp = static_cast((bits >> 52u) & 0x7FFu); @@ -24640,12 +24667,20 @@ inline decimal to_decimal(std::uint64_t bits) noexcept digit = digit < lowest ? lowest : digit; } integral += round_up ? 1u : 0u; - if (!round_up && !round_down) + // if the shorter candidate is outside the rounding interval: one digit more + return shortest_decimal{integral, dec_exp, static_cast(digit), !round_up && !round_down}; +} + +/// The shortest decimal in the rounding interval of a positive finite double +/// given by its bits, as one number. The significand can end in zeros. +inline decimal to_decimal(std::uint64_t bits) noexcept +{ + const shortest_decimal d = to_shortest(bits); + if (d.has_digit) { - // the shorter candidate is outside the rounding interval: one digit more - return decimal{(integral * 10) + digit, dec_exp}; + return decimal{(d.integral * 10) + d.digit, d.exponent}; } - return decimal{integral, dec_exp + 1}; + return decimal{d.integral, d.exponent + 1}; } } // namespace zmij @@ -25884,6 +25919,203 @@ inline char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept return end + (three ? 5 : 4); } +/*! +@brief the shortest decimal of a positive double (Zmij), as write_decimal() +writes it + +For a normal double, the shorter candidate has 15 or 16 digits: they are +converted at once (two halves of eight digits) and followed by the digit +after them, if there is one, without the multiplication and division by 10 +that counting the digits of one number would take. The fixed layouts move +the digits after the point by one byte. + +@return a pointer past the text; up to 41 bytes at @a first are written + (some beyond the returned end) +*/ +JSON_HEDLEY_NON_NULL(1) +JSON_HEDLEY_RETURNS_NON_NULL +inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcept +{ + const std::uint64_t sig = d.integral; + if (JSON_HEDLEY_UNLIKELY(sig < 100000000000000u || sig >= 10000000000000000u)) + { + // (subnormals) + return d.has_digit ? write_decimal(first, (sig * 10) + d.digit, d.exponent) : write_decimal(first, sig, d.exponent + 1); + } + const bool sixteen = sig >= 1000000000000000u; // (else 15 digits) + const int last = d.has_digit ? d.digit : 0; + const std::uint64_t upper = sig / 100000000u; +#if JSON_DTOA_SSE2 + // NOLINTBEGIN(portability-simd-intrinsics) + // the two halves in the 64-bit lanes, each as abcd * 2^32 + efgh, then as + // bytes (as eight_digit_bytes(), one lane each) + const __m128i x = _mm_set_epi64x(static_cast(sig - (upper * 100000000u)), static_cast(upper)); + const __m128i abcd = _mm_srli_epi64(_mm_mul_epu32(x, _mm_set1_epi64x(109951163)), 40); // 2^40 / 10000 + 1 + const __m128i abcd_efgh = _mm_add_epi64(x, _mm_mul_epu32(abcd, _mm_set1_epi64x(4294957296))); // 2^32 - 10000 + // 32-bit lanes in the order of the text: abcd, efgh of both halves + const __m128i fours = _mm_shuffle_epi32(abcd_efgh, _MM_SHUFFLE(2, 3, 0, 1)); + const __m128i ab = _mm_srli_epi16(_mm_mulhi_epu16(fours, _mm_set1_epi32(5243)), 3); + const __m128i ab_cd = _mm_or_si128(_mm_slli_epi32(_mm_sub_epi16(fours, _mm_mullo_epi16(ab, _mm_set1_epi32(100))), 16), ab); + // 16-bit lanes ab (< 100) -> bytes a, b: 256 * ab - 2559 * (ab / 10) + const __m128i bytes = _mm_sub_epi16(_mm_slli_epi16(ab_cd, 8), _mm_mullo_epi16(_mm_set1_epi16(2559), _mm_mulhi_epu16(ab_cd, _mm_set1_epi16(6554)))); + // the last digit that is not 0 (sig is not 0) + const auto nonzero = static_cast(_mm_movemask_epi8(_mm_cmpgt_epi8(bytes, _mm_setzero_si128()))); + const int digits = 63 - count_leading_zeros(nonzero) + (sixteen ? 1 : 0); // without trailing zeros + const __m128i chars = _mm_add_epi8(bytes, _mm_set1_epi8('0')); + // the 16 characters from the first digit + const __m128i s = sixteen ? chars : _mm_or_si128(_mm_srli_si128(chars, 1), _mm_slli_si128(_mm_cvtsi32_si128('0' + last), 15)); + const char s16 = static_cast(sixteen ? '0' + last : '0'); // the 17th + const auto store_16 = [&s](char* p) noexcept + { + std::memcpy(p, &s, 16); + }; + const char first_digit = static_cast(_mm_cvtsi128_si32(s)); + // NOLINTEND(portability-simd-intrinsics) +#elif JSON_DTOA_NEON + // as with SSE2: the halves in 32-bit lanes, then abcd, efgh of both + const uint32x2_t halves = vcreate_u32(upper | ((sig - (upper * 100000000u)) << 32u)); + const uint32x2_t abcd = vmovn_u64(vshrq_n_u64(vmull_n_u32(halves, static_cast(((std::uint64_t{1} << 40u) / 10000u) + 1u)), 40)); + const uint32x2_t efgh = vmls_n_u32(halves, abcd, 10000u); + const uint32x4_t fours = vcombine_u32(vzip1_u32(abcd, efgh), vzip2_u32(abcd, efgh)); + const uint32x4_t ab = vshrq_n_u32(vmulq_n_u32(fours, 5243u), 19); + const uint16x8_t ab_cd = vreinterpretq_u16_u32(vorrq_u32(ab, vshlq_n_u32(vmlsq_n_u32(fours, ab, 100u), 16))); + const uint16x8_t tens = vshrq_n_u16(vmulq_n_u16(ab_cd, 103u), 10); + const uint8x16_t bytes = vreinterpretq_u8_u16(vorrq_u16(tens, vshlq_n_u16(vmlsq_n_u16(ab_cd, tens, 10u), 8))); + // the last digit that is not 0 (sig is not 0): a nibble per byte + const std::uint64_t nonzero = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(bytes, bytes)), 4)), 0); + const int digits = ((63 - count_leading_zeros(nonzero)) / 4) + (sixteen ? 1 : 0); // without trailing zeros + const uint8x16_t chars = vaddq_u8(bytes, vdupq_n_u8('0')); + // the 16 characters from the first digit + const uint8x16_t s = sixteen ? chars : vextq_u8(chars, vdupq_n_u8(static_cast('0' + last)), 1); + const char s16 = static_cast(sixteen ? '0' + last : '0'); // the 17th + const auto store_16 = [&s](char* p) noexcept + { + vst1q_u8(reinterpret_cast(p), s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + }; + const auto first_digit = static_cast(vgetq_lane_u8(s, 0)); +#else + const std::uint64_t hi = eight_digit_bytes(upper); + const std::uint64_t lo = eight_digit_bytes(sig - (upper * 100000000u)); + // trailing zero digits: zero bytes (sig is not 0) + const int zeros = lo != 0 ? count_trailing_zeros(lo) / 8 : 8 + (count_trailing_zeros(hi) / 8); + const int digits = 15 - zeros + (sixteen ? 1 : 0); // without trailing zeros + // the 16 characters from the first digit + const std::uint64_t s_hi = (sixteen ? hi : (hi << 8u) | (lo >> 56u)) + 0x3030303030303030u; + const std::uint64_t s_lo = (sixteen ? lo : (lo << 8u) | static_cast(last)) + 0x3030303030303030u; + const char s16 = static_cast(sixteen ? '0' + last : '0'); // the 17th + const auto store_16 = [s_hi, s_lo](char* p) noexcept + { + store_msb_first(p, s_hi); + store_msb_first(p + 8, s_lo); + }; + const auto first_digit = static_cast(s_hi >> 56u); +#endif + const int len = d.has_digit ? 16 + (sixteen ? 1 : 0) : digits; // significant digits + const int n = 16 + (sixteen ? 1 : 0) + d.exponent; // digits before the point + + if (JSON_HEDLEY_LIKELY(n >= 1 && n <= 15)) + { + // "dig.its" and "digits[000].0": the digits after the point move by + // one byte ('0's follow the digits) +#if JSON_DTOA_SSE2 + // NOLINTBEGIN(portability-simd-intrinsics) + // (in the register: reading the digits back from memory right after + // storing them waits until the stores are done) + const __m128i at = _mm_set1_epi8(static_cast(n)); + const __m128i index = _mm_setr_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15); + const __m128i before = _mm_cmpgt_epi8(at, index); + const __m128i after = _mm_cmpgt_epi8(index, at); + const __m128i text = _mm_or_si128(_mm_or_si128(_mm_and_si128(s, before), _mm_and_si128(_mm_slli_si128(s, 1), after)), + _mm_andnot_si128(_mm_or_si128(before, after), _mm_set1_epi8('.'))); + std::memcpy(first, &text, 16); + first[16] = static_cast(_mm_extract_epi16(s, 7) >> 8); + first[17] = s16; + // NOLINTEND(portability-simd-intrinsics) +#elif JSON_DTOA_NEON + const uint8x16_t index = vcombine_u8(vcreate_u8(0x0706050403020100u), vcreate_u8(0x0F0E0D0C0B0A0908u)); + const uint8x16_t at = vdupq_n_u8(static_cast(n)); + const uint8x16_t after_point = vbslq_u8(vcgtq_u8(index, at), vextq_u8(vdupq_n_u8(0), s, 15), vdupq_n_u8('.')); + vst1q_u8(reinterpret_cast(first), vbslq_u8(vcltq_u8(index, at), s, after_point)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + first[16] = static_cast(vgetq_lane_u8(s, 15)); + first[17] = s16; +#else + store_16(first); + first[16] = s16; + std::uint64_t after_point[2]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read + std::memcpy(after_point, first + n, 16); + std::memcpy(first + n + 1, after_point, 16); + first[n] = '.'; +#endif + return first + (n >= len ? n + 2 : len + 1); + } + if (n <= 0 && n > -4) + { + // "0.[000]digits" + std::memset(first, '0', 8); + first[1] = '.'; + store_16(first + 2 - n); + first[18 - n] = s16; + return first + 2 - n + len; + } + // d.igitse+XX, with at least two exponent digits (as append_exponent()) + store_16(first + 1); + first[17] = s16; + first[0] = first_digit; + first[1] = '.'; + char* const end = first + (len == 1 ? 1 : len + 1); + const int e = n - 1; + const auto ea = static_cast(e < 0 ? -e : e); + const bool three = ea >= 100; + end[0] = 'e'; + end[1] = e < 0 ? '-' : '+'; + end[2] = static_cast('0' + (three ? ea / 100 : (ea / 10) % 10)); + end[3] = static_cast('0' + (three ? (ea / 10) % 10 : ea % 10)); + end[4] = static_cast('0' + (ea % 10)); + return end + (three ? 5 : 4); +} + +/// the powers of ten up to 10^16 +inline const std::array& powers_of_ten_16() noexcept +{ + static const std::array powers = + { + { + 1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u, + 100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u + } + }; + return powers; +} + +/*! +@brief digits * 10^exp, as write_decimal() writes it, for the digits of a +double that need no conversion (count digits, at most 15, the first not 0; +trailing zeros allowed): extended to 16 digits and written by write_shortest() + +@return a pointer past the text; up to 41 bytes at @a first are written + (some beyond the returned end) +*/ +JSON_HEDLEY_NON_NULL(1) +JSON_HEDLEY_RETURNS_NON_NULL +inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept +{ + JSON_ASSERT(digits >= powers_of_ten_16()[static_cast(count - 1)] && count <= 15); + const int scale = 16 - count; + return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast(scale)], exp - scale - 1, 0, false}); +} + +/// as write_short_decimal(), counting the digits (not 0, less than 10^15) +JSON_HEDLEY_NON_NULL(1) +JSON_HEDLEY_RETURNS_NON_NULL +inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept +{ + JSON_ASSERT(digits != 0 && digits < 1000000000000000u); + // floor(log10(2^bits)) + 1 digits, or one less + const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12; + const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast(log2_bound)] ? 1 : 0); + return write_short_decimal(first, digits, count, exp); +} + /// a positive finite float (other than double): Grisu2 and format_buffer() template JSON_HEDLEY_NON_NULL(1, 2) @@ -25915,7 +26147,7 @@ char* write_positive(char* first, const char* last, FloatType value) } /// a positive finite double: the shortest digits (Zmij), laid out by -/// write_decimal() (through a local buffer if [first, last) is shorter than +/// write_shortest() (through a local buffer if [first, last) is shorter than /// the 41 bytes it may write) JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL @@ -25925,13 +26157,13 @@ inline char* write_positive(char* first, const char* last, double value) "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); std::uint64_t bits = 0; std::memcpy(&bits, &value, sizeof(bits)); - const zmij::decimal d = zmij::to_decimal(bits); + const zmij::shortest_decimal d = zmij::to_shortest(bits); if (JSON_HEDLEY_LIKELY(last - first >= 41)) { - return write_decimal(first, d.significand, d.exponent); + return write_shortest(first, d); } std::array buf; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read - const auto len = static_cast(write_decimal(buf.data(), d.significand, d.exponent) - buf.data()); + const auto len = static_cast(write_shortest(buf.data(), d) - buf.data()); JSON_ASSERT(static_cast(last - first) >= len); std::memcpy(first, buf.data(), len); return first + len; @@ -27338,8 +27570,9 @@ class serializer /*! @brief dump an integer - Dump a given integer, appending it to @ref write_buffer. Works internally with - @a number_buffer. + Dump a given integer, appending it to @ref write_buffer (directly: copying + the digits from another buffer right after writing them waits until the + stores are done). @param[in] x integer number (signed or unsigned) to dump @tparam NumberType either @a number_integer_t or @a number_unsigned_t @@ -27374,33 +27607,57 @@ class serializer return; } - // use a pointer to fill the buffer - auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto) + // use a pointer to fill the buffer (room for as much as number_buffer holds) + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + number_buffer.size() > write_buffer.size())) + { + flush(); + } + auto* buffer_ptr = write_buffer.data() + write_buffer_pos; number_unsigned_t abs_value; - unsigned int n_chars{}; + // one byte for the minus sign + unsigned int n_chars = 0; if (is_negative_number(x)) { *buffer_ptr = '-'; abs_value = remove_sign(static_cast(x)); - - // account one more byte for the minus sign - n_chars = 1 + count_digits(abs_value); + n_chars = 1; } else { abs_value = static_cast(x); - n_chars = count_digits(abs_value); } + // up to 16 digits: eight at a time (as the digits of floats), written + // without leading zeros + if (abs_value < 10000000000000000u) + { + const std::uint64_t value = abs_value; + const std::uint64_t upper = value / 100000000u; + const std::uint64_t first = dtoa_impl::eight_digit_bytes(upper != 0 ? upper : value); + const auto leading = static_cast(count_leading_zeros(first) / 8); // (first is not 0) + char* const p = buffer_ptr + n_chars; + dtoa_impl::store_msb_first(p, (first << (8 * leading)) + 0x3030303030303030u); + n_chars += 8 - leading; + if (upper != 0) + { + dtoa_impl::store_msb_first(p + 8 - leading, dtoa_impl::eight_digit_bytes(value - (upper * 100000000u)) + 0x3030303030303030u); + n_chars += 8; + } + write_buffer_pos += n_chars; + return; + } + + n_chars += count_digits(abs_value); + // spare 1 byte for '\0' JSON_ASSERT(n_chars < number_buffer.size() - 1); // jump to the end to generate the string from backward, // so we later avoid reversing the result - buffer_ptr += static_cast(n_chars); + buffer_ptr += n_chars; // Fast int2ascii implementation inspired by "Fastware" talk by Andrei Alexandrescu // See: https://www.youtube.com/watch?v=o4-CwDo2zpg @@ -27423,14 +27680,13 @@ class serializer *(--buffer_ptr) = static_cast('0' + abs_value); } - put_buffer(number_buffer, n_chars); + write_buffer_pos += n_chars; } /*! @brief dump a floating-point number - Dump a given floating-point number, appending it to @ref write_buffer. Works internally - with @a number_buffer. + Dump a given floating-point number, appending it to @ref write_buffer. @param[in] x floating-point number to dump */ @@ -27457,10 +27713,15 @@ class serializer void dump_float(number_float_t x, std::true_type /*is_ieee_single_or_double*/) { - auto* begin = number_buffer.data(); + // directly into the write buffer: copying the text from number_buffer + // right after to_chars() wrote it waits until its stores are done + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + number_buffer.size() > write_buffer.size())) + { + flush(); + } + auto* begin = write_buffer.data() + write_buffer_pos; auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x); - - put_buffer(number_buffer, static_cast(end - begin)); + write_buffer_pos += static_cast(end - begin); } JSON_HEDLEY_NON_NULL(1) @@ -35143,6 +35404,8 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION +#undef JSON_DTOA_SSE2 +#undef JSON_DTOA_NEON #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 8c2f826c6..a85441d7e 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -5385,8 +5385,6 @@ NLOHMANN_JSON_NAMESPACE_END // #include -// #include - NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -5585,69 +5583,6 @@ struct dump_style bool source_numbers = false; ///< copy number tokens from the source }; -/*! -@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0, -at most 17 digits; up to 41 bytes are written at first) - -With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in -vector registers: output byte i is byte s + i of the digits (after '0's) -before the point, and byte s + i - 1 after it. The portable code writes the -digits to a buffer and copies them from there at another offset, and a load -that spans several recent stores waits until they reach the cache. -*/ -NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept -{ -#if NLOHMANN_VIEW_NEON - namespace dtoa = ::nlohmann::detail::dtoa_impl; - const std::uint64_t upper = digits / 100000000u; - const std::uint64_t b0 = upper / 100000000u; // one digit - const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u); - const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u); - // leading and trailing zero digits (as dtoa_impl::write_decimal()) - int leading = 7; - if (b0 == 0) - { - leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8); - } - int zeros = 16; - if (b2 != 0) - { - zeros = count_trailing_zeros(b2) / 8; - } - else if (b1 != 0) - { - zeros = 8 + (count_trailing_zeros(b1) / 8); - } - const int k = 24 - leading - zeros; // significant digits - const int n = k + exp + zeros; // position of the point after the first digit - if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15)) - { - static const std::array iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}}; - const int pad = n <= 0 ? 1 - n : 0; - const int len = k + pad; - const int point = n + pad; - // the 24 digit bytes in memory order, then '0's - const std::uint64_t zero_chars = 0x3030303030303030u; - const uint8x16x2_t table = {{ - vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))), - vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0')) - } - }; - const uint8x16_t s = vdupq_n_u8(static_cast(leading - pad)); - const uint8x16_t at_point = vdupq_n_u8(static_cast(point)); - for (std::size_t half = 0; half < 2; ++half) - { - const uint8x16_t i = vld1q_u8(iota.data() + (16 * half)); - // (+ 0xFF is - 1 after the point; indexes past the digits read a '0') - const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31)); - vst1q_u8(reinterpret_cast(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) - } - return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0" - } -#endif - return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp); -} - /*! @brief write a view's subtree as basic_json::dump() writes the value @@ -6149,7 +6084,11 @@ class view_serializer if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290) { *w = '-'; - return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast(d.exponent)); + w += d.negative ? 1 : 0; + // (without leading zeros, all digits of the token count) + const unsigned char lead = first[d.negative ? 1 : 0]; + return lead != '0' ? ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast(int_digits + frac_digits), static_cast(d.exponent)) + : ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast(d.exponent)); } return write_double_value_at(w, decimal_to_float(d)); // (without reading the token again) } @@ -6166,13 +6105,13 @@ class view_serializer /// a double as dump() writes it, at w (64 bytes of room) static char* write_double_value_at(char* w, double x) { - if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x))) + // (from the bits: without the checks of to_chars()) + std::uint64_t bits = 0; + std::memcpy(&bits, &x, sizeof(bits)); + if (NLOHMANN_VIEW_UNLIKELY((bits & 0x7FF0000000000000u) == 0x7FF0000000000000u)) { return write_text_at(w, "null", 4); } -#if NLOHMANN_VIEW_NEON - std::uint64_t bits = 0; - std::memcpy(&bits, &x, sizeof(bits)); *w = '-'; w += bits >> 63u; bits &= ~(std::uint64_t{1} << 63u); @@ -6180,11 +6119,7 @@ class view_serializer { return write_text_at(w, "0.0", 3); } - const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits); - return write_decimal(w, d.significand, d.exponent); -#else - return ::nlohmann::detail::to_chars(w, w + 64, x); -#endif + return ::nlohmann::detail::dtoa_impl::write_shortest(w, ::nlohmann::detail::zmij::to_shortest(bits)); } /// as serializer::dump_float()