Files
json/include/nlohmann/detail/bit_ops.hpp
T
Niels Lohmann 2dc6471968 Write doubles with the shortest digits (Żmij), digits in registers
Write doubles with the conversion of Zmij by Victor Zverovich (MIT),
ported to C++11 (detail/conversions/zmij.hpp). It finds the
shortest decimal that reads back as the same double, and the
closest one if there are several. Grisu2, used until now, is fast
but not always shortest: it sometimes writes a 17th digit where 16
suffice, or a last digit that is not the closest. The layout is
unchanged (1.5, 100.0, 1e+100, -0.0); float keeps Grisu2.

Digits are converted eight at a time with the BCD conversion of
Xiang JunBo, as in Zmij, and written with one byte swap per eight
digits and fixed-size moves instead of per-digit loops. Leading and
trailing zeros are counted from those bytes. to_chars() uses a
local buffer when the caller's is shorter than the 41 bytes this
may write. The powers of ten come from the number-parsing table,
adjusted where it holds values rounded up, and extended with Zmij's
compressed tables beyond 10^308.

write_shortest() converts its 16 digits in one vector register
(SSE2 on x86-64, NEON on 64-bit Arm, both baseline) and inserts the
decimal point inside the register, avoiding a store-forwarding
stall that cost about 25% of the time to write a double. dump()
writes floats and integers straight into the serializer's write
buffer instead of copying them from a member buffer, and small
integers eight digits at a time. read_eight_bytes() and
parse_eight_digits() are marked always-inline, which GCC had been
calling out of line in the number-parsing loops.

Of one million random doubles, about 0.14% are now written with
different digits, always to a value that still reads back as the
same double.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-10-11 10:27:35 +02:00

131 lines
4.5 KiB
C++

// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <cstdint> // uint64_t
#include <cstring> // memcpy
#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) && (!defined(__SIZEOF_INT128__) || (!defined(__GNUC__) && !defined(__clang__)))
#include <intrin0.h> // __umulh, _umul128, _BitScanForward64, _BitScanReverse64
#endif
#include <nlohmann/detail/abi_macros.hpp>
// Portable bit-level helpers for the number and string scanners. They use
// compiler builtins or platform-specific intrinsics where available and plain
// C++ otherwise, so they work regardless of byte order.
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
{
/// number of leading zero bits of x (x != 0)
inline int count_leading_zeros(std::uint64_t x) noexcept
{
#if defined(__GNUC__) || defined(__clang__)
return __builtin_clzll(x);
#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
unsigned long index = 0; // NOLINT(runtime/int): the type _BitScan*64 takes
_BitScanReverse64(&index, x);
return 63 - static_cast<int>(index);
#else
int n = 0;
for (int shift = 32; shift != 0; shift >>= 1)
{
if ((x >> (64 - shift)) == 0)
{
n += shift;
x <<= shift;
}
}
return n;
#endif
}
/// number of trailing zero bits of x (x != 0)
inline int count_trailing_zeros(std::uint64_t x) noexcept
{
#if defined(__GNUC__) || defined(__clang__)
return __builtin_ctzll(x);
#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
unsigned long index = 0; // NOLINT(runtime/int): the type _BitScan*64 takes
_BitScanForward64(&index, x);
return static_cast<int>(index);
#else
int n = 0;
for (int shift = 32; shift != 0; shift >>= 1)
{
if ((x << (64 - shift)) == 0)
{
n += shift;
x >>= shift;
}
}
return n;
#endif
}
/// the 128-bit product of two 64-bit numbers
struct uint128_parts
{
std::uint64_t low;
std::uint64_t high;
};
inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexcept
{
#if defined(__SIZEOF_INT128__)
__extension__ using uint128 = unsigned __int128;
const uint128 r = static_cast<uint128>(a) * b;
return {static_cast<std::uint64_t>(r), static_cast<std::uint64_t>(r >> 64u)};
#elif defined(_MSC_VER) && defined(_M_X64)
std::uint64_t high = 0;
const std::uint64_t low = _umul128(a, b, &high);
return {low, high};
#elif defined(_MSC_VER) && defined(_M_ARM64)
return {a * b, __umulh(a, b)};
#else
const std::uint64_t a_lo = a & 0xFFFFFFFFu;
const std::uint64_t a_hi = a >> 32u;
const std::uint64_t b_lo = b & 0xFFFFFFFFu;
const std::uint64_t b_hi = b >> 32u;
const std::uint64_t lo_lo = a_lo * b_lo;
const std::uint64_t hi_lo = a_hi * b_lo;
const std::uint64_t lo_hi = a_lo * b_hi;
const std::uint64_t hi_hi = a_hi * b_hi;
const std::uint64_t cross = (lo_lo >> 32u) + (hi_lo & 0xFFFFFFFFu) + lo_hi;
return {(cross << 32u) | (lo_lo & 0xFFFFFFFFu), (hi_lo >> 32u) + (cross >> 32u) + hi_hi};
#endif
}
/// eight bytes as a little-endian word (a single load on little-endian
/// targets; always inlined, as GCC otherwise calls it in the number loops)
JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
{
#if defined(_MSC_VER) || defined(__x86_64__) || defined(__i386__) || (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__)
// the byte order already matches (all MSVC targets are little-endian)
std::uint64_t result = 0;
std::memcpy(&result, b, sizeof(result));
return result;
#else
return static_cast<std::uint64_t>(b[0]) | (static_cast<std::uint64_t>(b[1]) << 8u)
| (static_cast<std::uint64_t>(b[2]) << 16u) | (static_cast<std::uint64_t>(b[3]) << 24u)
| (static_cast<std::uint64_t>(b[4]) << 32u) | (static_cast<std::uint64_t>(b[5]) << 40u)
| (static_cast<std::uint64_t>(b[6]) << 48u) | (static_cast<std::uint64_t>(b[7]) << 56u);
#endif
}
/// eight bytes as a little-endian word
JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const char* p) noexcept
{
return read_eight_bytes(reinterpret_cast<const unsigned char*>(p)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
}
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END