mirror of
https://github.com/nlohmann/json.git
synced 2026-10-06 14:40:32 +00:00
Give the library its own correctly rounded float converter for binary32 and binary64 (IEEE 754), and speed up the lexer's string and escape scanning. The converter splits a number token into sign, significand, and decimal exponent, then tries Clinger's fast path, then a templated Eisel-Lemire step, and falls back to an exact big-integer digit comparison for tokens with more than 19 significant digits whose two candidate values round differently. This replaces std::from_chars and strtod/strtof for both formats, so parsed values no longer depend on the C/C++ library or the current locale. The strtold fallback kept for other long double formats (x87, binary128) now also copies a multi-byte decimal point correctly, fixing #5660. eisel_lemire() and decimal_to_float() are always inlined so callers keep the whole conversion in their hot loop. The string-scanning kernels in string_scan.hpp find a stop byte with the trailing-zero count of the SWAR mask instead of a byte loop, and scalar_string_bulk_run() validates a run of multi-byte UTF-8 sequences one after another instead of re-searching after each one. get_codepoint() decodes a contiguous \uXXXX escape with one table lookup per byte instead of four range-checked get() calls; the streaming path and all error positions are unchanged. Adds 508 generated hard float-parsing cases with expected binary32 and binary64 bits, and kernel-comparison tests for the string scans and the escape table against byte-by-byte references. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
106 lines
3.2 KiB
C++
106 lines
3.2 KiB
C++
// __ _____ _____ _____
|
|
// __| | __| | | | JSON for Modern C++
|
|
// | | |__ | | | | | | version 3.12.0
|
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
|
//
|
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
#pragma once
|
|
|
|
#include <cstdint> // uint64_t
|
|
|
|
#include <nlohmann/detail/abi_macros.hpp>
|
|
|
|
// Portable bit-level helpers for the number and string scanners. They use
|
|
// compiler builtins where available and plain C++ otherwise, so they need no
|
|
// platform headers and work regardless of byte order.
|
|
|
|
NLOHMANN_JSON_NAMESPACE_BEGIN
|
|
namespace detail
|
|
{
|
|
|
|
/// number of leading zero bits of x (x != 0)
|
|
inline int count_leading_zeros(std::uint64_t x) noexcept
|
|
{
|
|
#if defined(__GNUC__) || defined(__clang__)
|
|
return __builtin_clzll(x);
|
|
#else
|
|
int n = 0;
|
|
for (int shift = 32; shift != 0; shift >>= 1)
|
|
{
|
|
if ((x >> (64 - shift)) == 0)
|
|
{
|
|
n += shift;
|
|
x <<= shift;
|
|
}
|
|
}
|
|
return n;
|
|
#endif
|
|
}
|
|
|
|
/// number of trailing zero bits of x (x != 0)
|
|
inline int count_trailing_zeros(std::uint64_t x) noexcept
|
|
{
|
|
#if defined(__GNUC__) || defined(__clang__)
|
|
return __builtin_ctzll(x);
|
|
#else
|
|
int n = 0;
|
|
for (int shift = 32; shift != 0; shift >>= 1)
|
|
{
|
|
if ((x << (64 - shift)) == 0)
|
|
{
|
|
n += shift;
|
|
x >>= shift;
|
|
}
|
|
}
|
|
return n;
|
|
#endif
|
|
}
|
|
|
|
/// the 128-bit product of two 64-bit numbers
|
|
struct uint128_parts
|
|
{
|
|
std::uint64_t low;
|
|
std::uint64_t high;
|
|
};
|
|
|
|
inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexcept
|
|
{
|
|
#if defined(__SIZEOF_INT128__)
|
|
__extension__ using uint128 = unsigned __int128;
|
|
const uint128 r = static_cast<uint128>(a) * b;
|
|
return {static_cast<std::uint64_t>(r), static_cast<std::uint64_t>(r >> 64u)};
|
|
#else
|
|
const std::uint64_t a_lo = a & 0xFFFFFFFFu;
|
|
const std::uint64_t a_hi = a >> 32u;
|
|
const std::uint64_t b_lo = b & 0xFFFFFFFFu;
|
|
const std::uint64_t b_hi = b >> 32u;
|
|
const std::uint64_t lo_lo = a_lo * b_lo;
|
|
const std::uint64_t hi_lo = a_hi * b_lo;
|
|
const std::uint64_t lo_hi = a_lo * b_hi;
|
|
const std::uint64_t hi_hi = a_hi * b_hi;
|
|
const std::uint64_t cross = (lo_lo >> 32u) + (hi_lo & 0xFFFFFFFFu) + lo_hi;
|
|
return {(cross << 32u) | (lo_lo & 0xFFFFFFFFu), (hi_lo >> 32u) + (cross >> 32u) + hi_hi};
|
|
#endif
|
|
}
|
|
|
|
/// eight bytes as a little-endian word (compilers fold this into one load on
|
|
/// little-endian targets)
|
|
inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
|
|
{
|
|
return static_cast<std::uint64_t>(b[0]) | (static_cast<std::uint64_t>(b[1]) << 8u)
|
|
| (static_cast<std::uint64_t>(b[2]) << 16u) | (static_cast<std::uint64_t>(b[3]) << 24u)
|
|
| (static_cast<std::uint64_t>(b[4]) << 32u) | (static_cast<std::uint64_t>(b[5]) << 40u)
|
|
| (static_cast<std::uint64_t>(b[6]) << 48u) | (static_cast<std::uint64_t>(b[7]) << 56u);
|
|
}
|
|
|
|
/// eight bytes as a little-endian word
|
|
inline std::uint64_t read_eight_bytes(const char* p) noexcept
|
|
{
|
|
return read_eight_bytes(reinterpret_cast<const unsigned char*>(p)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
|
}
|
|
|
|
} // namespace detail
|
|
NLOHMANN_JSON_NAMESPACE_END
|