Compare commits

..
Author SHA1 Message Date
Niels Lohmann ecf9df7c45 Convert the floats of json_view from the digit layout
The parser records where the integer digits, the fraction digits, and the
exponent of a float token are. For floats and doubles with at most 19
digits, the value is now read from that layout: the digits eight at a time,
without scanning the token, and rounded by the library's conversion core
(detail::decimal_to_float(): Clinger's fast path where both operands are
exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19
digits). It rounds correctly, so the values are those of parse(); other
tokens and types keep the library's conversion of the whole token.

get<double>(), materialize(), dump(), and comparisons use it. Traversing
canada.json (111,000 floats, every number converted): 0.95 -> 1.29 GB/s.

Tests add tokens around the limits (19 and 20 digits, 2^53, 10^22, and
those of float) to the bit-for-bit comparison with parse().

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-30 21:01:51 +02:00
6 changed files with 205 additions and 7 deletions
+1 -1
View File
@@ -81,7 +81,7 @@ BasicJsonType materialize(const document_data& d, const node* n)
++n;
break;
case value_t::number_float:
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d.str(*n), *n), no_token);
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d, *n), no_token);
++n;
break;
case value_t::boolean:
+94
View File
@@ -9,11 +9,15 @@
#pragma once
#include <cstddef> // size_t
#include <cstdint> // int64_t, uint64_t
#include <string> // string
#include <type_traits> // integral_constant
#include <nlohmann/json.hpp>
#include <nlohmann/detail/view/document_data.hpp>
#include <nlohmann/detail/view/macro_scope.hpp>
#include <nlohmann/detail/view/node.hpp>
#include <nlohmann/detail/view/scan.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -62,6 +66,96 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n)
return convert_float<FloatType>(first, last, dot, mantissa_end);
}
/*!
@brief the digits of a float token with at most 19 digits, from its layout
The digit layout recorded while parsing says where the integer digits, the
fraction digits, and the exponent are, so the digits are read eight at a
time without scanning.
@param[in] p first character of the token
@param[in] e end of the token
@param[in] limit end of the readable memory (the source text)
*/
NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept
{
const bool negative = *p == '-';
p += negative ? 1 : 0;
std::uint64_t w = parse_upto19(p, int_digits, limit);
p += int_digits;
std::int64_t q = 0;
if (frac_digits != 0)
{
w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit);
p += 1 + frac_digits;
q = -static_cast<std::int64_t>(frac_digits);
}
if (p != e)
{
// [eE][+-]digits; huge exponents saturate (the parser rejected overflow)
++p;
const bool exp_negative = *p == '-';
p += (*p == '-' || *p == '+') ? 1 : 0;
std::int64_t exp_value = 0;
for (; p != e; ++p)
{
if (exp_value < 0x10000000)
{
exp_value = (exp_value * 10) + (*p - '0');
}
}
q += exp_negative ? -exp_value : exp_value;
}
float_significand d;
d.w = w;
d.exponent = q;
d.negative = negative;
return d;
}
/*!
@brief the value of a float token with at most 19 digits, from its layout
The result is correctly rounded by the lexer's conversion
(detail::decimal_to_float(): Clinger's fast path where both operands are
exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19
digits), so it is the value parse() produces.
*/
template<typename FloatType>
NLOHMANN_VIEW_ALWAYS_INLINE FloatType layout_float(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept
{
return decimal_to_float<FloatType>(layout_decimal(p, e, int_digits, frac_digits, limit));
}
/// the value of the float token of a node, as parse() converts it; floats and
/// doubles with at most 19 digits are converted from the digit layout
template<typename FloatType>
FloatType float_value(const document_data& d, const node& n)
{
return float_value<FloatType>(d, n, std::integral_constant<bool, has_native_float_format<FloatType>::value> {});
}
template<typename FloatType>
FloatType float_value(const document_data& d, const node& n, std::true_type /*binary32 or binary64*/)
{
const unsigned int_digits = n.extra & 0xFFu;
const unsigned frac_digits = n.extra >> 8u;
if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many")
{
// (a float token not written by an edit is in the text)
const auto* const first = reinterpret_cast<const unsigned char*>(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
return layout_float<FloatType>(first, first + n.len, int_digits, frac_digits, reinterpret_cast<const unsigned char*>(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
}
return float_value<FloatType>(d.str(n), n);
}
template<typename FloatType>
FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/)
{
return float_value<FloatType>(d.str(n), n);
}
} // namespace view
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
+1 -1
View File
@@ -256,7 +256,7 @@ class view_serializer
}
else
{
write_float(float_value<number_float_t>(m_doc.str(n), n));
write_float(float_value<number_float_t>(m_doc, n));
}
break;
case value_t::object: // LCOV_EXCL_LINE (containers are written by dump())
+1 -1
View File
@@ -49,7 +49,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE T arithmetic_value(const document_data& d, const nod
case value_t::number_integer:
return static_cast<T>(static_cast<typename BasicJsonType::number_integer_t>(static_cast<std::int64_t>(integer_bits(n))));
case value_t::number_float:
return static_cast<T>(float_value<typename BasicJsonType::number_float_t>(d.str(n), n));
return static_cast<T>(float_value<typename BasicJsonType::number_float_t>(d, n));
case value_t::boolean:
return static_cast<T>((n.flags & node_flags::is_true) != 0);
case value_t::null:
+99 -3
View File
@@ -2537,13 +2537,19 @@ NLOHMANN_JSON_NAMESPACE_END
#include <cstddef> // size_t
#include <cstdint> // int64_t, uint64_t
#include <string> // string
#include <type_traits> // integral_constant
// #include <nlohmann/json.hpp>
// #include <nlohmann/detail/view/document_data.hpp>
// #include <nlohmann/detail/view/macro_scope.hpp>
// #include <nlohmann/detail/view/node.hpp>
// #include <nlohmann/detail/view/scan.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -2592,6 +2598,96 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n)
return convert_float<FloatType>(first, last, dot, mantissa_end);
}
/*!
@brief the digits of a float token with at most 19 digits, from its layout
The digit layout recorded while parsing says where the integer digits, the
fraction digits, and the exponent are, so the digits are read eight at a
time without scanning.
@param[in] p first character of the token
@param[in] e end of the token
@param[in] limit end of the readable memory (the source text)
*/
NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept
{
const bool negative = *p == '-';
p += negative ? 1 : 0;
std::uint64_t w = parse_upto19(p, int_digits, limit);
p += int_digits;
std::int64_t q = 0;
if (frac_digits != 0)
{
w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit);
p += 1 + frac_digits;
q = -static_cast<std::int64_t>(frac_digits);
}
if (p != e)
{
// [eE][+-]digits; huge exponents saturate (the parser rejected overflow)
++p;
const bool exp_negative = *p == '-';
p += (*p == '-' || *p == '+') ? 1 : 0;
std::int64_t exp_value = 0;
for (; p != e; ++p)
{
if (exp_value < 0x10000000)
{
exp_value = (exp_value * 10) + (*p - '0');
}
}
q += exp_negative ? -exp_value : exp_value;
}
float_significand d;
d.w = w;
d.exponent = q;
d.negative = negative;
return d;
}
/*!
@brief the value of a float token with at most 19 digits, from its layout
The result is correctly rounded by the lexer's conversion
(detail::decimal_to_float(): Clinger's fast path where both operands are
exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19
digits), so it is the value parse() produces.
*/
template<typename FloatType>
NLOHMANN_VIEW_ALWAYS_INLINE FloatType layout_float(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept
{
return decimal_to_float<FloatType>(layout_decimal(p, e, int_digits, frac_digits, limit));
}
/// the value of the float token of a node, as parse() converts it; floats and
/// doubles with at most 19 digits are converted from the digit layout
template<typename FloatType>
FloatType float_value(const document_data& d, const node& n)
{
return float_value<FloatType>(d, n, std::integral_constant<bool, has_native_float_format<FloatType>::value> {});
}
template<typename FloatType>
FloatType float_value(const document_data& d, const node& n, std::true_type /*binary32 or binary64*/)
{
const unsigned int_digits = n.extra & 0xFFu;
const unsigned frac_digits = n.extra >> 8u;
if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many")
{
// (a float token not written by an edit is in the text)
const auto* const first = reinterpret_cast<const unsigned char*>(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
return layout_float<FloatType>(first, first + n.len, int_digits, frac_digits, reinterpret_cast<const unsigned char*>(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
}
return float_value<FloatType>(d.str(n), n);
}
template<typename FloatType>
FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/)
{
return float_value<FloatType>(d.str(n), n);
}
} // namespace view
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
@@ -2660,7 +2756,7 @@ BasicJsonType materialize(const document_data& d, const node* n)
++n;
break;
case value_t::number_float:
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d.str(*n), *n), no_token);
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d, *n), no_token);
++n;
break;
case value_t::boolean:
@@ -3149,7 +3245,7 @@ class view_serializer
}
else
{
write_float(float_value<number_float_t>(m_doc.str(n), n));
write_float(float_value<number_float_t>(m_doc, n));
}
break;
case value_t::object: // LCOV_EXCL_LINE (containers are written by dump())
@@ -3482,7 +3578,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE T arithmetic_value(const document_data& d, const nod
case value_t::number_integer:
return static_cast<T>(static_cast<typename BasicJsonType::number_integer_t>(static_cast<std::int64_t>(integer_bits(n))));
case value_t::number_float:
return static_cast<T>(float_value<typename BasicJsonType::number_float_t>(d.str(n), n));
return static_cast<T>(float_value<typename BasicJsonType::number_float_t>(d, n));
case value_t::boolean:
return static_cast<T>((n.flags & node_flags::is_true) != 0);
case value_t::null:
+9 -1
View File
@@ -873,7 +873,15 @@ TEST_CASE("json_view values")
std::mt19937_64 rng(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed)
std::vector<std::string> tokens = {"0.1", "-0.0", "1e308", "1.7976931348623157e308", "2.2250738585072011e-308", "4.9e-324", "5e-324",
"0.1000000000000000055511151231257827021181583404541015625", "123456789012345678901234567890",
"9007199254740993", "1.00000000000000011102230246251565404236316680908203125", "7.2057594037927933e16"
"9007199254740993", "1.00000000000000011102230246251565404236316680908203125", "7.2057594037927933e16",
// around the limits of the conversion from the digit layout: 19 and 20
// digits, and those of Clinger's fast path (2^53, 10^22)
"1234567890.123456789", "1234567890.1234567891", "0.0000000000000000001", "123456789012345678.9",
"9007199254740992.0", "9007199254740993.0", "9007199254740994.0", "1.5e22", "1.5e23", "15e-22", "15e-23",
"1e-400", "0.0e0", "-0.0e-5", "12E+3", "12e-0",
// and of float: 2^24 + 1 and 2^24 + 3 (ties), the subnormal and normal limits
"16777217", "16777219", "1.4e-45", "7.006492321624085e-46", "7.006492321624086e-46",
"1.17549435e-38", "0.30000001192092896", "3.4028234e37"
};
for (int i = 0; i < 20000; ++i)
{