mirror of
https://github.com/nlohmann/json.git
synced 2026-10-03 13:10:33 +00:00
Convert floats independently of the C locale's decimal point
Under a locale whose decimal point is longer than one byte (e.g. U+066B in fa_IR.UTF-8 or ar_EG.UTF-8), every float that reached the strtod fallback was truncated at the decimal point: "3.14159265358979323846" became 3.0, and "1.5e400" became 1.0 instead of throwing. With libc++ and in C++11/14, that fallback was taken for most floats. The lexer now converts floats in this order: 1. std::from_chars, now also for float and double with libc++ 20 or later, which does not define __cpp_lib_to_chars (on Apple platforms only if the deployment target provides it); 2. Clinger's fast path (double only); 3. strtof_l/strtod_l/strtold_l with a "C" locale created once, on glibc, Apple platforms, and MSVC; 4. strtof/strtod/strtold with the decimal point of the current locale, which now puts a multi-byte decimal point into a copy of the token. If std::from_chars reports a value out of range, the result is derived from the token (+-infinity or +-0) instead of calling strtod, because implementations disagree on the stored value (P4168). Values that may be subnormal are left to the next step, because libstdc++ before GCC 13 reports some of them as out of range. The conversion helpers moved from the lexer to number_parse.hpp, so the last-resort path can be tested directly. Fixes #5660. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -9,10 +9,8 @@
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
#include <utility> // move
|
||||
@@ -217,18 +215,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
~lexer() = default;
|
||||
|
||||
private:
|
||||
/////////////////////
|
||||
// locales
|
||||
/////////////////////
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// scan functions
|
||||
/////////////////////
|
||||
@@ -1036,24 +1022,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan a number literal
|
||||
|
||||
@@ -1091,9 +1059,9 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
token_type::parse_error otherwise
|
||||
|
||||
@note The scanner is independent of the current locale: token_buffer
|
||||
always holds `.`. Only the std::strtod fallback of convert_number()
|
||||
depends on the locale, and it looks up the decimal point right
|
||||
before converting (see convert_float_locale_aware()).
|
||||
always holds `.`. Only the last-resort std::strtod fallback of
|
||||
convert_number() depends on the locale, and it looks up the decimal
|
||||
point right before converting (see parse_float_locale_aware()).
|
||||
*/
|
||||
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
|
||||
{
|
||||
@@ -1561,8 +1529,10 @@ scan_number_done:
|
||||
// this code is reached if we parse a floating-point number or if an
|
||||
// integer conversion above overflowed. Prefer std::from_chars
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod/strtold.
|
||||
// otherwise the exact Clinger fast path (double only); otherwise
|
||||
// strtof/strtod/strtold with the "C" locale where the C library offers
|
||||
// that; and only as a last resort strtof/strtod/strtold with the
|
||||
// decimal point of the current locale.
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
@@ -1575,63 +1545,13 @@ scan_number_done:
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
convert_float_locale_aware();
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the float in token_buffer with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
*/
|
||||
void convert_float_locale_aware()
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
if (parse_float_c_locale(num_begin, num_end, value_float))
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// get_string() hands the token to the SAX interface with '.'
|
||||
token_buffer[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
parse_float_locale_aware(token_buffer, decimal_point_position, value_float);
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
|
||||
@@ -10,9 +10,13 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <clocale> // LC_NUMERIC, LC_NUMERIC_MASK, newlocale, _create_locale
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtof_l, strtod_l, strtold_l, _strtof_l, _strtod_l, _strtold_l
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
#include <utility> // move
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
@@ -22,15 +26,61 @@
|
||||
// include with __has_include so such toolchains fall back to the scalar path.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__has_include)
|
||||
#if __has_include(<charconv>)
|
||||
#include <charconv> // from_chars (only used when __cpp_lib_to_chars is defined)
|
||||
#include <charconv> // from_chars
|
||||
#include <system_error> // errc
|
||||
|
||||
// std::from_chars is used for floating-point numbers
|
||||
// - for float, double, and long double if __cpp_lib_to_chars announces
|
||||
// complete support (only checked in C++17 or later: some standard
|
||||
// libraries, e.g. libstdc++ 15, define it even in C++14 mode, where
|
||||
// <charconv> is not included);
|
||||
// - for float and double with libc++ 20 or later, which does not define
|
||||
// __cpp_lib_to_chars because long double is missing. On Apple
|
||||
// platforms, the implementation is part of the system's libc++ and
|
||||
// only available when deploying to macOS/iOS 26 or later; for older
|
||||
// deployment targets, _LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT
|
||||
// is 0, and the fallbacks below are used.
|
||||
#if defined(__cpp_lib_to_chars)
|
||||
#define JSON_HAS_FLOAT_FROM_CHARS 1
|
||||
#define JSON_HAS_LONG_DOUBLE_FROM_CHARS 1
|
||||
#elif defined(_LIBCPP_VERSION) && defined(_LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT)
|
||||
#if _LIBCPP_VERSION >= 200000 && _LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT
|
||||
#define JSON_HAS_FLOAT_FROM_CHARS 1
|
||||
#endif
|
||||
#endif
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifndef JSON_HAS_FLOAT_FROM_CHARS
|
||||
#define JSON_HAS_FLOAT_FROM_CHARS 0
|
||||
#endif
|
||||
|
||||
#ifndef JSON_HAS_LONG_DOUBLE_FROM_CHARS
|
||||
#define JSON_HAS_LONG_DOUBLE_FROM_CHARS 0
|
||||
#endif
|
||||
|
||||
// strtof_l/strtod_l/strtold_l convert with a given locale object instead of the
|
||||
// global C locale. They are not part of ISO C or C++, so they are only used where
|
||||
// the C library is known to declare them: Microsoft's UCRT (as _strtod_l etc.),
|
||||
// Apple's libc (in <xlocale.h>, which must follow <cstdlib>), and glibc (as GNU
|
||||
// extensions, visible because g++ and clang++ define _GNU_SOURCE for C++).
|
||||
// Everything else, e.g. MinGW (whose runtime lacks them), musl (which declares
|
||||
// only some of them), Android, or uClibc, uses parse_float_locale_aware().
|
||||
#if defined(_MSC_VER) && !defined(__MINGW32__) && _MSC_VER >= 1900
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 1
|
||||
#elif defined(__APPLE__)
|
||||
#include <xlocale.h> // newlocale, strtof_l, strtod_l, strtold_l
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 1
|
||||
#elif defined(__GLIBC__) && defined(__USE_GNU) && !defined(__UCLIBC__)
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 1
|
||||
#else
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 0
|
||||
#endif
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
// already-validated number token into a value, where possible without the
|
||||
// locale/errno overhead of std::strtoull/std::strtod. They are free functions so
|
||||
// the lexer stays focused on scanning; see lexer::convert_number().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -263,27 +313,128 @@ bool parse_float_fast(const char* /*first*/, const char* /*last*/, FloatType& /*
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief derive the value of a number token that is out of range
|
||||
|
||||
The token [first, last) is a valid JSON number whose value cannot be
|
||||
represented by @a FloatType. The result follows from the token alone: a value
|
||||
of at least 1 can only overflow and becomes ±infinity (which the parser reports
|
||||
as out_of_range.406), a smaller one can only underflow and becomes ±0. The sign
|
||||
is taken from a leading '-', and the magnitude from the decimal exponent of the
|
||||
first nonzero digit.
|
||||
|
||||
A value slightly below the smallest normal number may still be representable
|
||||
as a subnormal number, which some implementations also report as out of range
|
||||
(libstdc++'s std::from_chars before GCC 13, which relies on the ERANGE of
|
||||
strtod for long double, and in GCC 11 for all types). Therefore ±0 is only
|
||||
returned if the value is below half the smallest subnormal number whatever its
|
||||
digits are.
|
||||
|
||||
@param[in] first pointer to the first character of the token
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] out ±infinity or ±0 on success
|
||||
@return true if @a out was set; false if the value may be a subnormal number,
|
||||
in which case the caller converts the token another way
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_out_of_range(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
const bool negative = first != last && *first == '-';
|
||||
const char* p = negative ? first + 1 : first;
|
||||
|
||||
// the decimal exponent of the first nonzero digit, from its position
|
||||
// relative to the decimal point
|
||||
std::int64_t exponent = 0;
|
||||
bool nonzero = false;
|
||||
for (; p != last && *p >= '0' && *p <= '9'; ++p)
|
||||
{
|
||||
if (nonzero)
|
||||
{
|
||||
++exponent;
|
||||
}
|
||||
else
|
||||
{
|
||||
nonzero = *p != '0';
|
||||
}
|
||||
}
|
||||
if (p != last && *p == '.')
|
||||
{
|
||||
for (++p; p != last && *p >= '0' && *p <= '9'; ++p)
|
||||
{
|
||||
if (!nonzero)
|
||||
{
|
||||
--exponent;
|
||||
nonzero = *p != '0';
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (nonzero && p != last && (*p == 'e' || *p == 'E'))
|
||||
{
|
||||
++p;
|
||||
const bool negative_exponent = p != last && *p == '-';
|
||||
if (p != last && (*p == '-' || *p == '+'))
|
||||
{
|
||||
++p;
|
||||
}
|
||||
// saturate: a larger exponent is far out of range for every type
|
||||
constexpr std::int64_t saturation = 100000000000000000; // 10^17
|
||||
std::int64_t explicit_exponent = 0;
|
||||
for (; p != last && *p >= '0' && *p <= '9'; ++p)
|
||||
{
|
||||
if (explicit_exponent < saturation)
|
||||
{
|
||||
explicit_exponent = (explicit_exponent * 10) + (*p - '0');
|
||||
}
|
||||
}
|
||||
exponent += negative_exponent ? -explicit_exponent : explicit_exponent;
|
||||
}
|
||||
|
||||
if (nonzero && exponent >= 0)
|
||||
{
|
||||
out = negative ? -std::numeric_limits<FloatType>::infinity() : std::numeric_limits<FloatType>::infinity();
|
||||
return true;
|
||||
}
|
||||
|
||||
// The value is below 10^(exponent + 1). It rounds to zero if that is at most
|
||||
// half the smallest subnormal number, 2^(min_exponent - digits - 1). The
|
||||
// bound rounds log10(2) up to 0.30103 and the product toward zero, and the
|
||||
// margin of 2 keeps it on the safe side.
|
||||
constexpr std::int64_t zero_exponent = (static_cast<std::int64_t>(std::numeric_limits<FloatType>::min_exponent - std::numeric_limits<FloatType>::digits - 1) * 30103 / 100000) - 2;
|
||||
if (!nonzero || exponent <= zero_exponent)
|
||||
{
|
||||
out = negative ? -FloatType(0) : FloatType(0);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse a float with std::from_chars (Eisel-Lemire) when available
|
||||
|
||||
std::from_chars is locale-independent, correctly rounded, and - via the
|
||||
Eisel-Lemire algorithm in modern standard libraries - much faster than strtod
|
||||
over the whole value range (not just the Clinger subset). It is used only when
|
||||
__cpp_lib_to_chars indicates full floating-point support and only when it
|
||||
consumes the entire token ([first, last)). An under-/overflow (result_out_of_range) also declines, so
|
||||
the caller's strtod fallback supplies the well-defined ±inf/0 result the parser
|
||||
expects (side-stepping the P4168 divergence between implementations).
|
||||
over the whole value range (not just the Clinger subset). It is used only where
|
||||
the standard library implements it for @a FloatType (see
|
||||
JSON_HAS_FLOAT_FROM_CHARS) and only when it consumes the entire token
|
||||
([first, last)).
|
||||
|
||||
For an under- or overflow (std::errc::result_out_of_range), implementations
|
||||
disagree on the value they store: libstdc++ leaves it unchanged, whereas libc++
|
||||
and the MSVC STL store ±0 or ±infinity (P4168). The result is therefore derived
|
||||
from the token, see parse_float_out_of_range().
|
||||
|
||||
@return true if the value was parsed exactly and fully; false to fall back
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
// JSON_HAS_CPP_17 must gate the use as well as the <charconv> include above:
|
||||
// some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even
|
||||
// in C++14 mode, where <charconv> is not included.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars)
|
||||
#if JSON_HAS_FLOAT_FROM_CHARS
|
||||
const auto result = std::from_chars(first, last, out);
|
||||
if (JSON_HEDLEY_UNLIKELY(result.ec == std::errc::result_out_of_range && result.ptr == last))
|
||||
{
|
||||
return parse_float_out_of_range(first, last, out);
|
||||
}
|
||||
return result.ec == std::errc() && result.ptr == last;
|
||||
#else
|
||||
static_cast<void>(first);
|
||||
@@ -293,5 +444,200 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out)
|
||||
#endif
|
||||
}
|
||||
|
||||
#if JSON_HAS_FLOAT_FROM_CHARS && !JSON_HAS_LONG_DOUBLE_FROM_CHARS
|
||||
/// libc++ implements std::from_chars for float and double, but not for long double
|
||||
inline bool parse_float_from_chars(const char* /*first*/, const char* /*last*/, long double& /*out*/) noexcept
|
||||
{
|
||||
return false;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if JSON_HAS_C_LOCALE_STRTOD
|
||||
#if defined(_MSC_VER)
|
||||
using c_locale_t = _locale_t;
|
||||
|
||||
/// the "C" locale for the numeric category, created on first use and never freed
|
||||
inline c_locale_t c_numeric_locale() noexcept
|
||||
{
|
||||
static const c_locale_t c_locale = _create_locale(LC_NUMERIC, "C");
|
||||
return c_locale;
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(float& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = _strtof_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = _strtod_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(long double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = _strtold_l(str, endptr, loc);
|
||||
}
|
||||
#else
|
||||
using c_locale_t = locale_t;
|
||||
|
||||
/// the "C" locale for the numeric category, created on first use and never freed
|
||||
inline c_locale_t c_numeric_locale() noexcept
|
||||
{
|
||||
static const c_locale_t c_locale = newlocale(LC_NUMERIC_MASK, "C", nullptr);
|
||||
return c_locale;
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(float& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = strtof_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = strtod_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(long double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = strtold_l(str, endptr, loc);
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/*!
|
||||
@brief parse a float with strtof_l/strtod_l/strtold_l in the "C" locale
|
||||
|
||||
These functions round correctly like strtod, but take the "C" locale as an
|
||||
argument instead of using the global one, so the decimal point is always '.'.
|
||||
The locale object is created on first use and never freed, so it remains valid
|
||||
for parsers that run during static destruction.
|
||||
|
||||
@param[in] first pointer to the first character of the token, which must be
|
||||
followed by a NUL character
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] out the parsed value (±infinity or ±0 if out of range)
|
||||
@return true if the value was parsed from the entire token; false if the C
|
||||
library offers no such functions (see JSON_HAS_C_LOCALE_STRTOD) or the
|
||||
locale could not be created, in which case the caller falls back to
|
||||
parse_float_locale_aware()
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_c_locale(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
#if JSON_HAS_C_LOCALE_STRTOD
|
||||
const c_locale_t loc = c_numeric_locale();
|
||||
if (JSON_HEDLEY_UNLIKELY(loc == nullptr))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness)
|
||||
strtof_c_locale(out, first, &endptr, loc);
|
||||
return endptr == last;
|
||||
#else
|
||||
static_cast<void>(first);
|
||||
static_cast<void>(last);
|
||||
static_cast<void>(out);
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_global_locale(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_global_locale(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_global_locale(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
inline std::string locale_decimal_point()
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr || *loc->decimal_point == '\0') ? "." : loc->decimal_point;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse a float with strtof/strtod/strtold in the current locale
|
||||
|
||||
This is the last resort for platforms without std::from_chars for @a FloatType
|
||||
and without parse_float_c_locale(). These functions expect the decimal point
|
||||
of the *current* locale, so the '.' in the token is replaced by it. It is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX handler,
|
||||
or another thread) must not truncate the value (#5198). A single-byte decimal
|
||||
point is substituted in place and restored afterwards, because the token is
|
||||
also handed to the SAX interface. A longer one (e.g., the two-byte U+066B of
|
||||
fa_IR.UTF-8 or ar_EG.UTF-8) is put into a copy of the token instead.
|
||||
|
||||
The token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the lookup
|
||||
and the call, and the conversion is repeated with the new decimal point. If it
|
||||
did not change, the value strtod parsed up to that point is kept.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
|
||||
@param[in,out] token the token, with '.' as decimal point
|
||||
@param[in] decimal_point_position the position of the '.' in @a token,
|
||||
or std::string::npos if it has none
|
||||
@param[out] out the parsed value
|
||||
*/
|
||||
template<typename StringType, typename FloatType>
|
||||
void parse_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& out)
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
std::string decimal_point = locale_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness)
|
||||
bool complete = false;
|
||||
if (!has_dot || decimal_point.size() == 1)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point[0] != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = static_cast<typename StringType::value_type>(decimal_point[0]);
|
||||
}
|
||||
strtof_global_locale(out, token.data(), &endptr);
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = '.';
|
||||
}
|
||||
complete = endptr == token.data() + token.size();
|
||||
}
|
||||
else
|
||||
{
|
||||
std::string copy(token.data(), token.size());
|
||||
copy.replace(decimal_point_position, 1, decimal_point);
|
||||
strtof_global_locale(out, copy.c_str(), &endptr);
|
||||
complete = endptr == copy.c_str() + copy.size();
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(complete))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
std::string current_decimal_point = locale_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = std::move(current_decimal_point);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -42,6 +42,9 @@
|
||||
#undef JSON_HAS_RANGES
|
||||
#undef JSON_HAS_STD_FORMAT
|
||||
#undef JSON_HAS_STATIC_RTTI
|
||||
#undef JSON_HAS_FLOAT_FROM_CHARS
|
||||
#undef JSON_HAS_LONG_DOUBLE_FROM_CHARS
|
||||
#undef JSON_HAS_C_LOCALE_STRTOD
|
||||
#undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
|
||||
#undef JSON_BRACE_INIT_COPY_SEMANTICS
|
||||
#undef JSON_PRECISE_STREAM_POSITION
|
||||
|
||||
Reference in New Issue
Block a user