diff --git a/docs/mkdocs/docs/api/basic_json/parse.md b/docs/mkdocs/docs/api/basic_json/parse.md index 20bb1c708..7e1f08bab 100644 --- a/docs/mkdocs/docs/api/basic_json/parse.md +++ b/docs/mkdocs/docs/api/basic_json/parse.md @@ -254,6 +254,8 @@ outside of a string, invalid) byte; see the [FAQ entry](../../home/faq.md#nul-by - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. - `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating it as end of input; planned to become the default in version 4.0.0. +- The result of converting floating-point numbers no longer depends on the C locale in version 3.13.0; before, a + locale whose decimal point is longer than one byte (e.g., `fa_IR.UTF-8`) truncated them at the decimal point. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/features/types/number_handling.md b/docs/mkdocs/docs/features/types/number_handling.md index cf37b044a..7f4810469 100644 --- a/docs/mkdocs/docs/features/types/number_handling.md +++ b/docs/mkdocs/docs/features/types/number_handling.md @@ -75,6 +75,13 @@ otherwise, it uses unsigned integer storage. [`std::strtoull`](https://en.cppreference.com/w/cpp/string/byte/strtoul), [`std::strtoll`](https://en.cppreference.com/w/cpp/string/byte/strtol), and [`std::strtod`](https://en.cppreference.com/w/cpp/string/byte/strtof), respectively. + - The result of converting floating-point numbers does not depend on the C locale (`LC_NUMERIC`). They are + converted with [`std::from_chars`](https://en.cppreference.com/w/cpp/utility/from_chars) where the standard + library implements it for the number type (including libc++ 20 or later for `#!c float` and `#!c double`), + otherwise with `strtod_l` and the "C" locale where the C library provides it (glibc, macOS, MSVC), and otherwise + with `std::strtod` and the decimal point of the current locale. Before version 3.13.0, the last way was used much + more often, and a locale whose decimal point is longer than one byte (e.g., `fa_IR.UTF-8`) truncated numbers at + the decimal point. !!! example "Examples" @@ -85,10 +92,11 @@ otherwise, it uses unsigned integer storage. ### Number limits - Any 64-bit signed or unsigned integer can be stored without loss of precision. -- Numbers exceeding the limits of `#!c double` (i.e., numbers that after conversion via -[`std::strtod`](https://en.cppreference.com/w/cpp/string/byte/strtof) are not satisfying +- Numbers exceeding the limits of `#!c double` (i.e., numbers that after conversion are not satisfying [`std::isfinite`](https://en.cppreference.com/w/cpp/numeric/math/isfinite) such as `#!c 1E400`) will throw exception [`json.exception.out_of_range.406`](../../home/exceptions.md#jsonexceptionout_of_range406) during parsing. +- Numbers too close to zero to be represented as `#!c double`, not even as subnormal number (such as `#!c 1E-400`), are +stored as `#!c 0.0`, or as `#!c -0.0` if they are negative. - Floating-point numbers are rounded to the next number representable as `double`. For instance `#!c 3.141592653589793238462643383279` is stored as [`0x400921fb54442d18`](https://float.exposed/0x400921fb54442d18). This is the same behavior as the code `#!c double x = 3.141592653589793238462643383279;`. diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index 00a964a17..f6f1dfbae 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -9,10 +9,8 @@ #pragma once #include // array -#include // localeconv #include // size_t #include // snprintf -#include // strtof, strtod, strtold, strtoll, strtoull #include // initializer_list #include // char_traits, string #include // move @@ -217,18 +215,6 @@ class lexer : public lexer_base ~lexer() = default; private: - ///////////////////// - // locales - ///////////////////// - - /// return the decimal point of the current locale - static char get_decimal_point() noexcept - { - const auto* loc = localeconv(); - JSON_ASSERT(loc != nullptr); - return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point); - } - ///////////////////// // scan functions ///////////////////// @@ -1036,24 +1022,6 @@ class lexer : public lexer_base } } - JSON_HEDLEY_NON_NULL(2) - static void strtof(float& f, const char* str, char** endptr) noexcept - { - f = std::strtof(str, endptr); - } - - JSON_HEDLEY_NON_NULL(2) - static void strtof(double& f, const char* str, char** endptr) noexcept - { - f = std::strtod(str, endptr); - } - - JSON_HEDLEY_NON_NULL(2) - static void strtof(long double& f, const char* str, char** endptr) noexcept - { - f = std::strtold(str, endptr); - } - /*! @brief scan a number literal @@ -1091,9 +1059,9 @@ class lexer : public lexer_base token_type::parse_error otherwise @note The scanner is independent of the current locale: token_buffer - always holds `.`. Only the std::strtod fallback of convert_number() - depends on the locale, and it looks up the decimal point right - before converting (see convert_float_locale_aware()). + always holds `.`. Only the last-resort std::strtod fallback of + convert_number() depends on the locale, and it looks up the decimal + point right before converting (see parse_float_locale_aware()). */ token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated. { @@ -1561,8 +1529,10 @@ scan_number_done: // this code is reached if we parse a floating-point number or if an // integer conversion above overflowed. Prefer std::from_chars // (Eisel-Lemire, locale-independent, correctly rounded) when available; - // otherwise the exact Clinger fast path (double only); otherwise the - // locale-aware strtof/strtod/strtold. + // otherwise the exact Clinger fast path (double only); otherwise + // strtof/strtod/strtold with the "C" locale where the C library offers + // that; and only as a last resort strtof/strtod/strtold with the + // decimal point of the current locale. if (parse_float_from_chars(num_begin, num_end, value_float)) { return token_type::value_float; @@ -1575,63 +1545,13 @@ scan_number_done: { return token_type::value_float; } - - convert_float_locale_aware(); - return token_type::value_float; - } - - /*! - @brief convert the float in token_buffer with strtof/strtod/strtold - - These functions expect the decimal point of the *current* locale, so it is - looked up right before the conversion instead of once when the lexer is - constructed: a locale change in between (by a parser callback, a SAX - handler, or another thread) must not truncate the value (#5198). The - token has been validated before, so if the conversion stops early and the - decimal point changed in the meantime, the locale changed between the - lookup and the call, and the conversion is repeated with the new decimal - point. If the decimal point did not change, a retry cannot succeed: the - locale's decimal point is not a single character (e.g., the two-byte - U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place. - The value strtod parsed up to that point is kept, as before this change. - - Note that changing the locale in another thread *while* strtod runs is - undefined behavior of the C library, which this function cannot prevent. - */ - void convert_float_locale_aware() - { - const bool has_dot = decimal_point_position != std::string::npos; - char decimal_point = get_decimal_point(); - for (;;) + if (parse_float_c_locale(num_begin, num_end, value_float)) { - const bool substitute = has_dot && decimal_point != '.'; - if (substitute) - { - token_buffer[decimal_point_position] = static_cast(decimal_point); - } - - char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) - strtof(value_float, token_buffer.data(), &endptr); - - if (substitute) - { - // get_string() hands the token to the SAX interface with '.' - token_buffer[decimal_point_position] = '.'; - } - - if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size())) - { - return; - } - - // retry only if the locale changed; otherwise, this would loop forever - const char current_decimal_point = get_decimal_point(); - if (current_decimal_point == decimal_point) - { - return; - } - decimal_point = current_decimal_point; + return token_type::value_float; } + + parse_float_locale_aware(token_buffer, decimal_point_position, value_float); + return token_type::value_float; } /*! diff --git a/include/nlohmann/detail/input/number_parse.hpp b/include/nlohmann/detail/input/number_parse.hpp index 25f6cac91..86e5df742 100644 --- a/include/nlohmann/detail/input/number_parse.hpp +++ b/include/nlohmann/detail/input/number_parse.hpp @@ -10,9 +10,13 @@ #include // array #include // FLT_EVAL_METHOD +#include // LC_NUMERIC, LC_NUMERIC_MASK, newlocale, _create_locale #include // size_t #include // int64_t, uint64_t +#include // strtof, strtod, strtold, strtof_l, strtod_l, strtold_l, _strtof_l, _strtod_l, _strtold_l #include // numeric_limits +#include // string +#include // move #include @@ -22,15 +26,61 @@ // include with __has_include so such toolchains fall back to the scalar path. #if defined(JSON_HAS_CPP_17) && defined(__has_include) #if __has_include() - #include // from_chars (only used when __cpp_lib_to_chars is defined) + #include // from_chars #include // errc + + // std::from_chars is used for floating-point numbers + // - for float, double, and long double if __cpp_lib_to_chars announces + // complete support (only checked in C++17 or later: some standard + // libraries, e.g. libstdc++ 15, define it even in C++14 mode, where + // is not included); + // - for float and double with libc++ 20 or later, which does not define + // __cpp_lib_to_chars because long double is missing. On Apple + // platforms, the implementation is part of the system's libc++ and + // only available when deploying to macOS/iOS 26 or later; for older + // deployment targets, _LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT + // is 0, and the fallbacks below are used. + #if defined(__cpp_lib_to_chars) + #define JSON_HAS_FLOAT_FROM_CHARS 1 + #define JSON_HAS_LONG_DOUBLE_FROM_CHARS 1 + #elif defined(_LIBCPP_VERSION) && defined(_LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT) + #if _LIBCPP_VERSION >= 200000 && _LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT + #define JSON_HAS_FLOAT_FROM_CHARS 1 + #endif + #endif #endif #endif +#ifndef JSON_HAS_FLOAT_FROM_CHARS + #define JSON_HAS_FLOAT_FROM_CHARS 0 +#endif + +#ifndef JSON_HAS_LONG_DOUBLE_FROM_CHARS + #define JSON_HAS_LONG_DOUBLE_FROM_CHARS 0 +#endif + +// strtof_l/strtod_l/strtold_l convert with a given locale object instead of the +// global C locale. They are not part of ISO C or C++, so they are only used where +// the C library is known to declare them: Microsoft's UCRT (as _strtod_l etc.), +// Apple's libc (in , which must follow ), and glibc (as GNU +// extensions, visible because g++ and clang++ define _GNU_SOURCE for C++). +// Everything else, e.g. MinGW (whose runtime lacks them), musl (which declares +// only some of them), Android, or uClibc, uses parse_float_locale_aware(). +#if defined(_MSC_VER) && !defined(__MINGW32__) && _MSC_VER >= 1900 + #define JSON_HAS_C_LOCALE_STRTOD 1 +#elif defined(__APPLE__) + #include // newlocale, strtof_l, strtod_l, strtold_l + #define JSON_HAS_C_LOCALE_STRTOD 1 +#elif defined(__GLIBC__) && defined(__USE_GNU) && !defined(__UCLIBC__) + #define JSON_HAS_C_LOCALE_STRTOD 1 +#else + #define JSON_HAS_C_LOCALE_STRTOD 0 +#endif + // This file contains the value-conversion helpers used by the lexer to turn an -// already-validated number token into a value, without the locale/errno -// overhead of std::strtoull/std::strtod. They are free functions so the lexer -// stays focused on scanning; see lexer::convert_number(). +// already-validated number token into a value, where possible without the +// locale/errno overhead of std::strtoull/std::strtod. They are free functions so +// the lexer stays focused on scanning; see lexer::convert_number(). NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -263,27 +313,128 @@ bool parse_float_fast(const char* /*first*/, const char* /*last*/, FloatType& /* return false; } +/*! +@brief derive the value of a number token that is out of range + +The token [first, last) is a valid JSON number whose value cannot be +represented by @a FloatType. The result follows from the token alone: a value +of at least 1 can only overflow and becomes ±infinity (which the parser reports +as out_of_range.406), a smaller one can only underflow and becomes ±0. The sign +is taken from a leading '-', and the magnitude from the decimal exponent of the +first nonzero digit. + +A value slightly below the smallest normal number may still be representable +as a subnormal number, which some implementations also report as out of range +(libstdc++'s std::from_chars before GCC 13, which relies on the ERANGE of +strtod for long double, and in GCC 11 for all types). Therefore ±0 is only +returned if the value is below half the smallest subnormal number whatever its +digits are. + +@param[in] first pointer to the first character of the token +@param[in] last pointer past the last character +@param[out] out ±infinity or ±0 on success +@return true if @a out was set; false if the value may be a subnormal number, + in which case the caller converts the token another way +*/ +template +bool parse_float_out_of_range(const char* first, const char* last, FloatType& out) noexcept +{ + const bool negative = first != last && *first == '-'; + const char* p = negative ? first + 1 : first; + + // the decimal exponent of the first nonzero digit, from its position + // relative to the decimal point + std::int64_t exponent = 0; + bool nonzero = false; + for (; p != last && *p >= '0' && *p <= '9'; ++p) + { + if (nonzero) + { + ++exponent; + } + else + { + nonzero = *p != '0'; + } + } + if (p != last && *p == '.') + { + for (++p; p != last && *p >= '0' && *p <= '9'; ++p) + { + if (!nonzero) + { + --exponent; + nonzero = *p != '0'; + } + } + } + + if (nonzero && p != last && (*p == 'e' || *p == 'E')) + { + ++p; + const bool negative_exponent = p != last && *p == '-'; + if (p != last && (*p == '-' || *p == '+')) + { + ++p; + } + // saturate: a larger exponent is far out of range for every type + constexpr std::int64_t saturation = 100000000000000000; // 10^17 + std::int64_t explicit_exponent = 0; + for (; p != last && *p >= '0' && *p <= '9'; ++p) + { + if (explicit_exponent < saturation) + { + explicit_exponent = (explicit_exponent * 10) + (*p - '0'); + } + } + exponent += negative_exponent ? -explicit_exponent : explicit_exponent; + } + + if (nonzero && exponent >= 0) + { + out = negative ? -std::numeric_limits::infinity() : std::numeric_limits::infinity(); + return true; + } + + // The value is below 10^(exponent + 1). It rounds to zero if that is at most + // half the smallest subnormal number, 2^(min_exponent - digits - 1). The + // bound rounds log10(2) up to 0.30103 and the product toward zero, and the + // margin of 2 keeps it on the safe side. + constexpr std::int64_t zero_exponent = (static_cast(std::numeric_limits::min_exponent - std::numeric_limits::digits - 1) * 30103 / 100000) - 2; + if (!nonzero || exponent <= zero_exponent) + { + out = negative ? -FloatType(0) : FloatType(0); + return true; + } + return false; +} + /*! @brief parse a float with std::from_chars (Eisel-Lemire) when available std::from_chars is locale-independent, correctly rounded, and - via the Eisel-Lemire algorithm in modern standard libraries - much faster than strtod -over the whole value range (not just the Clinger subset). It is used only when -__cpp_lib_to_chars indicates full floating-point support and only when it -consumes the entire token ([first, last)). An under-/overflow (result_out_of_range) also declines, so -the caller's strtod fallback supplies the well-defined ±inf/0 result the parser -expects (side-stepping the P4168 divergence between implementations). +over the whole value range (not just the Clinger subset). It is used only where +the standard library implements it for @a FloatType (see +JSON_HAS_FLOAT_FROM_CHARS) and only when it consumes the entire token +([first, last)). + +For an under- or overflow (std::errc::result_out_of_range), implementations +disagree on the value they store: libstdc++ leaves it unchanged, whereas libc++ +and the MSVC STL store ±0 or ±infinity (P4168). The result is therefore derived +from the token, see parse_float_out_of_range(). @return true if the value was parsed exactly and fully; false to fall back */ template bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept { - // JSON_HAS_CPP_17 must gate the use as well as the include above: - // some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even - // in C++14 mode, where is not included. -#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars) +#if JSON_HAS_FLOAT_FROM_CHARS const auto result = std::from_chars(first, last, out); + if (JSON_HEDLEY_UNLIKELY(result.ec == std::errc::result_out_of_range && result.ptr == last)) + { + return parse_float_out_of_range(first, last, out); + } return result.ec == std::errc() && result.ptr == last; #else static_cast(first); @@ -293,5 +444,200 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out) #endif } +#if JSON_HAS_FLOAT_FROM_CHARS && !JSON_HAS_LONG_DOUBLE_FROM_CHARS +/// libc++ implements std::from_chars for float and double, but not for long double +inline bool parse_float_from_chars(const char* /*first*/, const char* /*last*/, long double& /*out*/) noexcept +{ + return false; +} +#endif + +#if JSON_HAS_C_LOCALE_STRTOD +#if defined(_MSC_VER) +using c_locale_t = _locale_t; + +/// the "C" locale for the numeric category, created on first use and never freed +inline c_locale_t c_numeric_locale() noexcept +{ + static const c_locale_t c_locale = _create_locale(LC_NUMERIC, "C"); + return c_locale; +} + +inline void strtof_c_locale(float& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = _strtof_l(str, endptr, loc); +} + +inline void strtof_c_locale(double& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = _strtod_l(str, endptr, loc); +} + +inline void strtof_c_locale(long double& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = _strtold_l(str, endptr, loc); +} +#else +using c_locale_t = locale_t; + +/// the "C" locale for the numeric category, created on first use and never freed +inline c_locale_t c_numeric_locale() noexcept +{ + static const c_locale_t c_locale = newlocale(LC_NUMERIC_MASK, "C", nullptr); + return c_locale; +} + +inline void strtof_c_locale(float& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = strtof_l(str, endptr, loc); +} + +inline void strtof_c_locale(double& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = strtod_l(str, endptr, loc); +} + +inline void strtof_c_locale(long double& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = strtold_l(str, endptr, loc); +} +#endif +#endif + +/*! +@brief parse a float with strtof_l/strtod_l/strtold_l in the "C" locale + +These functions round correctly like strtod, but take the "C" locale as an +argument instead of using the global one, so the decimal point is always '.'. +The locale object is created on first use and never freed, so it remains valid +for parsers that run during static destruction. + +@param[in] first pointer to the first character of the token, which must be + followed by a NUL character +@param[in] last pointer past the last character +@param[out] out the parsed value (±infinity or ±0 if out of range) +@return true if the value was parsed from the entire token; false if the C + library offers no such functions (see JSON_HAS_C_LOCALE_STRTOD) or the + locale could not be created, in which case the caller falls back to + parse_float_locale_aware() +*/ +template +bool parse_float_c_locale(const char* first, const char* last, FloatType& out) noexcept +{ +#if JSON_HAS_C_LOCALE_STRTOD + const c_locale_t loc = c_numeric_locale(); + if (JSON_HEDLEY_UNLIKELY(loc == nullptr)) + { + return false; + } + char* endptr = nullptr; // NOLINT(misc-const-correctness) + strtof_c_locale(out, first, &endptr, loc); + return endptr == last; +#else + static_cast(first); + static_cast(last); + static_cast(out); + return false; +#endif +} + +JSON_HEDLEY_NON_NULL(2) +inline void strtof_global_locale(float& f, const char* str, char** endptr) noexcept +{ + f = std::strtof(str, endptr); +} + +JSON_HEDLEY_NON_NULL(2) +inline void strtof_global_locale(double& f, const char* str, char** endptr) noexcept +{ + f = std::strtod(str, endptr); +} + +JSON_HEDLEY_NON_NULL(2) +inline void strtof_global_locale(long double& f, const char* str, char** endptr) noexcept +{ + f = std::strtold(str, endptr); +} + +/// return the decimal point of the current locale +inline std::string locale_decimal_point() +{ + const auto* loc = localeconv(); + JSON_ASSERT(loc != nullptr); + return (loc->decimal_point == nullptr || *loc->decimal_point == '\0') ? "." : loc->decimal_point; +} + +/*! +@brief parse a float with strtof/strtod/strtold in the current locale + +This is the last resort for platforms without std::from_chars for @a FloatType +and without parse_float_c_locale(). These functions expect the decimal point +of the *current* locale, so the '.' in the token is replaced by it. It is +looked up right before the conversion instead of once when the lexer is +constructed: a locale change in between (by a parser callback, a SAX handler, +or another thread) must not truncate the value (#5198). A single-byte decimal +point is substituted in place and restored afterwards, because the token is +also handed to the SAX interface. A longer one (e.g., the two-byte U+066B of +fa_IR.UTF-8 or ar_EG.UTF-8) is put into a copy of the token instead. + +The token has been validated before, so if the conversion stops early and the +decimal point changed in the meantime, the locale changed between the lookup +and the call, and the conversion is repeated with the new decimal point. If it +did not change, the value strtod parsed up to that point is kept. + +Note that changing the locale in another thread *while* strtod runs is +undefined behavior of the C library, which this function cannot prevent. + +@param[in,out] token the token, with '.' as decimal point +@param[in] decimal_point_position the position of the '.' in @a token, + or std::string::npos if it has none +@param[out] out the parsed value +*/ +template +void parse_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& out) +{ + const bool has_dot = decimal_point_position != std::string::npos; + std::string decimal_point = locale_decimal_point(); + for (;;) + { + char* endptr = nullptr; // NOLINT(misc-const-correctness) + bool complete = false; + if (!has_dot || decimal_point.size() == 1) + { + const bool substitute = has_dot && decimal_point[0] != '.'; + if (substitute) + { + token[decimal_point_position] = static_cast(decimal_point[0]); + } + strtof_global_locale(out, token.data(), &endptr); + if (substitute) + { + token[decimal_point_position] = '.'; + } + complete = endptr == token.data() + token.size(); + } + else + { + std::string copy(token.data(), token.size()); + copy.replace(decimal_point_position, 1, decimal_point); + strtof_global_locale(out, copy.c_str(), &endptr); + complete = endptr == copy.c_str() + copy.size(); + } + + if (JSON_HEDLEY_LIKELY(complete)) + { + return; + } + + // retry only if the locale changed; otherwise, this would loop forever + std::string current_decimal_point = locale_decimal_point(); + if (current_decimal_point == decimal_point) + { + return; + } + decimal_point = std::move(current_decimal_point); + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index a73951a22..4a6b4481e 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -42,6 +42,9 @@ #undef JSON_HAS_RANGES #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI + #undef JSON_HAS_FLOAT_FROM_CHARS + #undef JSON_HAS_LONG_DOUBLE_FROM_CHARS + #undef JSON_HAS_C_LOCALE_STRTOD #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 576498738..5e1c200bd 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8472,10 +8472,8 @@ NLOHMANN_JSON_NAMESPACE_END #include // array -#include // localeconv #include // size_t #include // snprintf -#include // strtof, strtod, strtold, strtoll, strtoull #include // initializer_list #include // char_traits, string #include // move @@ -8496,9 +8494,13 @@ NLOHMANN_JSON_NAMESPACE_END #include // array #include // FLT_EVAL_METHOD +#include // LC_NUMERIC, LC_NUMERIC_MASK, newlocale, _create_locale #include // size_t #include // int64_t, uint64_t +#include // strtof, strtod, strtold, strtof_l, strtod_l, strtold_l, _strtof_l, _strtod_l, _strtold_l #include // numeric_limits +#include // string +#include // move // #include @@ -8509,15 +8511,61 @@ NLOHMANN_JSON_NAMESPACE_END // include with __has_include so such toolchains fall back to the scalar path. #if defined(JSON_HAS_CPP_17) && defined(__has_include) #if __has_include() - #include // from_chars (only used when __cpp_lib_to_chars is defined) + #include // from_chars #include // errc + + // std::from_chars is used for floating-point numbers + // - for float, double, and long double if __cpp_lib_to_chars announces + // complete support (only checked in C++17 or later: some standard + // libraries, e.g. libstdc++ 15, define it even in C++14 mode, where + // is not included); + // - for float and double with libc++ 20 or later, which does not define + // __cpp_lib_to_chars because long double is missing. On Apple + // platforms, the implementation is part of the system's libc++ and + // only available when deploying to macOS/iOS 26 or later; for older + // deployment targets, _LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT + // is 0, and the fallbacks below are used. + #if defined(__cpp_lib_to_chars) + #define JSON_HAS_FLOAT_FROM_CHARS 1 + #define JSON_HAS_LONG_DOUBLE_FROM_CHARS 1 + #elif defined(_LIBCPP_VERSION) && defined(_LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT) + #if _LIBCPP_VERSION >= 200000 && _LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT + #define JSON_HAS_FLOAT_FROM_CHARS 1 + #endif + #endif #endif #endif +#ifndef JSON_HAS_FLOAT_FROM_CHARS + #define JSON_HAS_FLOAT_FROM_CHARS 0 +#endif + +#ifndef JSON_HAS_LONG_DOUBLE_FROM_CHARS + #define JSON_HAS_LONG_DOUBLE_FROM_CHARS 0 +#endif + +// strtof_l/strtod_l/strtold_l convert with a given locale object instead of the +// global C locale. They are not part of ISO C or C++, so they are only used where +// the C library is known to declare them: Microsoft's UCRT (as _strtod_l etc.), +// Apple's libc (in , which must follow ), and glibc (as GNU +// extensions, visible because g++ and clang++ define _GNU_SOURCE for C++). +// Everything else, e.g. MinGW (whose runtime lacks them), musl (which declares +// only some of them), Android, or uClibc, uses parse_float_locale_aware(). +#if defined(_MSC_VER) && !defined(__MINGW32__) && _MSC_VER >= 1900 + #define JSON_HAS_C_LOCALE_STRTOD 1 +#elif defined(__APPLE__) + #include // newlocale, strtof_l, strtod_l, strtold_l + #define JSON_HAS_C_LOCALE_STRTOD 1 +#elif defined(__GLIBC__) && defined(__USE_GNU) && !defined(__UCLIBC__) + #define JSON_HAS_C_LOCALE_STRTOD 1 +#else + #define JSON_HAS_C_LOCALE_STRTOD 0 +#endif + // This file contains the value-conversion helpers used by the lexer to turn an -// already-validated number token into a value, without the locale/errno -// overhead of std::strtoull/std::strtod. They are free functions so the lexer -// stays focused on scanning; see lexer::convert_number(). +// already-validated number token into a value, where possible without the +// locale/errno overhead of std::strtoull/std::strtod. They are free functions so +// the lexer stays focused on scanning; see lexer::convert_number(). NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -8750,27 +8798,128 @@ bool parse_float_fast(const char* /*first*/, const char* /*last*/, FloatType& /* return false; } +/*! +@brief derive the value of a number token that is out of range + +The token [first, last) is a valid JSON number whose value cannot be +represented by @a FloatType. The result follows from the token alone: a value +of at least 1 can only overflow and becomes ±infinity (which the parser reports +as out_of_range.406), a smaller one can only underflow and becomes ±0. The sign +is taken from a leading '-', and the magnitude from the decimal exponent of the +first nonzero digit. + +A value slightly below the smallest normal number may still be representable +as a subnormal number, which some implementations also report as out of range +(libstdc++'s std::from_chars before GCC 13, which relies on the ERANGE of +strtod for long double, and in GCC 11 for all types). Therefore ±0 is only +returned if the value is below half the smallest subnormal number whatever its +digits are. + +@param[in] first pointer to the first character of the token +@param[in] last pointer past the last character +@param[out] out ±infinity or ±0 on success +@return true if @a out was set; false if the value may be a subnormal number, + in which case the caller converts the token another way +*/ +template +bool parse_float_out_of_range(const char* first, const char* last, FloatType& out) noexcept +{ + const bool negative = first != last && *first == '-'; + const char* p = negative ? first + 1 : first; + + // the decimal exponent of the first nonzero digit, from its position + // relative to the decimal point + std::int64_t exponent = 0; + bool nonzero = false; + for (; p != last && *p >= '0' && *p <= '9'; ++p) + { + if (nonzero) + { + ++exponent; + } + else + { + nonzero = *p != '0'; + } + } + if (p != last && *p == '.') + { + for (++p; p != last && *p >= '0' && *p <= '9'; ++p) + { + if (!nonzero) + { + --exponent; + nonzero = *p != '0'; + } + } + } + + if (nonzero && p != last && (*p == 'e' || *p == 'E')) + { + ++p; + const bool negative_exponent = p != last && *p == '-'; + if (p != last && (*p == '-' || *p == '+')) + { + ++p; + } + // saturate: a larger exponent is far out of range for every type + constexpr std::int64_t saturation = 100000000000000000; // 10^17 + std::int64_t explicit_exponent = 0; + for (; p != last && *p >= '0' && *p <= '9'; ++p) + { + if (explicit_exponent < saturation) + { + explicit_exponent = (explicit_exponent * 10) + (*p - '0'); + } + } + exponent += negative_exponent ? -explicit_exponent : explicit_exponent; + } + + if (nonzero && exponent >= 0) + { + out = negative ? -std::numeric_limits::infinity() : std::numeric_limits::infinity(); + return true; + } + + // The value is below 10^(exponent + 1). It rounds to zero if that is at most + // half the smallest subnormal number, 2^(min_exponent - digits - 1). The + // bound rounds log10(2) up to 0.30103 and the product toward zero, and the + // margin of 2 keeps it on the safe side. + constexpr std::int64_t zero_exponent = (static_cast(std::numeric_limits::min_exponent - std::numeric_limits::digits - 1) * 30103 / 100000) - 2; + if (!nonzero || exponent <= zero_exponent) + { + out = negative ? -FloatType(0) : FloatType(0); + return true; + } + return false; +} + /*! @brief parse a float with std::from_chars (Eisel-Lemire) when available std::from_chars is locale-independent, correctly rounded, and - via the Eisel-Lemire algorithm in modern standard libraries - much faster than strtod -over the whole value range (not just the Clinger subset). It is used only when -__cpp_lib_to_chars indicates full floating-point support and only when it -consumes the entire token ([first, last)). An under-/overflow (result_out_of_range) also declines, so -the caller's strtod fallback supplies the well-defined ±inf/0 result the parser -expects (side-stepping the P4168 divergence between implementations). +over the whole value range (not just the Clinger subset). It is used only where +the standard library implements it for @a FloatType (see +JSON_HAS_FLOAT_FROM_CHARS) and only when it consumes the entire token +([first, last)). + +For an under- or overflow (std::errc::result_out_of_range), implementations +disagree on the value they store: libstdc++ leaves it unchanged, whereas libc++ +and the MSVC STL store ±0 or ±infinity (P4168). The result is therefore derived +from the token, see parse_float_out_of_range(). @return true if the value was parsed exactly and fully; false to fall back */ template bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept { - // JSON_HAS_CPP_17 must gate the use as well as the include above: - // some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even - // in C++14 mode, where is not included. -#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars) +#if JSON_HAS_FLOAT_FROM_CHARS const auto result = std::from_chars(first, last, out); + if (JSON_HEDLEY_UNLIKELY(result.ec == std::errc::result_out_of_range && result.ptr == last)) + { + return parse_float_out_of_range(first, last, out); + } return result.ec == std::errc() && result.ptr == last; #else static_cast(first); @@ -8780,6 +8929,201 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out) #endif } +#if JSON_HAS_FLOAT_FROM_CHARS && !JSON_HAS_LONG_DOUBLE_FROM_CHARS +/// libc++ implements std::from_chars for float and double, but not for long double +inline bool parse_float_from_chars(const char* /*first*/, const char* /*last*/, long double& /*out*/) noexcept +{ + return false; +} +#endif + +#if JSON_HAS_C_LOCALE_STRTOD +#if defined(_MSC_VER) +using c_locale_t = _locale_t; + +/// the "C" locale for the numeric category, created on first use and never freed +inline c_locale_t c_numeric_locale() noexcept +{ + static const c_locale_t c_locale = _create_locale(LC_NUMERIC, "C"); + return c_locale; +} + +inline void strtof_c_locale(float& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = _strtof_l(str, endptr, loc); +} + +inline void strtof_c_locale(double& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = _strtod_l(str, endptr, loc); +} + +inline void strtof_c_locale(long double& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = _strtold_l(str, endptr, loc); +} +#else +using c_locale_t = locale_t; + +/// the "C" locale for the numeric category, created on first use and never freed +inline c_locale_t c_numeric_locale() noexcept +{ + static const c_locale_t c_locale = newlocale(LC_NUMERIC_MASK, "C", nullptr); + return c_locale; +} + +inline void strtof_c_locale(float& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = strtof_l(str, endptr, loc); +} + +inline void strtof_c_locale(double& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = strtod_l(str, endptr, loc); +} + +inline void strtof_c_locale(long double& f, const char* str, char** endptr, c_locale_t loc) noexcept +{ + f = strtold_l(str, endptr, loc); +} +#endif +#endif + +/*! +@brief parse a float with strtof_l/strtod_l/strtold_l in the "C" locale + +These functions round correctly like strtod, but take the "C" locale as an +argument instead of using the global one, so the decimal point is always '.'. +The locale object is created on first use and never freed, so it remains valid +for parsers that run during static destruction. + +@param[in] first pointer to the first character of the token, which must be + followed by a NUL character +@param[in] last pointer past the last character +@param[out] out the parsed value (±infinity or ±0 if out of range) +@return true if the value was parsed from the entire token; false if the C + library offers no such functions (see JSON_HAS_C_LOCALE_STRTOD) or the + locale could not be created, in which case the caller falls back to + parse_float_locale_aware() +*/ +template +bool parse_float_c_locale(const char* first, const char* last, FloatType& out) noexcept +{ +#if JSON_HAS_C_LOCALE_STRTOD + const c_locale_t loc = c_numeric_locale(); + if (JSON_HEDLEY_UNLIKELY(loc == nullptr)) + { + return false; + } + char* endptr = nullptr; // NOLINT(misc-const-correctness) + strtof_c_locale(out, first, &endptr, loc); + return endptr == last; +#else + static_cast(first); + static_cast(last); + static_cast(out); + return false; +#endif +} + +JSON_HEDLEY_NON_NULL(2) +inline void strtof_global_locale(float& f, const char* str, char** endptr) noexcept +{ + f = std::strtof(str, endptr); +} + +JSON_HEDLEY_NON_NULL(2) +inline void strtof_global_locale(double& f, const char* str, char** endptr) noexcept +{ + f = std::strtod(str, endptr); +} + +JSON_HEDLEY_NON_NULL(2) +inline void strtof_global_locale(long double& f, const char* str, char** endptr) noexcept +{ + f = std::strtold(str, endptr); +} + +/// return the decimal point of the current locale +inline std::string locale_decimal_point() +{ + const auto* loc = localeconv(); + JSON_ASSERT(loc != nullptr); + return (loc->decimal_point == nullptr || *loc->decimal_point == '\0') ? "." : loc->decimal_point; +} + +/*! +@brief parse a float with strtof/strtod/strtold in the current locale + +This is the last resort for platforms without std::from_chars for @a FloatType +and without parse_float_c_locale(). These functions expect the decimal point +of the *current* locale, so the '.' in the token is replaced by it. It is +looked up right before the conversion instead of once when the lexer is +constructed: a locale change in between (by a parser callback, a SAX handler, +or another thread) must not truncate the value (#5198). A single-byte decimal +point is substituted in place and restored afterwards, because the token is +also handed to the SAX interface. A longer one (e.g., the two-byte U+066B of +fa_IR.UTF-8 or ar_EG.UTF-8) is put into a copy of the token instead. + +The token has been validated before, so if the conversion stops early and the +decimal point changed in the meantime, the locale changed between the lookup +and the call, and the conversion is repeated with the new decimal point. If it +did not change, the value strtod parsed up to that point is kept. + +Note that changing the locale in another thread *while* strtod runs is +undefined behavior of the C library, which this function cannot prevent. + +@param[in,out] token the token, with '.' as decimal point +@param[in] decimal_point_position the position of the '.' in @a token, + or std::string::npos if it has none +@param[out] out the parsed value +*/ +template +void parse_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& out) +{ + const bool has_dot = decimal_point_position != std::string::npos; + std::string decimal_point = locale_decimal_point(); + for (;;) + { + char* endptr = nullptr; // NOLINT(misc-const-correctness) + bool complete = false; + if (!has_dot || decimal_point.size() == 1) + { + const bool substitute = has_dot && decimal_point[0] != '.'; + if (substitute) + { + token[decimal_point_position] = static_cast(decimal_point[0]); + } + strtof_global_locale(out, token.data(), &endptr); + if (substitute) + { + token[decimal_point_position] = '.'; + } + complete = endptr == token.data() + token.size(); + } + else + { + std::string copy(token.data(), token.size()); + copy.replace(decimal_point_position, 1, decimal_point); + strtof_global_locale(out, copy.c_str(), &endptr); + complete = endptr == copy.c_str() + copy.size(); + } + + if (JSON_HEDLEY_LIKELY(complete)) + { + return; + } + + // retry only if the locale changed; otherwise, this would loop forever + std::string current_decimal_point = locale_decimal_point(); + if (current_decimal_point == decimal_point) + { + return; + } + decimal_point = std::move(current_decimal_point); + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -9309,18 +9653,6 @@ class lexer : public lexer_base ~lexer() = default; private: - ///////////////////// - // locales - ///////////////////// - - /// return the decimal point of the current locale - static char get_decimal_point() noexcept - { - const auto* loc = localeconv(); - JSON_ASSERT(loc != nullptr); - return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point); - } - ///////////////////// // scan functions ///////////////////// @@ -10128,24 +10460,6 @@ class lexer : public lexer_base } } - JSON_HEDLEY_NON_NULL(2) - static void strtof(float& f, const char* str, char** endptr) noexcept - { - f = std::strtof(str, endptr); - } - - JSON_HEDLEY_NON_NULL(2) - static void strtof(double& f, const char* str, char** endptr) noexcept - { - f = std::strtod(str, endptr); - } - - JSON_HEDLEY_NON_NULL(2) - static void strtof(long double& f, const char* str, char** endptr) noexcept - { - f = std::strtold(str, endptr); - } - /*! @brief scan a number literal @@ -10183,9 +10497,9 @@ class lexer : public lexer_base token_type::parse_error otherwise @note The scanner is independent of the current locale: token_buffer - always holds `.`. Only the std::strtod fallback of convert_number() - depends on the locale, and it looks up the decimal point right - before converting (see convert_float_locale_aware()). + always holds `.`. Only the last-resort std::strtod fallback of + convert_number() depends on the locale, and it looks up the decimal + point right before converting (see parse_float_locale_aware()). */ token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated. { @@ -10653,8 +10967,10 @@ scan_number_done: // this code is reached if we parse a floating-point number or if an // integer conversion above overflowed. Prefer std::from_chars // (Eisel-Lemire, locale-independent, correctly rounded) when available; - // otherwise the exact Clinger fast path (double only); otherwise the - // locale-aware strtof/strtod/strtold. + // otherwise the exact Clinger fast path (double only); otherwise + // strtof/strtod/strtold with the "C" locale where the C library offers + // that; and only as a last resort strtof/strtod/strtold with the + // decimal point of the current locale. if (parse_float_from_chars(num_begin, num_end, value_float)) { return token_type::value_float; @@ -10667,63 +10983,13 @@ scan_number_done: { return token_type::value_float; } - - convert_float_locale_aware(); - return token_type::value_float; - } - - /*! - @brief convert the float in token_buffer with strtof/strtod/strtold - - These functions expect the decimal point of the *current* locale, so it is - looked up right before the conversion instead of once when the lexer is - constructed: a locale change in between (by a parser callback, a SAX - handler, or another thread) must not truncate the value (#5198). The - token has been validated before, so if the conversion stops early and the - decimal point changed in the meantime, the locale changed between the - lookup and the call, and the conversion is repeated with the new decimal - point. If the decimal point did not change, a retry cannot succeed: the - locale's decimal point is not a single character (e.g., the two-byte - U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place. - The value strtod parsed up to that point is kept, as before this change. - - Note that changing the locale in another thread *while* strtod runs is - undefined behavior of the C library, which this function cannot prevent. - */ - void convert_float_locale_aware() - { - const bool has_dot = decimal_point_position != std::string::npos; - char decimal_point = get_decimal_point(); - for (;;) + if (parse_float_c_locale(num_begin, num_end, value_float)) { - const bool substitute = has_dot && decimal_point != '.'; - if (substitute) - { - token_buffer[decimal_point_position] = static_cast(decimal_point); - } - - char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) - strtof(value_float, token_buffer.data(), &endptr); - - if (substitute) - { - // get_string() hands the token to the SAX interface with '.' - token_buffer[decimal_point_position] = '.'; - } - - if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size())) - { - return; - } - - // retry only if the locale changed; otherwise, this would loop forever - const char current_decimal_point = get_decimal_point(); - if (current_decimal_point == decimal_point) - { - return; - } - decimal_point = current_decimal_point; + return token_type::value_float; } + + parse_float_locale_aware(token_buffer, decimal_point_position, value_float); + return token_type::value_float; } /*! @@ -32743,6 +33009,9 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HAS_RANGES #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI + #undef JSON_HAS_FLOAT_FROM_CHARS + #undef JSON_HAS_LONG_DOUBLE_FROM_CHARS + #undef JSON_HAS_C_LOCALE_STRTOD #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index 5d52179d7..3bb8f4398 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -13,7 +13,10 @@ using nlohmann::json; #include // FLT_EVAL_METHOD +#include // signbit #include // strtod +#include // numeric_limits +#include // map #include // stringstream #include // string #include // vector @@ -700,3 +703,140 @@ TEST_CASE("parse_float_fast declines what it cannot convert exactly") CHECK_FALSE(fast("1e23", out)); CHECK_FALSE(fast("1e-23", out)); } + +namespace +{ +template +bool out_of_range_value(const std::string& s, FloatType& out) +{ + return nlohmann::detail::parse_float_out_of_range(s.data(), s.data() + s.size(), out); +} +} // namespace + +TEST_CASE("parse_float_out_of_range derives the value from the token") +{ + // std::from_chars reports numbers out of range without a portable value + // (P4168), so the value is derived from the token + const double inf = std::numeric_limits::infinity(); + double out = 1.0; + + SECTION("overflow") + { + CHECK(out_of_range_value("1e400", out)); + CHECK(out == inf); + CHECK(out_of_range_value("-1E+400", out)); + CHECK(out == -inf); + CHECK(out_of_range_value("123.456e306", out)); + CHECK(out == inf); + CHECK(out_of_range_value("0.001e99999999999999999999", out)); + CHECK(out == inf); + CHECK(out_of_range_value("-1" + std::string(400, '0'), out)); + CHECK(out == -inf); + } + + SECTION("underflow") + { + CHECK(out_of_range_value("1e-400", out)); + CHECK(out == 0.0); + CHECK(!std::signbit(out)); + CHECK(out_of_range_value("-1e-400", out)); + CHECK(out == 0.0); + CHECK(std::signbit(out)); + CHECK(out_of_range_value("0.00012e-321", out)); + CHECK(out == 0.0); + CHECK(out_of_range_value("-1234e-99999999999999999999", out)); + CHECK(std::signbit(out)); + CHECK(out_of_range_value("-0.0", out)); + CHECK(out == 0.0); + CHECK(std::signbit(out)); + } + + SECTION("possibly subnormal") + { + // some implementations report subnormal numbers as out of range; the + // caller then converts them another way + CHECK_FALSE(out_of_range_value("0.0012e-321", out)); + CHECK_FALSE(out_of_range_value("2.5e-320", out)); + CHECK_FALSE(out_of_range_value("-1e-310", out)); + } + + SECTION("float") + { + float f = 1.0f; + CHECK(out_of_range_value("-1e39", f)); + CHECK(f == -std::numeric_limits::infinity()); + CHECK(out_of_range_value("1e-47", f)); + CHECK(f == 0.0f); + CHECK_FALSE(out_of_range_value("1e-46", f)); + CHECK_FALSE(out_of_range_value("1e-40", f)); + } + + SECTION("long double") + { + long double ld = 1.0L; + CHECK(out_of_range_value("1e5000", ld)); + CHECK(ld == std::numeric_limits::infinity()); + CHECK(out_of_range_value("-1e-5000", ld)); + CHECK(ld == 0.0L); + CHECK(std::signbit(ld)); + } +} + +TEST_CASE("floating-point numbers out of range") +{ + // Whichever conversion the platform uses, an overflow throws, and an + // underflow yields a zero with the sign of the number. + using float_json = nlohmann::basic_json; + using long_double_json = nlohmann::basic_json; + + SECTION("double") + { + json _; + CHECK_THROWS_WITH_AS(_ = json::parse("1.5e400"), "[json.exception.out_of_range.406] number overflow parsing '1.5e400'", json::out_of_range&); + CHECK_THROWS_WITH_AS(_ = json::parse("-1.5e400"), "[json.exception.out_of_range.406] number overflow parsing '-1.5e400'", json::out_of_range&); + CHECK_THROWS_WITH_AS(_ = json::parse("1e99999999999999999999"), "[json.exception.out_of_range.406] number overflow parsing '1e99999999999999999999'", json::out_of_range&); + CHECK_THROWS_AS(_ = json::parse("1" + std::string(400, '0')), json::out_of_range&); + + const json zero = json::parse("1.5e-400"); + CHECK(zero == 0.0); + CHECK(!std::signbit(zero.get())); + const json negative_zero = json::parse("-1.5e-400"); + CHECK(negative_zero == 0.0); + CHECK(std::signbit(negative_zero.get())); + CHECK(std::signbit(json::parse("-0.0000000001e-99999999999999999999").get())); + + // around the smallest subnormal number + CHECK(json::parse("1e-324") == 0.0); + CHECK(json::parse("3e-324") == std::numeric_limits::denorm_min()); + CHECK(json::parse("-2.5e-320") == -2.5e-320); + } + + SECTION("float") + { + float_json _; + CHECK_THROWS_WITH_AS(_ = float_json::parse("1e39"), "[json.exception.out_of_range.406] number overflow parsing '1e39'", json::out_of_range&); + CHECK_THROWS_WITH_AS(_ = float_json::parse("-1e39"), "[json.exception.out_of_range.406] number overflow parsing '-1e39'", json::out_of_range&); + + const float_json zero = float_json::parse("1e-50"); + CHECK(zero == 0.0f); + CHECK(!std::signbit(zero.get())); + const float_json negative_zero = float_json::parse("-1e-50"); + CHECK(negative_zero == 0.0f); + CHECK(std::signbit(negative_zero.get())); + CHECK(float_json::parse("1e-45") == std::numeric_limits::denorm_min()); + } + + SECTION("long double") + { + long_double_json _; + CHECK_THROWS_WITH_AS(_ = long_double_json::parse("1e5000"), "[json.exception.out_of_range.406] number overflow parsing '1e5000'", json::out_of_range&); + CHECK_THROWS_WITH_AS(_ = long_double_json::parse("-1e5000"), "[json.exception.out_of_range.406] number overflow parsing '-1e5000'", json::out_of_range&); + + const long_double_json zero = long_double_json::parse("1e-5000"); + CHECK(zero == 0.0L); + CHECK(!std::signbit(zero.get())); + const long_double_json negative_zero = long_double_json::parse("-1e-5000"); + CHECK(negative_zero == 0.0L); + CHECK(std::signbit(negative_zero.get())); + } +} diff --git a/tests/src/unit-locale-cpp.cpp b/tests/src/unit-locale-cpp.cpp index 14f743a66..a8597ef6d 100644 --- a/tests/src/unit-locale-cpp.cpp +++ b/tests/src/unit-locale-cpp.cpp @@ -14,6 +14,8 @@ using nlohmann::json; #include #include +#include +#include #include #include #include @@ -257,10 +259,10 @@ struct LocaleSwitchingSax final: public nlohmann::json_sax TEST_CASE("locale changes between lexer construction and number conversion (#5198)") { - // The numbers are chosen so that the conversion also takes the strtod - // fallback, which honors the locale that is current at conversion time: - // too many significant digits for Clinger's fast path, an underflow that - // std::from_chars rejects, and a plain value. + // The numbers are chosen so that the conversion takes the slower paths: too + // many significant digits for Clinger's fast path, an underflow, and a plain + // value. Without std::from_chars and strtod_l, this is the strtod fallback, + // which honors the locale that is current at conversion time. const std::vector numbers = {"3.14159265358979323846", "1.5e-400", "12.34", "-0.000123456789012345678"}; std::string text = "["; for (const auto& n : numbers) @@ -346,41 +348,113 @@ TEST_CASE("locale changes between lexer construction and number conversion (#519 CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr); } +namespace +{ +// sets LC_NUMERIC to the first installed locale whose decimal point is longer +// than one byte, e.g. U+066B ARABIC DECIMAL SEPARATOR (two bytes in UTF-8) +const char* set_multi_byte_decimal_point_locale() +{ + const std::array names = {{"ar_EG.UTF-8", "ar_SA.UTF-8", "fa_IR.UTF-8", "ps_AF.UTF-8", "ar_EG", "fa_IR"}}; + for (const char* name : names) + { + if (std::setlocale(LC_NUMERIC, name) != nullptr && std::strlen(std::localeconv()->decimal_point) > 1) + { + return name; + } + } + return nullptr; +} +} // namespace + TEST_CASE("locale with a multi-byte decimal point") { - // Some locales use a decimal point that is not a single character, e.g. - // U+066B ARABIC DECIMAL SEPARATOR (two bytes in UTF-8). It cannot be - // substituted in place for '.', so the strtod fallback stops early. The - // conversion must still terminate rather than retry forever. - const std::array names = {{"ar_EG.UTF-8", "ar_SA.UTF-8", "fa_IR.UTF-8", "ps_AF.UTF-8", "ar_EG", "fa_IR"}}; - bool tested = false; + // Such a decimal point cannot be substituted in place for '.'; before + // #5660, the strtod fallback stopped there and returned the integer part. + const char* name = set_multi_byte_decimal_point_locale(); + if (name == nullptr) + { + MESSAGE("no locale with a multi-byte decimal point is usable"); + } + else + { + const std::string locale_name = name; + CAPTURE(locale_name); + + // too many significant digits for Clinger's fast path + CHECK(json::parse("3.141592653589793238462643383279") == 3.141592653589793); + CHECK(json::parse("1.7976931348623157e308") == (std::numeric_limits::max)()); + CHECK(json::accept("3.14159265358979323846")); + + // a subnormal number + CHECK(json::parse("-2.5e-320") == -2.5e-320); + + // out of range + json _; + CHECK_THROWS_WITH_AS(_ = json::parse("1.5e400"), "[json.exception.out_of_range.406] number overflow parsing '1.5e400'", json::out_of_range&); + CHECK(json::parse("1.5e-400") == 0.0); + + // float and long double as number_float_t + using float_json = nlohmann::basic_json; + using long_double_json = nlohmann::basic_json; + CHECK(float_json::parse("1.5") == 1.5f); + CHECK(long_double_json::parse("1.5") == 1.5L); + + // a value Clinger's fast path converts + CHECK(json::parse("12.5") == 12.5); + } + + CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr); +} + +TEST_CASE("conversion with the decimal point of the current locale") +{ + // parse_float_locale_aware() is the last resort for platforms without + // std::from_chars and strtod_l, so it is called directly here + const auto convert = [](std::string token, double & out) + { + nlohmann::detail::parse_float_locale_aware(token, token.find('.'), out); + // the token is also handed to the SAX interface and must keep its '.' + return token; + }; + + std::vector names = {"C", "de_DE", "de_DE.UTF-8"}; + const char* multi_byte = set_multi_byte_decimal_point_locale(); + if (multi_byte != nullptr) + { + names.push_back(multi_byte); + } + for (const char* name : names) { if (std::setlocale(LC_NUMERIC, name) == nullptr) { continue; } - const std::string decimal_point = std::localeconv()->decimal_point; - if (decimal_point.size() < 2) - { - continue; - } - CAPTURE(name); - tested = true; + const std::string locale_name = name; + CAPTURE(locale_name); - // too many significant digits for Clinger's fast path, and an underflow - // that std::from_chars rejects: both reach the strtod fallback - json j; - CHECK_NOTHROW(j = json::parse("[3.14159265358979323846, 1.5e-400, -0.000123456789012345678]")); - CHECK(j.is_array()); - CHECK(json::accept("3.14159265358979323846")); + double d = 0; + CHECK(convert("3.141592653589793238462643383279", d) == "3.141592653589793238462643383279"); + CHECK(d == 3.141592653589793); + CHECK(convert("-2.5e-320", d) == "-2.5e-320"); + CHECK(d == -2.5e-320); + CHECK(convert("12345678901234567890", d) == "12345678901234567890"); + CHECK(d == 12345678901234567890.0); - // a value the locale-independent paths convert is not affected - CHECK(json::parse("12.5") == 12.5); - } - if (!tested) - { - MESSAGE("no locale with a multi-byte decimal point is usable"); + float f = 0; + std::string token = "1.5"; + nlohmann::detail::parse_float_locale_aware(token, 1, f); + CHECK(f == 1.5f); + + long double ld = 0; + nlohmann::detail::parse_float_locale_aware(token, 1, ld); + CHECK(ld == 1.5L); + CHECK(token == "1.5"); + + // the lexer only passes valid tokens; for others, the conversion stops + // early, and the value parsed up to there is kept + CHECK(convert("1.5x", d) == "1.5x"); + CHECK(d == 1.5); } CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);