Convert floats independently of the C locale's decimal point

Under a locale whose decimal point is longer than one byte (e.g. U+066B
in fa_IR.UTF-8 or ar_EG.UTF-8), every float that reached the strtod
fallback was truncated at the decimal point: "3.14159265358979323846"
became 3.0, and "1.5e400" became 1.0 instead of throwing. With libc++
and in C++11/14, that fallback was taken for most floats.

The lexer now converts floats in this order:

1. std::from_chars, now also for float and double with libc++ 20 or
   later, which does not define __cpp_lib_to_chars (on Apple platforms
   only if the deployment target provides it);
2. Clinger's fast path (double only);
3. strtof_l/strtod_l/strtold_l with a "C" locale created once, on
   glibc, Apple platforms, and MSVC;
4. strtof/strtod/strtold with the decimal point of the current locale,
   which now puts a multi-byte decimal point into a copy of the token.

If std::from_chars reports a value out of range, the result is derived
from the token (+-infinity or +-0) instead of calling strtod, because
implementations disagree on the stored value (P4168). Values that may
be subnormal are left to the next step, because libstdc++ before
GCC 13 reports some of them as out of range.

The conversion helpers moved from the lexer to number_parse.hpp, so the
last-resort path can be tested directly.

Fixes #5660.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-30 09:51:50 +02:00
parent 633de8e44b
commit 436bfb1358
8 changed files with 1003 additions and 241 deletions
+140
View File
@@ -13,7 +13,10 @@
using nlohmann::json;
#include <cfloat> // FLT_EVAL_METHOD
#include <cmath> // signbit
#include <cstdlib> // strtod
#include <limits> // numeric_limits
#include <map> // map
#include <sstream> // stringstream
#include <string> // string
#include <vector> // vector
@@ -700,3 +703,140 @@ TEST_CASE("parse_float_fast declines what it cannot convert exactly")
CHECK_FALSE(fast("1e23", out));
CHECK_FALSE(fast("1e-23", out));
}
namespace
{
template<typename FloatType>
bool out_of_range_value(const std::string& s, FloatType& out)
{
return nlohmann::detail::parse_float_out_of_range(s.data(), s.data() + s.size(), out);
}
} // namespace
TEST_CASE("parse_float_out_of_range derives the value from the token")
{
// std::from_chars reports numbers out of range without a portable value
// (P4168), so the value is derived from the token
const double inf = std::numeric_limits<double>::infinity();
double out = 1.0;
SECTION("overflow")
{
CHECK(out_of_range_value("1e400", out));
CHECK(out == inf);
CHECK(out_of_range_value("-1E+400", out));
CHECK(out == -inf);
CHECK(out_of_range_value("123.456e306", out));
CHECK(out == inf);
CHECK(out_of_range_value("0.001e99999999999999999999", out));
CHECK(out == inf);
CHECK(out_of_range_value("-1" + std::string(400, '0'), out));
CHECK(out == -inf);
}
SECTION("underflow")
{
CHECK(out_of_range_value("1e-400", out));
CHECK(out == 0.0);
CHECK(!std::signbit(out));
CHECK(out_of_range_value("-1e-400", out));
CHECK(out == 0.0);
CHECK(std::signbit(out));
CHECK(out_of_range_value("0.00012e-321", out));
CHECK(out == 0.0);
CHECK(out_of_range_value("-1234e-99999999999999999999", out));
CHECK(std::signbit(out));
CHECK(out_of_range_value("-0.0", out));
CHECK(out == 0.0);
CHECK(std::signbit(out));
}
SECTION("possibly subnormal")
{
// some implementations report subnormal numbers as out of range; the
// caller then converts them another way
CHECK_FALSE(out_of_range_value("0.0012e-321", out));
CHECK_FALSE(out_of_range_value("2.5e-320", out));
CHECK_FALSE(out_of_range_value("-1e-310", out));
}
SECTION("float")
{
float f = 1.0f;
CHECK(out_of_range_value("-1e39", f));
CHECK(f == -std::numeric_limits<float>::infinity());
CHECK(out_of_range_value("1e-47", f));
CHECK(f == 0.0f);
CHECK_FALSE(out_of_range_value("1e-46", f));
CHECK_FALSE(out_of_range_value("1e-40", f));
}
SECTION("long double")
{
long double ld = 1.0L;
CHECK(out_of_range_value("1e5000", ld));
CHECK(ld == std::numeric_limits<long double>::infinity());
CHECK(out_of_range_value("-1e-5000", ld));
CHECK(ld == 0.0L);
CHECK(std::signbit(ld));
}
}
TEST_CASE("floating-point numbers out of range")
{
// Whichever conversion the platform uses, an overflow throws, and an
// underflow yields a zero with the sign of the number.
using float_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
using long_double_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, long double>;
SECTION("double")
{
json _;
CHECK_THROWS_WITH_AS(_ = json::parse("1.5e400"), "[json.exception.out_of_range.406] number overflow parsing '1.5e400'", json::out_of_range&);
CHECK_THROWS_WITH_AS(_ = json::parse("-1.5e400"), "[json.exception.out_of_range.406] number overflow parsing '-1.5e400'", json::out_of_range&);
CHECK_THROWS_WITH_AS(_ = json::parse("1e99999999999999999999"), "[json.exception.out_of_range.406] number overflow parsing '1e99999999999999999999'", json::out_of_range&);
CHECK_THROWS_AS(_ = json::parse("1" + std::string(400, '0')), json::out_of_range&);
const json zero = json::parse("1.5e-400");
CHECK(zero == 0.0);
CHECK(!std::signbit(zero.get<double>()));
const json negative_zero = json::parse("-1.5e-400");
CHECK(negative_zero == 0.0);
CHECK(std::signbit(negative_zero.get<double>()));
CHECK(std::signbit(json::parse("-0.0000000001e-99999999999999999999").get<double>()));
// around the smallest subnormal number
CHECK(json::parse("1e-324") == 0.0);
CHECK(json::parse("3e-324") == std::numeric_limits<double>::denorm_min());
CHECK(json::parse("-2.5e-320") == -2.5e-320);
}
SECTION("float")
{
float_json _;
CHECK_THROWS_WITH_AS(_ = float_json::parse("1e39"), "[json.exception.out_of_range.406] number overflow parsing '1e39'", json::out_of_range&);
CHECK_THROWS_WITH_AS(_ = float_json::parse("-1e39"), "[json.exception.out_of_range.406] number overflow parsing '-1e39'", json::out_of_range&);
const float_json zero = float_json::parse("1e-50");
CHECK(zero == 0.0f);
CHECK(!std::signbit(zero.get<float>()));
const float_json negative_zero = float_json::parse("-1e-50");
CHECK(negative_zero == 0.0f);
CHECK(std::signbit(negative_zero.get<float>()));
CHECK(float_json::parse("1e-45") == std::numeric_limits<float>::denorm_min());
}
SECTION("long double")
{
long_double_json _;
CHECK_THROWS_WITH_AS(_ = long_double_json::parse("1e5000"), "[json.exception.out_of_range.406] number overflow parsing '1e5000'", json::out_of_range&);
CHECK_THROWS_WITH_AS(_ = long_double_json::parse("-1e5000"), "[json.exception.out_of_range.406] number overflow parsing '-1e5000'", json::out_of_range&);
const long_double_json zero = long_double_json::parse("1e-5000");
CHECK(zero == 0.0L);
CHECK(!std::signbit(zero.get<long double>()));
const long_double_json negative_zero = long_double_json::parse("-1e-5000");
CHECK(negative_zero == 0.0L);
CHECK(std::signbit(negative_zero.get<long double>()));
}
}
+103 -29
View File
@@ -14,6 +14,8 @@ using nlohmann::json;
#include <array>
#include <clocale>
#include <cstring>
#include <limits>
#include <map>
#include <string>
#include <utility>
@@ -257,10 +259,10 @@ struct LocaleSwitchingSax final: public nlohmann::json_sax<json>
TEST_CASE("locale changes between lexer construction and number conversion (#5198)")
{
// The numbers are chosen so that the conversion also takes the strtod
// fallback, which honors the locale that is current at conversion time:
// too many significant digits for Clinger's fast path, an underflow that
// std::from_chars rejects, and a plain value.
// The numbers are chosen so that the conversion takes the slower paths: too
// many significant digits for Clinger's fast path, an underflow, and a plain
// value. Without std::from_chars and strtod_l, this is the strtod fallback,
// which honors the locale that is current at conversion time.
const std::vector<std::string> numbers = {"3.14159265358979323846", "1.5e-400", "12.34", "-0.000123456789012345678"};
std::string text = "[";
for (const auto& n : numbers)
@@ -346,41 +348,113 @@ TEST_CASE("locale changes between lexer construction and number conversion (#519
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
}
namespace
{
// sets LC_NUMERIC to the first installed locale whose decimal point is longer
// than one byte, e.g. U+066B ARABIC DECIMAL SEPARATOR (two bytes in UTF-8)
const char* set_multi_byte_decimal_point_locale()
{
const std::array<const char*, 6> names = {{"ar_EG.UTF-8", "ar_SA.UTF-8", "fa_IR.UTF-8", "ps_AF.UTF-8", "ar_EG", "fa_IR"}};
for (const char* name : names)
{
if (std::setlocale(LC_NUMERIC, name) != nullptr && std::strlen(std::localeconv()->decimal_point) > 1)
{
return name;
}
}
return nullptr;
}
} // namespace
TEST_CASE("locale with a multi-byte decimal point")
{
// Some locales use a decimal point that is not a single character, e.g.
// U+066B ARABIC DECIMAL SEPARATOR (two bytes in UTF-8). It cannot be
// substituted in place for '.', so the strtod fallback stops early. The
// conversion must still terminate rather than retry forever.
const std::array<const char*, 6> names = {{"ar_EG.UTF-8", "ar_SA.UTF-8", "fa_IR.UTF-8", "ps_AF.UTF-8", "ar_EG", "fa_IR"}};
bool tested = false;
// Such a decimal point cannot be substituted in place for '.'; before
// #5660, the strtod fallback stopped there and returned the integer part.
const char* name = set_multi_byte_decimal_point_locale();
if (name == nullptr)
{
MESSAGE("no locale with a multi-byte decimal point is usable");
}
else
{
const std::string locale_name = name;
CAPTURE(locale_name);
// too many significant digits for Clinger's fast path
CHECK(json::parse("3.141592653589793238462643383279") == 3.141592653589793);
CHECK(json::parse("1.7976931348623157e308") == (std::numeric_limits<double>::max)());
CHECK(json::accept("3.14159265358979323846"));
// a subnormal number
CHECK(json::parse("-2.5e-320") == -2.5e-320);
// out of range
json _;
CHECK_THROWS_WITH_AS(_ = json::parse("1.5e400"), "[json.exception.out_of_range.406] number overflow parsing '1.5e400'", json::out_of_range&);
CHECK(json::parse("1.5e-400") == 0.0);
// float and long double as number_float_t
using float_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
using long_double_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, long double>;
CHECK(float_json::parse("1.5") == 1.5f);
CHECK(long_double_json::parse("1.5") == 1.5L);
// a value Clinger's fast path converts
CHECK(json::parse("12.5") == 12.5);
}
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
}
TEST_CASE("conversion with the decimal point of the current locale")
{
// parse_float_locale_aware() is the last resort for platforms without
// std::from_chars and strtod_l, so it is called directly here
const auto convert = [](std::string token, double & out)
{
nlohmann::detail::parse_float_locale_aware(token, token.find('.'), out);
// the token is also handed to the SAX interface and must keep its '.'
return token;
};
std::vector<const char*> names = {"C", "de_DE", "de_DE.UTF-8"};
const char* multi_byte = set_multi_byte_decimal_point_locale();
if (multi_byte != nullptr)
{
names.push_back(multi_byte);
}
for (const char* name : names)
{
if (std::setlocale(LC_NUMERIC, name) == nullptr)
{
continue;
}
const std::string decimal_point = std::localeconv()->decimal_point;
if (decimal_point.size() < 2)
{
continue;
}
CAPTURE(name);
tested = true;
const std::string locale_name = name;
CAPTURE(locale_name);
// too many significant digits for Clinger's fast path, and an underflow
// that std::from_chars rejects: both reach the strtod fallback
json j;
CHECK_NOTHROW(j = json::parse("[3.14159265358979323846, 1.5e-400, -0.000123456789012345678]"));
CHECK(j.is_array());
CHECK(json::accept("3.14159265358979323846"));
double d = 0;
CHECK(convert("3.141592653589793238462643383279", d) == "3.141592653589793238462643383279");
CHECK(d == 3.141592653589793);
CHECK(convert("-2.5e-320", d) == "-2.5e-320");
CHECK(d == -2.5e-320);
CHECK(convert("12345678901234567890", d) == "12345678901234567890");
CHECK(d == 12345678901234567890.0);
// a value the locale-independent paths convert is not affected
CHECK(json::parse("12.5") == 12.5);
}
if (!tested)
{
MESSAGE("no locale with a multi-byte decimal point is usable");
float f = 0;
std::string token = "1.5";
nlohmann::detail::parse_float_locale_aware(token, 1, f);
CHECK(f == 1.5f);
long double ld = 0;
nlohmann::detail::parse_float_locale_aware(token, 1, ld);
CHECK(ld == 1.5L);
CHECK(token == "1.5");
// the lexer only passes valid tokens; for others, the conversion stops
// early, and the value parsed up to there is kept
CHECK(convert("1.5x", d) == "1.5x");
CHECK(d == 1.5);
}
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);