diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index 4f85f8132..00a964a17 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -1587,9 +1587,13 @@ scan_number_done: looked up right before the conversion instead of once when the lexer is constructed: a locale change in between (by a parser callback, a SAX handler, or another thread) must not truncate the value (#5198). The - token has been validated before, so if the conversion still stops early, - the locale changed between the lookup and the call, and the conversion is - repeated with the new decimal point. + token has been validated before, so if the conversion stops early and the + decimal point changed in the meantime, the locale changed between the + lookup and the call, and the conversion is repeated with the new decimal + point. If the decimal point did not change, a retry cannot succeed: the + locale's decimal point is not a single character (e.g., the two-byte + U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place. + The value strtod parsed up to that point is kept, as before this change. Note that changing the locale in another thread *while* strtod runs is undefined behavior of the C library, which this function cannot prevent. @@ -1597,9 +1601,9 @@ scan_number_done: void convert_float_locale_aware() { const bool has_dot = decimal_point_position != std::string::npos; + char decimal_point = get_decimal_point(); for (;;) { - const char decimal_point = get_decimal_point(); const bool substitute = has_dot && decimal_point != '.'; if (substitute) { @@ -1619,6 +1623,14 @@ scan_number_done: { return; } + + // retry only if the locale changed; otherwise, this would loop forever + const char current_decimal_point = get_decimal_point(); + if (current_decimal_point == decimal_point) + { + return; + } + decimal_point = current_decimal_point; } } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4975ac2a0..5b7c4ec19 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -10679,9 +10679,13 @@ scan_number_done: looked up right before the conversion instead of once when the lexer is constructed: a locale change in between (by a parser callback, a SAX handler, or another thread) must not truncate the value (#5198). The - token has been validated before, so if the conversion still stops early, - the locale changed between the lookup and the call, and the conversion is - repeated with the new decimal point. + token has been validated before, so if the conversion stops early and the + decimal point changed in the meantime, the locale changed between the + lookup and the call, and the conversion is repeated with the new decimal + point. If the decimal point did not change, a retry cannot succeed: the + locale's decimal point is not a single character (e.g., the two-byte + U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place. + The value strtod parsed up to that point is kept, as before this change. Note that changing the locale in another thread *while* strtod runs is undefined behavior of the C library, which this function cannot prevent. @@ -10689,9 +10693,9 @@ scan_number_done: void convert_float_locale_aware() { const bool has_dot = decimal_point_position != std::string::npos; + char decimal_point = get_decimal_point(); for (;;) { - const char decimal_point = get_decimal_point(); const bool substitute = has_dot && decimal_point != '.'; if (substitute) { @@ -10711,6 +10715,14 @@ scan_number_done: { return; } + + // retry only if the locale changed; otherwise, this would loop forever + const char current_decimal_point = get_decimal_point(); + if (current_decimal_point == decimal_point) + { + return; + } + decimal_point = current_decimal_point; } } diff --git a/tests/src/unit-locale-cpp.cpp b/tests/src/unit-locale-cpp.cpp index ef52b9ba4..b83e63bde 100644 --- a/tests/src/unit-locale-cpp.cpp +++ b/tests/src/unit-locale-cpp.cpp @@ -345,3 +345,43 @@ TEST_CASE("locale changes between lexer construction and number conversion (#519 std::setlocale(LC_NUMERIC, "C"); } + +TEST_CASE("locale with a multi-byte decimal point") +{ + // Some locales use a decimal point that is not a single character, e.g. + // U+066B ARABIC DECIMAL SEPARATOR (two bytes in UTF-8). It cannot be + // substituted in place for '.', so the strtod fallback stops early. The + // conversion must still terminate rather than retry forever. + const std::array names = {{"ar_EG.UTF-8", "ar_SA.UTF-8", "fa_IR.UTF-8", "ps_AF.UTF-8", "ar_EG", "fa_IR"}}; + bool tested = false; + for (const char* name : names) + { + if (std::setlocale(LC_NUMERIC, name) == nullptr) + { + continue; + } + const std::string decimal_point = std::localeconv()->decimal_point; + if (decimal_point.size() < 2) + { + continue; + } + CAPTURE(name); + tested = true; + + // too many significant digits for Clinger's fast path, and an underflow + // that std::from_chars rejects: both reach the strtod fallback + json j; + CHECK_NOTHROW(j = json::parse("[3.14159265358979323846, 1.5e-400, -0.000123456789012345678]")); + CHECK(j.is_array()); + CHECK(json::accept("3.14159265358979323846")); + + // a value the locale-independent paths convert is not affected + CHECK(json::parse("12.5") == 12.5); + } + if (!tested) + { + MESSAGE("no locale with a multi-byte decimal point is usable"); + } + + std::setlocale(LC_NUMERIC, "C"); +}