mirror of
https://github.com/nlohmann/json.git
synced 2026-08-02 15:32:17 +00:00
lexer: use std::from_chars when available, fall back to strto* otherwise
`std::from_chars` is locale-independent and thus avoids the TOCTOU issue #5198 when it's available. This commit introduces a `from_chars` utility in the `detail` namespace which merely wraps `std::from_chars` if it is available and emulates its API and behavior using the `strto*` family of functions when it is not. As of this commit, the `strto*`-based fallback is still locale-dependent and only matches `std::from_chars` when the "C" locale is used. Availability of `std::from_chars` for floating point numbers is varied. For instance, as of libc++ 22, `std::from_chars` for `long double` isn't supported and, consequently, the feature-test macro `__cpp_lib_to_chars` is not set, while the overloads for `float` and `double` are available. To avoid complicating matters further, the feature-test macro is taken as the authoritative indicator for (full) `std::from_chars` support and the fallback is used otherwise. If `std::from_chars` is available, our wrapper additionally tweaks its output in the case of floating-point out-of-range results. The behavior in this case is poorly (under-)specified and existing implementations do not agree. The wrapper's tweaks are intended to ensure consistent behavior across all toolchains in this edge case. In practice, this should only take effect when parsing out-of-range floats on libstdc++. The unit tests for the `from_chars` helper are intentionally compiled in both C++11 and C++17 modes. This ensures that the test expectations align with the actual behavior we're trying to model. The tests are not aiming to cover all of floating-point parsing comprehensively, as `std::from_chars` is specified to behave mostly the same as `strto*` and instead focuses on the edge cases where the two differ: - leading whitespace is not tolerated - `+` sign is not allowed (outside of the exponent) - out-parameter is left unchanged in case of `invalid_argument` and integral `result_out_of_range` errors Signed-off-by: Jonas Greitemann <jgreitemann@gmail.com>
This commit is contained in:
@@ -9,15 +9,15 @@
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
#include <system_error> // errc
|
||||
#include <utility> // move
|
||||
#include <vector> // vector
|
||||
|
||||
#include <nlohmann/detail/conversions/from_chars.hpp>
|
||||
#include <nlohmann/detail/input/input_adapters.hpp>
|
||||
#include <nlohmann/detail/input/position_t.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
@@ -152,7 +152,7 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
||||
: ia(std::move(adapter))
|
||||
, ignore_comments(ignore_comments_)
|
||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||
, decimal_point_char(static_cast<char_int_type>(from_chars_traits<number_float_t>::get_decimal_point()))
|
||||
{}
|
||||
|
||||
// deleted because of pointer members
|
||||
@@ -163,19 +163,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
~lexer() = default;
|
||||
|
||||
private:
|
||||
/////////////////////
|
||||
// locales
|
||||
/////////////////////
|
||||
|
||||
/// return the locale-dependent decimal point
|
||||
JSON_HEDLEY_PURE
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// scan functions
|
||||
/////////////////////
|
||||
@@ -941,24 +928,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan a number literal
|
||||
|
||||
@@ -1279,19 +1248,18 @@ scan_number_done:
|
||||
// we are done scanning a number)
|
||||
unget();
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
errno = 0;
|
||||
|
||||
// try to parse integers first and fall back to floats
|
||||
if (number_type == token_type::value_unsigned)
|
||||
{
|
||||
const auto x = std::strtoull(token_buffer.data(), &endptr, 10);
|
||||
unsigned long long x{}; // NOLINT(runtime/int)
|
||||
const auto res = ::nlohmann::detail::from_chars(token_buffer.data(), token_buffer.data() + token_buffer.size(), x);
|
||||
|
||||
// we checked the number format before
|
||||
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
|
||||
JSON_ASSERT(res.ptr == token_buffer.data() + token_buffer.size());
|
||||
|
||||
if (errno != ERANGE)
|
||||
if (res.ec != std::errc::result_out_of_range)
|
||||
{
|
||||
JSON_ASSERT(res.ec == std::errc{});
|
||||
value_unsigned = static_cast<number_unsigned_t>(x);
|
||||
if (value_unsigned == x)
|
||||
{
|
||||
@@ -1301,13 +1269,15 @@ scan_number_done:
|
||||
}
|
||||
else if (number_type == token_type::value_integer)
|
||||
{
|
||||
const auto x = std::strtoll(token_buffer.data(), &endptr, 10);
|
||||
long long x{}; // NOLINT(runtime/int)
|
||||
const auto res = ::nlohmann::detail::from_chars(token_buffer.data(), token_buffer.data() + token_buffer.size(), x);
|
||||
|
||||
// we checked the number format before
|
||||
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
|
||||
JSON_ASSERT(res.ptr == token_buffer.data() + token_buffer.size());
|
||||
|
||||
if (errno != ERANGE)
|
||||
if (res.ec != std::errc::result_out_of_range)
|
||||
{
|
||||
JSON_ASSERT(res.ec == std::errc{});
|
||||
value_integer = static_cast<number_integer_t>(x);
|
||||
if (value_integer == x)
|
||||
{
|
||||
@@ -1318,10 +1288,11 @@ scan_number_done:
|
||||
|
||||
// this code is reached if we parse a floating-point number or if an
|
||||
// integer conversion above failed
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
const auto res = ::nlohmann::detail::from_chars(token_buffer.data(), token_buffer.data() + token_buffer.size(), value_float);
|
||||
|
||||
// we checked the number format before
|
||||
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
|
||||
JSON_ASSERT(res.ptr == token_buffer.data() + token_buffer.size());
|
||||
JSON_ASSERT(res.ec != std::errc::invalid_argument);
|
||||
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user