Compare commits

...
Author SHA1 Message Date
Niels Lohmann f6c115a9a6 Move the float conversion chain out of the lexer
lexer::convert_number() converted float tokens with std::from_chars (when
available), Clinger's fast path, and the locale-aware strtod fallback, all
as lexer members. They are now free functions in number_parse.hpp:

- convert_float_fast(): std::from_chars, then Clinger's fast path, skipped
  when the mantissa has too many significant digits
- convert_float_locale_aware(): strtof/strtod/strtold with the decimal point
  of the current locale, retried when the locale changed (#5198)

so that other code converting JSON number tokens gets the same values. No
change in behavior; the lexer no longer includes <clocale> and <cstdlib>.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 03:49:24 +02:00
Niels Lohmann 33ef25099d Keep external headers as #include when amalgamating
A header that builds on json.hpp (such as the planned json_view.hpp) must
not inline json.hpp: its single-header version would contain a second copy
of the library, and that copy would change with every library change.

The optional config key "external" lists include paths that are kept as
#include directives. Only the first directive per path is kept; repeated
ones are commented out, as the tool already does for inlined headers.
config_json_view.json uses it for json_view.hpp; the existing configs do
not set it, and json.hpp and json_fwd.hpp regenerate byte-identically.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 03:49:24 +02:00
7 changed files with 416 additions and 311 deletions
+4 -151
View File
@@ -9,10 +9,8 @@
#pragma once
#include <array> // array
#include <clocale> // localeconv
#include <cstddef> // size_t
#include <cstdio> // snprintf
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
#include <initializer_list> // initializer_list
#include <string> // char_traits, string
#include <utility> // move
@@ -217,18 +215,6 @@ class lexer : public lexer_base<BasicJsonType>
~lexer() = default;
private:
/////////////////////
// locales
/////////////////////
/// return the decimal point of the current locale
static char get_decimal_point() noexcept
{
const auto* loc = localeconv();
JSON_ASSERT(loc != nullptr);
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
}
/////////////////////
// scan functions
/////////////////////
@@ -1036,24 +1022,6 @@ class lexer : public lexer_base<BasicJsonType>
}
}
JSON_HEDLEY_NON_NULL(2)
static void strtof(float& f, const char* str, char** endptr) noexcept
{
f = std::strtof(str, endptr);
}
JSON_HEDLEY_NON_NULL(2)
static void strtof(double& f, const char* str, char** endptr) noexcept
{
f = std::strtod(str, endptr);
}
JSON_HEDLEY_NON_NULL(2)
static void strtof(long double& f, const char* str, char** endptr) noexcept
{
f = std::strtold(str, endptr);
}
/*!
@brief scan a number literal
@@ -1093,7 +1061,7 @@ class lexer : public lexer_base<BasicJsonType>
@note The scanner is independent of the current locale: token_buffer
always holds `.`. Only the std::strtod fallback of convert_number()
depends on the locale, and it looks up the decimal point right
before converting (see convert_float_locale_aware()).
before converting (see detail::convert_float_locale_aware()).
*/
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
{
@@ -1424,59 +1392,6 @@ scan_number_done:
return token_type::uninitialized;
}
/*!
@brief check whether Clinger's fast path can still succeed for this token
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
more significant digits is at least 10^16 and therefore always exceeds it,
so calling the fast path would walk the token one extra time only to
decline before strtod has to run anyway.
Significant digits are the mantissa's digits from the first nonzero one on;
the sign, the decimal point, leading zeros, and the exponent do not count.
The answer is derived from indices - the digits are not scanned again - so
this stays off the hot path of the number scanners.
@param[in] mantissa_end offset just past the last mantissa byte in
token_buffer
@return false if parse_float_fast() is guaranteed to decline
*/
bool mantissa_fits_clinger(std::size_t mantissa_end) const
{
// 10^16 already exceeds 2^53, so 17 digits can never fit
constexpr std::size_t limit = 17;
const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u;
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
// a leading zero can only be a lone "0", which is not significant
const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u;
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
if (JSON_HEDLEY_LIKELY(digits < limit))
{
return true;
}
// Only a number below 1 can carry further insignificant zeros, and only
// while the count stays at the limit does removing them change the
// answer - so this loop is skipped for all but a few tokens. The
// fraction is located through decimal_point_position rather than by
// searching '.'.
if (lead_zero != 0)
{
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
for (std::size_t i = decimal_point_position + 1;
digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i)
{
--digits;
}
}
return digits < limit;
}
/*!
@brief convert the number text in token_buffer to its value and token type
@@ -1490,7 +1405,7 @@ scan_number_done:
token_buffer (the index of 'e'/'E', or
token_buffer.size() when there is no exponent);
used to skip Clinger's fast path when it cannot
possibly succeed - see mantissa_fits_clinger()
possibly succeed - see detail::mantissa_fits_clinger()
*/
token_type convert_number(token_type number_type, std::size_t mantissa_end)
{
@@ -1563,77 +1478,15 @@ scan_number_done:
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
// otherwise the exact Clinger fast path (double only); otherwise the
// locale-aware strtof/strtod/strtold.
if (parse_float_from_chars(num_begin, num_end, value_float))
{
return token_type::value_float;
}
// Skipping a fast path that cannot succeed is lossless and saves a full
// extra pass over the token's bytes, which otherwise shows up on
// high-precision inputs such as canada.json
if (mantissa_fits_clinger(mantissa_end)
&& parse_float_fast(num_begin, num_end, value_float))
if (convert_float_fast(num_begin, num_end, decimal_point_position, mantissa_end, value_float))
{
return token_type::value_float;
}
convert_float_locale_aware();
convert_float_locale_aware(token_buffer, decimal_point_position, value_float);
return token_type::value_float;
}
/*!
@brief convert the float in token_buffer with strtof/strtod/strtold
These functions expect the decimal point of the *current* locale, so it is
looked up right before the conversion instead of once when the lexer is
constructed: a locale change in between (by a parser callback, a SAX
handler, or another thread) must not truncate the value (#5198). The
token has been validated before, so if the conversion stops early and the
decimal point changed in the meantime, the locale changed between the
lookup and the call, and the conversion is repeated with the new decimal
point. If the decimal point did not change, a retry cannot succeed: the
locale's decimal point is not a single character (e.g., the two-byte
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
The value strtod parsed up to that point is kept, as before this change.
Note that changing the locale in another thread *while* strtod runs is
undefined behavior of the C library, which this function cannot prevent.
*/
void convert_float_locale_aware()
{
const bool has_dot = decimal_point_position != std::string::npos;
char decimal_point = get_decimal_point();
for (;;)
{
const bool substitute = has_dot && decimal_point != '.';
if (substitute)
{
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
}
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
strtof(value_float, token_buffer.data(), &endptr);
if (substitute)
{
// get_string() hands the token to the SAX interface with '.'
token_buffer[decimal_point_position] = '.';
}
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
{
return;
}
// retry only if the locale changed; otherwise, this would loop forever
const char current_decimal_point = get_decimal_point();
if (current_decimal_point == decimal_point)
{
return;
}
decimal_point = current_decimal_point;
}
}
/*!
@brief contiguous fast path for scanning a number
+184 -2
View File
@@ -10,9 +10,12 @@
#include <array> // array
#include <cfloat> // FLT_EVAL_METHOD
#include <clocale> // localeconv
#include <cstddef> // size_t
#include <cstdint> // int64_t, uint64_t
#include <cstdlib> // strtof, strtod, strtold
#include <limits> // numeric_limits
#include <string> // string
#include <nlohmann/detail/macro_scope.hpp>
@@ -29,8 +32,9 @@
// This file contains the value-conversion helpers used by the lexer to turn an
// already-validated number token into a value, without the locale/errno
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
// stays focused on scanning; see lexer::convert_number().
// overhead of std::strtoull/std::strtod where possible. They are free functions
// so the lexer stays focused on scanning (see lexer::convert_number()) and so
// that other parsers of JSON text can convert tokens exactly like it does.
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -293,5 +297,183 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out)
#endif
}
/*!
@brief check whether Clinger's fast path can still succeed for a float token
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
more significant digits is at least 10^16 and therefore always exceeds it,
so calling the fast path would walk the token one extra time only to
decline before strtod has to run anyway.
Significant digits are the mantissa's digits from the first nonzero one on;
the sign, the decimal point, leading zeros, and the exponent do not count.
The answer is derived from indices - the digits are not scanned again - so
this stays off the hot path of the number scanners.
@param[in] token the validated number token ('.' as decimal point)
@param[in] decimal_point_position index of the '.' in @a token, or
std::string::npos if there is none
@param[in] mantissa_end offset just past the last mantissa byte
@return false if parse_float_fast() is guaranteed to decline
*/
inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_position, std::size_t mantissa_end) noexcept
{
// 10^16 already exceeds 2^53, so 17 digits can never fit
constexpr std::size_t limit = 17;
const std::size_t neg = (token[0] == '-') ? 1u : 0u;
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
// a leading zero can only be a lone "0", which is not significant
const std::size_t lead_zero = (token[neg] == '0') ? 1u : 0u;
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
if (JSON_HEDLEY_LIKELY(digits < limit))
{
return true;
}
// Only a number below 1 can carry further insignificant zeros, and only
// while the count stays at the limit does removing them change the
// answer - so this loop is skipped for all but a few tokens. The
// fraction is located through decimal_point_position rather than by
// searching '.'.
if (lead_zero != 0)
{
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
for (std::size_t i = decimal_point_position + 1;
digits >= limit && i < mantissa_end && token[i] == '0'; ++i)
{
--digits;
}
}
return digits < limit;
}
/*!
@brief convert a validated float token without the C library, if possible
Tries std::from_chars (when available) and then Clinger's exact fast path
(double only), skipping the latter when it cannot succeed.
@param[in] first pointer to the first character of the token
@param[in] last pointer past the last character
@param[in] decimal_point_position index of the '.' in the token, or
std::string::npos if there is none
@param[in] mantissa_end offset just past the last mantissa byte (the
index of 'e'/'E', or the token length)
@param[out] value the converted value on success
@return true if the value was converted; false if convert_float_locale_aware()
must convert it
*/
template<typename FloatType>
bool convert_float_fast(const char* first, const char* last, std::size_t decimal_point_position,
std::size_t mantissa_end, FloatType& value) noexcept
{
if (parse_float_from_chars(first, last, value))
{
return true;
}
// Skipping a fast path that cannot succeed is lossless and saves a full
// extra pass over the token's bytes, which otherwise shows up on
// high-precision inputs such as canada.json
return mantissa_fits_clinger(first, decimal_point_position, mantissa_end)
&& parse_float_fast(first, last, value);
}
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
JSON_HEDLEY_NON_NULL(2)
inline void strtof_by_type(float& f, const char* str, char** endptr) noexcept
{
f = std::strtof(str, endptr);
}
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
JSON_HEDLEY_NON_NULL(2)
inline void strtof_by_type(double& f, const char* str, char** endptr) noexcept
{
f = std::strtod(str, endptr);
}
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
JSON_HEDLEY_NON_NULL(2)
inline void strtof_by_type(long double& f, const char* str, char** endptr) noexcept
{
f = std::strtold(str, endptr);
}
/// return the decimal point of the current locale
inline char get_decimal_point() noexcept
{
const auto* loc = localeconv();
JSON_ASSERT(loc != nullptr);
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
}
/*!
@brief convert a validated float token with strtof/strtod/strtold
These functions expect the decimal point of the *current* locale, so it is
looked up right before the conversion instead of once when the lexer is
constructed: a locale change in between (by a parser callback, a SAX
handler, or another thread) must not truncate the value (#5198). The
token has been validated before, so if the conversion stops early and the
decimal point changed in the meantime, the locale changed between the
lookup and the call, and the conversion is repeated with the new decimal
point. If the decimal point did not change, a retry cannot succeed: the
locale's decimal point is not a single character (e.g., the two-byte
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
The value strtod parsed up to that point is kept, as before this change.
Note that changing the locale in another thread *while* strtod runs is
undefined behavior of the C library, which this function cannot prevent.
@param[in,out] token the token with '.' as decimal point; its
decimal point is replaced during the
conversion and restored afterwards
(data() must be NUL-terminated)
@param[in] decimal_point_position index of the '.' in @a token, or
std::string::npos if there is none
@param[out] value the converted value
*/
template<typename StringType, typename FloatType>
void convert_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& value)
{
const bool has_dot = decimal_point_position != std::string::npos;
char decimal_point = get_decimal_point();
for (;;)
{
const bool substitute = has_dot && decimal_point != '.';
if (substitute)
{
token[decimal_point_position] = static_cast<typename StringType::value_type>(decimal_point);
}
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
strtof_by_type(value, token.data(), &endptr);
if (substitute)
{
// the caller hands the token on (e.g. to the SAX interface) with '.'
token[decimal_point_position] = '.';
}
if (JSON_HEDLEY_LIKELY(endptr == token.data() + token.size()))
{
return;
}
// retry only if the locale changed; otherwise, this would loop forever
const char current_decimal_point = get_decimal_point();
if (current_decimal_point == decimal_point)
{
return;
}
decimal_point = current_decimal_point;
}
}
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
+188 -153
View File
@@ -8472,10 +8472,8 @@ NLOHMANN_JSON_NAMESPACE_END
#include <array> // array
#include <clocale> // localeconv
#include <cstddef> // size_t
#include <cstdio> // snprintf
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
#include <initializer_list> // initializer_list
#include <string> // char_traits, string
#include <utility> // move
@@ -8496,9 +8494,12 @@ NLOHMANN_JSON_NAMESPACE_END
#include <array> // array
#include <cfloat> // FLT_EVAL_METHOD
#include <clocale> // localeconv
#include <cstddef> // size_t
#include <cstdint> // int64_t, uint64_t
#include <cstdlib> // strtof, strtod, strtold
#include <limits> // numeric_limits
#include <string> // string
// #include <nlohmann/detail/macro_scope.hpp>
@@ -8516,8 +8517,9 @@ NLOHMANN_JSON_NAMESPACE_END
// This file contains the value-conversion helpers used by the lexer to turn an
// already-validated number token into a value, without the locale/errno
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
// stays focused on scanning; see lexer::convert_number().
// overhead of std::strtoull/std::strtod where possible. They are free functions
// so the lexer stays focused on scanning (see lexer::convert_number()) and so
// that other parsers of JSON text can convert tokens exactly like it does.
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -8780,6 +8782,184 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out)
#endif
}
/*!
@brief check whether Clinger's fast path can still succeed for a float token
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
more significant digits is at least 10^16 and therefore always exceeds it,
so calling the fast path would walk the token one extra time only to
decline before strtod has to run anyway.
Significant digits are the mantissa's digits from the first nonzero one on;
the sign, the decimal point, leading zeros, and the exponent do not count.
The answer is derived from indices - the digits are not scanned again - so
this stays off the hot path of the number scanners.
@param[in] token the validated number token ('.' as decimal point)
@param[in] decimal_point_position index of the '.' in @a token, or
std::string::npos if there is none
@param[in] mantissa_end offset just past the last mantissa byte
@return false if parse_float_fast() is guaranteed to decline
*/
inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_position, std::size_t mantissa_end) noexcept
{
// 10^16 already exceeds 2^53, so 17 digits can never fit
constexpr std::size_t limit = 17;
const std::size_t neg = (token[0] == '-') ? 1u : 0u;
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
// a leading zero can only be a lone "0", which is not significant
const std::size_t lead_zero = (token[neg] == '0') ? 1u : 0u;
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
if (JSON_HEDLEY_LIKELY(digits < limit))
{
return true;
}
// Only a number below 1 can carry further insignificant zeros, and only
// while the count stays at the limit does removing them change the
// answer - so this loop is skipped for all but a few tokens. The
// fraction is located through decimal_point_position rather than by
// searching '.'.
if (lead_zero != 0)
{
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
for (std::size_t i = decimal_point_position + 1;
digits >= limit && i < mantissa_end && token[i] == '0'; ++i)
{
--digits;
}
}
return digits < limit;
}
/*!
@brief convert a validated float token without the C library, if possible
Tries std::from_chars (when available) and then Clinger's exact fast path
(double only), skipping the latter when it cannot succeed.
@param[in] first pointer to the first character of the token
@param[in] last pointer past the last character
@param[in] decimal_point_position index of the '.' in the token, or
std::string::npos if there is none
@param[in] mantissa_end offset just past the last mantissa byte (the
index of 'e'/'E', or the token length)
@param[out] value the converted value on success
@return true if the value was converted; false if convert_float_locale_aware()
must convert it
*/
template<typename FloatType>
bool convert_float_fast(const char* first, const char* last, std::size_t decimal_point_position,
std::size_t mantissa_end, FloatType& value) noexcept
{
if (parse_float_from_chars(first, last, value))
{
return true;
}
// Skipping a fast path that cannot succeed is lossless and saves a full
// extra pass over the token's bytes, which otherwise shows up on
// high-precision inputs such as canada.json
return mantissa_fits_clinger(first, decimal_point_position, mantissa_end)
&& parse_float_fast(first, last, value);
}
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
JSON_HEDLEY_NON_NULL(2)
inline void strtof_by_type(float& f, const char* str, char** endptr) noexcept
{
f = std::strtof(str, endptr);
}
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
JSON_HEDLEY_NON_NULL(2)
inline void strtof_by_type(double& f, const char* str, char** endptr) noexcept
{
f = std::strtod(str, endptr);
}
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
JSON_HEDLEY_NON_NULL(2)
inline void strtof_by_type(long double& f, const char* str, char** endptr) noexcept
{
f = std::strtold(str, endptr);
}
/// return the decimal point of the current locale
inline char get_decimal_point() noexcept
{
const auto* loc = localeconv();
JSON_ASSERT(loc != nullptr);
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
}
/*!
@brief convert a validated float token with strtof/strtod/strtold
These functions expect the decimal point of the *current* locale, so it is
looked up right before the conversion instead of once when the lexer is
constructed: a locale change in between (by a parser callback, a SAX
handler, or another thread) must not truncate the value (#5198). The
token has been validated before, so if the conversion stops early and the
decimal point changed in the meantime, the locale changed between the
lookup and the call, and the conversion is repeated with the new decimal
point. If the decimal point did not change, a retry cannot succeed: the
locale's decimal point is not a single character (e.g., the two-byte
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
The value strtod parsed up to that point is kept, as before this change.
Note that changing the locale in another thread *while* strtod runs is
undefined behavior of the C library, which this function cannot prevent.
@param[in,out] token the token with '.' as decimal point; its
decimal point is replaced during the
conversion and restored afterwards
(data() must be NUL-terminated)
@param[in] decimal_point_position index of the '.' in @a token, or
std::string::npos if there is none
@param[out] value the converted value
*/
template<typename StringType, typename FloatType>
void convert_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& value)
{
const bool has_dot = decimal_point_position != std::string::npos;
char decimal_point = get_decimal_point();
for (;;)
{
const bool substitute = has_dot && decimal_point != '.';
if (substitute)
{
token[decimal_point_position] = static_cast<typename StringType::value_type>(decimal_point);
}
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
strtof_by_type(value, token.data(), &endptr);
if (substitute)
{
// the caller hands the token on (e.g. to the SAX interface) with '.'
token[decimal_point_position] = '.';
}
if (JSON_HEDLEY_LIKELY(endptr == token.data() + token.size()))
{
return;
}
// retry only if the locale changed; otherwise, this would loop forever
const char current_decimal_point = get_decimal_point();
if (current_decimal_point == decimal_point)
{
return;
}
decimal_point = current_decimal_point;
}
}
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
@@ -9309,18 +9489,6 @@ class lexer : public lexer_base<BasicJsonType>
~lexer() = default;
private:
/////////////////////
// locales
/////////////////////
/// return the decimal point of the current locale
static char get_decimal_point() noexcept
{
const auto* loc = localeconv();
JSON_ASSERT(loc != nullptr);
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
}
/////////////////////
// scan functions
/////////////////////
@@ -10128,24 +10296,6 @@ class lexer : public lexer_base<BasicJsonType>
}
}
JSON_HEDLEY_NON_NULL(2)
static void strtof(float& f, const char* str, char** endptr) noexcept
{
f = std::strtof(str, endptr);
}
JSON_HEDLEY_NON_NULL(2)
static void strtof(double& f, const char* str, char** endptr) noexcept
{
f = std::strtod(str, endptr);
}
JSON_HEDLEY_NON_NULL(2)
static void strtof(long double& f, const char* str, char** endptr) noexcept
{
f = std::strtold(str, endptr);
}
/*!
@brief scan a number literal
@@ -10185,7 +10335,7 @@ class lexer : public lexer_base<BasicJsonType>
@note The scanner is independent of the current locale: token_buffer
always holds `.`. Only the std::strtod fallback of convert_number()
depends on the locale, and it looks up the decimal point right
before converting (see convert_float_locale_aware()).
before converting (see detail::convert_float_locale_aware()).
*/
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
{
@@ -10516,59 +10666,6 @@ scan_number_done:
return token_type::uninitialized;
}
/*!
@brief check whether Clinger's fast path can still succeed for this token
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
more significant digits is at least 10^16 and therefore always exceeds it,
so calling the fast path would walk the token one extra time only to
decline before strtod has to run anyway.
Significant digits are the mantissa's digits from the first nonzero one on;
the sign, the decimal point, leading zeros, and the exponent do not count.
The answer is derived from indices - the digits are not scanned again - so
this stays off the hot path of the number scanners.
@param[in] mantissa_end offset just past the last mantissa byte in
token_buffer
@return false if parse_float_fast() is guaranteed to decline
*/
bool mantissa_fits_clinger(std::size_t mantissa_end) const
{
// 10^16 already exceeds 2^53, so 17 digits can never fit
constexpr std::size_t limit = 17;
const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u;
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
// a leading zero can only be a lone "0", which is not significant
const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u;
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
if (JSON_HEDLEY_LIKELY(digits < limit))
{
return true;
}
// Only a number below 1 can carry further insignificant zeros, and only
// while the count stays at the limit does removing them change the
// answer - so this loop is skipped for all but a few tokens. The
// fraction is located through decimal_point_position rather than by
// searching '.'.
if (lead_zero != 0)
{
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
for (std::size_t i = decimal_point_position + 1;
digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i)
{
--digits;
}
}
return digits < limit;
}
/*!
@brief convert the number text in token_buffer to its value and token type
@@ -10582,7 +10679,7 @@ scan_number_done:
token_buffer (the index of 'e'/'E', or
token_buffer.size() when there is no exponent);
used to skip Clinger's fast path when it cannot
possibly succeed - see mantissa_fits_clinger()
possibly succeed - see detail::mantissa_fits_clinger()
*/
token_type convert_number(token_type number_type, std::size_t mantissa_end)
{
@@ -10655,77 +10752,15 @@ scan_number_done:
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
// otherwise the exact Clinger fast path (double only); otherwise the
// locale-aware strtof/strtod/strtold.
if (parse_float_from_chars(num_begin, num_end, value_float))
{
return token_type::value_float;
}
// Skipping a fast path that cannot succeed is lossless and saves a full
// extra pass over the token's bytes, which otherwise shows up on
// high-precision inputs such as canada.json
if (mantissa_fits_clinger(mantissa_end)
&& parse_float_fast(num_begin, num_end, value_float))
if (convert_float_fast(num_begin, num_end, decimal_point_position, mantissa_end, value_float))
{
return token_type::value_float;
}
convert_float_locale_aware();
convert_float_locale_aware(token_buffer, decimal_point_position, value_float);
return token_type::value_float;
}
/*!
@brief convert the float in token_buffer with strtof/strtod/strtold
These functions expect the decimal point of the *current* locale, so it is
looked up right before the conversion instead of once when the lexer is
constructed: a locale change in between (by a parser callback, a SAX
handler, or another thread) must not truncate the value (#5198). The
token has been validated before, so if the conversion stops early and the
decimal point changed in the meantime, the locale changed between the
lookup and the call, and the conversion is repeated with the new decimal
point. If the decimal point did not change, a retry cannot succeed: the
locale's decimal point is not a single character (e.g., the two-byte
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
The value strtod parsed up to that point is kept, as before this change.
Note that changing the locale in another thread *while* strtod runs is
undefined behavior of the C library, which this function cannot prevent.
*/
void convert_float_locale_aware()
{
const bool has_dot = decimal_point_position != std::string::npos;
char decimal_point = get_decimal_point();
for (;;)
{
const bool substitute = has_dot && decimal_point != '.';
if (substitute)
{
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
}
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
strtof(value_float, token_buffer.data(), &endptr);
if (substitute)
{
// get_string() hands the token to the SAX interface with '.'
token_buffer[decimal_point_position] = '.';
}
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
{
return;
}
// retry only if the locale changed; otherwise, this would loop forever
const char current_decimal_point = get_decimal_point();
if (current_decimal_point == decimal_point)
{
return;
}
decimal_point = current_decimal_point;
}
}
/*!
@brief contiguous fast path for scanning a number
+3
View File
@@ -8,3 +8,6 @@ The following changes have been made to the code with respect to <https://github
- membership check
- made function from `_is_within`
- removed unused variable `actual_path`
- Added the optional config key `external`: include paths listed there are kept as
`#include` directives instead of being inlined (the first directive per path; the
repeated ones are commented out).
+5
View File
@@ -57,6 +57,11 @@ Python v.2.7.0 or higher is required.
amalgamation. Have a look at `test/source.c.json` and `test/include.h.json`
to see two examples.
The optional `external` list names include paths that are kept as `#include`
directives instead of being inlined, e.g. `["nlohmann/json.hpp"]` for a header
that includes another amalgamated header. Only the first directive for each
of these paths is kept; the repeated ones are commented out.
* The `-s, --source` option should specify the path to the source directory.
This is useful for supporting separate source and build directories.
+23 -5
View File
@@ -62,6 +62,10 @@ class Amalgamation(object):
return None
def __init__(self, args):
# include paths that are kept as #include directives instead of
# being inlined (e.g. a header amalgamated on its own)
self.external = []
self.included_external = []
with open(args.config, 'r') as f:
config = json.loads(f.read())
for key in config:
@@ -220,11 +224,14 @@ class TranslationUnit(object):
while include_match:
if not _is_within(include_match, skippable_contexts):
include_path = include_match.group("path")
search_same_dir = include_match.group(1) == '"'
found_included_path = self.amalgamation.find_included_file(
include_path, self.file_dir if search_same_dir else None)
if found_included_path:
includes.append((include_match, found_included_path))
if include_path in self.amalgamation.external:
includes.append((include_match, None))
else:
search_same_dir = include_match.group(1) == '"'
found_included_path = self.amalgamation.find_included_file(
include_path, self.file_dir if search_same_dir else None)
if found_included_path:
includes.append((include_match, found_included_path))
include_match = self.include_pattern.search(self.content,
include_match.end())
@@ -235,6 +242,17 @@ class TranslationUnit(object):
for include in includes:
include_match, found_included_path = include
tmp_content += self.content[prev_end:include_match.start()]
if found_included_path is None:
# an external header: keep the first directive and comment
# out the repeated ones
include_path = include_match.group("path")
if include_path in self.amalgamation.included_external:
tmp_content += "// {0}".format(include_match.group(0))
else:
self.amalgamation.included_external.append(include_path)
tmp_content += include_match.group(0)
prev_end = include_match.end()
continue
tmp_content += "// {0}\n".format(include_match.group(0))
if found_included_path not in self.amalgamation.included_files:
t = TranslationUnit(found_included_path, self.amalgamation, False)
+9
View File
@@ -0,0 +1,9 @@
{
"project": "JSON for Modern C++",
"target": "single_include/nlohmann/json_view.hpp",
"sources": [
"include/nlohmann/json_view.hpp"
],
"include_paths": ["include"],
"external": ["nlohmann/json.hpp"]
}