mirror of
https://github.com/nlohmann/json.git
synced 2026-09-30 11:40:30 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f6c115a9a6 | ||
|
|
33ef25099d |
@@ -132,14 +132,8 @@ The library uses the following mapping from JSON values types to BJData types ac
|
||||
parsed back as a regular array,
|
||||
- every entry of `"_ArraySize_"` is a positive integer, and their product is representable as a `std::size_t`,
|
||||
- `"_ArrayData_"` is an array holding exactly that many elements, and
|
||||
- every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"`: for the integer types, a
|
||||
value that fits the named width; for `double`, any value; for `single`, a value that survives narrowing to
|
||||
`float` and back without change (for instance, `0.1` does not, since it is not exactly representable as
|
||||
`float`).
|
||||
|
||||
An annotated object is always read back with its keys in the order shown above, `"_ArrayType_"`, `"_ArraySize_"`,
|
||||
`"_ArrayData_"`, regardless of the order the ND-array's header stores them in on the wire. This matters for
|
||||
`ordered_json`, whose comparison takes key order into account.
|
||||
- every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"` (a floating-point number for
|
||||
`single` and `double`, an integer otherwise).
|
||||
|
||||
The current version of this library does not yet support automatic detection of and conversion from a nested JSON
|
||||
array input to a BJData ND-array.
|
||||
|
||||
@@ -2728,15 +2728,10 @@ class binary_reader
|
||||
is_ndarray can only return `true` when its initial value
|
||||
is `false`
|
||||
@param[in] prefix type marker if already read, otherwise set to 0
|
||||
@param[in] ndarray_dtype the element type marker of the enclosing bjdata ndarray if
|
||||
already known (it precedes the dimension vector read here),
|
||||
otherwise 0; used to emit the "_ArrayType_" annotation key
|
||||
before "_ArraySize_" if a dimension vector turns out to
|
||||
describe an ndarray
|
||||
|
||||
@return whether size determination completed
|
||||
*/
|
||||
bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0, char_int_type ndarray_dtype = 0)
|
||||
bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0)
|
||||
{
|
||||
if (prefix == 0)
|
||||
{
|
||||
@@ -2906,37 +2901,8 @@ class binary_reader
|
||||
}
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// the element type precedes the dimension vector (see get_ubjson_size_type)
|
||||
// and is passed down as ndarray_dtype; emit it here so the annotation keys
|
||||
// follow the documented _ArrayType_, _ArraySize_, _ArrayData_ order
|
||||
if (ndarray_dtype != 0)
|
||||
{
|
||||
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), ndarray_dtype, [](const bjd_type & p, char_int_type t)
|
||||
{
|
||||
return p.first < t;
|
||||
});
|
||||
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != ndarray_dtype))
|
||||
{
|
||||
auto last_token = get_token_string();
|
||||
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
|
||||
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
|
||||
}
|
||||
|
||||
string_t type_key = "_ArrayType_";
|
||||
string_t type = it->second; // sax->string() takes a reference
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(type_key) || !sax->string(type)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
string_t key = "_ArraySize_";
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(dim.size())))
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3) || !sax->key(key) || !sax->start_array(dim.size())))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -3037,7 +3003,7 @@ class binary_reader
|
||||
exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr));
|
||||
}
|
||||
|
||||
const bool is_error = get_ubjson_size_value(result.first, is_ndarray, 0, result.second);
|
||||
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
|
||||
// an ndarray was read here only if the flag flipped; when it was
|
||||
// seeded true, get_ubjson_size_value() already rejected the nested
|
||||
// dimension vector
|
||||
@@ -3273,17 +3239,30 @@ class binary_reader
|
||||
if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0)
|
||||
{
|
||||
size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker
|
||||
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t)
|
||||
{
|
||||
return p.first < t;
|
||||
});
|
||||
string_t key = "_ArrayType_";
|
||||
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second))
|
||||
{
|
||||
auto last_token = get_token_string();
|
||||
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
|
||||
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
|
||||
}
|
||||
|
||||
string_t type = it->second; // sax->string() takes a reference
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// the "_ArrayType_" and "_ArraySize_" annotation keys were already emitted by
|
||||
// get_ubjson_size_value() (the type marker is known before the dimension vector
|
||||
// that determines size_and_type.first is read, so it is emitted first there to
|
||||
// match the documented _ArrayType_, _ArraySize_, _ArrayData_ key order)
|
||||
if (size_and_type.second == 'C' || size_and_type.second == 'B')
|
||||
{
|
||||
size_and_type.second = 'U';
|
||||
}
|
||||
|
||||
string_t key = "_ArrayData_";
|
||||
key = "_ArrayData_";
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) ))
|
||||
{
|
||||
return false;
|
||||
|
||||
@@ -9,10 +9,8 @@
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
#include <utility> // move
|
||||
@@ -217,18 +215,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
~lexer() = default;
|
||||
|
||||
private:
|
||||
/////////////////////
|
||||
// locales
|
||||
/////////////////////
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// scan functions
|
||||
/////////////////////
|
||||
@@ -1036,24 +1022,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan a number literal
|
||||
|
||||
@@ -1093,7 +1061,7 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
@note The scanner is independent of the current locale: token_buffer
|
||||
always holds `.`. Only the std::strtod fallback of convert_number()
|
||||
depends on the locale, and it looks up the decimal point right
|
||||
before converting (see convert_float_locale_aware()).
|
||||
before converting (see detail::convert_float_locale_aware()).
|
||||
*/
|
||||
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
|
||||
{
|
||||
@@ -1424,59 +1392,6 @@ scan_number_done:
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for this token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] mantissa_end offset just past the last mantissa byte in
|
||||
token_buffer
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
bool mantissa_fits_clinger(std::size_t mantissa_end) const
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. The
|
||||
// fraction is located through decimal_point_position rather than by
|
||||
// searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the number text in token_buffer to its value and token type
|
||||
|
||||
@@ -1490,7 +1405,7 @@ scan_number_done:
|
||||
token_buffer (the index of 'e'/'E', or
|
||||
token_buffer.size() when there is no exponent);
|
||||
used to skip Clinger's fast path when it cannot
|
||||
possibly succeed - see mantissa_fits_clinger()
|
||||
possibly succeed - see detail::mantissa_fits_clinger()
|
||||
*/
|
||||
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||
{
|
||||
@@ -1563,77 +1478,15 @@ scan_number_done:
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod/strtold.
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
if (mantissa_fits_clinger(mantissa_end)
|
||||
&& parse_float_fast(num_begin, num_end, value_float))
|
||||
if (convert_float_fast(num_begin, num_end, decimal_point_position, mantissa_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
convert_float_locale_aware();
|
||||
convert_float_locale_aware(token_buffer, decimal_point_position, value_float);
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the float in token_buffer with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
*/
|
||||
void convert_float_locale_aware()
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// get_string() hands the token to the SAX interface with '.'
|
||||
token_buffer[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief contiguous fast path for scanning a number
|
||||
|
||||
|
||||
@@ -10,9 +10,12 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <cstdlib> // strtof, strtod, strtold
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
@@ -29,8 +32,9 @@
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
// overhead of std::strtoull/std::strtod where possible. They are free functions
|
||||
// so the lexer stays focused on scanning (see lexer::convert_number()) and so
|
||||
// that other parsers of JSON text can convert tokens exactly like it does.
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -293,5 +297,183 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out)
|
||||
#endif
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for a float token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] token the validated number token ('.' as decimal point)
|
||||
@param[in] decimal_point_position index of the '.' in @a token, or
|
||||
std::string::npos if there is none
|
||||
@param[in] mantissa_end offset just past the last mantissa byte
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_position, std::size_t mantissa_end) noexcept
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (token[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. The
|
||||
// fraction is located through decimal_point_position rather than by
|
||||
// searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert a validated float token without the C library, if possible
|
||||
|
||||
Tries std::from_chars (when available) and then Clinger's exact fast path
|
||||
(double only), skipping the latter when it cannot succeed.
|
||||
|
||||
@param[in] first pointer to the first character of the token
|
||||
@param[in] last pointer past the last character
|
||||
@param[in] decimal_point_position index of the '.' in the token, or
|
||||
std::string::npos if there is none
|
||||
@param[in] mantissa_end offset just past the last mantissa byte (the
|
||||
index of 'e'/'E', or the token length)
|
||||
@param[out] value the converted value on success
|
||||
@return true if the value was converted; false if convert_float_locale_aware()
|
||||
must convert it
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool convert_float_fast(const char* first, const char* last, std::size_t decimal_point_position,
|
||||
std::size_t mantissa_end, FloatType& value) noexcept
|
||||
{
|
||||
if (parse_float_from_chars(first, last, value))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
return mantissa_fits_clinger(first, decimal_point_position, mantissa_end)
|
||||
&& parse_float_fast(first, last, value);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
inline char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert a validated float token with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
|
||||
@param[in,out] token the token with '.' as decimal point; its
|
||||
decimal point is replaced during the
|
||||
conversion and restored afterwards
|
||||
(data() must be NUL-terminated)
|
||||
@param[in] decimal_point_position index of the '.' in @a token, or
|
||||
std::string::npos if there is none
|
||||
@param[out] value the converted value
|
||||
*/
|
||||
template<typename StringType, typename FloatType>
|
||||
void convert_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& value)
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = static_cast<typename StringType::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof_by_type(value, token.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// the caller hands the token on (e.g. to the SAX interface) with '.'
|
||||
token[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token.data() + token.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -1991,21 +1991,9 @@ class binary_writer
|
||||
case 'd':
|
||||
{
|
||||
const auto dval = el.template get<double>();
|
||||
#ifdef __GNUC__
|
||||
JSON_HEDLEY_DIAGNOSTIC_PUSH
|
||||
JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal")
|
||||
#endif
|
||||
// a value that would be rounded (rather than exactly represented) by the
|
||||
// narrowing to float is treated like an out-of-range integer element above;
|
||||
// this is the same criterion write_compact_float() uses for CBOR/MessagePack
|
||||
in_range = std::isnan(dval) ||
|
||||
in_range = !std::isfinite(dval) ||
|
||||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
|
||||
dval <= static_cast<double>((std::numeric_limits<float>::max)()) &&
|
||||
static_cast<double>(static_cast<float>(dval)) == dval) ||
|
||||
std::isinf(dval);
|
||||
#ifdef __GNUC__
|
||||
JSON_HEDLEY_DIAGNOSTIC_POP
|
||||
#endif
|
||||
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
|
||||
+211
-209
@@ -8472,10 +8472,8 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
#include <utility> // move
|
||||
@@ -8496,9 +8494,12 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <cstdlib> // strtof, strtod, strtold
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
|
||||
// #include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
@@ -8516,8 +8517,9 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
// overhead of std::strtoull/std::strtod where possible. They are free functions
|
||||
// so the lexer stays focused on scanning (see lexer::convert_number()) and so
|
||||
// that other parsers of JSON text can convert tokens exactly like it does.
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -8780,6 +8782,184 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out)
|
||||
#endif
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for a float token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] token the validated number token ('.' as decimal point)
|
||||
@param[in] decimal_point_position index of the '.' in @a token, or
|
||||
std::string::npos if there is none
|
||||
@param[in] mantissa_end offset just past the last mantissa byte
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_position, std::size_t mantissa_end) noexcept
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (token[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. The
|
||||
// fraction is located through decimal_point_position rather than by
|
||||
// searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert a validated float token without the C library, if possible
|
||||
|
||||
Tries std::from_chars (when available) and then Clinger's exact fast path
|
||||
(double only), skipping the latter when it cannot succeed.
|
||||
|
||||
@param[in] first pointer to the first character of the token
|
||||
@param[in] last pointer past the last character
|
||||
@param[in] decimal_point_position index of the '.' in the token, or
|
||||
std::string::npos if there is none
|
||||
@param[in] mantissa_end offset just past the last mantissa byte (the
|
||||
index of 'e'/'E', or the token length)
|
||||
@param[out] value the converted value on success
|
||||
@return true if the value was converted; false if convert_float_locale_aware()
|
||||
must convert it
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool convert_float_fast(const char* first, const char* last, std::size_t decimal_point_position,
|
||||
std::size_t mantissa_end, FloatType& value) noexcept
|
||||
{
|
||||
if (parse_float_from_chars(first, last, value))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
return mantissa_fits_clinger(first, decimal_point_position, mantissa_end)
|
||||
&& parse_float_fast(first, last, value);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
inline char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert a validated float token with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
|
||||
@param[in,out] token the token with '.' as decimal point; its
|
||||
decimal point is replaced during the
|
||||
conversion and restored afterwards
|
||||
(data() must be NUL-terminated)
|
||||
@param[in] decimal_point_position index of the '.' in @a token, or
|
||||
std::string::npos if there is none
|
||||
@param[out] value the converted value
|
||||
*/
|
||||
template<typename StringType, typename FloatType>
|
||||
void convert_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& value)
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = static_cast<typename StringType::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof_by_type(value, token.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// the caller hands the token on (e.g. to the SAX interface) with '.'
|
||||
token[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token.data() + token.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -9309,18 +9489,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
~lexer() = default;
|
||||
|
||||
private:
|
||||
/////////////////////
|
||||
// locales
|
||||
/////////////////////
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// scan functions
|
||||
/////////////////////
|
||||
@@ -10128,24 +10296,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan a number literal
|
||||
|
||||
@@ -10185,7 +10335,7 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
@note The scanner is independent of the current locale: token_buffer
|
||||
always holds `.`. Only the std::strtod fallback of convert_number()
|
||||
depends on the locale, and it looks up the decimal point right
|
||||
before converting (see convert_float_locale_aware()).
|
||||
before converting (see detail::convert_float_locale_aware()).
|
||||
*/
|
||||
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
|
||||
{
|
||||
@@ -10516,59 +10666,6 @@ scan_number_done:
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for this token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] mantissa_end offset just past the last mantissa byte in
|
||||
token_buffer
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
bool mantissa_fits_clinger(std::size_t mantissa_end) const
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. The
|
||||
// fraction is located through decimal_point_position rather than by
|
||||
// searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the number text in token_buffer to its value and token type
|
||||
|
||||
@@ -10582,7 +10679,7 @@ scan_number_done:
|
||||
token_buffer (the index of 'e'/'E', or
|
||||
token_buffer.size() when there is no exponent);
|
||||
used to skip Clinger's fast path when it cannot
|
||||
possibly succeed - see mantissa_fits_clinger()
|
||||
possibly succeed - see detail::mantissa_fits_clinger()
|
||||
*/
|
||||
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||
{
|
||||
@@ -10655,77 +10752,15 @@ scan_number_done:
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod/strtold.
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
if (mantissa_fits_clinger(mantissa_end)
|
||||
&& parse_float_fast(num_begin, num_end, value_float))
|
||||
if (convert_float_fast(num_begin, num_end, decimal_point_position, mantissa_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
convert_float_locale_aware();
|
||||
convert_float_locale_aware(token_buffer, decimal_point_position, value_float);
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the float in token_buffer with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
*/
|
||||
void convert_float_locale_aware()
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// get_string() hands the token to the SAX interface with '.'
|
||||
token_buffer[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief contiguous fast path for scanning a number
|
||||
|
||||
@@ -15495,15 +15530,10 @@ class binary_reader
|
||||
is_ndarray can only return `true` when its initial value
|
||||
is `false`
|
||||
@param[in] prefix type marker if already read, otherwise set to 0
|
||||
@param[in] ndarray_dtype the element type marker of the enclosing bjdata ndarray if
|
||||
already known (it precedes the dimension vector read here),
|
||||
otherwise 0; used to emit the "_ArrayType_" annotation key
|
||||
before "_ArraySize_" if a dimension vector turns out to
|
||||
describe an ndarray
|
||||
|
||||
@return whether size determination completed
|
||||
*/
|
||||
bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0, char_int_type ndarray_dtype = 0)
|
||||
bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0)
|
||||
{
|
||||
if (prefix == 0)
|
||||
{
|
||||
@@ -15673,37 +15703,8 @@ class binary_reader
|
||||
}
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// the element type precedes the dimension vector (see get_ubjson_size_type)
|
||||
// and is passed down as ndarray_dtype; emit it here so the annotation keys
|
||||
// follow the documented _ArrayType_, _ArraySize_, _ArrayData_ order
|
||||
if (ndarray_dtype != 0)
|
||||
{
|
||||
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), ndarray_dtype, [](const bjd_type & p, char_int_type t)
|
||||
{
|
||||
return p.first < t;
|
||||
});
|
||||
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != ndarray_dtype))
|
||||
{
|
||||
auto last_token = get_token_string();
|
||||
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
|
||||
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
|
||||
}
|
||||
|
||||
string_t type_key = "_ArrayType_";
|
||||
string_t type = it->second; // sax->string() takes a reference
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(type_key) || !sax->string(type)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
string_t key = "_ArraySize_";
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(dim.size())))
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3) || !sax->key(key) || !sax->start_array(dim.size())))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -15804,7 +15805,7 @@ class binary_reader
|
||||
exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr));
|
||||
}
|
||||
|
||||
const bool is_error = get_ubjson_size_value(result.first, is_ndarray, 0, result.second);
|
||||
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
|
||||
// an ndarray was read here only if the flag flipped; when it was
|
||||
// seeded true, get_ubjson_size_value() already rejected the nested
|
||||
// dimension vector
|
||||
@@ -16040,17 +16041,30 @@ class binary_reader
|
||||
if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0)
|
||||
{
|
||||
size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker
|
||||
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t)
|
||||
{
|
||||
return p.first < t;
|
||||
});
|
||||
string_t key = "_ArrayType_";
|
||||
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second))
|
||||
{
|
||||
auto last_token = get_token_string();
|
||||
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
|
||||
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
|
||||
}
|
||||
|
||||
string_t type = it->second; // sax->string() takes a reference
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// the "_ArrayType_" and "_ArraySize_" annotation keys were already emitted by
|
||||
// get_ubjson_size_value() (the type marker is known before the dimension vector
|
||||
// that determines size_and_type.first is read, so it is emitted first there to
|
||||
// match the documented _ArrayType_, _ArraySize_, _ArrayData_ key order)
|
||||
if (size_and_type.second == 'C' || size_and_type.second == 'B')
|
||||
{
|
||||
size_and_type.second = 'U';
|
||||
}
|
||||
|
||||
string_t key = "_ArrayData_";
|
||||
key = "_ArrayData_";
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) ))
|
||||
{
|
||||
return false;
|
||||
@@ -22342,21 +22356,9 @@ class binary_writer
|
||||
case 'd':
|
||||
{
|
||||
const auto dval = el.template get<double>();
|
||||
#ifdef __GNUC__
|
||||
JSON_HEDLEY_DIAGNOSTIC_PUSH
|
||||
JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal")
|
||||
#endif
|
||||
// a value that would be rounded (rather than exactly represented) by the
|
||||
// narrowing to float is treated like an out-of-range integer element above;
|
||||
// this is the same criterion write_compact_float() uses for CBOR/MessagePack
|
||||
in_range = std::isnan(dval) ||
|
||||
in_range = !std::isfinite(dval) ||
|
||||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
|
||||
dval <= static_cast<double>((std::numeric_limits<float>::max)()) &&
|
||||
static_cast<double>(static_cast<float>(dval)) == dval) ||
|
||||
std::isinf(dval);
|
||||
#ifdef __GNUC__
|
||||
JSON_HEDLEY_DIAGNOSTIC_POP
|
||||
#endif
|
||||
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
|
||||
@@ -11,7 +11,6 @@
|
||||
#define JSON_TESTS_PRIVATE
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
using ordered_json = nlohmann::ordered_json;
|
||||
|
||||
#include <algorithm>
|
||||
#include <climits>
|
||||
@@ -2295,33 +2294,29 @@ TEST_CASE("BJData")
|
||||
|
||||
SECTION("start_array() in ndarray _ArraySize_")
|
||||
{
|
||||
// _ArrayType_ (2 events: key + string) is now emitted before
|
||||
// _ArraySize_ (see GitHub issue #5661), which shifts the events
|
||||
// below later by the same 2 events
|
||||
std::vector<uint8_t> const v = {'[', '$', 'i', '#', '[', '$', 'i', '#', 'i', 2, 2, 1, 1, 2};
|
||||
SaxCountdown scp(4);
|
||||
SaxCountdown scp(2);
|
||||
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
|
||||
}
|
||||
|
||||
SECTION("number_integer() in ndarray _ArraySize_")
|
||||
{
|
||||
std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'i', '#', 'i', 2, 2, 1, 1, 2};
|
||||
SaxCountdown scp(5);
|
||||
SaxCountdown scp(3);
|
||||
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
|
||||
}
|
||||
|
||||
SECTION("key() in ndarray _ArrayType_")
|
||||
{
|
||||
// _ArrayType_ is emitted right after start_object(), before _ArraySize_
|
||||
std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'U', '#', 'i', 2, 2, 2, 1, 2, 3, 4};
|
||||
SaxCountdown scp(1);
|
||||
SaxCountdown scp(6);
|
||||
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
|
||||
}
|
||||
|
||||
SECTION("string() in ndarray _ArrayType_")
|
||||
{
|
||||
std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'U', '#', 'i', 2, 2, 2, 1, 2, 3, 4};
|
||||
SaxCountdown scp(2);
|
||||
SaxCountdown scp(7);
|
||||
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
|
||||
}
|
||||
|
||||
@@ -2924,22 +2919,6 @@ TEST_CASE("BJData")
|
||||
CHECK(out_single.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_single) == j_single);
|
||||
|
||||
// a double element that is finite and within the range of "single"
|
||||
// but is not exactly representable as a float, so narrowing it would
|
||||
// silently round it (0.1 is read back as 0.10000000149011612); this,
|
||||
// like the overflow case above, falls back to a plain object (see
|
||||
// GitHub issue #5661)
|
||||
json const j_single_rounded = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 0.1}}});
|
||||
const auto out_single_rounded = json::to_bjdata(j_single_rounded);
|
||||
CHECK(out_single_rounded.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_single_rounded) == j_single_rounded);
|
||||
|
||||
// a double element that underflows to 0 when narrowed to "single"
|
||||
json const j_single_underflow = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 1e-300}}});
|
||||
const auto out_single_underflow = json::to_bjdata(j_single_underflow);
|
||||
CHECK(out_single_underflow.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_single_underflow) == j_single_underflow);
|
||||
|
||||
// in-range boundary values still use the compact ndarray encoding
|
||||
json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {0, 255}}});
|
||||
CHECK(json::to_bjdata(j_uint8_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, 'i', 1, ']', 0, 255}));
|
||||
@@ -2953,23 +2932,6 @@ TEST_CASE("BJData")
|
||||
CHECK(json::from_bjdata(out_single_ok) == json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5f, -1.5f}}}));
|
||||
}
|
||||
|
||||
SECTION("ndarray annotation keys are read back in the documented order")
|
||||
{
|
||||
// from_bjdata() must emit the annotation object's keys in the order
|
||||
// used throughout the documentation, _ArrayType_, _ArraySize_,
|
||||
// _ArrayData_: the type marker precedes the dimension vector on the
|
||||
// wire (see get_ubjson_size_type()), so it is known, and emitted,
|
||||
// before _ArraySize_. For a plain json this key order is invisible
|
||||
// (its comparison ignores it), but for an ordered_json it is not (see
|
||||
// GitHub issue #5661).
|
||||
const ordered_json o = ordered_json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[2,2],"_ArrayData_":[1,2,3,4]})");
|
||||
const auto packed = ordered_json::to_bjdata(o);
|
||||
CHECK(packed.at(0) == '[');
|
||||
const ordered_json o_back = ordered_json::from_bjdata(packed);
|
||||
CHECK(o_back == o);
|
||||
CHECK(o_back.dump() == o.dump());
|
||||
}
|
||||
|
||||
SECTION("ndarray that would not be read back as an annotated object stays as object")
|
||||
{
|
||||
// the reader only restores an annotated object from an ND-array
|
||||
|
||||
@@ -8,3 +8,6 @@ The following changes have been made to the code with respect to <https://github
|
||||
- membership check
|
||||
- made function from `_is_within`
|
||||
- removed unused variable `actual_path`
|
||||
- Added the optional config key `external`: include paths listed there are kept as
|
||||
`#include` directives instead of being inlined (the first directive per path; the
|
||||
repeated ones are commented out).
|
||||
|
||||
@@ -57,6 +57,11 @@ Python v.2.7.0 or higher is required.
|
||||
amalgamation. Have a look at `test/source.c.json` and `test/include.h.json`
|
||||
to see two examples.
|
||||
|
||||
The optional `external` list names include paths that are kept as `#include`
|
||||
directives instead of being inlined, e.g. `["nlohmann/json.hpp"]` for a header
|
||||
that includes another amalgamated header. Only the first directive for each
|
||||
of these paths is kept; the repeated ones are commented out.
|
||||
|
||||
* The `-s, --source` option should specify the path to the source directory.
|
||||
This is useful for supporting separate source and build directories.
|
||||
|
||||
|
||||
@@ -62,6 +62,10 @@ class Amalgamation(object):
|
||||
return None
|
||||
|
||||
def __init__(self, args):
|
||||
# include paths that are kept as #include directives instead of
|
||||
# being inlined (e.g. a header amalgamated on its own)
|
||||
self.external = []
|
||||
self.included_external = []
|
||||
with open(args.config, 'r') as f:
|
||||
config = json.loads(f.read())
|
||||
for key in config:
|
||||
@@ -220,11 +224,14 @@ class TranslationUnit(object):
|
||||
while include_match:
|
||||
if not _is_within(include_match, skippable_contexts):
|
||||
include_path = include_match.group("path")
|
||||
search_same_dir = include_match.group(1) == '"'
|
||||
found_included_path = self.amalgamation.find_included_file(
|
||||
include_path, self.file_dir if search_same_dir else None)
|
||||
if found_included_path:
|
||||
includes.append((include_match, found_included_path))
|
||||
if include_path in self.amalgamation.external:
|
||||
includes.append((include_match, None))
|
||||
else:
|
||||
search_same_dir = include_match.group(1) == '"'
|
||||
found_included_path = self.amalgamation.find_included_file(
|
||||
include_path, self.file_dir if search_same_dir else None)
|
||||
if found_included_path:
|
||||
includes.append((include_match, found_included_path))
|
||||
|
||||
include_match = self.include_pattern.search(self.content,
|
||||
include_match.end())
|
||||
@@ -235,6 +242,17 @@ class TranslationUnit(object):
|
||||
for include in includes:
|
||||
include_match, found_included_path = include
|
||||
tmp_content += self.content[prev_end:include_match.start()]
|
||||
if found_included_path is None:
|
||||
# an external header: keep the first directive and comment
|
||||
# out the repeated ones
|
||||
include_path = include_match.group("path")
|
||||
if include_path in self.amalgamation.included_external:
|
||||
tmp_content += "// {0}".format(include_match.group(0))
|
||||
else:
|
||||
self.amalgamation.included_external.append(include_path)
|
||||
tmp_content += include_match.group(0)
|
||||
prev_end = include_match.end()
|
||||
continue
|
||||
tmp_content += "// {0}\n".format(include_match.group(0))
|
||||
if found_included_path not in self.amalgamation.included_files:
|
||||
t = TranslationUnit(found_included_path, self.amalgamation, False)
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"project": "JSON for Modern C++",
|
||||
"target": "single_include/nlohmann/json_view.hpp",
|
||||
"sources": [
|
||||
"include/nlohmann/json_view.hpp"
|
||||
],
|
||||
"include_paths": ["include"],
|
||||
"external": ["nlohmann/json.hpp"]
|
||||
}
|
||||
Reference in New Issue
Block a user