mirror of
https://github.com/nlohmann/json.git
synced 2026-09-29 19:20:30 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9505be15fd | ||
|
|
44325873ea | ||
|
|
0c630d4c30 | ||
|
|
04ed4f593c |
@@ -55,6 +55,10 @@ This implementation does exactly follow this approach, as it uses double precisi
|
||||
smaller than `-1.79769313486232e+308` and values greater than `1.79769313486232e+308` will be stored as NaN internally
|
||||
and be serialized to `null`.
|
||||
|
||||
During deserialization (from JSON text or any of the binary formats), a finite number that does not fit into
|
||||
`number_float_t` is rejected with [`out_of_range.406`](../../home/exceptions.md#jsonexceptionout_of_range406), for
|
||||
example a double-precision number in a binary format when `number_float_t` is `#!cpp float`.
|
||||
|
||||
#### Storage
|
||||
|
||||
Floating-point number values are stored directly inside a `basic_json` type.
|
||||
|
||||
@@ -47,8 +47,9 @@ With the default values for `NumberIntegerType` (`std::int64_t`), the default va
|
||||
|
||||
When the default type is used, the maximal integer number that can be stored is `9223372036854775807` (INT64_MAX) and
|
||||
the minimal integer number that can be stored is `-9223372036854775808` (INT64_MIN). Integer numbers that are out of
|
||||
range will yield over/underflow when used in a constructor. During deserialization, too large or small integer numbers
|
||||
will automatically be stored as [`number_unsigned_t`](number_unsigned_t.md) or [`number_float_t`](number_float_t.md).
|
||||
range will yield over/underflow when used in a constructor. During deserialization (from JSON text or any of the binary
|
||||
formats), too large or small integer numbers will automatically be stored as [`number_unsigned_t`](number_unsigned_t.md)
|
||||
or [`number_float_t`](number_float_t.md).
|
||||
|
||||
[RFC 8259](https://tools.ietf.org/html/rfc8259) further states:
|
||||
> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are
|
||||
|
||||
@@ -48,8 +48,9 @@ With the default values for `NumberUnsignedType` (`std::uint64_t`), the default
|
||||
|
||||
When the default type is used, the maximal integer number that can be stored is `18446744073709551615` (UINT64_MAX) and
|
||||
the minimal integer number that can be stored is `0`. Integer numbers that are out of range will yield over/underflow
|
||||
when used in a constructor. During deserialization, too large or small integer numbers will automatically be stored
|
||||
as [`number_integer_t`](number_integer_t.md) or [`number_float_t`](number_float_t.md).
|
||||
when used in a constructor. During deserialization (from JSON text or any of the binary formats), too large or small
|
||||
integer numbers will automatically be stored as [`number_integer_t`](number_integer_t.md) or
|
||||
[`number_float_t`](number_float_t.md).
|
||||
|
||||
[RFC 8259](https://tools.ietf.org/html/rfc8259) further states:
|
||||
> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are
|
||||
|
||||
@@ -168,9 +168,9 @@ The library maps CBOR types to JSON value types as follows:
|
||||
!!! warning "Negative integer overflow"
|
||||
|
||||
CBOR negative integers (major type 1) are decoded as `-1 - n`. If the encoded magnitude `n` is too large for the
|
||||
result to fit into `number_integer_t` (`std::int64_t` by default), parsing fails with a
|
||||
[`parse_error.112`](../../home/exceptions.md#jsonexceptionparse_error112) exception rather than overflowing
|
||||
silently.
|
||||
result to fit into `number_integer_t` (`std::int64_t` by default), the result is stored as `number_float_t`, like
|
||||
a too small integer in JSON text. For example, `-18446744073709551616` (`0x3B` followed by eight `0xFF` bytes) is
|
||||
stored as `-1.8446744073709552e+19`.
|
||||
|
||||
!!! warning "Object keys"
|
||||
|
||||
|
||||
@@ -331,9 +331,6 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde
|
||||
[json.exception.parse_error.112] parse error at byte 15: syntax error while parsing BSON binary: byte array length cannot be negative, is -1
|
||||
```
|
||||
```
|
||||
[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow
|
||||
```
|
||||
```
|
||||
[json.exception.parse_error.112] parse error at byte 5: syntax error while parsing BSON document: document size 6 does not match the number of bytes read (5)
|
||||
```
|
||||
|
||||
@@ -854,13 +851,18 @@ The JSON Patch operations 'remove' and 'add' cannot be applied to the root eleme
|
||||
|
||||
### json.exception.out_of_range.406
|
||||
|
||||
A parsed number could not be stored as without changing it to NaN or INF.
|
||||
A parsed number could not be stored without changing it to NaN or INF. For the binary formats, this happens when a
|
||||
finite floating-point number does not fit into [`number_float_t`](../api/basic_json/number_float_t.md), for example a
|
||||
double-precision number when `number_float_t` is `#!cpp float`.
|
||||
|
||||
!!! failure "Example message"
|
||||
!!! failure "Example messages"
|
||||
|
||||
```
|
||||
number overflow parsing '10E1000'
|
||||
```
|
||||
```
|
||||
[json.exception.out_of_range.406] syntax error while parsing CBOR value: number overflow
|
||||
```
|
||||
|
||||
### json.exception.out_of_range.407
|
||||
|
||||
|
||||
@@ -559,7 +559,7 @@ class binary_reader
|
||||
case 0x01: // double
|
||||
{
|
||||
double number{};
|
||||
return get_number<double, true>(input_format_t::bson, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number<double, true>(input_format_t::bson, number) && emit_float(input_format_t::bson, number);
|
||||
}
|
||||
|
||||
case 0x02: // string
|
||||
@@ -600,19 +600,19 @@ class binary_reader
|
||||
case 0x10: // int32
|
||||
{
|
||||
std::int32_t value{};
|
||||
return get_number<std::int32_t, true>(input_format_t::bson, value) && sax->number_integer(value);
|
||||
return get_number<std::int32_t, true>(input_format_t::bson, value) && emit_signed(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
case 0x12: // int64
|
||||
{
|
||||
std::int64_t value{};
|
||||
return get_number<std::int64_t, true>(input_format_t::bson, value) && sax->number_integer(value);
|
||||
return get_number<std::int64_t, true>(input_format_t::bson, value) && emit_signed(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
case 0x11: // uint64
|
||||
{
|
||||
std::uint64_t value{};
|
||||
return get_number<std::uint64_t, true>(input_format_t::bson, value) && sax->number_unsigned(value);
|
||||
return get_number<std::uint64_t, true>(input_format_t::bson, value) && emit_unsigned(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
default: // anything else is not supported (yet)
|
||||
@@ -638,14 +638,19 @@ class binary_reader
|
||||
{
|
||||
return false;
|
||||
}
|
||||
const auto max_val = static_cast<NumberType>((std::numeric_limits<number_integer_t>::max)());
|
||||
if (number > max_val)
|
||||
|
||||
// the value is -1 - number, which fits into number_integer_t
|
||||
// whenever number does
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_integer_t>(number)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
parse_error::create(112, chars_read,
|
||||
exception_message(input_format_t::cbor, "negative integer overflow", "value"), nullptr));
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
}
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
|
||||
// like the lexer does for JSON text, store a value too small for
|
||||
// number_integer_t as number_float_t; compute it as long double so
|
||||
// that emit_float sees a finite value and can detect an overflow of
|
||||
// number_float_t
|
||||
return emit_float(input_format_t::cbor, static_cast<long double>(-1) - static_cast<long double>(number));
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -702,25 +707,25 @@ class binary_reader
|
||||
case 0x18: // Unsigned integer (one-byte uint8_t follows)
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x19: // Unsigned integer (two-byte uint16_t follows)
|
||||
{
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x1A: // Unsigned integer (four-byte uint32_t follows)
|
||||
{
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x1B: // Unsigned integer (eight-byte uint64_t follows)
|
||||
{
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
// Negative integer -1-0x00..-1-0x17 (-1..-24)
|
||||
@@ -1165,13 +1170,13 @@ class binary_reader
|
||||
case 0xFA: // Single-Precision Float (four-byte IEEE 754)
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::cbor, number) && emit_float(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0xFB: // Double-Precision Float (eight-byte IEEE 754)
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::cbor, number) && emit_float(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
default: // anything else (0xFF is handled inside the other types)
|
||||
@@ -1935,61 +1940,61 @@ class binary_reader
|
||||
case 0xCA: // float 32
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::msgpack, number) && emit_float(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCB: // float 64
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::msgpack, number) && emit_float(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCC: // uint 8
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCD: // uint 16
|
||||
{
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCE: // uint 32
|
||||
{
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCF: // uint 64
|
||||
{
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD0: // int 8
|
||||
{
|
||||
std::int8_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD1: // int 16
|
||||
{
|
||||
std::int16_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD2: // int 32
|
||||
{
|
||||
std::int32_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD3: // int 64
|
||||
{
|
||||
std::int64_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xDC: // array 16
|
||||
@@ -2922,7 +2927,7 @@ class binary_reader
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->number_unsigned(static_cast<number_unsigned_t>(i))))
|
||||
if (JSON_HEDLEY_UNLIKELY(!emit_unsigned(input_format, i)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -3054,37 +3059,37 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'U':
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'i':
|
||||
{
|
||||
std::int8_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'I':
|
||||
{
|
||||
std::int16_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'l':
|
||||
{
|
||||
std::int32_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'L':
|
||||
{
|
||||
std::int64_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'u':
|
||||
@@ -3094,7 +3099,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'm':
|
||||
@@ -3104,7 +3109,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'M':
|
||||
@@ -3114,7 +3119,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'h':
|
||||
@@ -3172,13 +3177,13 @@ class binary_reader
|
||||
case 'd':
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format, number) && emit_float(input_format, number);
|
||||
}
|
||||
|
||||
case 'D':
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format, number) && emit_float(input_format, number);
|
||||
}
|
||||
|
||||
case 'H':
|
||||
@@ -3645,13 +3650,13 @@ class binary_reader
|
||||
case 0x8E: // binary32
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::bon8, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::bon8, number) && emit_float(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
case 0x8F: // binary64
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::bon8, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::bon8, number) && emit_float(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
case 0xF8:
|
||||
@@ -3717,7 +3722,9 @@ class binary_reader
|
||||
@brief pass an integer to the SAX parser
|
||||
|
||||
Non-negative integers are passed as unsigned, negative integers as signed
|
||||
numbers, like the other binary formats do.
|
||||
numbers, like the other binary formats do. A value that does not fit the
|
||||
number type is passed as described for @ref emit_unsigned and
|
||||
@ref emit_signed.
|
||||
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
@@ -3726,9 +3733,9 @@ class binary_reader
|
||||
{
|
||||
if (number >= 0)
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
return emit_unsigned(input_format_t::bon8, static_cast<std::uint64_t>(number));
|
||||
}
|
||||
return sax->number_integer(static_cast<number_integer_t>(number));
|
||||
return emit_signed(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -3785,8 +3792,7 @@ class binary_reader
|
||||
value = (value << 8) | static_cast<std::int64_t>(current);
|
||||
}
|
||||
|
||||
return negative ? sax->number_integer(static_cast<number_integer_t>(-(value + offset)))
|
||||
: sax->number_unsigned(static_cast<number_unsigned_t>(value + offset));
|
||||
return emit_bon8_integer(negative ? -(value + offset) : value + offset);
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -4083,6 +4089,88 @@ class binary_reader
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a signed integer read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a value that does not fit into
|
||||
number_integer_t is passed as number_unsigned_t if it is non-negative and
|
||||
fits there, and as number_float_t otherwise. With the default number
|
||||
types, every integer the binary formats can encode fits, so this only
|
||||
matters for narrower custom number types.
|
||||
|
||||
@tparam NumberType a signed integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if @a number overflows number_float_t (see
|
||||
@ref emit_float)
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_signed(const input_format_t format, const NumberType number)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_integer_t>(number)))
|
||||
{
|
||||
return sax->number_integer(static_cast<number_integer_t>(number));
|
||||
}
|
||||
if (value_in_range_of<number_unsigned_t>(number))
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_float(format, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass an unsigned integer read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a value that does not fit into
|
||||
number_unsigned_t is passed as number_float_t.
|
||||
|
||||
@tparam NumberType an unsigned integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if @a number overflows number_float_t (see
|
||||
@ref emit_float)
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_unsigned(const input_format_t format, const NumberType number)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_unsigned_t>(number)))
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_float(format, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a floating-point number read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a finite value that overflows
|
||||
number_float_t is rejected instead of silently becoming infinity. Infinity
|
||||
and NaN in the input are passed on unchanged. Integers only overflow if
|
||||
number_float_t cannot represent 2^64, e.g., a half-precision type.
|
||||
|
||||
@tparam NumberType a floating-point or integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the number
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if a finite @a number overflows number_float_t
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_float(const input_format_t format, const NumberType number)
|
||||
{
|
||||
const auto result = static_cast<number_float_t>(number);
|
||||
if (JSON_HEDLEY_UNLIKELY(std::isfinite(number) && !std::isfinite(result)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
out_of_range::create(406, exception_message(format, "number overflow", "value"), nullptr));
|
||||
}
|
||||
return sax->number_float(result, "");
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief create a string by reading characters from the input
|
||||
|
||||
|
||||
@@ -9,8 +9,10 @@
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
#include <utility> // move
|
||||
@@ -215,6 +217,18 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
~lexer() = default;
|
||||
|
||||
private:
|
||||
/////////////////////
|
||||
// locales
|
||||
/////////////////////
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// scan functions
|
||||
/////////////////////
|
||||
@@ -1022,6 +1036,24 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan a number literal
|
||||
|
||||
@@ -1061,7 +1093,7 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
@note The scanner is independent of the current locale: token_buffer
|
||||
always holds `.`. Only the std::strtod fallback of convert_number()
|
||||
depends on the locale, and it looks up the decimal point right
|
||||
before converting (see detail::convert_float_locale_aware()).
|
||||
before converting (see convert_float_locale_aware()).
|
||||
*/
|
||||
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
|
||||
{
|
||||
@@ -1392,6 +1424,59 @@ scan_number_done:
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for this token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] mantissa_end offset just past the last mantissa byte in
|
||||
token_buffer
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
bool mantissa_fits_clinger(std::size_t mantissa_end) const
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. The
|
||||
// fraction is located through decimal_point_position rather than by
|
||||
// searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the number text in token_buffer to its value and token type
|
||||
|
||||
@@ -1405,7 +1490,7 @@ scan_number_done:
|
||||
token_buffer (the index of 'e'/'E', or
|
||||
token_buffer.size() when there is no exponent);
|
||||
used to skip Clinger's fast path when it cannot
|
||||
possibly succeed - see detail::mantissa_fits_clinger()
|
||||
possibly succeed - see mantissa_fits_clinger()
|
||||
*/
|
||||
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||
{
|
||||
@@ -1478,15 +1563,77 @@ scan_number_done:
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod/strtold.
|
||||
if (convert_float_fast(num_begin, num_end, decimal_point_position, mantissa_end, value_float))
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
if (mantissa_fits_clinger(mantissa_end)
|
||||
&& parse_float_fast(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
convert_float_locale_aware(token_buffer, decimal_point_position, value_float);
|
||||
convert_float_locale_aware();
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the float in token_buffer with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
*/
|
||||
void convert_float_locale_aware()
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// get_string() hands the token to the SAX interface with '.'
|
||||
token_buffer[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief contiguous fast path for scanning a number
|
||||
|
||||
|
||||
@@ -10,12 +10,9 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <cstdlib> // strtof, strtod, strtold
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
@@ -32,9 +29,8 @@
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod where possible. They are free functions
|
||||
// so the lexer stays focused on scanning (see lexer::convert_number()) and so
|
||||
// that other parsers of JSON text can convert tokens exactly like it does.
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -297,183 +293,5 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out)
|
||||
#endif
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for a float token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] token the validated number token ('.' as decimal point)
|
||||
@param[in] decimal_point_position index of the '.' in @a token, or
|
||||
std::string::npos if there is none
|
||||
@param[in] mantissa_end offset just past the last mantissa byte
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_position, std::size_t mantissa_end) noexcept
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (token[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. The
|
||||
// fraction is located through decimal_point_position rather than by
|
||||
// searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert a validated float token without the C library, if possible
|
||||
|
||||
Tries std::from_chars (when available) and then Clinger's exact fast path
|
||||
(double only), skipping the latter when it cannot succeed.
|
||||
|
||||
@param[in] first pointer to the first character of the token
|
||||
@param[in] last pointer past the last character
|
||||
@param[in] decimal_point_position index of the '.' in the token, or
|
||||
std::string::npos if there is none
|
||||
@param[in] mantissa_end offset just past the last mantissa byte (the
|
||||
index of 'e'/'E', or the token length)
|
||||
@param[out] value the converted value on success
|
||||
@return true if the value was converted; false if convert_float_locale_aware()
|
||||
must convert it
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool convert_float_fast(const char* first, const char* last, std::size_t decimal_point_position,
|
||||
std::size_t mantissa_end, FloatType& value) noexcept
|
||||
{
|
||||
if (parse_float_from_chars(first, last, value))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
return mantissa_fits_clinger(first, decimal_point_position, mantissa_end)
|
||||
&& parse_float_fast(first, last, value);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
inline char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert a validated float token with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
|
||||
@param[in,out] token the token with '.' as decimal point; its
|
||||
decimal point is replaced during the
|
||||
conversion and restored afterwards
|
||||
(data() must be NUL-terminated)
|
||||
@param[in] decimal_point_position index of the '.' in @a token, or
|
||||
std::string::npos if there is none
|
||||
@param[out] value the converted value
|
||||
*/
|
||||
template<typename StringType, typename FloatType>
|
||||
void convert_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& value)
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = static_cast<typename StringType::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof_by_type(value, token.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// the caller hands the token on (e.g. to the SAX interface) with '.'
|
||||
token[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token.data() + token.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
+286
-233
@@ -8472,8 +8472,10 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
#include <utility> // move
|
||||
@@ -8494,12 +8496,9 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <cstdlib> // strtof, strtod, strtold
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
|
||||
// #include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
@@ -8517,9 +8516,8 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod where possible. They are free functions
|
||||
// so the lexer stays focused on scanning (see lexer::convert_number()) and so
|
||||
// that other parsers of JSON text can convert tokens exactly like it does.
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -8782,184 +8780,6 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out)
|
||||
#endif
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for a float token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] token the validated number token ('.' as decimal point)
|
||||
@param[in] decimal_point_position index of the '.' in @a token, or
|
||||
std::string::npos if there is none
|
||||
@param[in] mantissa_end offset just past the last mantissa byte
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_position, std::size_t mantissa_end) noexcept
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (token[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. The
|
||||
// fraction is located through decimal_point_position rather than by
|
||||
// searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert a validated float token without the C library, if possible
|
||||
|
||||
Tries std::from_chars (when available) and then Clinger's exact fast path
|
||||
(double only), skipping the latter when it cannot succeed.
|
||||
|
||||
@param[in] first pointer to the first character of the token
|
||||
@param[in] last pointer past the last character
|
||||
@param[in] decimal_point_position index of the '.' in the token, or
|
||||
std::string::npos if there is none
|
||||
@param[in] mantissa_end offset just past the last mantissa byte (the
|
||||
index of 'e'/'E', or the token length)
|
||||
@param[out] value the converted value on success
|
||||
@return true if the value was converted; false if convert_float_locale_aware()
|
||||
must convert it
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool convert_float_fast(const char* first, const char* last, std::size_t decimal_point_position,
|
||||
std::size_t mantissa_end, FloatType& value) noexcept
|
||||
{
|
||||
if (parse_float_from_chars(first, last, value))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
return mantissa_fits_clinger(first, decimal_point_position, mantissa_end)
|
||||
&& parse_float_fast(first, last, value);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
inline char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert a validated float token with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
|
||||
@param[in,out] token the token with '.' as decimal point; its
|
||||
decimal point is replaced during the
|
||||
conversion and restored afterwards
|
||||
(data() must be NUL-terminated)
|
||||
@param[in] decimal_point_position index of the '.' in @a token, or
|
||||
std::string::npos if there is none
|
||||
@param[out] value the converted value
|
||||
*/
|
||||
template<typename StringType, typename FloatType>
|
||||
void convert_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& value)
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = static_cast<typename StringType::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof_by_type(value, token.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// the caller hands the token on (e.g. to the SAX interface) with '.'
|
||||
token[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token.data() + token.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -9489,6 +9309,18 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
~lexer() = default;
|
||||
|
||||
private:
|
||||
/////////////////////
|
||||
// locales
|
||||
/////////////////////
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// scan functions
|
||||
/////////////////////
|
||||
@@ -10296,6 +10128,24 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan a number literal
|
||||
|
||||
@@ -10335,7 +10185,7 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
@note The scanner is independent of the current locale: token_buffer
|
||||
always holds `.`. Only the std::strtod fallback of convert_number()
|
||||
depends on the locale, and it looks up the decimal point right
|
||||
before converting (see detail::convert_float_locale_aware()).
|
||||
before converting (see convert_float_locale_aware()).
|
||||
*/
|
||||
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
|
||||
{
|
||||
@@ -10666,6 +10516,59 @@ scan_number_done:
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for this token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] mantissa_end offset just past the last mantissa byte in
|
||||
token_buffer
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
bool mantissa_fits_clinger(std::size_t mantissa_end) const
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. The
|
||||
// fraction is located through decimal_point_position rather than by
|
||||
// searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the number text in token_buffer to its value and token type
|
||||
|
||||
@@ -10679,7 +10582,7 @@ scan_number_done:
|
||||
token_buffer (the index of 'e'/'E', or
|
||||
token_buffer.size() when there is no exponent);
|
||||
used to skip Clinger's fast path when it cannot
|
||||
possibly succeed - see detail::mantissa_fits_clinger()
|
||||
possibly succeed - see mantissa_fits_clinger()
|
||||
*/
|
||||
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||
{
|
||||
@@ -10752,15 +10655,77 @@ scan_number_done:
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod/strtold.
|
||||
if (convert_float_fast(num_begin, num_end, decimal_point_position, mantissa_end, value_float))
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
if (mantissa_fits_clinger(mantissa_end)
|
||||
&& parse_float_fast(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
convert_float_locale_aware(token_buffer, decimal_point_position, value_float);
|
||||
convert_float_locale_aware();
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the float in token_buffer with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
*/
|
||||
void convert_float_locale_aware()
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// get_string() hands the token to the SAX interface with '.'
|
||||
token_buffer[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief contiguous fast path for scanning a number
|
||||
|
||||
@@ -13361,7 +13326,7 @@ class binary_reader
|
||||
case 0x01: // double
|
||||
{
|
||||
double number{};
|
||||
return get_number<double, true>(input_format_t::bson, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number<double, true>(input_format_t::bson, number) && emit_float(input_format_t::bson, number);
|
||||
}
|
||||
|
||||
case 0x02: // string
|
||||
@@ -13402,19 +13367,19 @@ class binary_reader
|
||||
case 0x10: // int32
|
||||
{
|
||||
std::int32_t value{};
|
||||
return get_number<std::int32_t, true>(input_format_t::bson, value) && sax->number_integer(value);
|
||||
return get_number<std::int32_t, true>(input_format_t::bson, value) && emit_signed(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
case 0x12: // int64
|
||||
{
|
||||
std::int64_t value{};
|
||||
return get_number<std::int64_t, true>(input_format_t::bson, value) && sax->number_integer(value);
|
||||
return get_number<std::int64_t, true>(input_format_t::bson, value) && emit_signed(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
case 0x11: // uint64
|
||||
{
|
||||
std::uint64_t value{};
|
||||
return get_number<std::uint64_t, true>(input_format_t::bson, value) && sax->number_unsigned(value);
|
||||
return get_number<std::uint64_t, true>(input_format_t::bson, value) && emit_unsigned(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
default: // anything else is not supported (yet)
|
||||
@@ -13440,14 +13405,19 @@ class binary_reader
|
||||
{
|
||||
return false;
|
||||
}
|
||||
const auto max_val = static_cast<NumberType>((std::numeric_limits<number_integer_t>::max)());
|
||||
if (number > max_val)
|
||||
|
||||
// the value is -1 - number, which fits into number_integer_t
|
||||
// whenever number does
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_integer_t>(number)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
parse_error::create(112, chars_read,
|
||||
exception_message(input_format_t::cbor, "negative integer overflow", "value"), nullptr));
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
}
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
|
||||
// like the lexer does for JSON text, store a value too small for
|
||||
// number_integer_t as number_float_t; compute it as long double so
|
||||
// that emit_float sees a finite value and can detect an overflow of
|
||||
// number_float_t
|
||||
return emit_float(input_format_t::cbor, static_cast<long double>(-1) - static_cast<long double>(number));
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -13504,25 +13474,25 @@ class binary_reader
|
||||
case 0x18: // Unsigned integer (one-byte uint8_t follows)
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x19: // Unsigned integer (two-byte uint16_t follows)
|
||||
{
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x1A: // Unsigned integer (four-byte uint32_t follows)
|
||||
{
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x1B: // Unsigned integer (eight-byte uint64_t follows)
|
||||
{
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
// Negative integer -1-0x00..-1-0x17 (-1..-24)
|
||||
@@ -13967,13 +13937,13 @@ class binary_reader
|
||||
case 0xFA: // Single-Precision Float (four-byte IEEE 754)
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::cbor, number) && emit_float(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0xFB: // Double-Precision Float (eight-byte IEEE 754)
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::cbor, number) && emit_float(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
default: // anything else (0xFF is handled inside the other types)
|
||||
@@ -14737,61 +14707,61 @@ class binary_reader
|
||||
case 0xCA: // float 32
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::msgpack, number) && emit_float(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCB: // float 64
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::msgpack, number) && emit_float(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCC: // uint 8
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCD: // uint 16
|
||||
{
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCE: // uint 32
|
||||
{
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCF: // uint 64
|
||||
{
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD0: // int 8
|
||||
{
|
||||
std::int8_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD1: // int 16
|
||||
{
|
||||
std::int16_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD2: // int 32
|
||||
{
|
||||
std::int32_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD3: // int 64
|
||||
{
|
||||
std::int64_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xDC: // array 16
|
||||
@@ -15724,7 +15694,7 @@ class binary_reader
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->number_unsigned(static_cast<number_unsigned_t>(i))))
|
||||
if (JSON_HEDLEY_UNLIKELY(!emit_unsigned(input_format, i)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -15856,37 +15826,37 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'U':
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'i':
|
||||
{
|
||||
std::int8_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'I':
|
||||
{
|
||||
std::int16_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'l':
|
||||
{
|
||||
std::int32_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'L':
|
||||
{
|
||||
std::int64_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'u':
|
||||
@@ -15896,7 +15866,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'm':
|
||||
@@ -15906,7 +15876,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'M':
|
||||
@@ -15916,7 +15886,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'h':
|
||||
@@ -15974,13 +15944,13 @@ class binary_reader
|
||||
case 'd':
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format, number) && emit_float(input_format, number);
|
||||
}
|
||||
|
||||
case 'D':
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format, number) && emit_float(input_format, number);
|
||||
}
|
||||
|
||||
case 'H':
|
||||
@@ -16447,13 +16417,13 @@ class binary_reader
|
||||
case 0x8E: // binary32
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::bon8, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::bon8, number) && emit_float(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
case 0x8F: // binary64
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::bon8, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::bon8, number) && emit_float(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
case 0xF8:
|
||||
@@ -16519,7 +16489,9 @@ class binary_reader
|
||||
@brief pass an integer to the SAX parser
|
||||
|
||||
Non-negative integers are passed as unsigned, negative integers as signed
|
||||
numbers, like the other binary formats do.
|
||||
numbers, like the other binary formats do. A value that does not fit the
|
||||
number type is passed as described for @ref emit_unsigned and
|
||||
@ref emit_signed.
|
||||
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
@@ -16528,9 +16500,9 @@ class binary_reader
|
||||
{
|
||||
if (number >= 0)
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
return emit_unsigned(input_format_t::bon8, static_cast<std::uint64_t>(number));
|
||||
}
|
||||
return sax->number_integer(static_cast<number_integer_t>(number));
|
||||
return emit_signed(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -16587,8 +16559,7 @@ class binary_reader
|
||||
value = (value << 8) | static_cast<std::int64_t>(current);
|
||||
}
|
||||
|
||||
return negative ? sax->number_integer(static_cast<number_integer_t>(-(value + offset)))
|
||||
: sax->number_unsigned(static_cast<number_unsigned_t>(value + offset));
|
||||
return emit_bon8_integer(negative ? -(value + offset) : value + offset);
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -16885,6 +16856,88 @@ class binary_reader
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a signed integer read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a value that does not fit into
|
||||
number_integer_t is passed as number_unsigned_t if it is non-negative and
|
||||
fits there, and as number_float_t otherwise. With the default number
|
||||
types, every integer the binary formats can encode fits, so this only
|
||||
matters for narrower custom number types.
|
||||
|
||||
@tparam NumberType a signed integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if @a number overflows number_float_t (see
|
||||
@ref emit_float)
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_signed(const input_format_t format, const NumberType number)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_integer_t>(number)))
|
||||
{
|
||||
return sax->number_integer(static_cast<number_integer_t>(number));
|
||||
}
|
||||
if (value_in_range_of<number_unsigned_t>(number))
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_float(format, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass an unsigned integer read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a value that does not fit into
|
||||
number_unsigned_t is passed as number_float_t.
|
||||
|
||||
@tparam NumberType an unsigned integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if @a number overflows number_float_t (see
|
||||
@ref emit_float)
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_unsigned(const input_format_t format, const NumberType number)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_unsigned_t>(number)))
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_float(format, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a floating-point number read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a finite value that overflows
|
||||
number_float_t is rejected instead of silently becoming infinity. Infinity
|
||||
and NaN in the input are passed on unchanged. Integers only overflow if
|
||||
number_float_t cannot represent 2^64, e.g., a half-precision type.
|
||||
|
||||
@tparam NumberType a floating-point or integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the number
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if a finite @a number overflows number_float_t
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_float(const input_format_t format, const NumberType number)
|
||||
{
|
||||
const auto result = static_cast<number_float_t>(number);
|
||||
if (JSON_HEDLEY_UNLIKELY(std::isfinite(number) && !std::isfinite(result)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
out_of_range::create(406, exception_message(format, "number overflow", "value"), nullptr));
|
||||
}
|
||||
return sax->number_float(result, "");
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief create a string by reading characters from the input
|
||||
|
||||
|
||||
@@ -11,7 +11,12 @@
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <cmath>
|
||||
#include <fstream>
|
||||
#include <limits>
|
||||
#include <map>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include "make_test_data_available.hpp"
|
||||
|
||||
TEST_CASE("Binary Formats" * doctest::skip())
|
||||
@@ -224,3 +229,139 @@ TEST_CASE("Binary Formats" * doctest::skip())
|
||||
CHECK((100.0 * double(ubjson_3_size) / double(json_size)) == Approx(89.450));
|
||||
}
|
||||
}
|
||||
|
||||
namespace
|
||||
{
|
||||
// the binary formats as function pointers for "Binary formats with narrow number types";
|
||||
// named functions rather than lambdas, because clang 3.5 cannot convert a lambda
|
||||
// to a function pointer in the braced initializer of the format table
|
||||
using narrow_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int32_t, std::uint32_t, float>;
|
||||
using bytes = std::vector<std::uint8_t>;
|
||||
|
||||
bytes encode_cbor(const json& j)
|
||||
{
|
||||
return json::to_cbor(j);
|
||||
}
|
||||
narrow_json decode_cbor(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_cbor(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
bytes encode_msgpack(const json& j)
|
||||
{
|
||||
return json::to_msgpack(j);
|
||||
}
|
||||
narrow_json decode_msgpack(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_msgpack(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
bytes encode_ubjson(const json& j)
|
||||
{
|
||||
return json::to_ubjson(j);
|
||||
}
|
||||
narrow_json decode_ubjson(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_ubjson(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
bytes encode_bjdata(const json& j)
|
||||
{
|
||||
return json::to_bjdata(j);
|
||||
}
|
||||
narrow_json decode_bjdata(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_bjdata(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
// BSON can only store numbers as object members
|
||||
bytes encode_bson(const json& j)
|
||||
{
|
||||
return json::to_bson(json{{"a", j}});
|
||||
}
|
||||
narrow_json decode_bson(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
const auto result = narrow_json::from_bson(v, true, allow_exceptions);
|
||||
return result.is_discarded() ? result : result.at("a");
|
||||
}
|
||||
|
||||
bytes encode_bon8(const json& j)
|
||||
{
|
||||
return json::to_bon8(j);
|
||||
}
|
||||
narrow_json decode_bon8(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_bon8(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("Binary formats with narrow number types")
|
||||
{
|
||||
// Numbers that do not fit the number types are handled like the lexer
|
||||
// handles them in JSON text: an integer that fits neither integer type is
|
||||
// stored as a floating-point number, and a finite floating-point number
|
||||
// that overflows number_float_t is rejected with out_of_range.406.
|
||||
struct binary_format
|
||||
{
|
||||
const char* name;
|
||||
bytes (*encode)(const json&);
|
||||
narrow_json (*decode)(const bytes&, bool);
|
||||
};
|
||||
|
||||
const std::vector<binary_format> formats =
|
||||
{
|
||||
{"CBOR", encode_cbor, decode_cbor},
|
||||
{"MessagePack", encode_msgpack, decode_msgpack},
|
||||
{"UBJSON", encode_ubjson, decode_ubjson},
|
||||
{"BJData", encode_bjdata, decode_bjdata},
|
||||
{"BSON", encode_bson, decode_bson},
|
||||
{"BON8", encode_bon8, decode_bon8},
|
||||
};
|
||||
|
||||
for (const auto& format : formats)
|
||||
{
|
||||
const std::string name = format.name;
|
||||
INFO("format := ", name);
|
||||
const auto roundtrip = [&format](const json & j)
|
||||
{
|
||||
return format.decode(format.encode(j), true);
|
||||
};
|
||||
|
||||
// integers that fit keep their type
|
||||
CHECK(roundtrip(json(-5)).is_number_integer());
|
||||
CHECK(roundtrip(json(-5)).get<std::int32_t>() == -5);
|
||||
CHECK(roundtrip(json(3000000000u)).is_number_unsigned());
|
||||
CHECK(roundtrip(json(3000000000u)).get<std::uint32_t>() == 3000000000u);
|
||||
|
||||
// integers that fit neither integer type are stored as float
|
||||
CHECK(roundtrip(json(5000000000u)).is_number_float());
|
||||
CHECK(roundtrip(json(5000000000u)).get<float>() == 5000000000.0f);
|
||||
if (name != "BON8") // BON8 cannot encode integers above INT64_MAX
|
||||
{
|
||||
CHECK(roundtrip(json(10000000000000000000u)).is_number_float());
|
||||
CHECK(roundtrip(json(10000000000000000000u)).get<float>() == 10000000000000000000.0f);
|
||||
}
|
||||
CHECK(roundtrip(json(-3000000000LL)).is_number_float());
|
||||
CHECK(roundtrip(json(-3000000000LL)).get<float>() == -3000000000.0f);
|
||||
CHECK(roundtrip(json(-5000000000LL)).is_number_float());
|
||||
CHECK(roundtrip(json(-5000000000LL)).get<float>() == -5000000000.0f);
|
||||
|
||||
// floating-point numbers that fit
|
||||
CHECK(roundtrip(json(1.5)).get<float>() == 1.5f);
|
||||
const auto just_above_max = std::nextafter(static_cast<double>((std::numeric_limits<float>::max)()),
|
||||
std::numeric_limits<double>::infinity());
|
||||
CHECK(roundtrip(json(just_above_max)).get<float>() == (std::numeric_limits<float>::max)());
|
||||
|
||||
// infinity and NaN are passed on
|
||||
CHECK(std::isinf(roundtrip(json(std::numeric_limits<double>::infinity())).get<float>()));
|
||||
CHECK(std::isnan(roundtrip(json(std::numeric_limits<double>::quiet_NaN())).get<float>()));
|
||||
|
||||
// finite floating-point numbers that overflow number_float_t are rejected
|
||||
const std::string message = "[json.exception.out_of_range.406] syntax error while parsing " + name
|
||||
+ " value: number overflow";
|
||||
CHECK_THROWS_WITH_AS(roundtrip(json(1e300)), message.c_str(), narrow_json::out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(roundtrip(json(-1e300)), message.c_str(), narrow_json::out_of_range&);
|
||||
CHECK(format.decode(format.encode(json(1e300)), false).is_discarded());
|
||||
}
|
||||
}
|
||||
|
||||
+16
-14
@@ -3187,7 +3187,8 @@ TEST_CASE("Tagged values")
|
||||
// CBOR encodes negative integers as: result = -1 - n
|
||||
// For type 0x3B, n is an 8-byte uint64_t. Valid range for n with
|
||||
// the default int64_t is [0, INT64_MAX], producing results in [INT64_MIN, -1].
|
||||
// When n > INT64_MAX, the result exceeds int64_t range and is rejected.
|
||||
// When n > INT64_MAX, the result exceeds int64_t range and is stored
|
||||
// as a floating-point number, as the lexer does for JSON text.
|
||||
|
||||
SECTION("n = 0 is valid (result = -1)")
|
||||
{
|
||||
@@ -3208,33 +3209,34 @@ TEST_CASE("Tagged values")
|
||||
CHECK(result.get<int64_t>() == (std::numeric_limits<int64_t>::min)());
|
||||
}
|
||||
|
||||
SECTION("n = INT64_MAX + 1 is rejected (overflow)")
|
||||
SECTION("n = INT64_MAX + 1 is stored as float")
|
||||
{
|
||||
// n = INT64_MAX + 1 (0x8000000000000000)
|
||||
// result = -1 - n = -9223372036854775809, which exceeds int64_t range
|
||||
// result = -1 - n = -9223372036854775809, which exceeds int64_t range;
|
||||
// the nearest double is -9223372036854775808.0
|
||||
const std::vector<uint8_t> input = {0x3B, 0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00};
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input),
|
||||
"[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow",
|
||||
json::parse_error);
|
||||
const auto result = json::from_cbor(input);
|
||||
CHECK(result.is_number_float());
|
||||
CHECK(result.get<double>() == -9223372036854775808.0);
|
||||
CHECK(result == json::parse("-9223372036854775809"));
|
||||
}
|
||||
|
||||
SECTION("n = UINT64_MAX is rejected (overflow)")
|
||||
SECTION("n = UINT64_MAX is stored as float")
|
||||
{
|
||||
// n = UINT64_MAX (0xFFFFFFFFFFFFFFFF)
|
||||
// result = -1 - n = -18446744073709551616, which exceeds int64_t range
|
||||
const std::vector<uint8_t> input = {0x3B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF};
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input),
|
||||
"[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow",
|
||||
json::parse_error);
|
||||
const auto result = json::from_cbor(input);
|
||||
CHECK(result.is_number_float());
|
||||
CHECK(result.get<double>() == -18446744073709551616.0);
|
||||
CHECK(result == json::parse("-18446744073709551616"));
|
||||
}
|
||||
|
||||
SECTION("overflow with allow_exceptions=false returns discarded")
|
||||
SECTION("overflow with allow_exceptions=false is not an error")
|
||||
{
|
||||
const std::vector<uint8_t> input = {0x3B, 0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00};
|
||||
const auto result = json::from_cbor(input, true, false);
|
||||
CHECK(result.is_discarded());
|
||||
CHECK(result.is_number_float());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -8,6 +8,3 @@ The following changes have been made to the code with respect to <https://github
|
||||
- membership check
|
||||
- made function from `_is_within`
|
||||
- removed unused variable `actual_path`
|
||||
- Added the optional config key `external`: include paths listed there are kept as
|
||||
`#include` directives instead of being inlined (the first directive per path; the
|
||||
repeated ones are commented out).
|
||||
|
||||
@@ -57,11 +57,6 @@ Python v.2.7.0 or higher is required.
|
||||
amalgamation. Have a look at `test/source.c.json` and `test/include.h.json`
|
||||
to see two examples.
|
||||
|
||||
The optional `external` list names include paths that are kept as `#include`
|
||||
directives instead of being inlined, e.g. `["nlohmann/json.hpp"]` for a header
|
||||
that includes another amalgamated header. Only the first directive for each
|
||||
of these paths is kept; the repeated ones are commented out.
|
||||
|
||||
* The `-s, --source` option should specify the path to the source directory.
|
||||
This is useful for supporting separate source and build directories.
|
||||
|
||||
|
||||
@@ -62,10 +62,6 @@ class Amalgamation(object):
|
||||
return None
|
||||
|
||||
def __init__(self, args):
|
||||
# include paths that are kept as #include directives instead of
|
||||
# being inlined (e.g. a header amalgamated on its own)
|
||||
self.external = []
|
||||
self.included_external = []
|
||||
with open(args.config, 'r') as f:
|
||||
config = json.loads(f.read())
|
||||
for key in config:
|
||||
@@ -224,14 +220,11 @@ class TranslationUnit(object):
|
||||
while include_match:
|
||||
if not _is_within(include_match, skippable_contexts):
|
||||
include_path = include_match.group("path")
|
||||
if include_path in self.amalgamation.external:
|
||||
includes.append((include_match, None))
|
||||
else:
|
||||
search_same_dir = include_match.group(1) == '"'
|
||||
found_included_path = self.amalgamation.find_included_file(
|
||||
include_path, self.file_dir if search_same_dir else None)
|
||||
if found_included_path:
|
||||
includes.append((include_match, found_included_path))
|
||||
search_same_dir = include_match.group(1) == '"'
|
||||
found_included_path = self.amalgamation.find_included_file(
|
||||
include_path, self.file_dir if search_same_dir else None)
|
||||
if found_included_path:
|
||||
includes.append((include_match, found_included_path))
|
||||
|
||||
include_match = self.include_pattern.search(self.content,
|
||||
include_match.end())
|
||||
@@ -242,17 +235,6 @@ class TranslationUnit(object):
|
||||
for include in includes:
|
||||
include_match, found_included_path = include
|
||||
tmp_content += self.content[prev_end:include_match.start()]
|
||||
if found_included_path is None:
|
||||
# an external header: keep the first directive and comment
|
||||
# out the repeated ones
|
||||
include_path = include_match.group("path")
|
||||
if include_path in self.amalgamation.included_external:
|
||||
tmp_content += "// {0}".format(include_match.group(0))
|
||||
else:
|
||||
self.amalgamation.included_external.append(include_path)
|
||||
tmp_content += include_match.group(0)
|
||||
prev_end = include_match.end()
|
||||
continue
|
||||
tmp_content += "// {0}\n".format(include_match.group(0))
|
||||
if found_included_path not in self.amalgamation.included_files:
|
||||
t = TranslationUnit(found_included_path, self.amalgamation, False)
|
||||
|
||||
@@ -1,9 +0,0 @@
|
||||
{
|
||||
"project": "JSON for Modern C++",
|
||||
"target": "single_include/nlohmann/json_view.hpp",
|
||||
"sources": [
|
||||
"include/nlohmann/json_view.hpp"
|
||||
],
|
||||
"include_paths": ["include"],
|
||||
"external": ["nlohmann/json.hpp"]
|
||||
}
|
||||
Reference in New Issue
Block a user