mirror of
https://github.com/nlohmann/json.git
synced 2026-10-03 05:00:30 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9505be15fd | ||
|
|
44325873ea | ||
|
|
0c630d4c30 | ||
|
|
04ed4f593c |
@@ -55,6 +55,10 @@ This implementation does exactly follow this approach, as it uses double precisi
|
||||
smaller than `-1.79769313486232e+308` and values greater than `1.79769313486232e+308` will be stored as NaN internally
|
||||
and be serialized to `null`.
|
||||
|
||||
During deserialization (from JSON text or any of the binary formats), a finite number that does not fit into
|
||||
`number_float_t` is rejected with [`out_of_range.406`](../../home/exceptions.md#jsonexceptionout_of_range406), for
|
||||
example a double-precision number in a binary format when `number_float_t` is `#!cpp float`.
|
||||
|
||||
#### Storage
|
||||
|
||||
Floating-point number values are stored directly inside a `basic_json` type.
|
||||
|
||||
@@ -47,8 +47,9 @@ With the default values for `NumberIntegerType` (`std::int64_t`), the default va
|
||||
|
||||
When the default type is used, the maximal integer number that can be stored is `9223372036854775807` (INT64_MAX) and
|
||||
the minimal integer number that can be stored is `-9223372036854775808` (INT64_MIN). Integer numbers that are out of
|
||||
range will yield over/underflow when used in a constructor. During deserialization, too large or small integer numbers
|
||||
will automatically be stored as [`number_unsigned_t`](number_unsigned_t.md) or [`number_float_t`](number_float_t.md).
|
||||
range will yield over/underflow when used in a constructor. During deserialization (from JSON text or any of the binary
|
||||
formats), too large or small integer numbers will automatically be stored as [`number_unsigned_t`](number_unsigned_t.md)
|
||||
or [`number_float_t`](number_float_t.md).
|
||||
|
||||
[RFC 8259](https://tools.ietf.org/html/rfc8259) further states:
|
||||
> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are
|
||||
|
||||
@@ -48,8 +48,9 @@ With the default values for `NumberUnsignedType` (`std::uint64_t`), the default
|
||||
|
||||
When the default type is used, the maximal integer number that can be stored is `18446744073709551615` (UINT64_MAX) and
|
||||
the minimal integer number that can be stored is `0`. Integer numbers that are out of range will yield over/underflow
|
||||
when used in a constructor. During deserialization, too large or small integer numbers will automatically be stored
|
||||
as [`number_integer_t`](number_integer_t.md) or [`number_float_t`](number_float_t.md).
|
||||
when used in a constructor. During deserialization (from JSON text or any of the binary formats), too large or small
|
||||
integer numbers will automatically be stored as [`number_integer_t`](number_integer_t.md) or
|
||||
[`number_float_t`](number_float_t.md).
|
||||
|
||||
[RFC 8259](https://tools.ietf.org/html/rfc8259) further states:
|
||||
> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are
|
||||
|
||||
@@ -254,8 +254,6 @@ outside of a string, invalid) byte; see the [FAQ entry](../../home/faq.md#nul-by
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
- `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating
|
||||
it as end of input; planned to become the default in version 4.0.0.
|
||||
- The result of converting floating-point numbers no longer depends on the C locale in version 3.13.0; before, a
|
||||
locale whose decimal point is longer than one byte (e.g., `fa_IR.UTF-8`) truncated them at the decimal point.
|
||||
|
||||
!!! warning "Deprecation"
|
||||
|
||||
|
||||
@@ -168,9 +168,9 @@ The library maps CBOR types to JSON value types as follows:
|
||||
!!! warning "Negative integer overflow"
|
||||
|
||||
CBOR negative integers (major type 1) are decoded as `-1 - n`. If the encoded magnitude `n` is too large for the
|
||||
result to fit into `number_integer_t` (`std::int64_t` by default), parsing fails with a
|
||||
[`parse_error.112`](../../home/exceptions.md#jsonexceptionparse_error112) exception rather than overflowing
|
||||
silently.
|
||||
result to fit into `number_integer_t` (`std::int64_t` by default), the result is stored as `number_float_t`, like
|
||||
a too small integer in JSON text. For example, `-18446744073709551616` (`0x3B` followed by eight `0xFF` bytes) is
|
||||
stored as `-1.8446744073709552e+19`.
|
||||
|
||||
!!! warning "Object keys"
|
||||
|
||||
|
||||
@@ -75,13 +75,6 @@ otherwise, it uses unsigned integer storage.
|
||||
[`std::strtoull`](https://en.cppreference.com/w/cpp/string/byte/strtoul),
|
||||
[`std::strtoll`](https://en.cppreference.com/w/cpp/string/byte/strtol), and
|
||||
[`std::strtod`](https://en.cppreference.com/w/cpp/string/byte/strtof), respectively.
|
||||
- The result of converting floating-point numbers does not depend on the C locale (`LC_NUMERIC`). They are
|
||||
converted with [`std::from_chars`](https://en.cppreference.com/w/cpp/utility/from_chars) where the standard
|
||||
library implements it for the number type (with libc++ 20 or later, only for `#!c float` and `#!c double`, and
|
||||
only where `strtod_l` is unavailable, because that is faster), otherwise with `strtod_l` and the "C" locale where
|
||||
the C library provides it (glibc, macOS, MSVC), and otherwise with `std::strtod` and the decimal point of the
|
||||
current locale. Before version 3.13.0, the last way was used much more often, and a locale whose decimal point
|
||||
is longer than one byte (e.g., `fa_IR.UTF-8`) truncated numbers at the decimal point.
|
||||
|
||||
!!! example "Examples"
|
||||
|
||||
@@ -92,11 +85,10 @@ otherwise, it uses unsigned integer storage.
|
||||
### Number limits
|
||||
|
||||
- Any 64-bit signed or unsigned integer can be stored without loss of precision.
|
||||
- Numbers exceeding the limits of `#!c double` (i.e., numbers that after conversion are not satisfying
|
||||
- Numbers exceeding the limits of `#!c double` (i.e., numbers that after conversion via
|
||||
[`std::strtod`](https://en.cppreference.com/w/cpp/string/byte/strtof) are not satisfying
|
||||
[`std::isfinite`](https://en.cppreference.com/w/cpp/numeric/math/isfinite) such as `#!c 1E400`) will throw exception
|
||||
[`json.exception.out_of_range.406`](../../home/exceptions.md#jsonexceptionout_of_range406) during parsing.
|
||||
- Numbers too close to zero to be represented as `#!c double`, not even as subnormal number (such as `#!c 1E-400`), are
|
||||
stored as `#!c 0.0`, or as `#!c -0.0` if they are negative.
|
||||
- Floating-point numbers are rounded to the next number representable as `double`. For instance
|
||||
`#!c 3.141592653589793238462643383279` is stored as [`0x400921fb54442d18`](https://float.exposed/0x400921fb54442d18).
|
||||
This is the same behavior as the code `#!c double x = 3.141592653589793238462643383279;`.
|
||||
|
||||
@@ -331,9 +331,6 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde
|
||||
[json.exception.parse_error.112] parse error at byte 15: syntax error while parsing BSON binary: byte array length cannot be negative, is -1
|
||||
```
|
||||
```
|
||||
[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow
|
||||
```
|
||||
```
|
||||
[json.exception.parse_error.112] parse error at byte 5: syntax error while parsing BSON document: document size 6 does not match the number of bytes read (5)
|
||||
```
|
||||
|
||||
@@ -854,13 +851,18 @@ The JSON Patch operations 'remove' and 'add' cannot be applied to the root eleme
|
||||
|
||||
### json.exception.out_of_range.406
|
||||
|
||||
A parsed number could not be stored as without changing it to NaN or INF.
|
||||
A parsed number could not be stored without changing it to NaN or INF. For the binary formats, this happens when a
|
||||
finite floating-point number does not fit into [`number_float_t`](../api/basic_json/number_float_t.md), for example a
|
||||
double-precision number when `number_float_t` is `#!cpp float`.
|
||||
|
||||
!!! failure "Example message"
|
||||
!!! failure "Example messages"
|
||||
|
||||
```
|
||||
number overflow parsing '10E1000'
|
||||
```
|
||||
```
|
||||
[json.exception.out_of_range.406] syntax error while parsing CBOR value: number overflow
|
||||
```
|
||||
|
||||
### json.exception.out_of_range.407
|
||||
|
||||
|
||||
@@ -559,7 +559,7 @@ class binary_reader
|
||||
case 0x01: // double
|
||||
{
|
||||
double number{};
|
||||
return get_number<double, true>(input_format_t::bson, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number<double, true>(input_format_t::bson, number) && emit_float(input_format_t::bson, number);
|
||||
}
|
||||
|
||||
case 0x02: // string
|
||||
@@ -600,19 +600,19 @@ class binary_reader
|
||||
case 0x10: // int32
|
||||
{
|
||||
std::int32_t value{};
|
||||
return get_number<std::int32_t, true>(input_format_t::bson, value) && sax->number_integer(value);
|
||||
return get_number<std::int32_t, true>(input_format_t::bson, value) && emit_signed(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
case 0x12: // int64
|
||||
{
|
||||
std::int64_t value{};
|
||||
return get_number<std::int64_t, true>(input_format_t::bson, value) && sax->number_integer(value);
|
||||
return get_number<std::int64_t, true>(input_format_t::bson, value) && emit_signed(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
case 0x11: // uint64
|
||||
{
|
||||
std::uint64_t value{};
|
||||
return get_number<std::uint64_t, true>(input_format_t::bson, value) && sax->number_unsigned(value);
|
||||
return get_number<std::uint64_t, true>(input_format_t::bson, value) && emit_unsigned(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
default: // anything else is not supported (yet)
|
||||
@@ -638,14 +638,19 @@ class binary_reader
|
||||
{
|
||||
return false;
|
||||
}
|
||||
const auto max_val = static_cast<NumberType>((std::numeric_limits<number_integer_t>::max)());
|
||||
if (number > max_val)
|
||||
|
||||
// the value is -1 - number, which fits into number_integer_t
|
||||
// whenever number does
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_integer_t>(number)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
parse_error::create(112, chars_read,
|
||||
exception_message(input_format_t::cbor, "negative integer overflow", "value"), nullptr));
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
}
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
|
||||
// like the lexer does for JSON text, store a value too small for
|
||||
// number_integer_t as number_float_t; compute it as long double so
|
||||
// that emit_float sees a finite value and can detect an overflow of
|
||||
// number_float_t
|
||||
return emit_float(input_format_t::cbor, static_cast<long double>(-1) - static_cast<long double>(number));
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -702,25 +707,25 @@ class binary_reader
|
||||
case 0x18: // Unsigned integer (one-byte uint8_t follows)
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x19: // Unsigned integer (two-byte uint16_t follows)
|
||||
{
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x1A: // Unsigned integer (four-byte uint32_t follows)
|
||||
{
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x1B: // Unsigned integer (eight-byte uint64_t follows)
|
||||
{
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
// Negative integer -1-0x00..-1-0x17 (-1..-24)
|
||||
@@ -1165,13 +1170,13 @@ class binary_reader
|
||||
case 0xFA: // Single-Precision Float (four-byte IEEE 754)
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::cbor, number) && emit_float(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0xFB: // Double-Precision Float (eight-byte IEEE 754)
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::cbor, number) && emit_float(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
default: // anything else (0xFF is handled inside the other types)
|
||||
@@ -1935,61 +1940,61 @@ class binary_reader
|
||||
case 0xCA: // float 32
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::msgpack, number) && emit_float(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCB: // float 64
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::msgpack, number) && emit_float(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCC: // uint 8
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCD: // uint 16
|
||||
{
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCE: // uint 32
|
||||
{
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCF: // uint 64
|
||||
{
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD0: // int 8
|
||||
{
|
||||
std::int8_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD1: // int 16
|
||||
{
|
||||
std::int16_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD2: // int 32
|
||||
{
|
||||
std::int32_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD3: // int 64
|
||||
{
|
||||
std::int64_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xDC: // array 16
|
||||
@@ -2922,7 +2927,7 @@ class binary_reader
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->number_unsigned(static_cast<number_unsigned_t>(i))))
|
||||
if (JSON_HEDLEY_UNLIKELY(!emit_unsigned(input_format, i)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -3054,37 +3059,37 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'U':
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'i':
|
||||
{
|
||||
std::int8_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'I':
|
||||
{
|
||||
std::int16_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'l':
|
||||
{
|
||||
std::int32_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'L':
|
||||
{
|
||||
std::int64_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'u':
|
||||
@@ -3094,7 +3099,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'm':
|
||||
@@ -3104,7 +3109,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'M':
|
||||
@@ -3114,7 +3119,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'h':
|
||||
@@ -3172,13 +3177,13 @@ class binary_reader
|
||||
case 'd':
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format, number) && emit_float(input_format, number);
|
||||
}
|
||||
|
||||
case 'D':
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format, number) && emit_float(input_format, number);
|
||||
}
|
||||
|
||||
case 'H':
|
||||
@@ -3645,13 +3650,13 @@ class binary_reader
|
||||
case 0x8E: // binary32
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::bon8, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::bon8, number) && emit_float(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
case 0x8F: // binary64
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::bon8, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::bon8, number) && emit_float(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
case 0xF8:
|
||||
@@ -3717,7 +3722,9 @@ class binary_reader
|
||||
@brief pass an integer to the SAX parser
|
||||
|
||||
Non-negative integers are passed as unsigned, negative integers as signed
|
||||
numbers, like the other binary formats do.
|
||||
numbers, like the other binary formats do. A value that does not fit the
|
||||
number type is passed as described for @ref emit_unsigned and
|
||||
@ref emit_signed.
|
||||
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
@@ -3726,9 +3733,9 @@ class binary_reader
|
||||
{
|
||||
if (number >= 0)
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
return emit_unsigned(input_format_t::bon8, static_cast<std::uint64_t>(number));
|
||||
}
|
||||
return sax->number_integer(static_cast<number_integer_t>(number));
|
||||
return emit_signed(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -3785,8 +3792,7 @@ class binary_reader
|
||||
value = (value << 8) | static_cast<std::int64_t>(current);
|
||||
}
|
||||
|
||||
return negative ? sax->number_integer(static_cast<number_integer_t>(-(value + offset)))
|
||||
: sax->number_unsigned(static_cast<number_unsigned_t>(value + offset));
|
||||
return emit_bon8_integer(negative ? -(value + offset) : value + offset);
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -4083,6 +4089,88 @@ class binary_reader
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a signed integer read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a value that does not fit into
|
||||
number_integer_t is passed as number_unsigned_t if it is non-negative and
|
||||
fits there, and as number_float_t otherwise. With the default number
|
||||
types, every integer the binary formats can encode fits, so this only
|
||||
matters for narrower custom number types.
|
||||
|
||||
@tparam NumberType a signed integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if @a number overflows number_float_t (see
|
||||
@ref emit_float)
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_signed(const input_format_t format, const NumberType number)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_integer_t>(number)))
|
||||
{
|
||||
return sax->number_integer(static_cast<number_integer_t>(number));
|
||||
}
|
||||
if (value_in_range_of<number_unsigned_t>(number))
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_float(format, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass an unsigned integer read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a value that does not fit into
|
||||
number_unsigned_t is passed as number_float_t.
|
||||
|
||||
@tparam NumberType an unsigned integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if @a number overflows number_float_t (see
|
||||
@ref emit_float)
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_unsigned(const input_format_t format, const NumberType number)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_unsigned_t>(number)))
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_float(format, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a floating-point number read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a finite value that overflows
|
||||
number_float_t is rejected instead of silently becoming infinity. Infinity
|
||||
and NaN in the input are passed on unchanged. Integers only overflow if
|
||||
number_float_t cannot represent 2^64, e.g., a half-precision type.
|
||||
|
||||
@tparam NumberType a floating-point or integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the number
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if a finite @a number overflows number_float_t
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_float(const input_format_t format, const NumberType number)
|
||||
{
|
||||
const auto result = static_cast<number_float_t>(number);
|
||||
if (JSON_HEDLEY_UNLIKELY(std::isfinite(number) && !std::isfinite(result)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
out_of_range::create(406, exception_message(format, "number overflow", "value"), nullptr));
|
||||
}
|
||||
return sax->number_float(result, "");
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief create a string by reading characters from the input
|
||||
|
||||
|
||||
@@ -9,8 +9,10 @@
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
#include <utility> // move
|
||||
@@ -215,6 +217,18 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
~lexer() = default;
|
||||
|
||||
private:
|
||||
/////////////////////
|
||||
// locales
|
||||
/////////////////////
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// scan functions
|
||||
/////////////////////
|
||||
@@ -1022,6 +1036,24 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan a number literal
|
||||
|
||||
@@ -1059,9 +1091,9 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
token_type::parse_error otherwise
|
||||
|
||||
@note The scanner is independent of the current locale: token_buffer
|
||||
always holds `.`. Only the last-resort std::strtod fallback of
|
||||
convert_number() depends on the locale, and it looks up the decimal
|
||||
point right before converting (see parse_float_locale_aware()).
|
||||
always holds `.`. Only the std::strtod fallback of convert_number()
|
||||
depends on the locale, and it looks up the decimal point right
|
||||
before converting (see convert_float_locale_aware()).
|
||||
*/
|
||||
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
|
||||
{
|
||||
@@ -1529,10 +1561,8 @@ scan_number_done:
|
||||
// this code is reached if we parse a floating-point number or if an
|
||||
// integer conversion above overflowed. Prefer std::from_chars
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise
|
||||
// strtof/strtod/strtold with the "C" locale where the C library offers
|
||||
// that; and only as a last resort strtof/strtod/strtold with the
|
||||
// decimal point of the current locale.
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod/strtold.
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
@@ -1545,15 +1575,65 @@ scan_number_done:
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
if (parse_float_c_locale(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
parse_float_locale_aware(token_buffer, decimal_point_position, value_float);
|
||||
convert_float_locale_aware();
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the float in token_buffer with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
*/
|
||||
void convert_float_locale_aware()
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// get_string() hands the token to the SAX interface with '.'
|
||||
token_buffer[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief contiguous fast path for scanning a number
|
||||
|
||||
|
||||
@@ -10,79 +10,27 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <clocale> // LC_NUMERIC, LC_NUMERIC_MASK, newlocale, _create_locale
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtof_l, strtod_l, strtold_l, _strtof_l, _strtod_l, _strtold_l
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
#include <utility> // move
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
// strtof_l/strtod_l/strtold_l convert with a given locale object instead of the
|
||||
// global C locale. They are not part of ISO C or C++, so they are only used where
|
||||
// the C library is known to declare them: Microsoft's UCRT (as _strtod_l etc.),
|
||||
// Apple's libc (in <xlocale.h>, which must follow <cstdlib>), and glibc (as GNU
|
||||
// extensions, visible because g++ and clang++ define _GNU_SOURCE for C++).
|
||||
// Everything else, e.g. MinGW (whose runtime lacks them), musl (which declares
|
||||
// only some of them), Android, or uClibc, uses parse_float_locale_aware().
|
||||
#if defined(_MSC_VER) && !defined(__MINGW32__) && _MSC_VER >= 1900
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 1
|
||||
#elif defined(__APPLE__)
|
||||
#include <xlocale.h> // newlocale, strtof_l, strtod_l, strtold_l
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 1
|
||||
#elif defined(__GLIBC__) && defined(__USE_GNU) && !defined(__UCLIBC__)
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 1
|
||||
#else
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 0
|
||||
#endif
|
||||
|
||||
// std::from_chars lives in <charconv>, but being in C++17 mode does not
|
||||
// guarantee the header exists: GCC 7 sets __cplusplus to C++17 yet ships no
|
||||
// <charconv> (added in GCC 8; floating-point support in GCC 11). Guard the
|
||||
// include with __has_include so such toolchains fall back to the scalar path.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__has_include)
|
||||
#if __has_include(<charconv>)
|
||||
#include <charconv> // from_chars
|
||||
#include <charconv> // from_chars (only used when __cpp_lib_to_chars is defined)
|
||||
#include <system_error> // errc
|
||||
|
||||
// std::from_chars is used for floating-point numbers
|
||||
// - for float, double, and long double if __cpp_lib_to_chars announces
|
||||
// complete support (only checked in C++17 or later: some standard
|
||||
// libraries, e.g. libstdc++ 15, define it even in C++14 mode, where
|
||||
// <charconv> is not included);
|
||||
// - for float and double with libc++ 20 or later, which does not define
|
||||
// __cpp_lib_to_chars because long double is missing, but only where
|
||||
// the C library offers no strtod_l: libc++'s implementation is slower
|
||||
// than Apple's strtod_l (by 1.3x to 2.8x per number), and it would be
|
||||
// tried before Clinger's fast path. On Apple platforms, it is also only
|
||||
// available when deploying to macOS/iOS 26 or later; for older
|
||||
// deployment targets, _LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT
|
||||
// is 0.
|
||||
#if defined(__cpp_lib_to_chars)
|
||||
#define JSON_HAS_FLOAT_FROM_CHARS 1
|
||||
#define JSON_HAS_LONG_DOUBLE_FROM_CHARS 1
|
||||
#elif !JSON_HAS_C_LOCALE_STRTOD && defined(_LIBCPP_VERSION) && defined(_LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT)
|
||||
#if _LIBCPP_VERSION >= 200000 && _LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT
|
||||
#define JSON_HAS_FLOAT_FROM_CHARS 1
|
||||
#endif
|
||||
#endif
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifndef JSON_HAS_FLOAT_FROM_CHARS
|
||||
#define JSON_HAS_FLOAT_FROM_CHARS 0
|
||||
#endif
|
||||
|
||||
#ifndef JSON_HAS_LONG_DOUBLE_FROM_CHARS
|
||||
#define JSON_HAS_LONG_DOUBLE_FROM_CHARS 0
|
||||
#endif
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, where possible without the
|
||||
// locale/errno overhead of std::strtoull/std::strtod. They are free functions so
|
||||
// the lexer stays focused on scanning; see lexer::convert_number().
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -315,128 +263,27 @@ bool parse_float_fast(const char* /*first*/, const char* /*last*/, FloatType& /*
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief derive the value of a number token that is out of range
|
||||
|
||||
The token [first, last) is a valid JSON number whose value cannot be
|
||||
represented by @a FloatType. The result follows from the token alone: a value
|
||||
of at least 1 can only overflow and becomes ±infinity (which the parser reports
|
||||
as out_of_range.406), a smaller one can only underflow and becomes ±0. The sign
|
||||
is taken from a leading '-', and the magnitude from the decimal exponent of the
|
||||
first nonzero digit.
|
||||
|
||||
A value slightly below the smallest normal number may still be representable
|
||||
as a subnormal number, which some implementations also report as out of range
|
||||
(libstdc++'s std::from_chars before GCC 13, which relies on the ERANGE of
|
||||
strtod for long double, and in GCC 11 for all types). Therefore ±0 is only
|
||||
returned if the value is below half the smallest subnormal number whatever its
|
||||
digits are.
|
||||
|
||||
@param[in] first pointer to the first character of the token
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] out ±infinity or ±0 on success
|
||||
@return true if @a out was set; false if the value may be a subnormal number,
|
||||
in which case the caller converts the token another way
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_out_of_range(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
const bool negative = first != last && *first == '-';
|
||||
const char* p = negative ? first + 1 : first;
|
||||
|
||||
// the decimal exponent of the first nonzero digit, from its position
|
||||
// relative to the decimal point
|
||||
std::int64_t exponent = 0;
|
||||
bool nonzero = false;
|
||||
for (; p != last && *p >= '0' && *p <= '9'; ++p)
|
||||
{
|
||||
if (nonzero)
|
||||
{
|
||||
++exponent;
|
||||
}
|
||||
else
|
||||
{
|
||||
nonzero = *p != '0';
|
||||
}
|
||||
}
|
||||
if (p != last && *p == '.')
|
||||
{
|
||||
for (++p; p != last && *p >= '0' && *p <= '9'; ++p)
|
||||
{
|
||||
if (!nonzero)
|
||||
{
|
||||
--exponent;
|
||||
nonzero = *p != '0';
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (nonzero && p != last && (*p == 'e' || *p == 'E'))
|
||||
{
|
||||
++p;
|
||||
const bool negative_exponent = p != last && *p == '-';
|
||||
if (p != last && (*p == '-' || *p == '+'))
|
||||
{
|
||||
++p;
|
||||
}
|
||||
// saturate: a larger exponent is far out of range for every type
|
||||
constexpr std::int64_t saturation = 100000000000000000; // 10^17
|
||||
std::int64_t explicit_exponent = 0;
|
||||
for (; p != last && *p >= '0' && *p <= '9'; ++p)
|
||||
{
|
||||
if (explicit_exponent < saturation)
|
||||
{
|
||||
explicit_exponent = (explicit_exponent * 10) + (*p - '0');
|
||||
}
|
||||
}
|
||||
exponent += negative_exponent ? -explicit_exponent : explicit_exponent;
|
||||
}
|
||||
|
||||
if (nonzero && exponent >= 0)
|
||||
{
|
||||
out = negative ? -std::numeric_limits<FloatType>::infinity() : std::numeric_limits<FloatType>::infinity();
|
||||
return true;
|
||||
}
|
||||
|
||||
// The value is below 10^(exponent + 1). It rounds to zero if that is at most
|
||||
// half the smallest subnormal number, 2^(min_exponent - digits - 1). The
|
||||
// bound rounds log10(2) up to 0.30103 and the product toward zero, and the
|
||||
// margin of 2 keeps it on the safe side.
|
||||
constexpr std::int64_t zero_exponent = (static_cast<std::int64_t>(std::numeric_limits<FloatType>::min_exponent - std::numeric_limits<FloatType>::digits - 1) * 30103 / 100000) - 2;
|
||||
if (!nonzero || exponent <= zero_exponent)
|
||||
{
|
||||
out = negative ? -FloatType(0) : FloatType(0);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse a float with std::from_chars (Eisel-Lemire) when available
|
||||
|
||||
std::from_chars is locale-independent, correctly rounded, and - via the
|
||||
Eisel-Lemire algorithm in modern standard libraries - much faster than strtod
|
||||
over the whole value range (not just the Clinger subset). It is used only where
|
||||
the standard library implements it for @a FloatType (see
|
||||
JSON_HAS_FLOAT_FROM_CHARS) and only when it consumes the entire token
|
||||
([first, last)).
|
||||
|
||||
For an under- or overflow (std::errc::result_out_of_range), implementations
|
||||
disagree on the value they store: libstdc++ leaves it unchanged, whereas libc++
|
||||
and the MSVC STL store ±0 or ±infinity (P4168). The result is therefore derived
|
||||
from the token, see parse_float_out_of_range().
|
||||
over the whole value range (not just the Clinger subset). It is used only when
|
||||
__cpp_lib_to_chars indicates full floating-point support and only when it
|
||||
consumes the entire token ([first, last)). An under-/overflow (result_out_of_range) also declines, so
|
||||
the caller's strtod fallback supplies the well-defined ±inf/0 result the parser
|
||||
expects (side-stepping the P4168 divergence between implementations).
|
||||
|
||||
@return true if the value was parsed exactly and fully; false to fall back
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
#if JSON_HAS_FLOAT_FROM_CHARS
|
||||
// JSON_HAS_CPP_17 must gate the use as well as the <charconv> include above:
|
||||
// some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even
|
||||
// in C++14 mode, where <charconv> is not included.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars)
|
||||
const auto result = std::from_chars(first, last, out);
|
||||
if (JSON_HEDLEY_UNLIKELY(result.ec == std::errc::result_out_of_range && result.ptr == last))
|
||||
{
|
||||
return parse_float_out_of_range(first, last, out);
|
||||
}
|
||||
return result.ec == std::errc() && result.ptr == last;
|
||||
#else
|
||||
static_cast<void>(first);
|
||||
@@ -446,200 +293,5 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out)
|
||||
#endif
|
||||
}
|
||||
|
||||
#if JSON_HAS_FLOAT_FROM_CHARS && !JSON_HAS_LONG_DOUBLE_FROM_CHARS
|
||||
/// libc++ implements std::from_chars for float and double, but not for long double
|
||||
inline bool parse_float_from_chars(const char* /*first*/, const char* /*last*/, long double& /*out*/) noexcept
|
||||
{
|
||||
return false;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if JSON_HAS_C_LOCALE_STRTOD
|
||||
#if defined(_MSC_VER)
|
||||
using c_locale_t = _locale_t;
|
||||
|
||||
/// the "C" locale for the numeric category, created on first use and never freed
|
||||
inline c_locale_t c_numeric_locale() noexcept
|
||||
{
|
||||
static const c_locale_t c_locale = _create_locale(LC_NUMERIC, "C");
|
||||
return c_locale;
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(float& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = _strtof_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = _strtod_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(long double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = _strtold_l(str, endptr, loc);
|
||||
}
|
||||
#else
|
||||
using c_locale_t = locale_t;
|
||||
|
||||
/// the "C" locale for the numeric category, created on first use and never freed
|
||||
inline c_locale_t c_numeric_locale() noexcept
|
||||
{
|
||||
static const c_locale_t c_locale = newlocale(LC_NUMERIC_MASK, "C", nullptr);
|
||||
return c_locale;
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(float& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = strtof_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = strtod_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(long double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = strtold_l(str, endptr, loc);
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/*!
|
||||
@brief parse a float with strtof_l/strtod_l/strtold_l in the "C" locale
|
||||
|
||||
These functions round correctly like strtod, but take the "C" locale as an
|
||||
argument instead of using the global one, so the decimal point is always '.'.
|
||||
The locale object is created on first use and never freed, so it remains valid
|
||||
for parsers that run during static destruction.
|
||||
|
||||
@param[in] first pointer to the first character of the token, which must be
|
||||
followed by a NUL character
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] out the parsed value (±infinity or ±0 if out of range)
|
||||
@return true if the value was parsed from the entire token; false if the C
|
||||
library offers no such functions (see JSON_HAS_C_LOCALE_STRTOD) or the
|
||||
locale could not be created, in which case the caller falls back to
|
||||
parse_float_locale_aware()
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_c_locale(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
#if JSON_HAS_C_LOCALE_STRTOD
|
||||
const c_locale_t loc = c_numeric_locale();
|
||||
if (JSON_HEDLEY_UNLIKELY(loc == nullptr))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness)
|
||||
strtof_c_locale(out, first, &endptr, loc);
|
||||
return endptr == last;
|
||||
#else
|
||||
static_cast<void>(first);
|
||||
static_cast<void>(last);
|
||||
static_cast<void>(out);
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_global_locale(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_global_locale(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_global_locale(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
inline std::string locale_decimal_point()
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr || *loc->decimal_point == '\0') ? "." : loc->decimal_point;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse a float with strtof/strtod/strtold in the current locale
|
||||
|
||||
This is the last resort for platforms without std::from_chars for @a FloatType
|
||||
and without parse_float_c_locale(). These functions expect the decimal point
|
||||
of the *current* locale, so the '.' in the token is replaced by it. It is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX handler,
|
||||
or another thread) must not truncate the value (#5198). A single-byte decimal
|
||||
point is substituted in place and restored afterwards, because the token is
|
||||
also handed to the SAX interface. A longer one (e.g., the two-byte U+066B of
|
||||
fa_IR.UTF-8 or ar_EG.UTF-8) is put into a copy of the token instead.
|
||||
|
||||
The token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the lookup
|
||||
and the call, and the conversion is repeated with the new decimal point. If it
|
||||
did not change, the value strtod parsed up to that point is kept.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
|
||||
@param[in,out] token the token, with '.' as decimal point
|
||||
@param[in] decimal_point_position the position of the '.' in @a token,
|
||||
or std::string::npos if it has none
|
||||
@param[out] out the parsed value
|
||||
*/
|
||||
template<typename StringType, typename FloatType>
|
||||
void parse_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& out)
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
std::string decimal_point = locale_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness)
|
||||
bool complete = false;
|
||||
if (!has_dot || decimal_point.size() == 1)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point[0] != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = static_cast<typename StringType::value_type>(decimal_point[0]);
|
||||
}
|
||||
strtof_global_locale(out, token.data(), &endptr);
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = '.';
|
||||
}
|
||||
complete = endptr == token.data() + token.size();
|
||||
}
|
||||
else
|
||||
{
|
||||
std::string copy(token.data(), token.size());
|
||||
copy.replace(decimal_point_position, 1, decimal_point);
|
||||
strtof_global_locale(out, copy.c_str(), &endptr);
|
||||
complete = endptr == copy.c_str() + copy.size();
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(complete))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
std::string current_decimal_point = locale_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = std::move(current_decimal_point);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -42,9 +42,6 @@
|
||||
#undef JSON_HAS_RANGES
|
||||
#undef JSON_HAS_STD_FORMAT
|
||||
#undef JSON_HAS_STATIC_RTTI
|
||||
#undef JSON_HAS_FLOAT_FROM_CHARS
|
||||
#undef JSON_HAS_LONG_DOUBLE_FROM_CHARS
|
||||
#undef JSON_HAS_C_LOCALE_STRTOD
|
||||
#undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
|
||||
#undef JSON_BRACE_INIT_COPY_SEMANTICS
|
||||
#undef JSON_PRECISE_STREAM_POSITION
|
||||
|
||||
+238
-421
@@ -8472,8 +8472,10 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
#include <utility> // move
|
||||
@@ -8494,80 +8496,28 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <clocale> // LC_NUMERIC, LC_NUMERIC_MASK, newlocale, _create_locale
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtof_l, strtod_l, strtold_l, _strtof_l, _strtod_l, _strtold_l
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
#include <utility> // move
|
||||
|
||||
// #include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
|
||||
// strtof_l/strtod_l/strtold_l convert with a given locale object instead of the
|
||||
// global C locale. They are not part of ISO C or C++, so they are only used where
|
||||
// the C library is known to declare them: Microsoft's UCRT (as _strtod_l etc.),
|
||||
// Apple's libc (in <xlocale.h>, which must follow <cstdlib>), and glibc (as GNU
|
||||
// extensions, visible because g++ and clang++ define _GNU_SOURCE for C++).
|
||||
// Everything else, e.g. MinGW (whose runtime lacks them), musl (which declares
|
||||
// only some of them), Android, or uClibc, uses parse_float_locale_aware().
|
||||
#if defined(_MSC_VER) && !defined(__MINGW32__) && _MSC_VER >= 1900
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 1
|
||||
#elif defined(__APPLE__)
|
||||
#include <xlocale.h> // newlocale, strtof_l, strtod_l, strtold_l
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 1
|
||||
#elif defined(__GLIBC__) && defined(__USE_GNU) && !defined(__UCLIBC__)
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 1
|
||||
#else
|
||||
#define JSON_HAS_C_LOCALE_STRTOD 0
|
||||
#endif
|
||||
|
||||
// std::from_chars lives in <charconv>, but being in C++17 mode does not
|
||||
// guarantee the header exists: GCC 7 sets __cplusplus to C++17 yet ships no
|
||||
// <charconv> (added in GCC 8; floating-point support in GCC 11). Guard the
|
||||
// include with __has_include so such toolchains fall back to the scalar path.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__has_include)
|
||||
#if __has_include(<charconv>)
|
||||
#include <charconv> // from_chars
|
||||
#include <charconv> // from_chars (only used when __cpp_lib_to_chars is defined)
|
||||
#include <system_error> // errc
|
||||
|
||||
// std::from_chars is used for floating-point numbers
|
||||
// - for float, double, and long double if __cpp_lib_to_chars announces
|
||||
// complete support (only checked in C++17 or later: some standard
|
||||
// libraries, e.g. libstdc++ 15, define it even in C++14 mode, where
|
||||
// <charconv> is not included);
|
||||
// - for float and double with libc++ 20 or later, which does not define
|
||||
// __cpp_lib_to_chars because long double is missing, but only where
|
||||
// the C library offers no strtod_l: libc++'s implementation is slower
|
||||
// than Apple's strtod_l (by 1.3x to 2.8x per number), and it would be
|
||||
// tried before Clinger's fast path. On Apple platforms, it is also only
|
||||
// available when deploying to macOS/iOS 26 or later; for older
|
||||
// deployment targets, _LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT
|
||||
// is 0.
|
||||
#if defined(__cpp_lib_to_chars)
|
||||
#define JSON_HAS_FLOAT_FROM_CHARS 1
|
||||
#define JSON_HAS_LONG_DOUBLE_FROM_CHARS 1
|
||||
#elif !JSON_HAS_C_LOCALE_STRTOD && defined(_LIBCPP_VERSION) && defined(_LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT)
|
||||
#if _LIBCPP_VERSION >= 200000 && _LIBCPP_AVAILABILITY_HAS_FROM_CHARS_FLOATING_POINT
|
||||
#define JSON_HAS_FLOAT_FROM_CHARS 1
|
||||
#endif
|
||||
#endif
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifndef JSON_HAS_FLOAT_FROM_CHARS
|
||||
#define JSON_HAS_FLOAT_FROM_CHARS 0
|
||||
#endif
|
||||
|
||||
#ifndef JSON_HAS_LONG_DOUBLE_FROM_CHARS
|
||||
#define JSON_HAS_LONG_DOUBLE_FROM_CHARS 0
|
||||
#endif
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, where possible without the
|
||||
// locale/errno overhead of std::strtoull/std::strtod. They are free functions so
|
||||
// the lexer stays focused on scanning; see lexer::convert_number().
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -8800,128 +8750,27 @@ bool parse_float_fast(const char* /*first*/, const char* /*last*/, FloatType& /*
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief derive the value of a number token that is out of range
|
||||
|
||||
The token [first, last) is a valid JSON number whose value cannot be
|
||||
represented by @a FloatType. The result follows from the token alone: a value
|
||||
of at least 1 can only overflow and becomes ±infinity (which the parser reports
|
||||
as out_of_range.406), a smaller one can only underflow and becomes ±0. The sign
|
||||
is taken from a leading '-', and the magnitude from the decimal exponent of the
|
||||
first nonzero digit.
|
||||
|
||||
A value slightly below the smallest normal number may still be representable
|
||||
as a subnormal number, which some implementations also report as out of range
|
||||
(libstdc++'s std::from_chars before GCC 13, which relies on the ERANGE of
|
||||
strtod for long double, and in GCC 11 for all types). Therefore ±0 is only
|
||||
returned if the value is below half the smallest subnormal number whatever its
|
||||
digits are.
|
||||
|
||||
@param[in] first pointer to the first character of the token
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] out ±infinity or ±0 on success
|
||||
@return true if @a out was set; false if the value may be a subnormal number,
|
||||
in which case the caller converts the token another way
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_out_of_range(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
const bool negative = first != last && *first == '-';
|
||||
const char* p = negative ? first + 1 : first;
|
||||
|
||||
// the decimal exponent of the first nonzero digit, from its position
|
||||
// relative to the decimal point
|
||||
std::int64_t exponent = 0;
|
||||
bool nonzero = false;
|
||||
for (; p != last && *p >= '0' && *p <= '9'; ++p)
|
||||
{
|
||||
if (nonzero)
|
||||
{
|
||||
++exponent;
|
||||
}
|
||||
else
|
||||
{
|
||||
nonzero = *p != '0';
|
||||
}
|
||||
}
|
||||
if (p != last && *p == '.')
|
||||
{
|
||||
for (++p; p != last && *p >= '0' && *p <= '9'; ++p)
|
||||
{
|
||||
if (!nonzero)
|
||||
{
|
||||
--exponent;
|
||||
nonzero = *p != '0';
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (nonzero && p != last && (*p == 'e' || *p == 'E'))
|
||||
{
|
||||
++p;
|
||||
const bool negative_exponent = p != last && *p == '-';
|
||||
if (p != last && (*p == '-' || *p == '+'))
|
||||
{
|
||||
++p;
|
||||
}
|
||||
// saturate: a larger exponent is far out of range for every type
|
||||
constexpr std::int64_t saturation = 100000000000000000; // 10^17
|
||||
std::int64_t explicit_exponent = 0;
|
||||
for (; p != last && *p >= '0' && *p <= '9'; ++p)
|
||||
{
|
||||
if (explicit_exponent < saturation)
|
||||
{
|
||||
explicit_exponent = (explicit_exponent * 10) + (*p - '0');
|
||||
}
|
||||
}
|
||||
exponent += negative_exponent ? -explicit_exponent : explicit_exponent;
|
||||
}
|
||||
|
||||
if (nonzero && exponent >= 0)
|
||||
{
|
||||
out = negative ? -std::numeric_limits<FloatType>::infinity() : std::numeric_limits<FloatType>::infinity();
|
||||
return true;
|
||||
}
|
||||
|
||||
// The value is below 10^(exponent + 1). It rounds to zero if that is at most
|
||||
// half the smallest subnormal number, 2^(min_exponent - digits - 1). The
|
||||
// bound rounds log10(2) up to 0.30103 and the product toward zero, and the
|
||||
// margin of 2 keeps it on the safe side.
|
||||
constexpr std::int64_t zero_exponent = (static_cast<std::int64_t>(std::numeric_limits<FloatType>::min_exponent - std::numeric_limits<FloatType>::digits - 1) * 30103 / 100000) - 2;
|
||||
if (!nonzero || exponent <= zero_exponent)
|
||||
{
|
||||
out = negative ? -FloatType(0) : FloatType(0);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse a float with std::from_chars (Eisel-Lemire) when available
|
||||
|
||||
std::from_chars is locale-independent, correctly rounded, and - via the
|
||||
Eisel-Lemire algorithm in modern standard libraries - much faster than strtod
|
||||
over the whole value range (not just the Clinger subset). It is used only where
|
||||
the standard library implements it for @a FloatType (see
|
||||
JSON_HAS_FLOAT_FROM_CHARS) and only when it consumes the entire token
|
||||
([first, last)).
|
||||
|
||||
For an under- or overflow (std::errc::result_out_of_range), implementations
|
||||
disagree on the value they store: libstdc++ leaves it unchanged, whereas libc++
|
||||
and the MSVC STL store ±0 or ±infinity (P4168). The result is therefore derived
|
||||
from the token, see parse_float_out_of_range().
|
||||
over the whole value range (not just the Clinger subset). It is used only when
|
||||
__cpp_lib_to_chars indicates full floating-point support and only when it
|
||||
consumes the entire token ([first, last)). An under-/overflow (result_out_of_range) also declines, so
|
||||
the caller's strtod fallback supplies the well-defined ±inf/0 result the parser
|
||||
expects (side-stepping the P4168 divergence between implementations).
|
||||
|
||||
@return true if the value was parsed exactly and fully; false to fall back
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
#if JSON_HAS_FLOAT_FROM_CHARS
|
||||
// JSON_HAS_CPP_17 must gate the use as well as the <charconv> include above:
|
||||
// some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even
|
||||
// in C++14 mode, where <charconv> is not included.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars)
|
||||
const auto result = std::from_chars(first, last, out);
|
||||
if (JSON_HEDLEY_UNLIKELY(result.ec == std::errc::result_out_of_range && result.ptr == last))
|
||||
{
|
||||
return parse_float_out_of_range(first, last, out);
|
||||
}
|
||||
return result.ec == std::errc() && result.ptr == last;
|
||||
#else
|
||||
static_cast<void>(first);
|
||||
@@ -8931,201 +8780,6 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out)
|
||||
#endif
|
||||
}
|
||||
|
||||
#if JSON_HAS_FLOAT_FROM_CHARS && !JSON_HAS_LONG_DOUBLE_FROM_CHARS
|
||||
/// libc++ implements std::from_chars for float and double, but not for long double
|
||||
inline bool parse_float_from_chars(const char* /*first*/, const char* /*last*/, long double& /*out*/) noexcept
|
||||
{
|
||||
return false;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if JSON_HAS_C_LOCALE_STRTOD
|
||||
#if defined(_MSC_VER)
|
||||
using c_locale_t = _locale_t;
|
||||
|
||||
/// the "C" locale for the numeric category, created on first use and never freed
|
||||
inline c_locale_t c_numeric_locale() noexcept
|
||||
{
|
||||
static const c_locale_t c_locale = _create_locale(LC_NUMERIC, "C");
|
||||
return c_locale;
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(float& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = _strtof_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = _strtod_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(long double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = _strtold_l(str, endptr, loc);
|
||||
}
|
||||
#else
|
||||
using c_locale_t = locale_t;
|
||||
|
||||
/// the "C" locale for the numeric category, created on first use and never freed
|
||||
inline c_locale_t c_numeric_locale() noexcept
|
||||
{
|
||||
static const c_locale_t c_locale = newlocale(LC_NUMERIC_MASK, "C", nullptr);
|
||||
return c_locale;
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(float& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = strtof_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = strtod_l(str, endptr, loc);
|
||||
}
|
||||
|
||||
inline void strtof_c_locale(long double& f, const char* str, char** endptr, c_locale_t loc) noexcept
|
||||
{
|
||||
f = strtold_l(str, endptr, loc);
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/*!
|
||||
@brief parse a float with strtof_l/strtod_l/strtold_l in the "C" locale
|
||||
|
||||
These functions round correctly like strtod, but take the "C" locale as an
|
||||
argument instead of using the global one, so the decimal point is always '.'.
|
||||
The locale object is created on first use and never freed, so it remains valid
|
||||
for parsers that run during static destruction.
|
||||
|
||||
@param[in] first pointer to the first character of the token, which must be
|
||||
followed by a NUL character
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] out the parsed value (±infinity or ±0 if out of range)
|
||||
@return true if the value was parsed from the entire token; false if the C
|
||||
library offers no such functions (see JSON_HAS_C_LOCALE_STRTOD) or the
|
||||
locale could not be created, in which case the caller falls back to
|
||||
parse_float_locale_aware()
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_c_locale(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
#if JSON_HAS_C_LOCALE_STRTOD
|
||||
const c_locale_t loc = c_numeric_locale();
|
||||
if (JSON_HEDLEY_UNLIKELY(loc == nullptr))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness)
|
||||
strtof_c_locale(out, first, &endptr, loc);
|
||||
return endptr == last;
|
||||
#else
|
||||
static_cast<void>(first);
|
||||
static_cast<void>(last);
|
||||
static_cast<void>(out);
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_global_locale(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_global_locale(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_global_locale(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
inline std::string locale_decimal_point()
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr || *loc->decimal_point == '\0') ? "." : loc->decimal_point;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse a float with strtof/strtod/strtold in the current locale
|
||||
|
||||
This is the last resort for platforms without std::from_chars for @a FloatType
|
||||
and without parse_float_c_locale(). These functions expect the decimal point
|
||||
of the *current* locale, so the '.' in the token is replaced by it. It is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX handler,
|
||||
or another thread) must not truncate the value (#5198). A single-byte decimal
|
||||
point is substituted in place and restored afterwards, because the token is
|
||||
also handed to the SAX interface. A longer one (e.g., the two-byte U+066B of
|
||||
fa_IR.UTF-8 or ar_EG.UTF-8) is put into a copy of the token instead.
|
||||
|
||||
The token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the lookup
|
||||
and the call, and the conversion is repeated with the new decimal point. If it
|
||||
did not change, the value strtod parsed up to that point is kept.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
|
||||
@param[in,out] token the token, with '.' as decimal point
|
||||
@param[in] decimal_point_position the position of the '.' in @a token,
|
||||
or std::string::npos if it has none
|
||||
@param[out] out the parsed value
|
||||
*/
|
||||
template<typename StringType, typename FloatType>
|
||||
void parse_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& out)
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
std::string decimal_point = locale_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness)
|
||||
bool complete = false;
|
||||
if (!has_dot || decimal_point.size() == 1)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point[0] != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = static_cast<typename StringType::value_type>(decimal_point[0]);
|
||||
}
|
||||
strtof_global_locale(out, token.data(), &endptr);
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = '.';
|
||||
}
|
||||
complete = endptr == token.data() + token.size();
|
||||
}
|
||||
else
|
||||
{
|
||||
std::string copy(token.data(), token.size());
|
||||
copy.replace(decimal_point_position, 1, decimal_point);
|
||||
strtof_global_locale(out, copy.c_str(), &endptr);
|
||||
complete = endptr == copy.c_str() + copy.size();
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(complete))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
std::string current_decimal_point = locale_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = std::move(current_decimal_point);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -9655,6 +9309,18 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
~lexer() = default;
|
||||
|
||||
private:
|
||||
/////////////////////
|
||||
// locales
|
||||
/////////////////////
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// scan functions
|
||||
/////////////////////
|
||||
@@ -10462,6 +10128,24 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan a number literal
|
||||
|
||||
@@ -10499,9 +10183,9 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
token_type::parse_error otherwise
|
||||
|
||||
@note The scanner is independent of the current locale: token_buffer
|
||||
always holds `.`. Only the last-resort std::strtod fallback of
|
||||
convert_number() depends on the locale, and it looks up the decimal
|
||||
point right before converting (see parse_float_locale_aware()).
|
||||
always holds `.`. Only the std::strtod fallback of convert_number()
|
||||
depends on the locale, and it looks up the decimal point right
|
||||
before converting (see convert_float_locale_aware()).
|
||||
*/
|
||||
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
|
||||
{
|
||||
@@ -10969,10 +10653,8 @@ scan_number_done:
|
||||
// this code is reached if we parse a floating-point number or if an
|
||||
// integer conversion above overflowed. Prefer std::from_chars
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise
|
||||
// strtof/strtod/strtold with the "C" locale where the C library offers
|
||||
// that; and only as a last resort strtof/strtod/strtold with the
|
||||
// decimal point of the current locale.
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod/strtold.
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
@@ -10985,15 +10667,65 @@ scan_number_done:
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
if (parse_float_c_locale(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
parse_float_locale_aware(token_buffer, decimal_point_position, value_float);
|
||||
convert_float_locale_aware();
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the float in token_buffer with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
*/
|
||||
void convert_float_locale_aware()
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// get_string() hands the token to the SAX interface with '.'
|
||||
token_buffer[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief contiguous fast path for scanning a number
|
||||
|
||||
@@ -13594,7 +13326,7 @@ class binary_reader
|
||||
case 0x01: // double
|
||||
{
|
||||
double number{};
|
||||
return get_number<double, true>(input_format_t::bson, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number<double, true>(input_format_t::bson, number) && emit_float(input_format_t::bson, number);
|
||||
}
|
||||
|
||||
case 0x02: // string
|
||||
@@ -13635,19 +13367,19 @@ class binary_reader
|
||||
case 0x10: // int32
|
||||
{
|
||||
std::int32_t value{};
|
||||
return get_number<std::int32_t, true>(input_format_t::bson, value) && sax->number_integer(value);
|
||||
return get_number<std::int32_t, true>(input_format_t::bson, value) && emit_signed(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
case 0x12: // int64
|
||||
{
|
||||
std::int64_t value{};
|
||||
return get_number<std::int64_t, true>(input_format_t::bson, value) && sax->number_integer(value);
|
||||
return get_number<std::int64_t, true>(input_format_t::bson, value) && emit_signed(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
case 0x11: // uint64
|
||||
{
|
||||
std::uint64_t value{};
|
||||
return get_number<std::uint64_t, true>(input_format_t::bson, value) && sax->number_unsigned(value);
|
||||
return get_number<std::uint64_t, true>(input_format_t::bson, value) && emit_unsigned(input_format_t::bson, value);
|
||||
}
|
||||
|
||||
default: // anything else is not supported (yet)
|
||||
@@ -13673,14 +13405,19 @@ class binary_reader
|
||||
{
|
||||
return false;
|
||||
}
|
||||
const auto max_val = static_cast<NumberType>((std::numeric_limits<number_integer_t>::max)());
|
||||
if (number > max_val)
|
||||
|
||||
// the value is -1 - number, which fits into number_integer_t
|
||||
// whenever number does
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_integer_t>(number)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
parse_error::create(112, chars_read,
|
||||
exception_message(input_format_t::cbor, "negative integer overflow", "value"), nullptr));
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
}
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
|
||||
// like the lexer does for JSON text, store a value too small for
|
||||
// number_integer_t as number_float_t; compute it as long double so
|
||||
// that emit_float sees a finite value and can detect an overflow of
|
||||
// number_float_t
|
||||
return emit_float(input_format_t::cbor, static_cast<long double>(-1) - static_cast<long double>(number));
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -13737,25 +13474,25 @@ class binary_reader
|
||||
case 0x18: // Unsigned integer (one-byte uint8_t follows)
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x19: // Unsigned integer (two-byte uint16_t follows)
|
||||
{
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x1A: // Unsigned integer (four-byte uint32_t follows)
|
||||
{
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0x1B: // Unsigned integer (eight-byte uint64_t follows)
|
||||
{
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
// Negative integer -1-0x00..-1-0x17 (-1..-24)
|
||||
@@ -14200,13 +13937,13 @@ class binary_reader
|
||||
case 0xFA: // Single-Precision Float (four-byte IEEE 754)
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::cbor, number) && emit_float(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
case 0xFB: // Double-Precision Float (eight-byte IEEE 754)
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::cbor, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::cbor, number) && emit_float(input_format_t::cbor, number);
|
||||
}
|
||||
|
||||
default: // anything else (0xFF is handled inside the other types)
|
||||
@@ -14970,61 +14707,61 @@ class binary_reader
|
||||
case 0xCA: // float 32
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::msgpack, number) && emit_float(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCB: // float 64
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::msgpack, number) && emit_float(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCC: // uint 8
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCD: // uint 16
|
||||
{
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCE: // uint 32
|
||||
{
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xCF: // uint 64
|
||||
{
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD0: // int 8
|
||||
{
|
||||
std::int8_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD1: // int 16
|
||||
{
|
||||
std::int16_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD2: // int 32
|
||||
{
|
||||
std::int32_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xD3: // int 64
|
||||
{
|
||||
std::int64_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
}
|
||||
|
||||
case 0xDC: // array 16
|
||||
@@ -15957,7 +15694,7 @@ class binary_reader
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->number_unsigned(static_cast<number_unsigned_t>(i))))
|
||||
if (JSON_HEDLEY_UNLIKELY(!emit_unsigned(input_format, i)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -16089,37 +15826,37 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'U':
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'i':
|
||||
{
|
||||
std::int8_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'I':
|
||||
{
|
||||
std::int16_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'l':
|
||||
{
|
||||
std::int32_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'L':
|
||||
{
|
||||
std::int64_t number{};
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
}
|
||||
|
||||
case 'u':
|
||||
@@ -16129,7 +15866,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'm':
|
||||
@@ -16139,7 +15876,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'M':
|
||||
@@ -16149,7 +15886,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
}
|
||||
|
||||
case 'h':
|
||||
@@ -16207,13 +15944,13 @@ class binary_reader
|
||||
case 'd':
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format, number) && emit_float(input_format, number);
|
||||
}
|
||||
|
||||
case 'D':
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format, number) && emit_float(input_format, number);
|
||||
}
|
||||
|
||||
case 'H':
|
||||
@@ -16680,13 +16417,13 @@ class binary_reader
|
||||
case 0x8E: // binary32
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::bon8, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::bon8, number) && emit_float(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
case 0x8F: // binary64
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::bon8, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
return get_number(input_format_t::bon8, number) && emit_float(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
case 0xF8:
|
||||
@@ -16752,7 +16489,9 @@ class binary_reader
|
||||
@brief pass an integer to the SAX parser
|
||||
|
||||
Non-negative integers are passed as unsigned, negative integers as signed
|
||||
numbers, like the other binary formats do.
|
||||
numbers, like the other binary formats do. A value that does not fit the
|
||||
number type is passed as described for @ref emit_unsigned and
|
||||
@ref emit_signed.
|
||||
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
@@ -16761,9 +16500,9 @@ class binary_reader
|
||||
{
|
||||
if (number >= 0)
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
return emit_unsigned(input_format_t::bon8, static_cast<std::uint64_t>(number));
|
||||
}
|
||||
return sax->number_integer(static_cast<number_integer_t>(number));
|
||||
return emit_signed(input_format_t::bon8, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -16820,8 +16559,7 @@ class binary_reader
|
||||
value = (value << 8) | static_cast<std::int64_t>(current);
|
||||
}
|
||||
|
||||
return negative ? sax->number_integer(static_cast<number_integer_t>(-(value + offset)))
|
||||
: sax->number_unsigned(static_cast<number_unsigned_t>(value + offset));
|
||||
return emit_bon8_integer(negative ? -(value + offset) : value + offset);
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -17118,6 +16856,88 @@ class binary_reader
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a signed integer read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a value that does not fit into
|
||||
number_integer_t is passed as number_unsigned_t if it is non-negative and
|
||||
fits there, and as number_float_t otherwise. With the default number
|
||||
types, every integer the binary formats can encode fits, so this only
|
||||
matters for narrower custom number types.
|
||||
|
||||
@tparam NumberType a signed integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if @a number overflows number_float_t (see
|
||||
@ref emit_float)
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_signed(const input_format_t format, const NumberType number)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_integer_t>(number)))
|
||||
{
|
||||
return sax->number_integer(static_cast<number_integer_t>(number));
|
||||
}
|
||||
if (value_in_range_of<number_unsigned_t>(number))
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_float(format, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass an unsigned integer read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a value that does not fit into
|
||||
number_unsigned_t is passed as number_float_t.
|
||||
|
||||
@tparam NumberType an unsigned integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if @a number overflows number_float_t (see
|
||||
@ref emit_float)
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_unsigned(const input_format_t format, const NumberType number)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_unsigned_t>(number)))
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_float(format, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a floating-point number read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a finite value that overflows
|
||||
number_float_t is rejected instead of silently becoming infinity. Infinity
|
||||
and NaN in the input are passed on unchanged. Integers only overflow if
|
||||
number_float_t cannot represent 2^64, e.g., a half-precision type.
|
||||
|
||||
@tparam NumberType a floating-point or integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the number
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if a finite @a number overflows number_float_t
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_float(const input_format_t format, const NumberType number)
|
||||
{
|
||||
const auto result = static_cast<number_float_t>(number);
|
||||
if (JSON_HEDLEY_UNLIKELY(std::isfinite(number) && !std::isfinite(result)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
out_of_range::create(406, exception_message(format, "number overflow", "value"), nullptr));
|
||||
}
|
||||
return sax->number_float(result, "");
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief create a string by reading characters from the input
|
||||
|
||||
@@ -33011,9 +32831,6 @@ struct formatter<nlohmann::NLOHMANN_BASIC_JSON_TPL, char> // NOLINT(cert-dcl58-c
|
||||
#undef JSON_HAS_RANGES
|
||||
#undef JSON_HAS_STD_FORMAT
|
||||
#undef JSON_HAS_STATIC_RTTI
|
||||
#undef JSON_HAS_FLOAT_FROM_CHARS
|
||||
#undef JSON_HAS_LONG_DOUBLE_FROM_CHARS
|
||||
#undef JSON_HAS_C_LOCALE_STRTOD
|
||||
#undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
|
||||
#undef JSON_BRACE_INIT_COPY_SEMANTICS
|
||||
#undef JSON_PRECISE_STREAM_POSITION
|
||||
|
||||
@@ -11,7 +11,12 @@
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <cmath>
|
||||
#include <fstream>
|
||||
#include <limits>
|
||||
#include <map>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include "make_test_data_available.hpp"
|
||||
|
||||
TEST_CASE("Binary Formats" * doctest::skip())
|
||||
@@ -224,3 +229,139 @@ TEST_CASE("Binary Formats" * doctest::skip())
|
||||
CHECK((100.0 * double(ubjson_3_size) / double(json_size)) == Approx(89.450));
|
||||
}
|
||||
}
|
||||
|
||||
namespace
|
||||
{
|
||||
// the binary formats as function pointers for "Binary formats with narrow number types";
|
||||
// named functions rather than lambdas, because clang 3.5 cannot convert a lambda
|
||||
// to a function pointer in the braced initializer of the format table
|
||||
using narrow_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int32_t, std::uint32_t, float>;
|
||||
using bytes = std::vector<std::uint8_t>;
|
||||
|
||||
bytes encode_cbor(const json& j)
|
||||
{
|
||||
return json::to_cbor(j);
|
||||
}
|
||||
narrow_json decode_cbor(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_cbor(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
bytes encode_msgpack(const json& j)
|
||||
{
|
||||
return json::to_msgpack(j);
|
||||
}
|
||||
narrow_json decode_msgpack(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_msgpack(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
bytes encode_ubjson(const json& j)
|
||||
{
|
||||
return json::to_ubjson(j);
|
||||
}
|
||||
narrow_json decode_ubjson(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_ubjson(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
bytes encode_bjdata(const json& j)
|
||||
{
|
||||
return json::to_bjdata(j);
|
||||
}
|
||||
narrow_json decode_bjdata(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_bjdata(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
// BSON can only store numbers as object members
|
||||
bytes encode_bson(const json& j)
|
||||
{
|
||||
return json::to_bson(json{{"a", j}});
|
||||
}
|
||||
narrow_json decode_bson(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
const auto result = narrow_json::from_bson(v, true, allow_exceptions);
|
||||
return result.is_discarded() ? result : result.at("a");
|
||||
}
|
||||
|
||||
bytes encode_bon8(const json& j)
|
||||
{
|
||||
return json::to_bon8(j);
|
||||
}
|
||||
narrow_json decode_bon8(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_bon8(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("Binary formats with narrow number types")
|
||||
{
|
||||
// Numbers that do not fit the number types are handled like the lexer
|
||||
// handles them in JSON text: an integer that fits neither integer type is
|
||||
// stored as a floating-point number, and a finite floating-point number
|
||||
// that overflows number_float_t is rejected with out_of_range.406.
|
||||
struct binary_format
|
||||
{
|
||||
const char* name;
|
||||
bytes (*encode)(const json&);
|
||||
narrow_json (*decode)(const bytes&, bool);
|
||||
};
|
||||
|
||||
const std::vector<binary_format> formats =
|
||||
{
|
||||
{"CBOR", encode_cbor, decode_cbor},
|
||||
{"MessagePack", encode_msgpack, decode_msgpack},
|
||||
{"UBJSON", encode_ubjson, decode_ubjson},
|
||||
{"BJData", encode_bjdata, decode_bjdata},
|
||||
{"BSON", encode_bson, decode_bson},
|
||||
{"BON8", encode_bon8, decode_bon8},
|
||||
};
|
||||
|
||||
for (const auto& format : formats)
|
||||
{
|
||||
const std::string name = format.name;
|
||||
INFO("format := ", name);
|
||||
const auto roundtrip = [&format](const json & j)
|
||||
{
|
||||
return format.decode(format.encode(j), true);
|
||||
};
|
||||
|
||||
// integers that fit keep their type
|
||||
CHECK(roundtrip(json(-5)).is_number_integer());
|
||||
CHECK(roundtrip(json(-5)).get<std::int32_t>() == -5);
|
||||
CHECK(roundtrip(json(3000000000u)).is_number_unsigned());
|
||||
CHECK(roundtrip(json(3000000000u)).get<std::uint32_t>() == 3000000000u);
|
||||
|
||||
// integers that fit neither integer type are stored as float
|
||||
CHECK(roundtrip(json(5000000000u)).is_number_float());
|
||||
CHECK(roundtrip(json(5000000000u)).get<float>() == 5000000000.0f);
|
||||
if (name != "BON8") // BON8 cannot encode integers above INT64_MAX
|
||||
{
|
||||
CHECK(roundtrip(json(10000000000000000000u)).is_number_float());
|
||||
CHECK(roundtrip(json(10000000000000000000u)).get<float>() == 10000000000000000000.0f);
|
||||
}
|
||||
CHECK(roundtrip(json(-3000000000LL)).is_number_float());
|
||||
CHECK(roundtrip(json(-3000000000LL)).get<float>() == -3000000000.0f);
|
||||
CHECK(roundtrip(json(-5000000000LL)).is_number_float());
|
||||
CHECK(roundtrip(json(-5000000000LL)).get<float>() == -5000000000.0f);
|
||||
|
||||
// floating-point numbers that fit
|
||||
CHECK(roundtrip(json(1.5)).get<float>() == 1.5f);
|
||||
const auto just_above_max = std::nextafter(static_cast<double>((std::numeric_limits<float>::max)()),
|
||||
std::numeric_limits<double>::infinity());
|
||||
CHECK(roundtrip(json(just_above_max)).get<float>() == (std::numeric_limits<float>::max)());
|
||||
|
||||
// infinity and NaN are passed on
|
||||
CHECK(std::isinf(roundtrip(json(std::numeric_limits<double>::infinity())).get<float>()));
|
||||
CHECK(std::isnan(roundtrip(json(std::numeric_limits<double>::quiet_NaN())).get<float>()));
|
||||
|
||||
// finite floating-point numbers that overflow number_float_t are rejected
|
||||
const std::string message = "[json.exception.out_of_range.406] syntax error while parsing " + name
|
||||
+ " value: number overflow";
|
||||
CHECK_THROWS_WITH_AS(roundtrip(json(1e300)), message.c_str(), narrow_json::out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(roundtrip(json(-1e300)), message.c_str(), narrow_json::out_of_range&);
|
||||
CHECK(format.decode(format.encode(json(1e300)), false).is_discarded());
|
||||
}
|
||||
}
|
||||
|
||||
+16
-14
@@ -3187,7 +3187,8 @@ TEST_CASE("Tagged values")
|
||||
// CBOR encodes negative integers as: result = -1 - n
|
||||
// For type 0x3B, n is an 8-byte uint64_t. Valid range for n with
|
||||
// the default int64_t is [0, INT64_MAX], producing results in [INT64_MIN, -1].
|
||||
// When n > INT64_MAX, the result exceeds int64_t range and is rejected.
|
||||
// When n > INT64_MAX, the result exceeds int64_t range and is stored
|
||||
// as a floating-point number, as the lexer does for JSON text.
|
||||
|
||||
SECTION("n = 0 is valid (result = -1)")
|
||||
{
|
||||
@@ -3208,33 +3209,34 @@ TEST_CASE("Tagged values")
|
||||
CHECK(result.get<int64_t>() == (std::numeric_limits<int64_t>::min)());
|
||||
}
|
||||
|
||||
SECTION("n = INT64_MAX + 1 is rejected (overflow)")
|
||||
SECTION("n = INT64_MAX + 1 is stored as float")
|
||||
{
|
||||
// n = INT64_MAX + 1 (0x8000000000000000)
|
||||
// result = -1 - n = -9223372036854775809, which exceeds int64_t range
|
||||
// result = -1 - n = -9223372036854775809, which exceeds int64_t range;
|
||||
// the nearest double is -9223372036854775808.0
|
||||
const std::vector<uint8_t> input = {0x3B, 0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00};
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input),
|
||||
"[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow",
|
||||
json::parse_error);
|
||||
const auto result = json::from_cbor(input);
|
||||
CHECK(result.is_number_float());
|
||||
CHECK(result.get<double>() == -9223372036854775808.0);
|
||||
CHECK(result == json::parse("-9223372036854775809"));
|
||||
}
|
||||
|
||||
SECTION("n = UINT64_MAX is rejected (overflow)")
|
||||
SECTION("n = UINT64_MAX is stored as float")
|
||||
{
|
||||
// n = UINT64_MAX (0xFFFFFFFFFFFFFFFF)
|
||||
// result = -1 - n = -18446744073709551616, which exceeds int64_t range
|
||||
const std::vector<uint8_t> input = {0x3B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF};
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input),
|
||||
"[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow",
|
||||
json::parse_error);
|
||||
const auto result = json::from_cbor(input);
|
||||
CHECK(result.is_number_float());
|
||||
CHECK(result.get<double>() == -18446744073709551616.0);
|
||||
CHECK(result == json::parse("-18446744073709551616"));
|
||||
}
|
||||
|
||||
SECTION("overflow with allow_exceptions=false returns discarded")
|
||||
SECTION("overflow with allow_exceptions=false is not an error")
|
||||
{
|
||||
const std::vector<uint8_t> input = {0x3B, 0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00};
|
||||
const auto result = json::from_cbor(input, true, false);
|
||||
CHECK(result.is_discarded());
|
||||
CHECK(result.is_number_float());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -13,10 +13,7 @@
|
||||
using nlohmann::json;
|
||||
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <cmath> // signbit
|
||||
#include <cstdlib> // strtod
|
||||
#include <limits> // numeric_limits
|
||||
#include <map> // map
|
||||
#include <sstream> // stringstream
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
@@ -703,140 +700,3 @@ TEST_CASE("parse_float_fast declines what it cannot convert exactly")
|
||||
CHECK_FALSE(fast("1e23", out));
|
||||
CHECK_FALSE(fast("1e-23", out));
|
||||
}
|
||||
|
||||
namespace
|
||||
{
|
||||
template<typename FloatType>
|
||||
bool out_of_range_value(const std::string& s, FloatType& out)
|
||||
{
|
||||
return nlohmann::detail::parse_float_out_of_range(s.data(), s.data() + s.size(), out);
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("parse_float_out_of_range derives the value from the token")
|
||||
{
|
||||
// std::from_chars reports numbers out of range without a portable value
|
||||
// (P4168), so the value is derived from the token
|
||||
const double inf = std::numeric_limits<double>::infinity();
|
||||
double out = 1.0;
|
||||
|
||||
SECTION("overflow")
|
||||
{
|
||||
CHECK(out_of_range_value("1e400", out));
|
||||
CHECK(out == inf);
|
||||
CHECK(out_of_range_value("-1E+400", out));
|
||||
CHECK(out == -inf);
|
||||
CHECK(out_of_range_value("123.456e306", out));
|
||||
CHECK(out == inf);
|
||||
CHECK(out_of_range_value("0.001e99999999999999999999", out));
|
||||
CHECK(out == inf);
|
||||
CHECK(out_of_range_value("-1" + std::string(400, '0'), out));
|
||||
CHECK(out == -inf);
|
||||
}
|
||||
|
||||
SECTION("underflow")
|
||||
{
|
||||
CHECK(out_of_range_value("1e-400", out));
|
||||
CHECK(out == 0.0);
|
||||
CHECK(!std::signbit(out));
|
||||
CHECK(out_of_range_value("-1e-400", out));
|
||||
CHECK(out == 0.0);
|
||||
CHECK(std::signbit(out));
|
||||
CHECK(out_of_range_value("0.00012e-321", out));
|
||||
CHECK(out == 0.0);
|
||||
CHECK(out_of_range_value("-1234e-99999999999999999999", out));
|
||||
CHECK(std::signbit(out));
|
||||
CHECK(out_of_range_value("-0.0", out));
|
||||
CHECK(out == 0.0);
|
||||
CHECK(std::signbit(out));
|
||||
}
|
||||
|
||||
SECTION("possibly subnormal")
|
||||
{
|
||||
// some implementations report subnormal numbers as out of range; the
|
||||
// caller then converts them another way
|
||||
CHECK_FALSE(out_of_range_value("0.0012e-321", out));
|
||||
CHECK_FALSE(out_of_range_value("2.5e-320", out));
|
||||
CHECK_FALSE(out_of_range_value("-1e-310", out));
|
||||
}
|
||||
|
||||
SECTION("float")
|
||||
{
|
||||
float f = 1.0f;
|
||||
CHECK(out_of_range_value("-1e39", f));
|
||||
CHECK(f == -std::numeric_limits<float>::infinity());
|
||||
CHECK(out_of_range_value("1e-47", f));
|
||||
CHECK(f == 0.0f);
|
||||
CHECK_FALSE(out_of_range_value("1e-46", f));
|
||||
CHECK_FALSE(out_of_range_value("1e-40", f));
|
||||
}
|
||||
|
||||
SECTION("long double")
|
||||
{
|
||||
long double ld = 1.0L;
|
||||
CHECK(out_of_range_value("1e5000", ld));
|
||||
CHECK(ld == std::numeric_limits<long double>::infinity());
|
||||
CHECK(out_of_range_value("-1e-5000", ld));
|
||||
CHECK(ld == 0.0L);
|
||||
CHECK(std::signbit(ld));
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("floating-point numbers out of range")
|
||||
{
|
||||
// Whichever conversion the platform uses, an overflow throws, and an
|
||||
// underflow yields a zero with the sign of the number.
|
||||
using float_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
|
||||
using long_double_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, long double>;
|
||||
|
||||
SECTION("double")
|
||||
{
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse("1.5e400"), "[json.exception.out_of_range.406] number overflow parsing '1.5e400'", json::out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse("-1.5e400"), "[json.exception.out_of_range.406] number overflow parsing '-1.5e400'", json::out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse("1e99999999999999999999"), "[json.exception.out_of_range.406] number overflow parsing '1e99999999999999999999'", json::out_of_range&);
|
||||
CHECK_THROWS_AS(_ = json::parse("1" + std::string(400, '0')), json::out_of_range&);
|
||||
|
||||
const json zero = json::parse("1.5e-400");
|
||||
CHECK(zero == 0.0);
|
||||
CHECK(!std::signbit(zero.get<double>()));
|
||||
const json negative_zero = json::parse("-1.5e-400");
|
||||
CHECK(negative_zero == 0.0);
|
||||
CHECK(std::signbit(negative_zero.get<double>()));
|
||||
CHECK(std::signbit(json::parse("-0.0000000001e-99999999999999999999").get<double>()));
|
||||
|
||||
// around the smallest subnormal number
|
||||
CHECK(json::parse("1e-324") == 0.0);
|
||||
CHECK(json::parse("3e-324") == std::numeric_limits<double>::denorm_min());
|
||||
CHECK(json::parse("-2.5e-320") == -2.5e-320);
|
||||
}
|
||||
|
||||
SECTION("float")
|
||||
{
|
||||
float_json _;
|
||||
CHECK_THROWS_WITH_AS(_ = float_json::parse("1e39"), "[json.exception.out_of_range.406] number overflow parsing '1e39'", json::out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = float_json::parse("-1e39"), "[json.exception.out_of_range.406] number overflow parsing '-1e39'", json::out_of_range&);
|
||||
|
||||
const float_json zero = float_json::parse("1e-50");
|
||||
CHECK(zero == 0.0f);
|
||||
CHECK(!std::signbit(zero.get<float>()));
|
||||
const float_json negative_zero = float_json::parse("-1e-50");
|
||||
CHECK(negative_zero == 0.0f);
|
||||
CHECK(std::signbit(negative_zero.get<float>()));
|
||||
CHECK(float_json::parse("1e-45") == std::numeric_limits<float>::denorm_min());
|
||||
}
|
||||
|
||||
SECTION("long double")
|
||||
{
|
||||
long_double_json _;
|
||||
CHECK_THROWS_WITH_AS(_ = long_double_json::parse("1e5000"), "[json.exception.out_of_range.406] number overflow parsing '1e5000'", json::out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = long_double_json::parse("-1e5000"), "[json.exception.out_of_range.406] number overflow parsing '-1e5000'", json::out_of_range&);
|
||||
|
||||
const long_double_json zero = long_double_json::parse("1e-5000");
|
||||
CHECK(zero == 0.0L);
|
||||
CHECK(!std::signbit(zero.get<long double>()));
|
||||
const long_double_json negative_zero = long_double_json::parse("-1e-5000");
|
||||
CHECK(negative_zero == 0.0L);
|
||||
CHECK(std::signbit(negative_zero.get<long double>()));
|
||||
}
|
||||
}
|
||||
|
||||
+29
-103
@@ -14,8 +14,6 @@ using nlohmann::json;
|
||||
|
||||
#include <array>
|
||||
#include <clocale>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
#include <map>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
@@ -259,10 +257,10 @@ struct LocaleSwitchingSax final: public nlohmann::json_sax<json>
|
||||
|
||||
TEST_CASE("locale changes between lexer construction and number conversion (#5198)")
|
||||
{
|
||||
// The numbers are chosen so that the conversion takes the slower paths: too
|
||||
// many significant digits for Clinger's fast path, an underflow, and a plain
|
||||
// value. Without std::from_chars and strtod_l, this is the strtod fallback,
|
||||
// which honors the locale that is current at conversion time.
|
||||
// The numbers are chosen so that the conversion also takes the strtod
|
||||
// fallback, which honors the locale that is current at conversion time:
|
||||
// too many significant digits for Clinger's fast path, an underflow that
|
||||
// std::from_chars rejects, and a plain value.
|
||||
const std::vector<std::string> numbers = {"3.14159265358979323846", "1.5e-400", "12.34", "-0.000123456789012345678"};
|
||||
std::string text = "[";
|
||||
for (const auto& n : numbers)
|
||||
@@ -348,113 +346,41 @@ TEST_CASE("locale changes between lexer construction and number conversion (#519
|
||||
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
|
||||
}
|
||||
|
||||
namespace
|
||||
{
|
||||
// sets LC_NUMERIC to the first installed locale whose decimal point is longer
|
||||
// than one byte, e.g. U+066B ARABIC DECIMAL SEPARATOR (two bytes in UTF-8)
|
||||
const char* set_multi_byte_decimal_point_locale()
|
||||
{
|
||||
const std::array<const char*, 6> names = {{"ar_EG.UTF-8", "ar_SA.UTF-8", "fa_IR.UTF-8", "ps_AF.UTF-8", "ar_EG", "fa_IR"}};
|
||||
for (const char* name : names)
|
||||
{
|
||||
if (std::setlocale(LC_NUMERIC, name) != nullptr && std::strlen(std::localeconv()->decimal_point) > 1)
|
||||
{
|
||||
return name;
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("locale with a multi-byte decimal point")
|
||||
{
|
||||
// Such a decimal point cannot be substituted in place for '.'; before
|
||||
// #5660, the strtod fallback stopped there and returned the integer part.
|
||||
const char* name = set_multi_byte_decimal_point_locale();
|
||||
if (name == nullptr)
|
||||
{
|
||||
MESSAGE("no locale with a multi-byte decimal point is usable");
|
||||
}
|
||||
else
|
||||
{
|
||||
const std::string locale_name = name;
|
||||
CAPTURE(locale_name);
|
||||
|
||||
// too many significant digits for Clinger's fast path
|
||||
CHECK(json::parse("3.141592653589793238462643383279") == 3.141592653589793);
|
||||
CHECK(json::parse("1.7976931348623157e308") == (std::numeric_limits<double>::max)());
|
||||
CHECK(json::accept("3.14159265358979323846"));
|
||||
|
||||
// a subnormal number
|
||||
CHECK(json::parse("-2.5e-320") == -2.5e-320);
|
||||
|
||||
// out of range
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse("1.5e400"), "[json.exception.out_of_range.406] number overflow parsing '1.5e400'", json::out_of_range&);
|
||||
CHECK(json::parse("1.5e-400") == 0.0);
|
||||
|
||||
// float and long double as number_float_t
|
||||
using float_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
|
||||
using long_double_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, long double>;
|
||||
CHECK(float_json::parse("1.5") == 1.5f);
|
||||
CHECK(long_double_json::parse("1.5") == 1.5L);
|
||||
|
||||
// a value Clinger's fast path converts
|
||||
CHECK(json::parse("12.5") == 12.5);
|
||||
}
|
||||
|
||||
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
|
||||
}
|
||||
|
||||
TEST_CASE("conversion with the decimal point of the current locale")
|
||||
{
|
||||
// parse_float_locale_aware() is the last resort for platforms without
|
||||
// std::from_chars and strtod_l, so it is called directly here
|
||||
const auto convert = [](std::string token, double & out)
|
||||
{
|
||||
nlohmann::detail::parse_float_locale_aware(token, token.find('.'), out);
|
||||
// the token is also handed to the SAX interface and must keep its '.'
|
||||
return token;
|
||||
};
|
||||
|
||||
std::vector<const char*> names = {"C", "de_DE", "de_DE.UTF-8"};
|
||||
const char* multi_byte = set_multi_byte_decimal_point_locale();
|
||||
if (multi_byte != nullptr)
|
||||
{
|
||||
names.push_back(multi_byte);
|
||||
}
|
||||
|
||||
// Some locales use a decimal point that is not a single character, e.g.
|
||||
// U+066B ARABIC DECIMAL SEPARATOR (two bytes in UTF-8). It cannot be
|
||||
// substituted in place for '.', so the strtod fallback stops early. The
|
||||
// conversion must still terminate rather than retry forever.
|
||||
const std::array<const char*, 6> names = {{"ar_EG.UTF-8", "ar_SA.UTF-8", "fa_IR.UTF-8", "ps_AF.UTF-8", "ar_EG", "fa_IR"}};
|
||||
bool tested = false;
|
||||
for (const char* name : names)
|
||||
{
|
||||
if (std::setlocale(LC_NUMERIC, name) == nullptr)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
const std::string locale_name = name;
|
||||
CAPTURE(locale_name);
|
||||
const std::string decimal_point = std::localeconv()->decimal_point;
|
||||
if (decimal_point.size() < 2)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
CAPTURE(name);
|
||||
tested = true;
|
||||
|
||||
double d = 0;
|
||||
CHECK(convert("3.141592653589793238462643383279", d) == "3.141592653589793238462643383279");
|
||||
CHECK(d == 3.141592653589793);
|
||||
CHECK(convert("-2.5e-320", d) == "-2.5e-320");
|
||||
CHECK(d == -2.5e-320);
|
||||
CHECK(convert("12345678901234567890", d) == "12345678901234567890");
|
||||
CHECK(d == 12345678901234567890.0);
|
||||
// too many significant digits for Clinger's fast path, and an underflow
|
||||
// that std::from_chars rejects: both reach the strtod fallback
|
||||
json j;
|
||||
CHECK_NOTHROW(j = json::parse("[3.14159265358979323846, 1.5e-400, -0.000123456789012345678]"));
|
||||
CHECK(j.is_array());
|
||||
CHECK(json::accept("3.14159265358979323846"));
|
||||
|
||||
float f = 0;
|
||||
std::string token = "1.5";
|
||||
nlohmann::detail::parse_float_locale_aware(token, 1, f);
|
||||
CHECK(f == 1.5f);
|
||||
|
||||
long double ld = 0;
|
||||
nlohmann::detail::parse_float_locale_aware(token, 1, ld);
|
||||
CHECK(ld == 1.5L);
|
||||
CHECK(token == "1.5");
|
||||
|
||||
// the lexer only passes valid tokens; for others, the conversion stops
|
||||
// early, and the value parsed up to there is kept
|
||||
CHECK(convert("1.5x", d) == "1.5x");
|
||||
CHECK(d == 1.5);
|
||||
// a value the locale-independent paths convert is not affected
|
||||
CHECK(json::parse("12.5") == 12.5);
|
||||
}
|
||||
if (!tested)
|
||||
{
|
||||
MESSAGE("no locale with a multi-byte decimal point is usable");
|
||||
}
|
||||
|
||||
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
|
||||
|
||||
Reference in New Issue
Block a user