mirror of
https://github.com/nlohmann/json.git
synced 2026-09-30 19:50:34 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4356d0fe50 | ||
|
|
32d5540d37 | ||
|
|
e0d1123acd | ||
|
|
b3753fbb1a | ||
|
|
5e41bf284a | ||
|
|
50d8e71b9f | ||
|
|
c620c01063 |
@@ -55,10 +55,6 @@ This implementation does exactly follow this approach, as it uses double precisi
|
||||
smaller than `-1.79769313486232e+308` and values greater than `1.79769313486232e+308` will be stored as NaN internally
|
||||
and be serialized to `null`.
|
||||
|
||||
During deserialization (from JSON text or any of the binary formats), a finite number that does not fit into
|
||||
`number_float_t` is rejected with [`out_of_range.406`](../../home/exceptions.md#jsonexceptionout_of_range406), for
|
||||
example a double-precision number in a binary format when `number_float_t` is `#!cpp float`.
|
||||
|
||||
#### Storage
|
||||
|
||||
Floating-point number values are stored directly inside a `basic_json` type.
|
||||
|
||||
@@ -47,9 +47,8 @@ With the default values for `NumberIntegerType` (`std::int64_t`), the default va
|
||||
|
||||
When the default type is used, the maximal integer number that can be stored is `9223372036854775807` (INT64_MAX) and
|
||||
the minimal integer number that can be stored is `-9223372036854775808` (INT64_MIN). Integer numbers that are out of
|
||||
range will yield over/underflow when used in a constructor. During deserialization (from JSON text or any of the binary
|
||||
formats), too large or small integer numbers will automatically be stored as [`number_unsigned_t`](number_unsigned_t.md)
|
||||
or [`number_float_t`](number_float_t.md).
|
||||
range will yield over/underflow when used in a constructor. During deserialization, too large or small integer numbers
|
||||
will automatically be stored as [`number_unsigned_t`](number_unsigned_t.md) or [`number_float_t`](number_float_t.md).
|
||||
|
||||
[RFC 8259](https://tools.ietf.org/html/rfc8259) further states:
|
||||
> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are
|
||||
|
||||
@@ -48,9 +48,8 @@ With the default values for `NumberUnsignedType` (`std::uint64_t`), the default
|
||||
|
||||
When the default type is used, the maximal integer number that can be stored is `18446744073709551615` (UINT64_MAX) and
|
||||
the minimal integer number that can be stored is `0`. Integer numbers that are out of range will yield over/underflow
|
||||
when used in a constructor. During deserialization (from JSON text or any of the binary formats), too large or small
|
||||
integer numbers will automatically be stored as [`number_integer_t`](number_integer_t.md) or
|
||||
[`number_float_t`](number_float_t.md).
|
||||
when used in a constructor. During deserialization, too large or small integer numbers will automatically be stored
|
||||
as [`number_integer_t`](number_integer_t.md) or [`number_float_t`](number_float_t.md).
|
||||
|
||||
[RFC 8259](https://tools.ietf.org/html/rfc8259) further states:
|
||||
> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are
|
||||
|
||||
@@ -168,9 +168,9 @@ The library maps CBOR types to JSON value types as follows:
|
||||
!!! warning "Negative integer overflow"
|
||||
|
||||
CBOR negative integers (major type 1) are decoded as `-1 - n`. If the encoded magnitude `n` is too large for the
|
||||
result to fit into `number_integer_t` (`std::int64_t` by default), the result is stored as `number_float_t`, like
|
||||
a too small integer in JSON text. For example, `-18446744073709551616` (`0x3B` followed by eight `0xFF` bytes) is
|
||||
stored as `-1.8446744073709552e+19`.
|
||||
result to fit into `number_integer_t` (`std::int64_t` by default), parsing fails with a
|
||||
[`parse_error.112`](../../home/exceptions.md#jsonexceptionparse_error112) exception rather than overflowing
|
||||
silently.
|
||||
|
||||
!!! warning "Object keys"
|
||||
|
||||
|
||||
@@ -331,6 +331,9 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde
|
||||
[json.exception.parse_error.112] parse error at byte 15: syntax error while parsing BSON binary: byte array length cannot be negative, is -1
|
||||
```
|
||||
```
|
||||
[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow
|
||||
```
|
||||
```
|
||||
[json.exception.parse_error.112] parse error at byte 5: syntax error while parsing BSON document: document size 6 does not match the number of bytes read (5)
|
||||
```
|
||||
|
||||
@@ -851,18 +854,13 @@ The JSON Patch operations 'remove' and 'add' cannot be applied to the root eleme
|
||||
|
||||
### json.exception.out_of_range.406
|
||||
|
||||
A parsed number could not be stored without changing it to NaN or INF. For the binary formats, this happens when a
|
||||
finite floating-point number does not fit into [`number_float_t`](../api/basic_json/number_float_t.md), for example a
|
||||
double-precision number when `number_float_t` is `#!cpp float`.
|
||||
A parsed number could not be stored as without changing it to NaN or INF.
|
||||
|
||||
!!! failure "Example messages"
|
||||
!!! failure "Example message"
|
||||
|
||||
```
|
||||
number overflow parsing '10E1000'
|
||||
```
|
||||
```
|
||||
[json.exception.out_of_range.406] syntax error while parsing CBOR value: number overflow
|
||||
```
|
||||
|
||||
### json.exception.out_of_range.407
|
||||
|
||||
|
||||
@@ -559,7 +559,7 @@ class binary_reader
|
||||
case 0x01: // double
|
||||
{
|
||||
double number{};
|
||||
return get_number<double, true>(input_format_t::bson, number) && emit_float(input_format_t::bson, number);
|
||||
return get_number<double, true>(input_format_t::bson, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
}
|
||||
|
||||
case 0x02: // string
|
||||
@@ -600,19 +600,19 @@ class binary_reader
|
||||
case 0x10: // int32
|
||||
{
|
||||
std::int32_t value{};
|
||||
return get_number<std::int32_t, true>(input_format_t::bson, value) && emit_signed(input_format_t::bson, value);
|
||||
return get_number<std::int32_t, true>(input_format_t::bson, value) && sax->number_integer(value);
|
||||
}
|
||||
|
||||
case 0x12: // int64
|
||||
{
|
||||
std::int64_t value{};
|
||||
return get_number<std::int64_t, true>(input_format_t::bson, value) && emit_signed(input_format_t::bson, value);
|
||||
return get_number<std::int64_t, true>(input_format_t::bson, value) && sax->number_integer(value);
|
||||
}
|
||||
|
||||
case 0x11: // uint64
|
||||
{
|
||||
std::uint64_t value{};
|
||||
return get_number<std::uint64_t, true>(input_format_t::bson, value) && emit_unsigned(input_format_t::bson, value);
|
||||
return get_number<std::uint64_t, true>(input_format_t::bson, value) && sax->number_unsigned(value);
|
||||
}
|
||||
|
||||
default: // anything else is not supported (yet)
|
||||
@@ -638,19 +638,14 @@ class binary_reader
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// the value is -1 - number, which fits into number_integer_t
|
||||
// whenever number does
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_integer_t>(number)))
|
||||
const auto max_val = static_cast<NumberType>((std::numeric_limits<number_integer_t>::max)());
|
||||
if (number > max_val)
|
||||
{
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
parse_error::create(112, chars_read,
|
||||
exception_message(input_format_t::cbor, "negative integer overflow", "value"), nullptr));
|
||||
}
|
||||
|
||||
// like the lexer does for JSON text, store a value too small for
|
||||
// number_integer_t as number_float_t; compute it as long double so
|
||||
// that emit_float sees a finite value and can detect an overflow of
|
||||
// number_float_t
|
||||
return emit_float(input_format_t::cbor, static_cast<long double>(-1) - static_cast<long double>(number));
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -707,25 +702,25 @@ class binary_reader
|
||||
case 0x18: // Unsigned integer (one-byte uint8_t follows)
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 0x19: // Unsigned integer (two-byte uint16_t follows)
|
||||
{
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 0x1A: // Unsigned integer (four-byte uint32_t follows)
|
||||
{
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 0x1B: // Unsigned integer (eight-byte uint64_t follows)
|
||||
{
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format_t::cbor, number) && emit_unsigned(input_format_t::cbor, number);
|
||||
return get_number(input_format_t::cbor, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
// Negative integer -1-0x00..-1-0x17 (-1..-24)
|
||||
@@ -1170,13 +1165,13 @@ class binary_reader
|
||||
case 0xFA: // Single-Precision Float (four-byte IEEE 754)
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::cbor, number) && emit_float(input_format_t::cbor, number);
|
||||
return get_number(input_format_t::cbor, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
}
|
||||
|
||||
case 0xFB: // Double-Precision Float (eight-byte IEEE 754)
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::cbor, number) && emit_float(input_format_t::cbor, number);
|
||||
return get_number(input_format_t::cbor, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
}
|
||||
|
||||
default: // anything else (0xFF is handled inside the other types)
|
||||
@@ -1940,61 +1935,61 @@ class binary_reader
|
||||
case 0xCA: // float 32
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::msgpack, number) && emit_float(input_format_t::msgpack, number);
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
}
|
||||
|
||||
case 0xCB: // float 64
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::msgpack, number) && emit_float(input_format_t::msgpack, number);
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
}
|
||||
|
||||
case 0xCC: // uint 8
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 0xCD: // uint 16
|
||||
{
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 0xCE: // uint 32
|
||||
{
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 0xCF: // uint 64
|
||||
{
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && emit_unsigned(input_format_t::msgpack, number);
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 0xD0: // int 8
|
||||
{
|
||||
std::int8_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
}
|
||||
|
||||
case 0xD1: // int 16
|
||||
{
|
||||
std::int16_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
}
|
||||
|
||||
case 0xD2: // int 32
|
||||
{
|
||||
std::int32_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
}
|
||||
|
||||
case 0xD3: // int 64
|
||||
{
|
||||
std::int64_t number{};
|
||||
return get_number(input_format_t::msgpack, number) && emit_signed(input_format_t::msgpack, number);
|
||||
return get_number(input_format_t::msgpack, number) && sax->number_integer(number);
|
||||
}
|
||||
|
||||
case 0xDC: // array 16
|
||||
@@ -2927,7 +2922,7 @@ class binary_reader
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!emit_unsigned(input_format, i)))
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->number_unsigned(static_cast<number_unsigned_t>(i))))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -3059,37 +3054,37 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 'U':
|
||||
{
|
||||
std::uint8_t number{};
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 'i':
|
||||
{
|
||||
std::int8_t number{};
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
}
|
||||
|
||||
case 'I':
|
||||
{
|
||||
std::int16_t number{};
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
}
|
||||
|
||||
case 'l':
|
||||
{
|
||||
std::int32_t number{};
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
}
|
||||
|
||||
case 'L':
|
||||
{
|
||||
std::int64_t number{};
|
||||
return get_number(input_format, number) && emit_signed(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_integer(number);
|
||||
}
|
||||
|
||||
case 'u':
|
||||
@@ -3099,7 +3094,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint16_t number{};
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 'm':
|
||||
@@ -3109,7 +3104,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint32_t number{};
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 'M':
|
||||
@@ -3119,7 +3114,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint64_t number{};
|
||||
return get_number(input_format, number) && emit_unsigned(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_unsigned(number);
|
||||
}
|
||||
|
||||
case 'h':
|
||||
@@ -3177,13 +3172,13 @@ class binary_reader
|
||||
case 'd':
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format, number) && emit_float(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
}
|
||||
|
||||
case 'D':
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format, number) && emit_float(input_format, number);
|
||||
return get_number(input_format, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
}
|
||||
|
||||
case 'H':
|
||||
@@ -3349,8 +3344,8 @@ class binary_reader
|
||||
return enter_object(detail::unknown_size());
|
||||
}
|
||||
|
||||
// Note, no reader for UBJSON binary types is implemented because they do
|
||||
// not exist
|
||||
// Note, UBJSON has no binary type of its own; BJData, which shares this
|
||||
// reader, decodes optimized 'B' arrays as binary in get_ubjson_array().
|
||||
|
||||
bool get_ubjson_high_precision_number()
|
||||
{
|
||||
@@ -3650,13 +3645,13 @@ class binary_reader
|
||||
case 0x8E: // binary32
|
||||
{
|
||||
float number{};
|
||||
return get_number(input_format_t::bon8, number) && emit_float(input_format_t::bon8, number);
|
||||
return get_number(input_format_t::bon8, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
}
|
||||
|
||||
case 0x8F: // binary64
|
||||
{
|
||||
double number{};
|
||||
return get_number(input_format_t::bon8, number) && emit_float(input_format_t::bon8, number);
|
||||
return get_number(input_format_t::bon8, number) && sax->number_float(static_cast<number_float_t>(number), "");
|
||||
}
|
||||
|
||||
case 0xF8:
|
||||
@@ -3722,9 +3717,7 @@ class binary_reader
|
||||
@brief pass an integer to the SAX parser
|
||||
|
||||
Non-negative integers are passed as unsigned, negative integers as signed
|
||||
numbers, like the other binary formats do. A value that does not fit the
|
||||
number type is passed as described for @ref emit_unsigned and
|
||||
@ref emit_signed.
|
||||
numbers, like the other binary formats do.
|
||||
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
@@ -3733,9 +3726,9 @@ class binary_reader
|
||||
{
|
||||
if (number >= 0)
|
||||
{
|
||||
return emit_unsigned(input_format_t::bon8, static_cast<std::uint64_t>(number));
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_signed(input_format_t::bon8, number);
|
||||
return sax->number_integer(static_cast<number_integer_t>(number));
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -3792,7 +3785,8 @@ class binary_reader
|
||||
value = (value << 8) | static_cast<std::int64_t>(current);
|
||||
}
|
||||
|
||||
return emit_bon8_integer(negative ? -(value + offset) : value + offset);
|
||||
return negative ? sax->number_integer(static_cast<number_integer_t>(-(value + offset)))
|
||||
: sax->number_unsigned(static_cast<number_unsigned_t>(value + offset));
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -4058,7 +4052,7 @@ class binary_reader
|
||||
#endif
|
||||
}
|
||||
|
||||
/*
|
||||
/*!
|
||||
@brief read a number from the input
|
||||
|
||||
@tparam NumberType the type of the number
|
||||
@@ -4068,10 +4062,10 @@ class binary_reader
|
||||
@return whether conversion completed
|
||||
|
||||
@note This function needs to respect the system's endianness, because
|
||||
bytes in CBOR, MessagePack, and UBJSON are stored in network order
|
||||
(big endian) and therefore need reordering on little endian systems.
|
||||
On the other hand, BSON and BJData use little endian and should reorder
|
||||
on big endian systems.
|
||||
bytes in CBOR, MessagePack, UBJSON, and BON8 are stored in network
|
||||
order (big endian) and therefore need reordering on little endian
|
||||
systems. On the other hand, BSON and BJData use little endian and
|
||||
should reorder on big endian systems.
|
||||
*/
|
||||
template<typename NumberType, bool InputIsLittleEndian = false>
|
||||
bool get_number(const input_format_t format, NumberType& result)
|
||||
@@ -4089,88 +4083,6 @@ class binary_reader
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a signed integer read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a value that does not fit into
|
||||
number_integer_t is passed as number_unsigned_t if it is non-negative and
|
||||
fits there, and as number_float_t otherwise. With the default number
|
||||
types, every integer the binary formats can encode fits, so this only
|
||||
matters for narrower custom number types.
|
||||
|
||||
@tparam NumberType a signed integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if @a number overflows number_float_t (see
|
||||
@ref emit_float)
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_signed(const input_format_t format, const NumberType number)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_integer_t>(number)))
|
||||
{
|
||||
return sax->number_integer(static_cast<number_integer_t>(number));
|
||||
}
|
||||
if (value_in_range_of<number_unsigned_t>(number))
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_float(format, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass an unsigned integer read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a value that does not fit into
|
||||
number_unsigned_t is passed as number_float_t.
|
||||
|
||||
@tparam NumberType an unsigned integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the integer
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if @a number overflows number_float_t (see
|
||||
@ref emit_float)
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_unsigned(const input_format_t format, const NumberType number)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<number_unsigned_t>(number)))
|
||||
{
|
||||
return sax->number_unsigned(static_cast<number_unsigned_t>(number));
|
||||
}
|
||||
return emit_float(format, number);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a floating-point number read from the input to the SAX parser
|
||||
|
||||
Like the lexer does for JSON text, a finite value that overflows
|
||||
number_float_t is rejected instead of silently becoming infinity. Infinity
|
||||
and NaN in the input are passed on unchanged. Integers only overflow if
|
||||
number_float_t cannot represent 2^64, e.g., a half-precision type.
|
||||
|
||||
@tparam NumberType a floating-point or integer type
|
||||
@param[in] format the current format (for diagnostics)
|
||||
@param[in] number the number
|
||||
@return whether the SAX parser accepted the value
|
||||
|
||||
@throw out_of_range.406 if a finite @a number overflows number_float_t
|
||||
*/
|
||||
template<typename NumberType>
|
||||
bool emit_float(const input_format_t format, const NumberType number)
|
||||
{
|
||||
const auto result = static_cast<number_float_t>(number);
|
||||
if (JSON_HEDLEY_UNLIKELY(std::isfinite(number) && !std::isfinite(result)))
|
||||
{
|
||||
return sax->parse_error(chars_read, get_token_string(),
|
||||
out_of_range::create(406, exception_message(format, "number overflow", "value"), nullptr));
|
||||
}
|
||||
return sax->number_float(result, "");
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief create a string by reading characters from the input
|
||||
|
||||
|
||||
@@ -8,12 +8,12 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <algorithm> // min
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstring> // strlen
|
||||
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
|
||||
#include <memory> // shared_ptr, make_shared, addressof
|
||||
#include <numeric> // accumulate
|
||||
#include <streambuf> // streambuf
|
||||
#include <string> // string, char_traits
|
||||
#include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer
|
||||
@@ -28,6 +28,7 @@
|
||||
#include <nlohmann/detail/iterators/iterator_traits.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||
#include <nlohmann/detail/string_utils.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -82,8 +83,9 @@ class file_input_adapter
|
||||
};
|
||||
|
||||
/*!
|
||||
Input adapter for a (caching) istream. Ignores a UFT Byte Order Mark at
|
||||
beginning of input. Does not support changing the underlying std::streambuf
|
||||
Input adapter for a (caching) istream. Does not skip a UTF Byte Order Mark
|
||||
itself; that is done by the lexer's skip_bom(). Does not support changing
|
||||
the underlying std::streambuf
|
||||
in mid-input. Maintains underlying std::istream and std::streambuf to support
|
||||
subsequent use of standard std::istream operations to process any input
|
||||
characters following those used in parsing the JSON input. Clears the
|
||||
@@ -448,32 +450,14 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
|
||||
// get the current character
|
||||
const auto wc = input.get_character();
|
||||
|
||||
// UTF-32 to UTF-8 encoding
|
||||
if (wc < 0x80)
|
||||
if (wc <= 0x10FFFF)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
||||
utf8_bytes_filled = 1;
|
||||
}
|
||||
else if (wc <= 0x7FF)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u) & 0x1Fu));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
||||
utf8_bytes_filled = 2;
|
||||
}
|
||||
else if (wc <= 0xFFFF)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u) & 0x0Fu));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
||||
utf8_bytes_filled = 3;
|
||||
}
|
||||
else if (wc <= 0x10FFFF)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | ((static_cast<unsigned int>(wc) >> 18u) & 0x07u));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 12u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
|
||||
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
||||
utf8_bytes_filled = 4;
|
||||
// UTF-32 to UTF-8 encoding
|
||||
utf8_bytes_filled = 0;
|
||||
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
|
||||
{
|
||||
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -510,24 +494,15 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
|
||||
// get the current character
|
||||
const auto wc = input.get_character();
|
||||
|
||||
// UTF-16 to UTF-8 encoding
|
||||
if (wc < 0x80)
|
||||
if (0xD800 > wc || wc >= 0xE000)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
||||
utf8_bytes_filled = 1;
|
||||
}
|
||||
else if (wc <= 0x7FF)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u)));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
||||
utf8_bytes_filled = 2;
|
||||
}
|
||||
else if (0xD800 > wc || wc >= 0xE000)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u)));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
||||
utf8_bytes_filled = 3;
|
||||
// a UTF-16 code unit outside the surrogate range is a valid
|
||||
// code point (at most U+FFFF) on its own
|
||||
utf8_bytes_filled = 0;
|
||||
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
|
||||
{
|
||||
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -545,11 +520,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
|
||||
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
|
||||
{
|
||||
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu));
|
||||
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu));
|
||||
utf8_bytes_filled = 4;
|
||||
utf8_bytes_filled = 0;
|
||||
encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
|
||||
{
|
||||
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
|
||||
});
|
||||
valid_pair = true;
|
||||
}
|
||||
}
|
||||
@@ -862,9 +837,9 @@ auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) /
|
||||
return input_adapter(array, array + N);
|
||||
}
|
||||
|
||||
// This class only handles inputs of input_buffer_adapter type.
|
||||
// It's required so that expressions like {ptr, len} can be implicitly cast
|
||||
// to the correct adapter.
|
||||
// This class only handles inputs that construct a contiguous_bytes_input_adapter
|
||||
// (e.g. span_input_adapter). It's required so that expressions like {ptr, len}
|
||||
// can be implicitly cast to the correct adapter.
|
||||
class span_input_adapter
|
||||
{
|
||||
public:
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
|
||||
#include <algorithm> // find_if, min
|
||||
#include <cstddef>
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
#include <type_traits> // enable_if_t
|
||||
#include <utility> // move, pair
|
||||
@@ -175,6 +176,88 @@ template<typename ArrayType>
|
||||
inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/)
|
||||
{}
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
/*!
|
||||
@brief set the diagnostic positions of a value the DOM SAX parsers just stored
|
||||
|
||||
Shared by json_sax_dom_parser and json_sax_dom_callback_parser. basic_json
|
||||
befriends this struct, as the position members are private.
|
||||
*/
|
||||
struct diagnostic_positions
|
||||
{
|
||||
/*!
|
||||
@param[in,out] v the value that was just parsed
|
||||
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
|
||||
*/
|
||||
template<typename BasicJsonType, typename LexerType>
|
||||
static void set_from_lexer(BasicJsonType& v, LexerType* lexer)
|
||||
{
|
||||
if (lexer)
|
||||
{
|
||||
// Lexer has read past the current field value, so set the end position to the current position.
|
||||
// The start position will be set below based on the length of the string representation
|
||||
// of the value.
|
||||
v.end_position = lexer->get_position();
|
||||
|
||||
switch (v.type())
|
||||
{
|
||||
case value_t::boolean:
|
||||
{
|
||||
// 4 and 5 are the string length of "true" and "false"
|
||||
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::null:
|
||||
{
|
||||
// 4 is the string length of "null"
|
||||
v.start_position = v.end_position - 4;
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
// escape sequences make the token longer than the value it
|
||||
// parses to, so the start position cannot be derived from
|
||||
// the value; use the offset the lexer recorded instead
|
||||
v.start_position = lexer->get_token_start_position();
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::discarded:
|
||||
{
|
||||
// an object or array the callback of
|
||||
// json_sax_dom_callback_parser rejected has no position
|
||||
v.end_position = std::string::npos;
|
||||
v.start_position = v.end_position;
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::binary:
|
||||
case value_t::number_integer:
|
||||
case value_t::number_unsigned:
|
||||
case value_t::number_float:
|
||||
{
|
||||
v.start_position = v.end_position - lexer->get_string().size();
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::object:
|
||||
case value_t::array:
|
||||
{
|
||||
// object and array are handled in start_object() and start_array() handlers
|
||||
// skip setting the values here.
|
||||
break;
|
||||
}
|
||||
default: // LCOV_EXCL_LINE
|
||||
// Handle all possible types discretely, default handler should never be reached.
|
||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
#endif
|
||||
|
||||
/*!
|
||||
@brief SAX implementation to create a JSON value from SAX events
|
||||
|
||||
@@ -376,76 +459,6 @@ class json_sax_dom_parser
|
||||
|
||||
private:
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
|
||||
{
|
||||
if (m_lexer_ref)
|
||||
{
|
||||
// Lexer has read past the current field value, so set the end position to the current position.
|
||||
// The start position will be set below based on the length of the string representation
|
||||
// of the value.
|
||||
v.end_position = m_lexer_ref->get_position();
|
||||
|
||||
switch (v.type())
|
||||
{
|
||||
case value_t::boolean:
|
||||
{
|
||||
// 4 and 5 are the string length of "true" and "false"
|
||||
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::null:
|
||||
{
|
||||
// 4 is the string length of "null"
|
||||
v.start_position = v.end_position - 4;
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
// escape sequences make the token longer than the value it
|
||||
// parses to, so the start position cannot be derived from
|
||||
// the value; use the offset the lexer recorded instead
|
||||
v.start_position = m_lexer_ref->get_token_start_position();
|
||||
break;
|
||||
}
|
||||
|
||||
// As we handle the start and end positions for values created during parsing,
|
||||
// we do not expect the following value type to be called. Regardless, set the positions
|
||||
// in case this is created manually or through a different constructor. Exclude from lcov
|
||||
// since the exact condition of this switch is esoteric.
|
||||
// LCOV_EXCL_START
|
||||
case value_t::discarded:
|
||||
{
|
||||
v.end_position = std::string::npos;
|
||||
v.start_position = v.end_position;
|
||||
break;
|
||||
}
|
||||
// LCOV_EXCL_STOP
|
||||
case value_t::binary:
|
||||
case value_t::number_integer:
|
||||
case value_t::number_unsigned:
|
||||
case value_t::number_float:
|
||||
{
|
||||
v.start_position = v.end_position - m_lexer_ref->get_string().size();
|
||||
break;
|
||||
}
|
||||
case value_t::object:
|
||||
case value_t::array:
|
||||
{
|
||||
// object and array are handled in start_object() and start_array() handlers
|
||||
// skip setting the values here.
|
||||
break;
|
||||
}
|
||||
default: // LCOV_EXCL_LINE
|
||||
// Handle all possible types discretely, default handler should never be reached.
|
||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
/*!
|
||||
@invariant If the ref stack is empty, then the passed value will be the new
|
||||
root.
|
||||
@@ -461,7 +474,7 @@ class json_sax_dom_parser
|
||||
root = BasicJsonType(std::forward<Value>(v));
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
handle_diagnostic_positions_for_json_value(root);
|
||||
diagnostic_positions::set_from_lexer(root, m_lexer_ref);
|
||||
#endif
|
||||
|
||||
return &root;
|
||||
@@ -474,7 +487,7 @@ class json_sax_dom_parser
|
||||
ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v));
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
handle_diagnostic_positions_for_json_value(ref_stack.back()->m_data.m_value.array->back());
|
||||
diagnostic_positions::set_from_lexer(ref_stack.back()->m_data.m_value.array->back(), m_lexer_ref);
|
||||
#endif
|
||||
|
||||
return &(ref_stack.back()->m_data.m_value.array->back());
|
||||
@@ -485,7 +498,7 @@ class json_sax_dom_parser
|
||||
*object_element = BasicJsonType(std::forward<Value>(v));
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
handle_diagnostic_positions_for_json_value(*object_element);
|
||||
diagnostic_positions::set_from_lexer(*object_element, m_lexer_ref);
|
||||
#endif
|
||||
|
||||
return object_element;
|
||||
@@ -662,7 +675,7 @@ class json_sax_dom_callback_parser
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
// Set start/end positions for discarded object.
|
||||
handle_diagnostic_positions_for_json_value(*ref_stack.back());
|
||||
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
@@ -778,7 +791,7 @@ class json_sax_dom_callback_parser
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
// Set start/end positions for discarded array.
|
||||
handle_diagnostic_positions_for_json_value(*ref_stack.back());
|
||||
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
@@ -831,72 +844,6 @@ class json_sax_dom_callback_parser
|
||||
|
||||
private:
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
|
||||
{
|
||||
if (m_lexer_ref)
|
||||
{
|
||||
// Lexer has read past the current field value, so set the end position to the current position.
|
||||
// The start position will be set below based on the length of the string representation
|
||||
// of the value.
|
||||
v.end_position = m_lexer_ref->get_position();
|
||||
|
||||
switch (v.type())
|
||||
{
|
||||
case value_t::boolean:
|
||||
{
|
||||
// 4 and 5 are the string length of "true" and "false"
|
||||
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::null:
|
||||
{
|
||||
// 4 is the string length of "null"
|
||||
v.start_position = v.end_position - 4;
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
// escape sequences make the token longer than the value it
|
||||
// parses to, so the start position cannot be derived from
|
||||
// the value; use the offset the lexer recorded instead
|
||||
v.start_position = m_lexer_ref->get_token_start_position();
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::discarded:
|
||||
{
|
||||
v.end_position = std::string::npos;
|
||||
v.start_position = v.end_position;
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::binary:
|
||||
case value_t::number_integer:
|
||||
case value_t::number_unsigned:
|
||||
case value_t::number_float:
|
||||
{
|
||||
v.start_position = v.end_position - m_lexer_ref->get_string().size();
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::object:
|
||||
case value_t::array:
|
||||
{
|
||||
// object and array are handled in start_object() and start_array() handlers
|
||||
// skip setting the values here.
|
||||
break;
|
||||
}
|
||||
default: // LCOV_EXCL_LINE
|
||||
// Handle all possible types discretely, default handler should never be reached.
|
||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
/// if there is a pending duplicate-key stash entry for this exact slot,
|
||||
/// remove it from the stash; if restore_value is true, the stashed
|
||||
/// previous value is moved back into the slot first (use this when the
|
||||
@@ -1018,7 +965,7 @@ class json_sax_dom_callback_parser
|
||||
auto value = BasicJsonType(std::forward<Value>(v));
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
handle_diagnostic_positions_for_json_value(value);
|
||||
diagnostic_positions::set_from_lexer(value, m_lexer_ref);
|
||||
#endif
|
||||
|
||||
// check callback
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
@@ -24,6 +25,7 @@
|
||||
#include <nlohmann/detail/input/string_scan.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||
#include <nlohmann/detail/string_utils.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -500,32 +502,10 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
|
||||
|
||||
// translate codepoint into bytes
|
||||
if (codepoint < 0x80)
|
||||
encode_utf8(static_cast<std::uint32_t>(codepoint), [this](std::uint32_t byte)
|
||||
{
|
||||
// 1-byte characters: 0xxxxxxx (ASCII)
|
||||
add(static_cast<char_int_type>(codepoint));
|
||||
}
|
||||
else if (codepoint <= 0x7FF)
|
||||
{
|
||||
// 2-byte characters: 110xxxxx 10xxxxxx
|
||||
add(static_cast<char_int_type>(0xC0u | (static_cast<unsigned int>(codepoint) >> 6u)));
|
||||
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
|
||||
}
|
||||
else if (codepoint <= 0xFFFF)
|
||||
{
|
||||
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
|
||||
add(static_cast<char_int_type>(0xE0u | (static_cast<unsigned int>(codepoint) >> 12u)));
|
||||
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
|
||||
}
|
||||
else
|
||||
{
|
||||
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
|
||||
add(static_cast<char_int_type>(0xF0u | (static_cast<unsigned int>(codepoint) >> 18u)));
|
||||
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 12u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
|
||||
}
|
||||
add(static_cast<char_int_type>(byte));
|
||||
});
|
||||
|
||||
break;
|
||||
}
|
||||
@@ -1494,45 +1474,30 @@ scan_number_done:
|
||||
*/
|
||||
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||
{
|
||||
// If the caller does not need the converted value (only whether the
|
||||
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
||||
// unsigned/integer token can be reported without calling
|
||||
// strtoull()/strtoll() at all, *provided* we can already tell from
|
||||
// the digit count alone that the conversion cannot overflow 64 bits.
|
||||
// Such tokens are always finite and are accepted unconditionally by
|
||||
// the parser regardless of their actual value (parser::sax_parse_internal()
|
||||
// never checks finiteness for value_unsigned/value_integer), so the
|
||||
// classification below is all that is needed.
|
||||
// accept() only needs to know whether the input is valid, so it sets
|
||||
// discard_number_values (see json.hpp), and an integer token whose
|
||||
// digit count shows that it fits is reported without calling
|
||||
// convert_integer(). A number with up to 18 digits always fits into
|
||||
// both std::uint64_t and std::int64_t (18 nines is about 1e18, below
|
||||
// INT64_MAX, which is about 9.2e18). Longer tokens take the exact path
|
||||
// below, including the fallback to floating point when the value does
|
||||
// not fit.
|
||||
//
|
||||
// A decimal number with up to 18 digits is always representable in
|
||||
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
||||
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
||||
// could not have set errno to ERANGE for it. Numbers with more digits
|
||||
// (rare in practice) fall through to the exact code below, unchanged,
|
||||
// so their handling -- including reclassification to value_float when
|
||||
// the value overflows 64 bits, and rejection when it is not even
|
||||
// finite as a double -- is bit-for-bit identical to before this
|
||||
// optimization.
|
||||
// With a narrower number_unsigned_t/number_integer_t (e.g.
|
||||
// std::uint32_t), the exact path would reclassify some of these tokens
|
||||
// as (finite) floats, while this check reports integers. That does not
|
||||
// change the result of accept(): it always parses through
|
||||
// json_sax_acceptor, whose number callbacks discard their argument and
|
||||
// return true, and the parser rejects neither integers nor finite
|
||||
// floats. value_unsigned/value_integer are left unset here, so a caller
|
||||
// that reads the converted value must not set discard_number_values.
|
||||
//
|
||||
// Note this reasons about std::uint64_t/std::int64_t, not about
|
||||
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
||||
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
||||
// *only* because discard_number_values is exclusively set by
|
||||
// accept() (see json.hpp), and accept() always parses through the
|
||||
// library's own json_sax_acceptor -- never a user-supplied SAX
|
||||
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
||||
// callbacks unconditionally discard their argument and return true.
|
||||
// So for every caller that can reach this branch, neither the token
|
||||
// classification below nor the eventual (possibly narrowed, and on
|
||||
// this fast path left stale/unset) value_unsigned/value_integer is
|
||||
// ever consulted -- an unsigned/integer token is accepted outright,
|
||||
// and even a >18-digit token that this fast path deliberately falls
|
||||
// through for is, once reclassified to value_float, still finite
|
||||
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
||||
// or number_integer_t regardless of that type's width. If this
|
||||
// function is ever taught to run with discard_number_values true for
|
||||
// a caller that *does* read the converted value, this reasoning (and
|
||||
// the fast path below) would need to be revisited.
|
||||
// On contiguous input, scan_number_bulk_contiguous() converts integer
|
||||
// tokens itself and does not pass them to this function, unless
|
||||
// JSON_DIAGNOSTIC_POSITIONS is enabled. This check is therefore only
|
||||
// reached for input without bulk access (e.g. streams), with
|
||||
// JSON_DIAGNOSTIC_POSITIONS, or when scan_number_bulk_contiguous()
|
||||
// falls back to scan_number().
|
||||
if (discard_number_values)
|
||||
{
|
||||
constexpr std::size_t safe_digit_count = 18;
|
||||
@@ -2023,7 +1988,7 @@ scan_number_done:
|
||||
return value_float;
|
||||
}
|
||||
|
||||
/// return current string value (implicitly resets the token; useful only once)
|
||||
/// return current string value
|
||||
string_t& get_string()
|
||||
{
|
||||
// a number token holds '.' regardless of the locale (#4084)
|
||||
@@ -2325,11 +2290,11 @@ scan_number_done:
|
||||
/// the position of the decimal point in token_buffer
|
||||
std::size_t decimal_point_position = std::string::npos;
|
||||
|
||||
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
||||
/// token classification and never looks at the converted numeric value;
|
||||
/// when set, scan_number() may skip strtoull()/strtoll() for
|
||||
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
||||
/// fit into 64 bits (see scan_number())
|
||||
/// whether the caller only needs the token types and never looks at the
|
||||
/// converted numeric values; set only by accept(), which parses through
|
||||
/// json_sax_acceptor. When set, convert_number() skips converting integer
|
||||
/// tokens whose digit count guarantees that they fit into 64 bits (see
|
||||
/// there)
|
||||
const bool discard_number_values = false;
|
||||
};
|
||||
|
||||
|
||||
@@ -54,7 +54,8 @@ using parser_callback_t =
|
||||
/*!
|
||||
@brief syntax analysis
|
||||
|
||||
This class implements a recursive descent parser.
|
||||
This class implements an iterative parser that keeps the open containers on
|
||||
an explicit stack and reports what it reads as SAX events.
|
||||
*/
|
||||
template<typename BasicJsonType, typename InputAdapterType>
|
||||
class parser
|
||||
@@ -98,28 +99,9 @@ class parser
|
||||
if (callback)
|
||||
{
|
||||
json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer);
|
||||
sax_parse_internal(&sdp);
|
||||
|
||||
if (strict)
|
||||
{
|
||||
// in strict mode, input must be completely read
|
||||
if (get_token() != token_type::end_of_input)
|
||||
{
|
||||
sdp.parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(),
|
||||
exception_message(token_type::end_of_input, "value"), nullptr));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// the caller keeps using the input: position it right after
|
||||
// the value by leaving the character that terminated it
|
||||
m_lexer.release_lookahead();
|
||||
}
|
||||
|
||||
// in case of an error, return a discarded value
|
||||
if (sdp.is_errored())
|
||||
if (!parse_dom(sdp, strict))
|
||||
{
|
||||
result = value_t::discarded;
|
||||
return;
|
||||
@@ -135,26 +117,9 @@ class parser
|
||||
else
|
||||
{
|
||||
json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer);
|
||||
sax_parse_internal(&sdp);
|
||||
|
||||
if (strict)
|
||||
{
|
||||
// in strict mode, input must be completely read
|
||||
if (get_token() != token_type::end_of_input)
|
||||
{
|
||||
sdp.parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// see above
|
||||
m_lexer.release_lookahead();
|
||||
}
|
||||
|
||||
// in case of an error, return a discarded value
|
||||
if (sdp.is_errored())
|
||||
if (!parse_dom(sdp, strict))
|
||||
{
|
||||
result = value_t::discarded;
|
||||
return;
|
||||
@@ -207,6 +172,46 @@ class parser
|
||||
}
|
||||
|
||||
private:
|
||||
/*!
|
||||
@brief run a DOM SAX parser to completion and position the lexer
|
||||
|
||||
Shared by both branches of @ref parse(): builds no SAX parser itself,
|
||||
but drives an already-constructed @a json_sax_dom_parser or
|
||||
@ref json_sax_dom_callback_parser through @ref sax_parse_internal(),
|
||||
then applies the strict-EOF check (reporting parse_error.101 through
|
||||
@a sdp on failure) or, in non-strict mode, releases the lookahead so
|
||||
the caller can keep reading the input right after the parsed value.
|
||||
|
||||
@param[in,out] sdp the DOM SAX parser to run
|
||||
@param[in] strict whether to expect the last token to be EOF
|
||||
@return whether @a sdp did not report an error
|
||||
*/
|
||||
template<typename DomSax>
|
||||
bool parse_dom(DomSax& sdp, const bool strict)
|
||||
{
|
||||
sax_parse_internal(&sdp);
|
||||
|
||||
if (strict)
|
||||
{
|
||||
// in strict mode, input must be completely read
|
||||
if (get_token() != token_type::end_of_input)
|
||||
{
|
||||
sdp.parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(),
|
||||
exception_message(token_type::end_of_input, "value"), nullptr));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// the caller keeps using the input: position it right after
|
||||
// the value by leaving the character that terminated it
|
||||
m_lexer.release_lookahead();
|
||||
}
|
||||
|
||||
return !sdp.is_errored();
|
||||
}
|
||||
|
||||
template<typename SAX>
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
bool sax_parse_internal(SAX* sax)
|
||||
@@ -439,8 +444,9 @@ class parser
|
||||
|
||||
// We are done with this array. Before we can parse a
|
||||
// new value, we need to evaluate the new state first.
|
||||
// By setting skip_to_state_evaluation to false, we
|
||||
// are effectively jumping to the beginning of this if.
|
||||
// By setting skip_to_state_evaluation to true, the next
|
||||
// iteration skips parsing a value and evaluates the
|
||||
// enclosing state directly.
|
||||
JSON_ASSERT(!states.empty());
|
||||
states.pop_back();
|
||||
skip_to_state_evaluation = true;
|
||||
@@ -500,8 +506,9 @@ class parser
|
||||
|
||||
// We are done with this object. Before we can parse a
|
||||
// new value, we need to evaluate the new state first.
|
||||
// By setting skip_to_state_evaluation to false, we
|
||||
// are effectively jumping to the beginning of this if.
|
||||
// By setting skip_to_state_evaluation to true, the next
|
||||
// iteration skips parsing a value and evaluates the
|
||||
// enclosing state directly.
|
||||
JSON_ASSERT(!states.empty());
|
||||
states.pop_back();
|
||||
skip_to_state_evaluation = true;
|
||||
|
||||
@@ -36,6 +36,61 @@ StringType to_string(std::size_t value)
|
||||
return result;
|
||||
}
|
||||
|
||||
///////////////////
|
||||
// UTF-8 encoding //
|
||||
///////////////////
|
||||
|
||||
/*!
|
||||
@brief encode a Unicode code point as UTF-8
|
||||
|
||||
Used to turn a decoded code point back into bytes: by the wide-string input
|
||||
adapters in input_adapters.hpp (one code point per UTF-32 unit, per UTF-16
|
||||
unit outside the surrogate range, and per valid UTF-16 surrogate pair), and
|
||||
by the lexer's `\uXXXX`/`\uXXXX\uYYYY` handling in lexer.hpp. Passing a
|
||||
code point above U+10FFFF, or one in the surrogate range U+D800..U+DFFF, is
|
||||
undefined behavior; callers are expected to have rejected those already
|
||||
(the wide-string adapters pass malformed units through unencoded instead of
|
||||
calling this function, and the lexer rejects unpaired surrogates before
|
||||
reaching it).
|
||||
|
||||
@tparam Out a callable invoked with one byte (as std::uint32_t, 0x00..0xFF)
|
||||
at a time, most significant byte first
|
||||
@param[in] cp the code point to encode (at most U+10FFFF)
|
||||
@param[in] out called once for each byte of the UTF-8 encoding of @a cp
|
||||
*/
|
||||
template<typename Out>
|
||||
void encode_utf8(std::uint32_t cp, Out&& out)
|
||||
{
|
||||
JSON_ASSERT(cp <= 0x10FFFF);
|
||||
|
||||
if (cp < 0x80)
|
||||
{
|
||||
// 1-byte characters: 0xxxxxxx (ASCII)
|
||||
out(cp);
|
||||
}
|
||||
else if (cp <= 0x7FF)
|
||||
{
|
||||
// 2-byte characters: 110xxxxx 10xxxxxx
|
||||
out(0xC0u | (cp >> 6u));
|
||||
out(0x80u | (cp & 0x3Fu));
|
||||
}
|
||||
else if (cp <= 0xFFFF)
|
||||
{
|
||||
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
|
||||
out(0xE0u | (cp >> 12u));
|
||||
out(0x80u | ((cp >> 6u) & 0x3Fu));
|
||||
out(0x80u | (cp & 0x3Fu));
|
||||
}
|
||||
else
|
||||
{
|
||||
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
|
||||
out(0xF0u | (cp >> 18u));
|
||||
out(0x80u | ((cp >> 12u) & 0x3Fu));
|
||||
out(0x80u | ((cp >> 6u) & 0x3Fu));
|
||||
out(0x80u | (cp & 0x3Fu));
|
||||
}
|
||||
}
|
||||
|
||||
///////////////////
|
||||
// UTF-8 decoding //
|
||||
///////////////////
|
||||
@@ -51,11 +106,23 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
|
||||
written by Björn Hoehrmann. See
|
||||
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
|
||||
|
||||
This decoder is the single source of truth for UTF-8 validation in this
|
||||
library: it is used both by the serializer (to escape and, in strict mode,
|
||||
reject ill-formed UTF-8 when dumping a string) and by the binary readers
|
||||
(to reject ill-formed UTF-8 in CBOR/MessagePack/BSON/UBJSON text strings at
|
||||
decode time; see @ref is_valid_utf8 below).
|
||||
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four
|
||||
places, which differ in speed, diagnostics, and how they read the input:
|
||||
|
||||
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in
|
||||
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR,
|
||||
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
|
||||
text strings at decode time).
|
||||
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
|
||||
diagnostic for each kind of error.
|
||||
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
|
||||
bulk string scan, the bulk path of the BON8 reader, and the BON8 writer.
|
||||
They must accept exactly what the lexer's switch accepts.
|
||||
- the byte path of binary_reader::get_bon8_string(): BON8 input without bulk
|
||||
access, and the bytes the bulk path leaves to it.
|
||||
|
||||
All four must accept the same set of sequences, so a change to one needs a
|
||||
matching change to the others.
|
||||
|
||||
@param[in,out] state the current decoder state
|
||||
@param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT)
|
||||
|
||||
@@ -149,6 +149,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
friend class ::nlohmann::detail::json_sax_dom_parser;
|
||||
template<typename BasicJsonType, typename InputAdapterType>
|
||||
friend class ::nlohmann::detail::json_sax_dom_callback_parser;
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
friend struct ::nlohmann::detail::diagnostic_positions;
|
||||
#endif
|
||||
friend class ::nlohmann::detail::exception;
|
||||
|
||||
/// workaround type for MSVC
|
||||
|
||||
+330
-452
File diff suppressed because it is too large
Load Diff
@@ -11,12 +11,7 @@
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <cmath>
|
||||
#include <fstream>
|
||||
#include <limits>
|
||||
#include <map>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include "make_test_data_available.hpp"
|
||||
|
||||
TEST_CASE("Binary Formats" * doctest::skip())
|
||||
@@ -229,139 +224,3 @@ TEST_CASE("Binary Formats" * doctest::skip())
|
||||
CHECK((100.0 * double(ubjson_3_size) / double(json_size)) == Approx(89.450));
|
||||
}
|
||||
}
|
||||
|
||||
namespace
|
||||
{
|
||||
// the binary formats as function pointers for "Binary formats with narrow number types";
|
||||
// named functions rather than lambdas, because clang 3.5 cannot convert a lambda
|
||||
// to a function pointer in the braced initializer of the format table
|
||||
using narrow_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int32_t, std::uint32_t, float>;
|
||||
using bytes = std::vector<std::uint8_t>;
|
||||
|
||||
bytes encode_cbor(const json& j)
|
||||
{
|
||||
return json::to_cbor(j);
|
||||
}
|
||||
narrow_json decode_cbor(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_cbor(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
bytes encode_msgpack(const json& j)
|
||||
{
|
||||
return json::to_msgpack(j);
|
||||
}
|
||||
narrow_json decode_msgpack(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_msgpack(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
bytes encode_ubjson(const json& j)
|
||||
{
|
||||
return json::to_ubjson(j);
|
||||
}
|
||||
narrow_json decode_ubjson(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_ubjson(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
bytes encode_bjdata(const json& j)
|
||||
{
|
||||
return json::to_bjdata(j);
|
||||
}
|
||||
narrow_json decode_bjdata(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_bjdata(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
// BSON can only store numbers as object members
|
||||
bytes encode_bson(const json& j)
|
||||
{
|
||||
return json::to_bson(json{{"a", j}});
|
||||
}
|
||||
narrow_json decode_bson(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
const auto result = narrow_json::from_bson(v, true, allow_exceptions);
|
||||
return result.is_discarded() ? result : result.at("a");
|
||||
}
|
||||
|
||||
bytes encode_bon8(const json& j)
|
||||
{
|
||||
return json::to_bon8(j);
|
||||
}
|
||||
narrow_json decode_bon8(const bytes& v, bool allow_exceptions)
|
||||
{
|
||||
return narrow_json::from_bon8(v, true, allow_exceptions);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("Binary formats with narrow number types")
|
||||
{
|
||||
// Numbers that do not fit the number types are handled like the lexer
|
||||
// handles them in JSON text: an integer that fits neither integer type is
|
||||
// stored as a floating-point number, and a finite floating-point number
|
||||
// that overflows number_float_t is rejected with out_of_range.406.
|
||||
struct binary_format
|
||||
{
|
||||
const char* name;
|
||||
bytes (*encode)(const json&);
|
||||
narrow_json (*decode)(const bytes&, bool);
|
||||
};
|
||||
|
||||
const std::vector<binary_format> formats =
|
||||
{
|
||||
{"CBOR", encode_cbor, decode_cbor},
|
||||
{"MessagePack", encode_msgpack, decode_msgpack},
|
||||
{"UBJSON", encode_ubjson, decode_ubjson},
|
||||
{"BJData", encode_bjdata, decode_bjdata},
|
||||
{"BSON", encode_bson, decode_bson},
|
||||
{"BON8", encode_bon8, decode_bon8},
|
||||
};
|
||||
|
||||
for (const auto& format : formats)
|
||||
{
|
||||
const std::string name = format.name;
|
||||
INFO("format := ", name);
|
||||
const auto roundtrip = [&format](const json & j)
|
||||
{
|
||||
return format.decode(format.encode(j), true);
|
||||
};
|
||||
|
||||
// integers that fit keep their type
|
||||
CHECK(roundtrip(json(-5)).is_number_integer());
|
||||
CHECK(roundtrip(json(-5)).get<std::int32_t>() == -5);
|
||||
CHECK(roundtrip(json(3000000000u)).is_number_unsigned());
|
||||
CHECK(roundtrip(json(3000000000u)).get<std::uint32_t>() == 3000000000u);
|
||||
|
||||
// integers that fit neither integer type are stored as float
|
||||
CHECK(roundtrip(json(5000000000u)).is_number_float());
|
||||
CHECK(roundtrip(json(5000000000u)).get<float>() == 5000000000.0f);
|
||||
if (name != "BON8") // BON8 cannot encode integers above INT64_MAX
|
||||
{
|
||||
CHECK(roundtrip(json(10000000000000000000u)).is_number_float());
|
||||
CHECK(roundtrip(json(10000000000000000000u)).get<float>() == 10000000000000000000.0f);
|
||||
}
|
||||
CHECK(roundtrip(json(-3000000000LL)).is_number_float());
|
||||
CHECK(roundtrip(json(-3000000000LL)).get<float>() == -3000000000.0f);
|
||||
CHECK(roundtrip(json(-5000000000LL)).is_number_float());
|
||||
CHECK(roundtrip(json(-5000000000LL)).get<float>() == -5000000000.0f);
|
||||
|
||||
// floating-point numbers that fit
|
||||
CHECK(roundtrip(json(1.5)).get<float>() == 1.5f);
|
||||
const auto just_above_max = std::nextafter(static_cast<double>((std::numeric_limits<float>::max)()),
|
||||
std::numeric_limits<double>::infinity());
|
||||
CHECK(roundtrip(json(just_above_max)).get<float>() == (std::numeric_limits<float>::max)());
|
||||
|
||||
// infinity and NaN are passed on
|
||||
CHECK(std::isinf(roundtrip(json(std::numeric_limits<double>::infinity())).get<float>()));
|
||||
CHECK(std::isnan(roundtrip(json(std::numeric_limits<double>::quiet_NaN())).get<float>()));
|
||||
|
||||
// finite floating-point numbers that overflow number_float_t are rejected
|
||||
const std::string message = "[json.exception.out_of_range.406] syntax error while parsing " + name
|
||||
+ " value: number overflow";
|
||||
CHECK_THROWS_WITH_AS(roundtrip(json(1e300)), message.c_str(), narrow_json::out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(roundtrip(json(-1e300)), message.c_str(), narrow_json::out_of_range&);
|
||||
CHECK(format.decode(format.encode(json(1e300)), false).is_discarded());
|
||||
}
|
||||
}
|
||||
|
||||
+14
-16
@@ -3187,8 +3187,7 @@ TEST_CASE("Tagged values")
|
||||
// CBOR encodes negative integers as: result = -1 - n
|
||||
// For type 0x3B, n is an 8-byte uint64_t. Valid range for n with
|
||||
// the default int64_t is [0, INT64_MAX], producing results in [INT64_MIN, -1].
|
||||
// When n > INT64_MAX, the result exceeds int64_t range and is stored
|
||||
// as a floating-point number, as the lexer does for JSON text.
|
||||
// When n > INT64_MAX, the result exceeds int64_t range and is rejected.
|
||||
|
||||
SECTION("n = 0 is valid (result = -1)")
|
||||
{
|
||||
@@ -3209,34 +3208,33 @@ TEST_CASE("Tagged values")
|
||||
CHECK(result.get<int64_t>() == (std::numeric_limits<int64_t>::min)());
|
||||
}
|
||||
|
||||
SECTION("n = INT64_MAX + 1 is stored as float")
|
||||
SECTION("n = INT64_MAX + 1 is rejected (overflow)")
|
||||
{
|
||||
// n = INT64_MAX + 1 (0x8000000000000000)
|
||||
// result = -1 - n = -9223372036854775809, which exceeds int64_t range;
|
||||
// the nearest double is -9223372036854775808.0
|
||||
// result = -1 - n = -9223372036854775809, which exceeds int64_t range
|
||||
const std::vector<uint8_t> input = {0x3B, 0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00};
|
||||
const auto result = json::from_cbor(input);
|
||||
CHECK(result.is_number_float());
|
||||
CHECK(result.get<double>() == -9223372036854775808.0);
|
||||
CHECK(result == json::parse("-9223372036854775809"));
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input),
|
||||
"[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow",
|
||||
json::parse_error);
|
||||
}
|
||||
|
||||
SECTION("n = UINT64_MAX is stored as float")
|
||||
SECTION("n = UINT64_MAX is rejected (overflow)")
|
||||
{
|
||||
// n = UINT64_MAX (0xFFFFFFFFFFFFFFFF)
|
||||
// result = -1 - n = -18446744073709551616, which exceeds int64_t range
|
||||
const std::vector<uint8_t> input = {0x3B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF};
|
||||
const auto result = json::from_cbor(input);
|
||||
CHECK(result.is_number_float());
|
||||
CHECK(result.get<double>() == -18446744073709551616.0);
|
||||
CHECK(result == json::parse("-18446744073709551616"));
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input),
|
||||
"[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow",
|
||||
json::parse_error);
|
||||
}
|
||||
|
||||
SECTION("overflow with allow_exceptions=false is not an error")
|
||||
SECTION("overflow with allow_exceptions=false returns discarded")
|
||||
{
|
||||
const std::vector<uint8_t> input = {0x3B, 0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00};
|
||||
const auto result = json::from_cbor(input, true, false);
|
||||
CHECK(result.is_number_float());
|
||||
CHECK(result.is_discarded());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user