mirror of
https://github.com/nlohmann/json.git
synced 2026-09-07 16:57:59 +00:00
Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ee69490b28 | ||
|
|
3970ecccd1 | ||
|
|
00e5d21041 |
@@ -69,6 +69,12 @@ The library uses the following mapping from JSON values types to UBJSON types ac
|
|||||||
Note that `use_size = true` alone may result in larger representations - the benefit of this parameter is that the
|
Note that `use_size = true` alone may result in larger representations - the benefit of this parameter is that the
|
||||||
receiving side is immediately informed on the number of elements of the container.
|
receiving side is immediately informed on the number of elements of the container.
|
||||||
|
|
||||||
|
An array whose type marker is `Z` (null), `T` (true) or `F` (false) stores no payload at all, because the marker
|
||||||
|
already is the value. Its declared count is therefore the only thing that decides how much memory the receiving side
|
||||||
|
allocates, and a handful of bytes can describe billions of elements. `from_ubjson` rejects such an array with
|
||||||
|
[`out_of_range.408`](../../home/exceptions.md#jsonexceptionout_of_range408) when the count exceeds 1,048,576, and
|
||||||
|
`to_ubjson` writes longer arrays of these types without the annotation, so any value it produces can be read back.
|
||||||
|
|
||||||
!!! info "Binary values"
|
!!! info "Binary values"
|
||||||
|
|
||||||
If the JSON data contains the binary type, the value stored is a list of integers, as suggested by the UBJSON
|
If the JSON data contains the binary type, the value stored is a list of integers, as suggested by the UBJSON
|
||||||
|
|||||||
@@ -868,6 +868,12 @@ The size of an array or object in a [binary format](../features/binary_formats/i
|
|||||||
the size following `#` for [UBJSON](../features/binary_formats/ubjson.md)/[BJData](../features/binary_formats/bjdata.md),
|
the size following `#` for [UBJSON](../features/binary_formats/ubjson.md)/[BJData](../features/binary_formats/bjdata.md),
|
||||||
or the encoded length for [CBOR](../features/binary_formats/cbor.md).
|
or the encoded length for [CBOR](../features/binary_formats/cbor.md).
|
||||||
|
|
||||||
|
The exception is also thrown for a [UBJSON](../features/binary_formats/ubjson.md) array of a type that is encoded by its
|
||||||
|
marker alone (`Z`, `T` or `F`) whose declared count exceeds 1,048,576. Such an array has no payload, so its count alone
|
||||||
|
decides how much memory is allocated, and a handful of bytes would otherwise describe billions of values.
|
||||||
|
[`to_ubjson`](../api/basic_json/to_ubjson.md) writes longer arrays of these types without the size and type annotation,
|
||||||
|
so any value it produces can still be read back.
|
||||||
|
|
||||||
!!! failure "Example messages"
|
!!! failure "Example messages"
|
||||||
|
|
||||||
```
|
```
|
||||||
@@ -879,6 +885,9 @@ or the encoded length for [CBOR](../features/binary_formats/cbor.md).
|
|||||||
```
|
```
|
||||||
[json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive map size
|
[json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive map size
|
||||||
```
|
```
|
||||||
|
```
|
||||||
|
[json.exception.out_of_range.408] syntax error while parsing UBJSON size: excessive array size
|
||||||
|
```
|
||||||
|
|
||||||
### json.exception.out_of_range.409
|
### json.exception.out_of_range.409
|
||||||
|
|
||||||
|
|||||||
@@ -58,6 +58,26 @@ inline bool little_endianness(int num = 1) noexcept
|
|||||||
return *reinterpret_cast<char*>(&num) == 1;
|
return *reinterpret_cast<char*>(&num) == 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief largest element count accepted for a UBJSON container of a valueless type
|
||||||
|
|
||||||
|
An element of type 'Z' (null), 'T' (true) or 'F' (false) is encoded by its
|
||||||
|
type marker alone, so an optimized container of one of those types has no
|
||||||
|
payload at all and its declared count is the only thing that decides how much
|
||||||
|
is allocated: `[$Z#L` followed by a large count turns some ten bytes of input
|
||||||
|
into that many values (see #2793, which reports 35 GB and 150 seconds). Every
|
||||||
|
other type costs at least one byte per element and is bounded by the end of
|
||||||
|
the input.
|
||||||
|
|
||||||
|
This is a sanity bound rather than a security boundary, and it is far above
|
||||||
|
any container met in practice. @ref binary_writer falls back to the
|
||||||
|
unoptimized encoding for longer containers, so that a value serialized by
|
||||||
|
this library can always be read back.
|
||||||
|
|
||||||
|
@sa https://github.com/nlohmann/json/issues/2793
|
||||||
|
*/
|
||||||
|
JSON_INLINE_VARIABLE constexpr std::size_t max_valueless_container_size = 1 << 20;
|
||||||
|
|
||||||
///////////////////
|
///////////////////
|
||||||
// binary reader //
|
// binary reader //
|
||||||
///////////////////
|
///////////////////
|
||||||
@@ -996,23 +1016,21 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief reads a CBOR string
|
@brief reads a definite-length CBOR string
|
||||||
|
|
||||||
This function first reads starting bytes to determine the expected
|
Reads everything @ref get_cbor_string accepts except the indefinite-length
|
||||||
string length and then copies this number of bytes into a string.
|
form, which that function handles itself. The bytes are appended to @a
|
||||||
Additionally, CBOR's strings with indefinite lengths are supported.
|
result, so consecutive chunks of an indefinite-length string can be read
|
||||||
|
into the same string.
|
||||||
|
|
||||||
@param[out] result created string
|
@param[out] result string the bytes are appended to
|
||||||
|
|
||||||
@return whether string creation completed
|
@return whether string creation completed
|
||||||
*/
|
|
||||||
bool get_cbor_string(string_t& result)
|
|
||||||
{
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "string")))
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
|
@pre @a current is not EOF
|
||||||
|
*/
|
||||||
|
bool get_cbor_string_chunk(string_t& result)
|
||||||
|
{
|
||||||
switch (current)
|
switch (current)
|
||||||
{
|
{
|
||||||
// UTF-8 string (0x00..0x17 bytes follow)
|
// UTF-8 string (0x00..0x17 bytes follow)
|
||||||
@@ -1068,20 +1086,6 @@ class binary_reader
|
|||||||
return get_number(input_format_t::cbor, len) && get_string(input_format_t::cbor, len, result);
|
return get_number(input_format_t::cbor, len) && get_string(input_format_t::cbor, len, result);
|
||||||
}
|
}
|
||||||
|
|
||||||
case 0x7F: // UTF-8 string (indefinite length)
|
|
||||||
{
|
|
||||||
while (get() != 0xFF)
|
|
||||||
{
|
|
||||||
string_t chunk;
|
|
||||||
if (!get_cbor_string(chunk))
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
result.append(chunk);
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
default:
|
default:
|
||||||
{
|
{
|
||||||
auto last_token = get_token_string();
|
auto last_token = get_token_string();
|
||||||
@@ -1092,23 +1096,82 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief reads a CBOR byte array
|
@brief reads a CBOR string
|
||||||
|
|
||||||
This function first reads starting bytes to determine the expected
|
This function first reads starting bytes to determine the expected
|
||||||
byte array length and then copies this number of bytes into the byte array.
|
string length and then copies this number of bytes into a string.
|
||||||
Additionally, CBOR's byte arrays with indefinite lengths are supported.
|
Additionally, CBOR's strings with indefinite lengths are supported.
|
||||||
|
|
||||||
@param[out] result created byte array
|
@param[out] result created string
|
||||||
|
|
||||||
|
@return whether string creation completed
|
||||||
|
*/
|
||||||
|
bool get_cbor_string(string_t& result)
|
||||||
|
{
|
||||||
|
// number of indefinite-length strings that have been opened and not
|
||||||
|
// closed yet. RFC 8949, Section 3.2.3 does not permit nesting them,
|
||||||
|
// but this reader has always accepted it, so the open levels are
|
||||||
|
// counted instead of recursed through, which overflowed the stack for
|
||||||
|
// an input of repeated 0x7F bytes (see #5104). Every chunk is appended
|
||||||
|
// to the same result, so no per-level state is needed.
|
||||||
|
std::size_t open = 0;
|
||||||
|
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "string")))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (current == 0x7F) // UTF-8 string (indefinite length)
|
||||||
|
{
|
||||||
|
++open;
|
||||||
|
get();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// a break marker closes the innermost indefinite-length string;
|
||||||
|
// outside of one it is not a string and falls through to the error
|
||||||
|
if (open != 0 && current == 0xFF)
|
||||||
|
{
|
||||||
|
if (--open == 0)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
get();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result)))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (open == 0)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
get();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief reads a definite-length CBOR byte array
|
||||||
|
|
||||||
|
Reads everything @ref get_cbor_binary accepts except the indefinite-length
|
||||||
|
form, which that function handles itself. The bytes are appended to @a
|
||||||
|
result, so consecutive chunks of an indefinite-length byte array can be
|
||||||
|
read into the same byte array.
|
||||||
|
|
||||||
|
@param[out] result byte array the bytes are appended to
|
||||||
|
|
||||||
@return whether byte array creation completed
|
@return whether byte array creation completed
|
||||||
*/
|
|
||||||
bool get_cbor_binary(binary_t& result)
|
|
||||||
{
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "binary")))
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
|
@pre @a current is not EOF
|
||||||
|
*/
|
||||||
|
bool get_cbor_binary_chunk(binary_t& result)
|
||||||
|
{
|
||||||
switch (current)
|
switch (current)
|
||||||
{
|
{
|
||||||
// Binary data (0x00..0x17 bytes follow)
|
// Binary data (0x00..0x17 bytes follow)
|
||||||
@@ -1168,20 +1231,6 @@ class binary_reader
|
|||||||
get_binary(input_format_t::cbor, len, result);
|
get_binary(input_format_t::cbor, len, result);
|
||||||
}
|
}
|
||||||
|
|
||||||
case 0x5F: // Binary data (indefinite length)
|
|
||||||
{
|
|
||||||
while (get() != 0xFF)
|
|
||||||
{
|
|
||||||
binary_t chunk;
|
|
||||||
if (!get_cbor_binary(chunk))
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
result.insert(result.end(), chunk.begin(), chunk.end());
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
default:
|
default:
|
||||||
{
|
{
|
||||||
auto last_token = get_token_string();
|
auto last_token = get_token_string();
|
||||||
@@ -1191,6 +1240,63 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief reads a CBOR byte array
|
||||||
|
|
||||||
|
This function first reads starting bytes to determine the expected
|
||||||
|
byte array length and then copies this number of bytes into the byte array.
|
||||||
|
Additionally, CBOR's byte arrays with indefinite lengths are supported.
|
||||||
|
|
||||||
|
@param[out] result created byte array
|
||||||
|
|
||||||
|
@return whether byte array creation completed
|
||||||
|
*/
|
||||||
|
bool get_cbor_binary(binary_t& result)
|
||||||
|
{
|
||||||
|
// the open indefinite-length byte arrays are counted rather than
|
||||||
|
// recursed through, for the reason given in @ref get_cbor_string
|
||||||
|
std::size_t open = 0;
|
||||||
|
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "binary")))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (current == 0x5F) // Binary data (indefinite length)
|
||||||
|
{
|
||||||
|
++open;
|
||||||
|
get();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// a break marker closes the innermost indefinite-length byte
|
||||||
|
// array; outside of one it falls through to the error below
|
||||||
|
if (open != 0 && current == 0xFF)
|
||||||
|
{
|
||||||
|
if (--open == 0)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
get();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!get_cbor_binary_chunk(result)))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (open == 0)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
get();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief narrow a definite CBOR array/map length to std::size_t
|
@brief narrow a definite CBOR array/map length to std::size_t
|
||||||
|
|
||||||
@@ -2391,7 +2497,12 @@ class binary_reader
|
|||||||
{
|
{
|
||||||
result.first = npos; // size
|
result.first = npos; // size
|
||||||
result.second = 0; // type
|
result.second = 0; // type
|
||||||
bool is_ndarray = false;
|
// seed the flag with the caller's context: inside an ndarray dimension
|
||||||
|
// vector another ndarray is not allowed, and get_ubjson_size_value()
|
||||||
|
// rejects it up front instead of reading it and reporting afterwards.
|
||||||
|
// Seeding it with `false` made every '#' of a "[#[#[..." chain descend
|
||||||
|
// another level, which overflowed the stack (see #5104).
|
||||||
|
bool is_ndarray = inside_ndarray;
|
||||||
|
|
||||||
get_ignore_noop();
|
get_ignore_noop();
|
||||||
|
|
||||||
@@ -2424,13 +2535,11 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
|
|
||||||
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
|
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
|
||||||
if (input_format == input_format_t::bjdata && is_ndarray)
|
// an ndarray was read here only if the flag flipped; when it was
|
||||||
|
// seeded true, get_ubjson_size_value() already rejected the nested
|
||||||
|
// dimension vector
|
||||||
|
if (input_format == input_format_t::bjdata && is_ndarray && !inside_ndarray)
|
||||||
{
|
{
|
||||||
if (inside_ndarray)
|
|
||||||
{
|
|
||||||
return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read,
|
|
||||||
exception_message(input_format, "ndarray can not be recursive", "size"), nullptr));
|
|
||||||
}
|
|
||||||
result.second |= (1 << 8); // use bit 8 to indicate ndarray, all UBJSON and BJData markers should be ASCII letters
|
result.second |= (1 << 8); // use bit 8 to indicate ndarray, all UBJSON and BJData markers should be ASCII letters
|
||||||
}
|
}
|
||||||
return is_error;
|
return is_error;
|
||||||
@@ -2439,7 +2548,7 @@ class binary_reader
|
|||||||
if (current == '#')
|
if (current == '#')
|
||||||
{
|
{
|
||||||
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
|
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
|
||||||
if (input_format == input_format_t::bjdata && is_ndarray)
|
if (input_format == input_format_t::bjdata && is_ndarray && !inside_ndarray)
|
||||||
{
|
{
|
||||||
return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read,
|
return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read,
|
||||||
exception_message(input_format, "ndarray requires both type and size", "size"), nullptr));
|
exception_message(input_format, "ndarray requires both type and size", "size"), nullptr));
|
||||||
@@ -2710,6 +2819,17 @@ class binary_reader
|
|||||||
|
|
||||||
if (size_and_type.first != npos)
|
if (size_and_type.first != npos)
|
||||||
{
|
{
|
||||||
|
// reading an element of a valueless type consumes no input, so the
|
||||||
|
// declared count alone decides how much is allocated; the check is
|
||||||
|
// made before the start event so that no container is opened that
|
||||||
|
// is then abandoned. See @ref max_valueless_container_size.
|
||||||
|
if (JSON_HEDLEY_UNLIKELY((size_and_type.second == 'Z' || size_and_type.second == 'T' || size_and_type.second == 'F')
|
||||||
|
&& size_and_type.first > max_valueless_container_size))
|
||||||
|
{
|
||||||
|
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
|
||||||
|
exception_message(input_format, "excessive array size", "size"), nullptr));
|
||||||
|
}
|
||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!sax->start_array(size_and_type.first)))
|
if (JSON_HEDLEY_UNLIKELY(!sax->start_array(size_and_type.first)))
|
||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
|
|||||||
@@ -826,7 +826,17 @@ class binary_writer
|
|||||||
|
|
||||||
std::vector<CharType> bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type
|
std::vector<CharType> bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type
|
||||||
|
|
||||||
if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end()))
|
// an optimized array of a valueless type carries no payload, so a
|
||||||
|
// reader has nothing but the declared count to bound the allocation
|
||||||
|
// by and refuses an excessive one. Write the unoptimized form for
|
||||||
|
// those, at one byte per element, so the result can be read back.
|
||||||
|
// Objects are not affected: every element is preceded by its key.
|
||||||
|
const bool valueless_type = (first_prefix == 'Z' || first_prefix == 'T' || first_prefix == 'F');
|
||||||
|
const bool excessive_valueless = valueless_type
|
||||||
|
&& j.m_data.m_value.array->size() > detail::max_valueless_container_size;
|
||||||
|
|
||||||
|
if (same_prefix && !excessive_valueless
|
||||||
|
&& !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end()))
|
||||||
{
|
{
|
||||||
prefix_required = false;
|
prefix_required = false;
|
||||||
oa->write_character(to_char_type('$'));
|
oa->write_character(to_char_type('$'));
|
||||||
|
|||||||
@@ -10745,6 +10745,26 @@ inline bool little_endianness(int num = 1) noexcept
|
|||||||
return *reinterpret_cast<char*>(&num) == 1;
|
return *reinterpret_cast<char*>(&num) == 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief largest element count accepted for a UBJSON container of a valueless type
|
||||||
|
|
||||||
|
An element of type 'Z' (null), 'T' (true) or 'F' (false) is encoded by its
|
||||||
|
type marker alone, so an optimized container of one of those types has no
|
||||||
|
payload at all and its declared count is the only thing that decides how much
|
||||||
|
is allocated: `[$Z#L` followed by a large count turns some ten bytes of input
|
||||||
|
into that many values (see #2793, which reports 35 GB and 150 seconds). Every
|
||||||
|
other type costs at least one byte per element and is bounded by the end of
|
||||||
|
the input.
|
||||||
|
|
||||||
|
This is a sanity bound rather than a security boundary, and it is far above
|
||||||
|
any container met in practice. @ref binary_writer falls back to the
|
||||||
|
unoptimized encoding for longer containers, so that a value serialized by
|
||||||
|
this library can always be read back.
|
||||||
|
|
||||||
|
@sa https://github.com/nlohmann/json/issues/2793
|
||||||
|
*/
|
||||||
|
JSON_INLINE_VARIABLE constexpr std::size_t max_valueless_container_size = 1 << 20;
|
||||||
|
|
||||||
///////////////////
|
///////////////////
|
||||||
// binary reader //
|
// binary reader //
|
||||||
///////////////////
|
///////////////////
|
||||||
@@ -11683,23 +11703,21 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief reads a CBOR string
|
@brief reads a definite-length CBOR string
|
||||||
|
|
||||||
This function first reads starting bytes to determine the expected
|
Reads everything @ref get_cbor_string accepts except the indefinite-length
|
||||||
string length and then copies this number of bytes into a string.
|
form, which that function handles itself. The bytes are appended to @a
|
||||||
Additionally, CBOR's strings with indefinite lengths are supported.
|
result, so consecutive chunks of an indefinite-length string can be read
|
||||||
|
into the same string.
|
||||||
|
|
||||||
@param[out] result created string
|
@param[out] result string the bytes are appended to
|
||||||
|
|
||||||
@return whether string creation completed
|
@return whether string creation completed
|
||||||
*/
|
|
||||||
bool get_cbor_string(string_t& result)
|
|
||||||
{
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "string")))
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
|
@pre @a current is not EOF
|
||||||
|
*/
|
||||||
|
bool get_cbor_string_chunk(string_t& result)
|
||||||
|
{
|
||||||
switch (current)
|
switch (current)
|
||||||
{
|
{
|
||||||
// UTF-8 string (0x00..0x17 bytes follow)
|
// UTF-8 string (0x00..0x17 bytes follow)
|
||||||
@@ -11755,20 +11773,6 @@ class binary_reader
|
|||||||
return get_number(input_format_t::cbor, len) && get_string(input_format_t::cbor, len, result);
|
return get_number(input_format_t::cbor, len) && get_string(input_format_t::cbor, len, result);
|
||||||
}
|
}
|
||||||
|
|
||||||
case 0x7F: // UTF-8 string (indefinite length)
|
|
||||||
{
|
|
||||||
while (get() != 0xFF)
|
|
||||||
{
|
|
||||||
string_t chunk;
|
|
||||||
if (!get_cbor_string(chunk))
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
result.append(chunk);
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
default:
|
default:
|
||||||
{
|
{
|
||||||
auto last_token = get_token_string();
|
auto last_token = get_token_string();
|
||||||
@@ -11779,23 +11783,82 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief reads a CBOR byte array
|
@brief reads a CBOR string
|
||||||
|
|
||||||
This function first reads starting bytes to determine the expected
|
This function first reads starting bytes to determine the expected
|
||||||
byte array length and then copies this number of bytes into the byte array.
|
string length and then copies this number of bytes into a string.
|
||||||
Additionally, CBOR's byte arrays with indefinite lengths are supported.
|
Additionally, CBOR's strings with indefinite lengths are supported.
|
||||||
|
|
||||||
@param[out] result created byte array
|
@param[out] result created string
|
||||||
|
|
||||||
|
@return whether string creation completed
|
||||||
|
*/
|
||||||
|
bool get_cbor_string(string_t& result)
|
||||||
|
{
|
||||||
|
// number of indefinite-length strings that have been opened and not
|
||||||
|
// closed yet. RFC 8949, Section 3.2.3 does not permit nesting them,
|
||||||
|
// but this reader has always accepted it, so the open levels are
|
||||||
|
// counted instead of recursed through, which overflowed the stack for
|
||||||
|
// an input of repeated 0x7F bytes (see #5104). Every chunk is appended
|
||||||
|
// to the same result, so no per-level state is needed.
|
||||||
|
std::size_t open = 0;
|
||||||
|
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "string")))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (current == 0x7F) // UTF-8 string (indefinite length)
|
||||||
|
{
|
||||||
|
++open;
|
||||||
|
get();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// a break marker closes the innermost indefinite-length string;
|
||||||
|
// outside of one it is not a string and falls through to the error
|
||||||
|
if (open != 0 && current == 0xFF)
|
||||||
|
{
|
||||||
|
if (--open == 0)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
get();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result)))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (open == 0)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
get();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief reads a definite-length CBOR byte array
|
||||||
|
|
||||||
|
Reads everything @ref get_cbor_binary accepts except the indefinite-length
|
||||||
|
form, which that function handles itself. The bytes are appended to @a
|
||||||
|
result, so consecutive chunks of an indefinite-length byte array can be
|
||||||
|
read into the same byte array.
|
||||||
|
|
||||||
|
@param[out] result byte array the bytes are appended to
|
||||||
|
|
||||||
@return whether byte array creation completed
|
@return whether byte array creation completed
|
||||||
*/
|
|
||||||
bool get_cbor_binary(binary_t& result)
|
|
||||||
{
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "binary")))
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
|
@pre @a current is not EOF
|
||||||
|
*/
|
||||||
|
bool get_cbor_binary_chunk(binary_t& result)
|
||||||
|
{
|
||||||
switch (current)
|
switch (current)
|
||||||
{
|
{
|
||||||
// Binary data (0x00..0x17 bytes follow)
|
// Binary data (0x00..0x17 bytes follow)
|
||||||
@@ -11855,20 +11918,6 @@ class binary_reader
|
|||||||
get_binary(input_format_t::cbor, len, result);
|
get_binary(input_format_t::cbor, len, result);
|
||||||
}
|
}
|
||||||
|
|
||||||
case 0x5F: // Binary data (indefinite length)
|
|
||||||
{
|
|
||||||
while (get() != 0xFF)
|
|
||||||
{
|
|
||||||
binary_t chunk;
|
|
||||||
if (!get_cbor_binary(chunk))
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
result.insert(result.end(), chunk.begin(), chunk.end());
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
default:
|
default:
|
||||||
{
|
{
|
||||||
auto last_token = get_token_string();
|
auto last_token = get_token_string();
|
||||||
@@ -11878,6 +11927,63 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief reads a CBOR byte array
|
||||||
|
|
||||||
|
This function first reads starting bytes to determine the expected
|
||||||
|
byte array length and then copies this number of bytes into the byte array.
|
||||||
|
Additionally, CBOR's byte arrays with indefinite lengths are supported.
|
||||||
|
|
||||||
|
@param[out] result created byte array
|
||||||
|
|
||||||
|
@return whether byte array creation completed
|
||||||
|
*/
|
||||||
|
bool get_cbor_binary(binary_t& result)
|
||||||
|
{
|
||||||
|
// the open indefinite-length byte arrays are counted rather than
|
||||||
|
// recursed through, for the reason given in @ref get_cbor_string
|
||||||
|
std::size_t open = 0;
|
||||||
|
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "binary")))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (current == 0x5F) // Binary data (indefinite length)
|
||||||
|
{
|
||||||
|
++open;
|
||||||
|
get();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// a break marker closes the innermost indefinite-length byte
|
||||||
|
// array; outside of one it falls through to the error below
|
||||||
|
if (open != 0 && current == 0xFF)
|
||||||
|
{
|
||||||
|
if (--open == 0)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
get();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!get_cbor_binary_chunk(result)))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (open == 0)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
get();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief narrow a definite CBOR array/map length to std::size_t
|
@brief narrow a definite CBOR array/map length to std::size_t
|
||||||
|
|
||||||
@@ -13078,7 +13184,12 @@ class binary_reader
|
|||||||
{
|
{
|
||||||
result.first = npos; // size
|
result.first = npos; // size
|
||||||
result.second = 0; // type
|
result.second = 0; // type
|
||||||
bool is_ndarray = false;
|
// seed the flag with the caller's context: inside an ndarray dimension
|
||||||
|
// vector another ndarray is not allowed, and get_ubjson_size_value()
|
||||||
|
// rejects it up front instead of reading it and reporting afterwards.
|
||||||
|
// Seeding it with `false` made every '#' of a "[#[#[..." chain descend
|
||||||
|
// another level, which overflowed the stack (see #5104).
|
||||||
|
bool is_ndarray = inside_ndarray;
|
||||||
|
|
||||||
get_ignore_noop();
|
get_ignore_noop();
|
||||||
|
|
||||||
@@ -13111,13 +13222,11 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
|
|
||||||
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
|
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
|
||||||
if (input_format == input_format_t::bjdata && is_ndarray)
|
// an ndarray was read here only if the flag flipped; when it was
|
||||||
|
// seeded true, get_ubjson_size_value() already rejected the nested
|
||||||
|
// dimension vector
|
||||||
|
if (input_format == input_format_t::bjdata && is_ndarray && !inside_ndarray)
|
||||||
{
|
{
|
||||||
if (inside_ndarray)
|
|
||||||
{
|
|
||||||
return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read,
|
|
||||||
exception_message(input_format, "ndarray can not be recursive", "size"), nullptr));
|
|
||||||
}
|
|
||||||
result.second |= (1 << 8); // use bit 8 to indicate ndarray, all UBJSON and BJData markers should be ASCII letters
|
result.second |= (1 << 8); // use bit 8 to indicate ndarray, all UBJSON and BJData markers should be ASCII letters
|
||||||
}
|
}
|
||||||
return is_error;
|
return is_error;
|
||||||
@@ -13126,7 +13235,7 @@ class binary_reader
|
|||||||
if (current == '#')
|
if (current == '#')
|
||||||
{
|
{
|
||||||
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
|
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
|
||||||
if (input_format == input_format_t::bjdata && is_ndarray)
|
if (input_format == input_format_t::bjdata && is_ndarray && !inside_ndarray)
|
||||||
{
|
{
|
||||||
return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read,
|
return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read,
|
||||||
exception_message(input_format, "ndarray requires both type and size", "size"), nullptr));
|
exception_message(input_format, "ndarray requires both type and size", "size"), nullptr));
|
||||||
@@ -13397,6 +13506,17 @@ class binary_reader
|
|||||||
|
|
||||||
if (size_and_type.first != npos)
|
if (size_and_type.first != npos)
|
||||||
{
|
{
|
||||||
|
// reading an element of a valueless type consumes no input, so the
|
||||||
|
// declared count alone decides how much is allocated; the check is
|
||||||
|
// made before the start event so that no container is opened that
|
||||||
|
// is then abandoned. See @ref max_valueless_container_size.
|
||||||
|
if (JSON_HEDLEY_UNLIKELY((size_and_type.second == 'Z' || size_and_type.second == 'T' || size_and_type.second == 'F')
|
||||||
|
&& size_and_type.first > max_valueless_container_size))
|
||||||
|
{
|
||||||
|
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
|
||||||
|
exception_message(input_format, "excessive array size", "size"), nullptr));
|
||||||
|
}
|
||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!sax->start_array(size_and_type.first)))
|
if (JSON_HEDLEY_UNLIKELY(!sax->start_array(size_and_type.first)))
|
||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
@@ -17834,7 +17954,17 @@ class binary_writer
|
|||||||
|
|
||||||
std::vector<CharType> bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type
|
std::vector<CharType> bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type
|
||||||
|
|
||||||
if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end()))
|
// an optimized array of a valueless type carries no payload, so a
|
||||||
|
// reader has nothing but the declared count to bound the allocation
|
||||||
|
// by and refuses an excessive one. Write the unoptimized form for
|
||||||
|
// those, at one byte per element, so the result can be read back.
|
||||||
|
// Objects are not affected: every element is preceded by its key.
|
||||||
|
const bool valueless_type = (first_prefix == 'Z' || first_prefix == 'T' || first_prefix == 'F');
|
||||||
|
const bool excessive_valueless = valueless_type
|
||||||
|
&& j.m_data.m_value.array->size() > detail::max_valueless_container_size;
|
||||||
|
|
||||||
|
if (same_prefix && !excessive_valueless
|
||||||
|
&& !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end()))
|
||||||
{
|
{
|
||||||
prefix_required = false;
|
prefix_required = false;
|
||||||
oa->write_character(to_char_type('$'));
|
oa->write_character(to_char_type('$'));
|
||||||
|
|||||||
@@ -3288,8 +3288,10 @@ TEST_CASE("BJData")
|
|||||||
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR1), "[json.exception.parse_error.113] parse error at byte 6: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR1), "[json.exception.parse_error.113] parse error at byte 6: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
|
||||||
CHECK(json::from_bjdata(vR1, true, false).is_discarded());
|
CHECK(json::from_bjdata(vR1, true, false).is_discarded());
|
||||||
|
|
||||||
|
// a dimension vector that opens another one is rejected where the
|
||||||
|
// nested '[' is read, rather than after it has been descended into
|
||||||
std::vector<uint8_t> const vR2 = {'[', '$', 'i', '#', '[', '#', '[', 'i', 1, ']', ']', 1};
|
std::vector<uint8_t> const vR2 = {'[', '$', 'i', '#', '[', '#', '[', 'i', 1, ']', ']', 1};
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR2), "[json.exception.parse_error.113] parse error at byte 11: syntax error while parsing BJData size: expected length type specification (U, i, u, I, m, l, M, L) after '#'; last byte: 0x5D", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR2), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
|
||||||
CHECK(json::from_bjdata(vR2, true, false).is_discarded());
|
CHECK(json::from_bjdata(vR2, true, false).is_discarded());
|
||||||
|
|
||||||
std::vector<uint8_t> const vR3 = {'[', '#', '[', 'i', '2', 'i', 2, ']'};
|
std::vector<uint8_t> const vR3 = {'[', '#', '[', 'i', '2', 'i', 2, ']'};
|
||||||
@@ -3297,7 +3299,7 @@ TEST_CASE("BJData")
|
|||||||
CHECK(json::from_bjdata(vR3, true, false).is_discarded());
|
CHECK(json::from_bjdata(vR3, true, false).is_discarded());
|
||||||
|
|
||||||
std::vector<uint8_t> const vR4 = {'[', '$', 'i', '#', '[', '$', 'i', '#', '[', 'i', 1, ']', 1};
|
std::vector<uint8_t> const vR4 = {'[', '$', 'i', '#', '[', '$', 'i', '#', '[', 'i', 1, ']', 1};
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR4), "[json.exception.parse_error.110] parse error at byte 14: syntax error while parsing BJData number: unexpected end of input", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR4), "[json.exception.parse_error.113] parse error at byte 9: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
|
||||||
CHECK(json::from_bjdata(vR4, true, false).is_discarded());
|
CHECK(json::from_bjdata(vR4, true, false).is_discarded());
|
||||||
|
|
||||||
std::vector<uint8_t> const vR5 = {'[', '$', 'i', '#', '[', '[', '[', ']', ']', ']'};
|
std::vector<uint8_t> const vR5 = {'[', '$', 'i', '#', '[', '[', '[', ']', ']', ']'};
|
||||||
@@ -3305,12 +3307,25 @@ TEST_CASE("BJData")
|
|||||||
CHECK(json::from_bjdata(vR5, true, false).is_discarded());
|
CHECK(json::from_bjdata(vR5, true, false).is_discarded());
|
||||||
|
|
||||||
std::vector<uint8_t> const vR6 = {'[', '$', 'i', '#', '[', '$', 'i', '#', '[', 'i', '2', 'i', 2, ']'};
|
std::vector<uint8_t> const vR6 = {'[', '$', 'i', '#', '[', '$', 'i', '#', '[', 'i', '2', 'i', 2, ']'};
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR6), "[json.exception.parse_error.112] parse error at byte 14: syntax error while parsing BJData size: ndarray can not be recursive", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR6), "[json.exception.parse_error.113] parse error at byte 9: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
|
||||||
CHECK(json::from_bjdata(vR6, true, false).is_discarded());
|
CHECK(json::from_bjdata(vR6, true, false).is_discarded());
|
||||||
|
|
||||||
std::vector<uint8_t> const vH = {'[', 'H', '[', '#', '[', '$', 'i', '#', '[', 'i', '2', 'i', 2, ']'};
|
std::vector<uint8_t> const vH = {'[', 'H', '[', '#', '[', '$', 'i', '#', '[', 'i', '2', 'i', 2, ']'};
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vH), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vH), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
|
||||||
CHECK(json::from_bjdata(vH, true, false).is_discarded());
|
CHECK(json::from_bjdata(vH, true, false).is_discarded());
|
||||||
|
|
||||||
|
// Every "#[" of this chain used to open another dimension vector
|
||||||
|
// and cost several stack frames before anything was rejected, so a
|
||||||
|
// long enough chain crashed the process (see #5104). The nested
|
||||||
|
// vector is refused where it is read, so the length is irrelevant.
|
||||||
|
std::vector<uint8_t> vRdeep = {'['};
|
||||||
|
for (std::size_t i = 0; i < 100000; ++i)
|
||||||
|
{
|
||||||
|
vRdeep.push_back('#');
|
||||||
|
vRdeep.push_back('[');
|
||||||
|
}
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vRdeep), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
|
||||||
|
CHECK(json::from_bjdata(vRdeep, true, false).is_discarded());
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("objects")
|
SECTION("objects")
|
||||||
|
|||||||
@@ -2035,6 +2035,58 @@ TEST_CASE("CBOR definite length equal to the indefinite-length sentinel")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("CBOR indefinite-length strings do not recurse per chunk")
|
||||||
|
{
|
||||||
|
// Reading an indefinite-length string or byte array used to call itself
|
||||||
|
// once per chunk, so a payload of repeated 0x7F (or 0x5F) bytes exhausted
|
||||||
|
// the call stack before any of the input was rejected. The open levels are
|
||||||
|
// counted now, and the levels below prove the reader still reads the same
|
||||||
|
// values and reports the same errors at the same byte offsets.
|
||||||
|
json _;
|
||||||
|
|
||||||
|
SECTION("many open levels are reported, not crashed on")
|
||||||
|
{
|
||||||
|
const std::vector<uint8_t> input(200000, 0x7F);
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 200001: syntax error while parsing CBOR string: unexpected end of input", json::parse_error&);
|
||||||
|
CHECK(json::from_cbor(input, true, false).is_discarded());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("many open levels are reported, not crashed on (binary)")
|
||||||
|
{
|
||||||
|
const std::vector<uint8_t> input(200000, 0x5F);
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 200001: syntax error while parsing CBOR binary: unexpected end of input", json::parse_error&);
|
||||||
|
CHECK(json::from_cbor(input, true, false).is_discarded());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("chunks are still concatenated")
|
||||||
|
{
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0xFF})) == json(""));
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0x61, 0x61, 0xFF})) == json("a"));
|
||||||
|
// nested indefinite-length strings are concatenated across levels
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0x7F, 0x61, 0x61, 0xFF, 0x61, 0x62, 0xFF})) == json("ab"));
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0x7F, 0x7F, 0x61, 0x7A, 0xFF, 0xFF, 0xFF})) == json("z"));
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0xA1, 0x7F, 0x61, 0x61, 0xFF, 0x01})) == json({{"a", 1}}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("chunks are still concatenated (binary)")
|
||||||
|
{
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x5F, 0x41, 0x61, 0xFF})) == json::binary({0x61}));
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x5F, 0x5F, 0x41, 0x61, 0xFF, 0x41, 0x62, 0xFF})) == json::binary({0x61, 0x62}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("a chunk that is not a string is still rejected")
|
||||||
|
{
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7F, 0x7F, 0x00})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x00", json::parse_error&);
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x5F, 0x5F, 0x00})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR binary: expected length specification (0x40-0x5B) or indefinite binary array type (0x5F); last byte: 0x00", json::parse_error&);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("a break marker outside an indefinite-length string is not a string")
|
||||||
|
{
|
||||||
|
// 0xFF only closes a string that was opened; on its own it is not one
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0xFF, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF", json::parse_error&);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("CBOR roundtrips" * doctest::skip())
|
TEST_CASE("CBOR roundtrips" * doctest::skip())
|
||||||
{
|
{
|
||||||
SECTION("input from flynn")
|
SECTION("input from flynn")
|
||||||
|
|||||||
@@ -2149,6 +2149,67 @@ TEST_CASE("UBJSON")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("UBJSON optimized arrays of a valueless type are bounded")
|
||||||
|
{
|
||||||
|
// An element of type 'Z', 'T' or 'F' is encoded by its marker alone, so an
|
||||||
|
// optimized array of one of those has no payload and the declared count is
|
||||||
|
// the only thing deciding how much is allocated. Ten bytes used to produce
|
||||||
|
// billions of values (#2793); every other type costs at least one byte per
|
||||||
|
// element and is bounded by the end of the input.
|
||||||
|
json _;
|
||||||
|
|
||||||
|
SECTION("an excessive count is rejected")
|
||||||
|
{
|
||||||
|
// 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value
|
||||||
|
for (const auto marker :
|
||||||
|
{'Z', 'T', 'F'
|
||||||
|
})
|
||||||
|
{
|
||||||
|
const std::vector<uint8_t> input = {'[', '$', static_cast<uint8_t>(marker), '#', 'l', 0x7F, 0xFF, 0xFF, 0xFF};
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(input), "[json.exception.out_of_range.408] syntax error while parsing UBJSON size: excessive array size", json::out_of_range&);
|
||||||
|
CHECK(json::from_ubjson(input, true, false).is_discarded());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("ordinary counts are unaffected")
|
||||||
|
{
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', 'Z', '#', 'i', 3})) == json({nullptr, nullptr, nullptr}));
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', 'T', '#', 'i', 2})) == json({true, true}));
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', 'F', '#', 'i', 2})) == json({false, false}));
|
||||||
|
// 'N' is a no-op rather than a value, and still yields an empty array
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', 'N', '#', 'i', 2})) == json::array());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("a type with a payload is unaffected")
|
||||||
|
{
|
||||||
|
// A count past the limit is not rejected for 'U', which costs a byte
|
||||||
|
// per element and is bounded by the end of the input instead. The
|
||||||
|
// count is kept just past the limit rather than made huge, because a
|
||||||
|
// count that also exceeds the array's max_size() is reported as
|
||||||
|
// out_of_range before the input runs out, and max_size() depends on
|
||||||
|
// the width of std::size_t.
|
||||||
|
const std::vector<uint8_t> input = {'[', '$', 'U', '#', 'l', 0x00, 0x10, 0x00, 0x01};
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(input), "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing UBJSON number: unexpected end of input", json::parse_error&);
|
||||||
|
CHECK(json::from_ubjson(input, true, false).is_discarded());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("the writer stays within what the reader accepts")
|
||||||
|
{
|
||||||
|
// below the limit the optimized form is used and is tiny; above it the
|
||||||
|
// writer falls back so that the result can still be read back
|
||||||
|
json const at_limit(1048576, nullptr);
|
||||||
|
const auto v_at_limit = json::to_ubjson(at_limit, true, true);
|
||||||
|
CHECK(v_at_limit.size() == 9);
|
||||||
|
CHECK(v_at_limit.at(1) == '$');
|
||||||
|
CHECK(json::from_ubjson(v_at_limit) == at_limit);
|
||||||
|
|
||||||
|
json const above_limit(1048577, nullptr);
|
||||||
|
const auto v_above_limit = json::to_ubjson(above_limit, true, true);
|
||||||
|
CHECK(v_above_limit.at(1) != '$');
|
||||||
|
CHECK(json::from_ubjson(v_above_limit) == above_limit);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("Universal Binary JSON Specification Examples 1")
|
TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||||
{
|
{
|
||||||
SECTION("Null Value")
|
SECTION("Null Value")
|
||||||
|
|||||||
Reference in New Issue
Block a user