mirror of
https://github.com/nlohmann/json.git
synced 2026-09-28 10:40:30 +00:00
Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b2666c334e | ||
|
|
ad39cda092 | ||
|
|
8aabb9981f |
@@ -363,7 +363,7 @@ std::cout << j_string << " == " << serialized_string << std::endl;
|
|||||||
|
|
||||||
[`.dump()`](https://json.nlohmann.me/api/basic_json/dump/) returns the originally stored string value.
|
[`.dump()`](https://json.nlohmann.me/api/basic_json/dump/) returns the originally stored string value.
|
||||||
|
|
||||||
Note the library only supports UTF-8. When you store strings with different encodings in the library, calling [`dump()`](https://json.nlohmann.me/api/basic_json/dump/) may throw an exception unless `json::error_handler_t::replace` or `json::error_handler_t::ignore` are used as error handlers.
|
Note the library only supports UTF-8. When you store strings with different encodings in the library, calling [`dump()`](https://json.nlohmann.me/api/basic_json/dump/) may throw an exception unless `json::error_handler_t::replace`, `json::error_handler_t::ignore`, or `json::error_handler_t::keep` are used as error handlers.
|
||||||
|
|
||||||
#### To/from streams (e.g., files, string streams)
|
#### To/from streams (e.g., files, string streams)
|
||||||
|
|
||||||
@@ -1915,7 +1915,7 @@ The library supports **Unicode input** as follows:
|
|||||||
- [Unicode noncharacters](https://www.unicode.org/faq/private_use.html#nonchar1) will not be replaced by the library.
|
- [Unicode noncharacters](https://www.unicode.org/faq/private_use.html#nonchar1) will not be replaced by the library.
|
||||||
- Invalid surrogates (e.g., incomplete pairs such as `\uDEAD`) will yield parse errors.
|
- Invalid surrogates (e.g., incomplete pairs such as `\uDEAD`) will yield parse errors.
|
||||||
- The strings stored in the library are UTF-8 encoded. When using the default string type (`std::string`), note that its length/size functions return the number of stored bytes rather than the number of characters or glyphs.
|
- The strings stored in the library are UTF-8 encoded. When using the default string type (`std::string`), note that its length/size functions return the number of stored bytes rather than the number of characters or glyphs.
|
||||||
- When you store strings with different encodings in the library, calling [`dump()`](https://json.nlohmann.me/api/basic_json/dump/) may throw an exception unless `json::error_handler_t::replace` or `json::error_handler_t::ignore` are used as error handlers.
|
- When you store strings with different encodings in the library, calling [`dump()`](https://json.nlohmann.me/api/basic_json/dump/) may throw an exception unless `json::error_handler_t::replace`, `json::error_handler_t::ignore`, or `json::error_handler_t::keep` are used as error handlers.
|
||||||
- To store wide strings (e.g., `std::wstring`), you need to convert them to a UTF-8 encoded `std::string` before, see [an example](https://json.nlohmann.me/home/faq/#wide-string-handling).
|
- To store wide strings (e.g., `std::wstring`), you need to convert them to a UTF-8 encoded `std::string` before, see [an example](https://json.nlohmann.me/home/faq/#wide-string-handling).
|
||||||
|
|
||||||
### Comments in JSON
|
### Comments in JSON
|
||||||
|
|||||||
@@ -25,10 +25,15 @@ and `ensure_ascii` parameters.
|
|||||||
result consists of ASCII characters only.
|
result consists of ASCII characters only.
|
||||||
|
|
||||||
`error_handler` (in)
|
`error_handler` (in)
|
||||||
: how to react on decoding errors; there are three possible values (see [`error_handler_t`](error_handler_t.md):
|
: how to react on decoding errors; there are four possible values (see [`error_handler_t`](error_handler_t.md)):
|
||||||
`strict` (throws an exception in case a decoding error occurs; default), `replace` (replace invalid UTF-8 sequences
|
|
||||||
with U+FFFD), and `ignore` (ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the
|
- `strict`: throw a [`type_error`](../../home/exceptions.md#type-errors) exception in case a decoding error occurs
|
||||||
output unchanged, and invalid bytes are dropped)).
|
(default),
|
||||||
|
- `replace`: replace invalid UTF-8 sequences with U+FFFD (� REPLACEMENT CHARACTER),
|
||||||
|
- `ignore`: ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the output unchanged,
|
||||||
|
and invalid bytes are dropped, and
|
||||||
|
- `keep`: keep invalid UTF-8 sequences during serialization; all bytes are copied to the output unchanged, so the
|
||||||
|
result is not valid UTF-8.
|
||||||
|
|
||||||
## Return value
|
## Return value
|
||||||
|
|
||||||
@@ -94,3 +99,4 @@ Binary values are serialized as an object containing two keys:
|
|||||||
- Indentation character `indent_char`, option `ensure_ascii` and exceptions added in version 3.0.0.
|
- Indentation character `indent_char`, option `ensure_ascii` and exceptions added in version 3.0.0.
|
||||||
- Error handlers added in version 3.4.0.
|
- Error handlers added in version 3.4.0.
|
||||||
- Serialization of binary values added in version 3.8.0.
|
- Serialization of binary values added in version 3.8.0.
|
||||||
|
- Error handler value `keep` added in version 3.13.0.
|
||||||
|
|||||||
@@ -4,12 +4,13 @@
|
|||||||
enum class error_handler_t {
|
enum class error_handler_t {
|
||||||
strict,
|
strict,
|
||||||
replace,
|
replace,
|
||||||
ignore
|
ignore,
|
||||||
|
keep
|
||||||
};
|
};
|
||||||
```
|
```
|
||||||
|
|
||||||
This enumeration is used in the [`dump`](dump.md) function to choose how to treat decoding errors while serializing a
|
This enumeration is used in the [`dump`](dump.md) function to choose how to treat decoding errors while serializing a
|
||||||
`basic_json` value. Three values are differentiated:
|
`basic_json` value. Four values are differentiated:
|
||||||
|
|
||||||
strict
|
strict
|
||||||
: throw a `type_error` exception in case of invalid UTF-8
|
: throw a `type_error` exception in case of invalid UTF-8
|
||||||
@@ -20,6 +21,12 @@ replace
|
|||||||
ignore
|
ignore
|
||||||
: ignore invalid UTF-8 sequences; all valid bytes are copied to the output unchanged, and invalid bytes are dropped
|
: ignore invalid UTF-8 sequences; all valid bytes are copied to the output unchanged, and invalid bytes are dropped
|
||||||
|
|
||||||
|
keep
|
||||||
|
: keep invalid UTF-8 sequences; all bytes are copied to the output unchanged. Valid characters are still escaped as
|
||||||
|
usual (e.g., `"`, `\\`, and control characters), so the result has valid JSON syntax, but it is not valid UTF-8.
|
||||||
|
In particular, [`parse`](parse.md) rejects it, and with `ensure_ascii` set to `true`, the invalid bytes are the
|
||||||
|
only non-ASCII bytes of the output.
|
||||||
|
|
||||||
## Examples
|
## Examples
|
||||||
|
|
||||||
??? example
|
??? example
|
||||||
@@ -40,3 +47,4 @@ ignore
|
|||||||
## Version history
|
## Version history
|
||||||
|
|
||||||
- Added in version 3.4.0.
|
- Added in version 3.4.0.
|
||||||
|
- Added value `keep` in version 3.13.0.
|
||||||
|
|||||||
@@ -80,8 +80,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
|
|||||||
the end of the file was not reached when `strict` was set to true
|
the end of the file was not reached when `strict` was set to true
|
||||||
- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from CBOR were
|
- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from CBOR were
|
||||||
used in the given input or if the input is not valid CBOR
|
used in the given input or if the input is not valid CBOR
|
||||||
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other
|
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string was expected as a map key,
|
||||||
types are not supported, as JSON object keys are always strings) or a string is malformed
|
but not found
|
||||||
|
|
||||||
## Complexity
|
## Complexity
|
||||||
|
|
||||||
|
|||||||
@@ -73,8 +73,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
|
|||||||
the end of the file was not reached when `strict` was set to true
|
the end of the file was not reached when `strict` was set to true
|
||||||
- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from
|
- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from
|
||||||
MessagePack were used in the given input or if the input is not valid MessagePack
|
MessagePack were used in the given input or if the input is not valid MessagePack
|
||||||
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other
|
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string was expected as a map key,
|
||||||
types are not supported, as JSON object keys are always strings) or a string is malformed
|
but not found
|
||||||
|
|
||||||
## Complexity
|
## Complexity
|
||||||
|
|
||||||
|
|||||||
@@ -1,3 +1,4 @@
|
|||||||
|
#include <iomanip>
|
||||||
#include <iostream>
|
#include <iostream>
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
@@ -21,4 +22,12 @@ int main()
|
|||||||
<< "\nstring with ignored invalid characters: "
|
<< "\nstring with ignored invalid characters: "
|
||||||
<< j_invalid.dump(-1, ' ', false, json::error_handler_t::ignore)
|
<< j_invalid.dump(-1, ' ', false, json::error_handler_t::ignore)
|
||||||
<< '\n';
|
<< '\n';
|
||||||
|
|
||||||
|
// the invalid byte is kept; print the result byte-wise to make it visible
|
||||||
|
std::cout << "string with kept invalid characters:";
|
||||||
|
for (const unsigned char c : j_invalid.dump(-1, ' ', false, json::error_handler_t::keep))
|
||||||
|
{
|
||||||
|
std::cout << ' ' << std::hex << std::setw(2) << std::setfill('0') << static_cast<int>(c);
|
||||||
|
}
|
||||||
|
std::cout << '\n';
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,3 +1,4 @@
|
|||||||
[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9
|
[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9
|
||||||
string with replaced invalid characters: "ä�ü"
|
string with replaced invalid characters: "ä�ü"
|
||||||
string with ignored invalid characters: "äü"
|
string with ignored invalid characters: "äü"
|
||||||
|
string with kept invalid characters: 22 c3 a4 a9 c3 bc 22
|
||||||
|
|||||||
@@ -174,20 +174,7 @@ The library maps CBOR types to JSON value types as follows:
|
|||||||
|
|
||||||
!!! warning "Object keys"
|
!!! warning "Object keys"
|
||||||
|
|
||||||
CBOR allows map keys of any type, whereas JSON only allows strings as keys in object values. Therefore, CBOR maps
|
CBOR allows map keys of any type, whereas JSON only allows strings as keys in object values. Therefore, CBOR maps with keys other than UTF-8 strings are rejected.
|
||||||
with keys other than text strings (major type 3) are rejected with a
|
|
||||||
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions` set
|
|
||||||
to `false`, a discarded value) naming the type of the key that was found, for instance:
|
|
||||||
|
|
||||||
```
|
|
||||||
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found an unsigned integer; last byte: 0x01
|
|
||||||
```
|
|
||||||
|
|
||||||
This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed
|
|
||||||
on. This is a deliberate restriction of the library's JSON value model, not an oversight: formats built on CBOR
|
|
||||||
maps with integer keys, such as COSE ([RFC 9052](https://www.rfc-editor.org/rfc/rfc9052.html)) or CWT
|
|
||||||
([RFC 8392](https://www.rfc-editor.org/rfc/rfc8392.html)), cannot be read with this library and need a
|
|
||||||
general-purpose CBOR library instead.
|
|
||||||
|
|
||||||
!!! warning "UTF-8 validation of text strings"
|
!!! warning "UTF-8 validation of text strings"
|
||||||
|
|
||||||
|
|||||||
@@ -138,21 +138,6 @@ The library maps MessagePack types to JSON value types as follows:
|
|||||||
|
|
||||||
Any MessagePack output created by `to_msgpack` can be successfully parsed by `from_msgpack`.
|
Any MessagePack output created by `to_msgpack` can be successfully parsed by `from_msgpack`.
|
||||||
|
|
||||||
!!! warning "Object keys"
|
|
||||||
|
|
||||||
MessagePack allows map keys of any type, whereas JSON only allows strings as keys in object values. Like the
|
|
||||||
JSON-compatible [profile](https://github.com/msgpack/msgpack/blob/master/spec.md#profile) sketched in the
|
|
||||||
MessagePack specification, this library restricts map keys to `str` values. Maps with keys of any other type are
|
|
||||||
rejected with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with
|
|
||||||
`allow_exceptions` set to `false`, a discarded value) naming the type of the key that was found, for instance:
|
|
||||||
|
|
||||||
```
|
|
||||||
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found nil; last byte: 0xC0
|
|
||||||
```
|
|
||||||
|
|
||||||
This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed
|
|
||||||
on. Such input needs a general-purpose MessagePack library instead.
|
|
||||||
|
|
||||||
!!! warning "UTF-8 validation of string values"
|
!!! warning "UTF-8 validation of string values"
|
||||||
|
|
||||||
The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8.
|
The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8.
|
||||||
|
|||||||
@@ -64,6 +64,7 @@ serialization fails by default. The fourth argument of `dump` selects an
|
|||||||
- `strict` (default) — throw a [`type_error.316`](../home/exceptions.md#jsonexceptiontype_error316) exception.
|
- `strict` (default) — throw a [`type_error.316`](../home/exceptions.md#jsonexceptiontype_error316) exception.
|
||||||
- `replace` — replace invalid bytes with the Unicode replacement character U+FFFD (`�`).
|
- `replace` — replace invalid bytes with the Unicode replacement character U+FFFD (`�`).
|
||||||
- `ignore` — silently drop invalid bytes.
|
- `ignore` — silently drop invalid bytes.
|
||||||
|
- `keep` — copy invalid bytes to the output unchanged; the result is not valid UTF-8.
|
||||||
|
|
||||||
??? example
|
??? example
|
||||||
|
|
||||||
|
|||||||
@@ -343,20 +343,13 @@ A string could not be read from a [binary format](../features/binary_formats/ind
|
|||||||
string was read where one was required (for instance as a map key), the string's length specification is invalid, or
|
string was read where one was required (for instance as a map key), the string's length specification is invalid, or
|
||||||
the string's bytes are not valid UTF-8.
|
the string's bytes are not valid UTF-8.
|
||||||
|
|
||||||
CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other
|
|
||||||
type (for instance integers or `null`) are therefore not supported; see the notes on
|
|
||||||
[CBOR](../features/binary_formats/cbor.md) and [MessagePack](../features/binary_formats/messagepack.md).
|
|
||||||
|
|
||||||
!!! failure "Example messages"
|
!!! failure "Example messages"
|
||||||
|
|
||||||
```
|
```
|
||||||
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found an unsigned integer; last byte: 0x01
|
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF
|
||||||
```
|
```
|
||||||
```
|
```
|
||||||
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found nil; last byte: 0xC0
|
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack string: expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0xFF
|
||||||
```
|
|
||||||
```
|
|
||||||
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C
|
|
||||||
```
|
```
|
||||||
```
|
```
|
||||||
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing UBJSON char: byte after 'C' must be in range 0x00..0x7F; last byte: 0x82
|
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing UBJSON char: byte after 'C' must be in range 0x00..0x7F; last byte: 0x82
|
||||||
@@ -762,6 +755,7 @@ The `dump()` function only works with UTF-8 encoded strings; that is, if you ass
|
|||||||
- Pass an error handler as last parameter to the `dump()` function to avoid this exception:
|
- Pass an error handler as last parameter to the `dump()` function to avoid this exception:
|
||||||
- `json::error_handler_t::replace` will replace invalid bytes sequences with `U+FFFD`
|
- `json::error_handler_t::replace` will replace invalid bytes sequences with `U+FFFD`
|
||||||
- `json::error_handler_t::ignore` will silently ignore invalid byte sequences
|
- `json::error_handler_t::ignore` will silently ignore invalid byte sequences
|
||||||
|
- `json::error_handler_t::keep` will copy invalid byte sequences to the output unchanged
|
||||||
|
|
||||||
### json.exception.type_error.317
|
### json.exception.type_error.317
|
||||||
|
|
||||||
|
|||||||
@@ -85,7 +85,7 @@ The library supports **Unicode input** as follows:
|
|||||||
- The library will not replace [Unicode noncharacters](http://www.unicode.org/faq/private_use.html#nonchar1).
|
- The library will not replace [Unicode noncharacters](http://www.unicode.org/faq/private_use.html#nonchar1).
|
||||||
- Invalid surrogates (e.g., incomplete pairs such as `\uDEAD`) will yield parse errors.
|
- Invalid surrogates (e.g., incomplete pairs such as `\uDEAD`) will yield parse errors.
|
||||||
- The strings stored in the library are UTF-8 encoded. When using the default string type (`std::string`), note that its length/size functions return the number of stored bytes rather than the number of characters or glyphs.
|
- The strings stored in the library are UTF-8 encoded. When using the default string type (`std::string`), note that its length/size functions return the number of stored bytes rather than the number of characters or glyphs.
|
||||||
- When you store strings with different encodings in the library, calling [`dump()`](https://nlohmann.github.io/json/classnlohmann_1_1basic__json_a50ec80b02d0f3f51130d4abb5d1cfdc5.html#a50ec80b02d0f3f51130d4abb5d1cfdc5) may throw an exception unless `json::error_handler_t::replace` or `json::error_handler_t::ignore` are used as error handlers.
|
- When you store strings with different encodings in the library, calling [`dump()`](https://nlohmann.github.io/json/classnlohmann_1_1basic__json_a50ec80b02d0f3f51130d4abb5d1cfdc5.html#a50ec80b02d0f3f51130d4abb5d1cfdc5) may throw an exception unless `json::error_handler_t::replace`, `json::error_handler_t::ignore`, or `json::error_handler_t::keep` are used as error handlers.
|
||||||
|
|
||||||
In most cases, the parser is right to complain, because the input is not UTF-8 encoded. This is especially true for Microsoft Windows, where Latin-1 or ISO 8859-1 is often the standard encoding.
|
In most cases, the parser is right to complain, because the input is not UTF-8 encoded. This is especially true for Microsoft Windows, where Latin-1 or ISO 8859-1 is often the standard encoding.
|
||||||
|
|
||||||
|
|||||||
@@ -1324,80 +1324,6 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
|
||||||
@brief reads a CBOR object key
|
|
||||||
|
|
||||||
RFC 8949 allows any data item as a map key, but only strings have a
|
|
||||||
counterpart in JSON. A key of any other type is rejected with a message
|
|
||||||
naming that type, rather than the one @ref get_cbor_string gives for a
|
|
||||||
malformed string.
|
|
||||||
|
|
||||||
@param[out] result created key
|
|
||||||
|
|
||||||
@return whether key creation completed
|
|
||||||
*/
|
|
||||||
bool get_cbor_object_key(string_t& result)
|
|
||||||
{
|
|
||||||
// EOF and major type 3 (text string) are left to get_cbor_string
|
|
||||||
if (current == char_traits<char_type>::eof() || (static_cast<unsigned int>(current) & 0xE0u) == 0x60u)
|
|
||||||
{
|
|
||||||
return get_cbor_string(result);
|
|
||||||
}
|
|
||||||
|
|
||||||
const char* found = nullptr;
|
|
||||||
switch (static_cast<unsigned int>(current) >> 5u)
|
|
||||||
{
|
|
||||||
case 0:
|
|
||||||
found = "an unsigned integer";
|
|
||||||
break;
|
|
||||||
case 1:
|
|
||||||
found = "a negative integer";
|
|
||||||
break;
|
|
||||||
case 2:
|
|
||||||
found = "a byte string";
|
|
||||||
break;
|
|
||||||
case 4:
|
|
||||||
found = "an array";
|
|
||||||
break;
|
|
||||||
case 5:
|
|
||||||
found = "a map";
|
|
||||||
break;
|
|
||||||
case 6:
|
|
||||||
found = "a tag";
|
|
||||||
break;
|
|
||||||
default: // major type 7
|
|
||||||
switch (current)
|
|
||||||
{
|
|
||||||
case 0xF4:
|
|
||||||
case 0xF5:
|
|
||||||
found = "a boolean";
|
|
||||||
break;
|
|
||||||
case 0xF6:
|
|
||||||
found = "null";
|
|
||||||
break;
|
|
||||||
case 0xF7:
|
|
||||||
found = "undefined";
|
|
||||||
break;
|
|
||||||
case 0xF9:
|
|
||||||
case 0xFA:
|
|
||||||
case 0xFB:
|
|
||||||
found = "a floating-point number";
|
|
||||||
break;
|
|
||||||
case 0xFF:
|
|
||||||
found = "a break stop code";
|
|
||||||
break;
|
|
||||||
default:
|
|
||||||
found = "a simple value";
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
auto last_token = get_token_string();
|
|
||||||
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
|
|
||||||
exception_message(input_format_t::cbor, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr));
|
|
||||||
}
|
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief reads a definite-length CBOR byte array
|
@brief reads a definite-length CBOR byte array
|
||||||
|
|
||||||
@@ -1642,7 +1568,7 @@ class binary_reader
|
|||||||
if (top.is_object)
|
if (top.is_object)
|
||||||
{
|
{
|
||||||
key.clear();
|
key.clear();
|
||||||
if (JSON_HEDLEY_UNLIKELY(!get_cbor_object_key(key) || !sax->key(key)))
|
if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key)))
|
||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
@@ -2143,98 +2069,6 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
|
||||||
@brief reads a MessagePack object key
|
|
||||||
|
|
||||||
The MessagePack specification allows any type as a map key, but only
|
|
||||||
strings have a counterpart in JSON. A key of any other type is rejected
|
|
||||||
with a message naming that type, rather than the one @ref
|
|
||||||
get_msgpack_string gives for a malformed string.
|
|
||||||
|
|
||||||
@param[out] result created key
|
|
||||||
|
|
||||||
@return whether key creation completed
|
|
||||||
*/
|
|
||||||
bool get_msgpack_object_key(string_t& result)
|
|
||||||
{
|
|
||||||
const char* found = nullptr;
|
|
||||||
switch (current)
|
|
||||||
{
|
|
||||||
case 0xC0:
|
|
||||||
found = "nil";
|
|
||||||
break;
|
|
||||||
case 0xC2:
|
|
||||||
case 0xC3:
|
|
||||||
found = "a boolean";
|
|
||||||
break;
|
|
||||||
case 0xCA:
|
|
||||||
case 0xCB:
|
|
||||||
found = "a float";
|
|
||||||
break;
|
|
||||||
case 0xC4:
|
|
||||||
case 0xC5:
|
|
||||||
case 0xC6:
|
|
||||||
found = "a bin";
|
|
||||||
break;
|
|
||||||
case 0xC7:
|
|
||||||
case 0xC8:
|
|
||||||
case 0xC9:
|
|
||||||
case 0xD4:
|
|
||||||
case 0xD5:
|
|
||||||
case 0xD6:
|
|
||||||
case 0xD7:
|
|
||||||
case 0xD8:
|
|
||||||
found = "an ext";
|
|
||||||
break;
|
|
||||||
case 0xCC:
|
|
||||||
case 0xCD:
|
|
||||||
case 0xCE:
|
|
||||||
case 0xCF:
|
|
||||||
case 0xD0:
|
|
||||||
case 0xD1:
|
|
||||||
case 0xD2:
|
|
||||||
case 0xD3:
|
|
||||||
found = "an integer";
|
|
||||||
break;
|
|
||||||
case 0xDC:
|
|
||||||
case 0xDD:
|
|
||||||
found = "an array";
|
|
||||||
break;
|
|
||||||
case 0xDE:
|
|
||||||
case 0xDF:
|
|
||||||
found = "a map";
|
|
||||||
break;
|
|
||||||
default:
|
|
||||||
// fixint, fixmap, and fixarray; strings, EOF, and the unused
|
|
||||||
// byte 0xC1 are left to get_msgpack_string
|
|
||||||
if (current == char_traits<char_type>::eof())
|
|
||||||
{
|
|
||||||
return get_msgpack_string(result);
|
|
||||||
}
|
|
||||||
if (current <= 0x7F || current >= 0xE0)
|
|
||||||
{
|
|
||||||
found = "an integer";
|
|
||||||
}
|
|
||||||
else if (current <= 0x8F)
|
|
||||||
{
|
|
||||||
found = "a map";
|
|
||||||
}
|
|
||||||
else if (current <= 0x9F)
|
|
||||||
{
|
|
||||||
found = "an array";
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
return get_msgpack_string(result);
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
auto last_token = get_token_string();
|
|
||||||
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
|
|
||||||
exception_message(input_format_t::msgpack, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr));
|
|
||||||
}
|
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief reads a MessagePack byte array
|
@brief reads a MessagePack byte array
|
||||||
|
|
||||||
@@ -2397,7 +2231,7 @@ class binary_reader
|
|||||||
{
|
{
|
||||||
get();
|
get();
|
||||||
key.clear();
|
key.clear();
|
||||||
if (JSON_HEDLEY_UNLIKELY(!get_msgpack_object_key(key) || !sax->key(key)))
|
if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key)))
|
||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -48,7 +48,8 @@ enum class error_handler_t
|
|||||||
{
|
{
|
||||||
strict, ///< throw a type_error exception in case of invalid UTF-8
|
strict, ///< throw a type_error exception in case of invalid UTF-8
|
||||||
replace, ///< replace invalid UTF-8 sequences with U+FFFD
|
replace, ///< replace invalid UTF-8 sequences with U+FFFD
|
||||||
ignore ///< ignore invalid UTF-8 sequences
|
ignore, ///< ignore invalid UTF-8 sequences
|
||||||
|
keep ///< keep invalid UTF-8 sequences; their bytes are copied unchanged
|
||||||
};
|
};
|
||||||
|
|
||||||
template<typename BasicJsonType>
|
template<typename BasicJsonType>
|
||||||
@@ -1019,6 +1020,47 @@ class serializer
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
case error_handler_t::keep:
|
||||||
|
{
|
||||||
|
// drop whatever the incomplete sequence left in
|
||||||
|
// the buffer (only copied if !EnsureAscii) and copy
|
||||||
|
// the ill-formed bytes from the input instead
|
||||||
|
bytes = bytes_after_last_accept;
|
||||||
|
|
||||||
|
if (undumped_chars > 0)
|
||||||
|
{
|
||||||
|
// the pending bytes of the incomplete sequence
|
||||||
|
// are ill-formed; the current byte may be OK for
|
||||||
|
// itself, so we would like to read it again
|
||||||
|
for (std::size_t j = i - undumped_chars; j < i; ++j)
|
||||||
|
{
|
||||||
|
string_buffer[bytes++] = s[j];
|
||||||
|
}
|
||||||
|
--i;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// the current byte cannot start any sequence
|
||||||
|
string_buffer[bytes++] = s[i];
|
||||||
|
}
|
||||||
|
|
||||||
|
// write buffer and reset index; there must be 13 bytes
|
||||||
|
// left, as this is the maximal number of bytes to be
|
||||||
|
// written ("\uxxxx\uxxxx\0") for one code point
|
||||||
|
if (string_buffer.size() - bytes < 13)
|
||||||
|
{
|
||||||
|
put_buffer(string_buffer, bytes);
|
||||||
|
bytes = 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bytes_after_last_accept = bytes;
|
||||||
|
undumped_chars = 0;
|
||||||
|
|
||||||
|
// continue processing the string
|
||||||
|
state = UTF8_ACCEPT;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
default: // LCOV_EXCL_LINE
|
default: // LCOV_EXCL_LINE
|
||||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
||||||
}
|
}
|
||||||
@@ -1064,6 +1106,15 @@ class serializer
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
case error_handler_t::keep:
|
||||||
|
{
|
||||||
|
// write all accepted bytes
|
||||||
|
put_buffer(string_buffer, bytes_after_last_accept);
|
||||||
|
// copy the bytes of the incomplete sequence unchanged
|
||||||
|
put_string(s, s.size() - undumped_chars, s.size());
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
case error_handler_t::replace:
|
case error_handler_t::replace:
|
||||||
{
|
{
|
||||||
// write all accepted bytes
|
// write all accepted bytes
|
||||||
|
|||||||
@@ -14059,80 +14059,6 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
|
||||||
@brief reads a CBOR object key
|
|
||||||
|
|
||||||
RFC 8949 allows any data item as a map key, but only strings have a
|
|
||||||
counterpart in JSON. A key of any other type is rejected with a message
|
|
||||||
naming that type, rather than the one @ref get_cbor_string gives for a
|
|
||||||
malformed string.
|
|
||||||
|
|
||||||
@param[out] result created key
|
|
||||||
|
|
||||||
@return whether key creation completed
|
|
||||||
*/
|
|
||||||
bool get_cbor_object_key(string_t& result)
|
|
||||||
{
|
|
||||||
// EOF and major type 3 (text string) are left to get_cbor_string
|
|
||||||
if (current == char_traits<char_type>::eof() || (static_cast<unsigned int>(current) & 0xE0u) == 0x60u)
|
|
||||||
{
|
|
||||||
return get_cbor_string(result);
|
|
||||||
}
|
|
||||||
|
|
||||||
const char* found = nullptr;
|
|
||||||
switch (static_cast<unsigned int>(current) >> 5u)
|
|
||||||
{
|
|
||||||
case 0:
|
|
||||||
found = "an unsigned integer";
|
|
||||||
break;
|
|
||||||
case 1:
|
|
||||||
found = "a negative integer";
|
|
||||||
break;
|
|
||||||
case 2:
|
|
||||||
found = "a byte string";
|
|
||||||
break;
|
|
||||||
case 4:
|
|
||||||
found = "an array";
|
|
||||||
break;
|
|
||||||
case 5:
|
|
||||||
found = "a map";
|
|
||||||
break;
|
|
||||||
case 6:
|
|
||||||
found = "a tag";
|
|
||||||
break;
|
|
||||||
default: // major type 7
|
|
||||||
switch (current)
|
|
||||||
{
|
|
||||||
case 0xF4:
|
|
||||||
case 0xF5:
|
|
||||||
found = "a boolean";
|
|
||||||
break;
|
|
||||||
case 0xF6:
|
|
||||||
found = "null";
|
|
||||||
break;
|
|
||||||
case 0xF7:
|
|
||||||
found = "undefined";
|
|
||||||
break;
|
|
||||||
case 0xF9:
|
|
||||||
case 0xFA:
|
|
||||||
case 0xFB:
|
|
||||||
found = "a floating-point number";
|
|
||||||
break;
|
|
||||||
case 0xFF:
|
|
||||||
found = "a break stop code";
|
|
||||||
break;
|
|
||||||
default:
|
|
||||||
found = "a simple value";
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
auto last_token = get_token_string();
|
|
||||||
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
|
|
||||||
exception_message(input_format_t::cbor, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr));
|
|
||||||
}
|
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief reads a definite-length CBOR byte array
|
@brief reads a definite-length CBOR byte array
|
||||||
|
|
||||||
@@ -14377,7 +14303,7 @@ class binary_reader
|
|||||||
if (top.is_object)
|
if (top.is_object)
|
||||||
{
|
{
|
||||||
key.clear();
|
key.clear();
|
||||||
if (JSON_HEDLEY_UNLIKELY(!get_cbor_object_key(key) || !sax->key(key)))
|
if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key)))
|
||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
@@ -14878,98 +14804,6 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
|
||||||
@brief reads a MessagePack object key
|
|
||||||
|
|
||||||
The MessagePack specification allows any type as a map key, but only
|
|
||||||
strings have a counterpart in JSON. A key of any other type is rejected
|
|
||||||
with a message naming that type, rather than the one @ref
|
|
||||||
get_msgpack_string gives for a malformed string.
|
|
||||||
|
|
||||||
@param[out] result created key
|
|
||||||
|
|
||||||
@return whether key creation completed
|
|
||||||
*/
|
|
||||||
bool get_msgpack_object_key(string_t& result)
|
|
||||||
{
|
|
||||||
const char* found = nullptr;
|
|
||||||
switch (current)
|
|
||||||
{
|
|
||||||
case 0xC0:
|
|
||||||
found = "nil";
|
|
||||||
break;
|
|
||||||
case 0xC2:
|
|
||||||
case 0xC3:
|
|
||||||
found = "a boolean";
|
|
||||||
break;
|
|
||||||
case 0xCA:
|
|
||||||
case 0xCB:
|
|
||||||
found = "a float";
|
|
||||||
break;
|
|
||||||
case 0xC4:
|
|
||||||
case 0xC5:
|
|
||||||
case 0xC6:
|
|
||||||
found = "a bin";
|
|
||||||
break;
|
|
||||||
case 0xC7:
|
|
||||||
case 0xC8:
|
|
||||||
case 0xC9:
|
|
||||||
case 0xD4:
|
|
||||||
case 0xD5:
|
|
||||||
case 0xD6:
|
|
||||||
case 0xD7:
|
|
||||||
case 0xD8:
|
|
||||||
found = "an ext";
|
|
||||||
break;
|
|
||||||
case 0xCC:
|
|
||||||
case 0xCD:
|
|
||||||
case 0xCE:
|
|
||||||
case 0xCF:
|
|
||||||
case 0xD0:
|
|
||||||
case 0xD1:
|
|
||||||
case 0xD2:
|
|
||||||
case 0xD3:
|
|
||||||
found = "an integer";
|
|
||||||
break;
|
|
||||||
case 0xDC:
|
|
||||||
case 0xDD:
|
|
||||||
found = "an array";
|
|
||||||
break;
|
|
||||||
case 0xDE:
|
|
||||||
case 0xDF:
|
|
||||||
found = "a map";
|
|
||||||
break;
|
|
||||||
default:
|
|
||||||
// fixint, fixmap, and fixarray; strings, EOF, and the unused
|
|
||||||
// byte 0xC1 are left to get_msgpack_string
|
|
||||||
if (current == char_traits<char_type>::eof())
|
|
||||||
{
|
|
||||||
return get_msgpack_string(result);
|
|
||||||
}
|
|
||||||
if (current <= 0x7F || current >= 0xE0)
|
|
||||||
{
|
|
||||||
found = "an integer";
|
|
||||||
}
|
|
||||||
else if (current <= 0x8F)
|
|
||||||
{
|
|
||||||
found = "a map";
|
|
||||||
}
|
|
||||||
else if (current <= 0x9F)
|
|
||||||
{
|
|
||||||
found = "an array";
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
return get_msgpack_string(result);
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
auto last_token = get_token_string();
|
|
||||||
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
|
|
||||||
exception_message(input_format_t::msgpack, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr));
|
|
||||||
}
|
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief reads a MessagePack byte array
|
@brief reads a MessagePack byte array
|
||||||
|
|
||||||
@@ -15132,7 +14966,7 @@ class binary_reader
|
|||||||
{
|
{
|
||||||
get();
|
get();
|
||||||
key.clear();
|
key.clear();
|
||||||
if (JSON_HEDLEY_UNLIKELY(!get_msgpack_object_key(key) || !sax->key(key)))
|
if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key)))
|
||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
@@ -24097,7 +23931,8 @@ enum class error_handler_t
|
|||||||
{
|
{
|
||||||
strict, ///< throw a type_error exception in case of invalid UTF-8
|
strict, ///< throw a type_error exception in case of invalid UTF-8
|
||||||
replace, ///< replace invalid UTF-8 sequences with U+FFFD
|
replace, ///< replace invalid UTF-8 sequences with U+FFFD
|
||||||
ignore ///< ignore invalid UTF-8 sequences
|
ignore, ///< ignore invalid UTF-8 sequences
|
||||||
|
keep ///< keep invalid UTF-8 sequences; their bytes are copied unchanged
|
||||||
};
|
};
|
||||||
|
|
||||||
template<typename BasicJsonType>
|
template<typename BasicJsonType>
|
||||||
@@ -25068,6 +24903,47 @@ class serializer
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
case error_handler_t::keep:
|
||||||
|
{
|
||||||
|
// drop whatever the incomplete sequence left in
|
||||||
|
// the buffer (only copied if !EnsureAscii) and copy
|
||||||
|
// the ill-formed bytes from the input instead
|
||||||
|
bytes = bytes_after_last_accept;
|
||||||
|
|
||||||
|
if (undumped_chars > 0)
|
||||||
|
{
|
||||||
|
// the pending bytes of the incomplete sequence
|
||||||
|
// are ill-formed; the current byte may be OK for
|
||||||
|
// itself, so we would like to read it again
|
||||||
|
for (std::size_t j = i - undumped_chars; j < i; ++j)
|
||||||
|
{
|
||||||
|
string_buffer[bytes++] = s[j];
|
||||||
|
}
|
||||||
|
--i;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// the current byte cannot start any sequence
|
||||||
|
string_buffer[bytes++] = s[i];
|
||||||
|
}
|
||||||
|
|
||||||
|
// write buffer and reset index; there must be 13 bytes
|
||||||
|
// left, as this is the maximal number of bytes to be
|
||||||
|
// written ("\uxxxx\uxxxx\0") for one code point
|
||||||
|
if (string_buffer.size() - bytes < 13)
|
||||||
|
{
|
||||||
|
put_buffer(string_buffer, bytes);
|
||||||
|
bytes = 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bytes_after_last_accept = bytes;
|
||||||
|
undumped_chars = 0;
|
||||||
|
|
||||||
|
// continue processing the string
|
||||||
|
state = UTF8_ACCEPT;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
default: // LCOV_EXCL_LINE
|
default: // LCOV_EXCL_LINE
|
||||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
||||||
}
|
}
|
||||||
@@ -25113,6 +24989,15 @@ class serializer
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
case error_handler_t::keep:
|
||||||
|
{
|
||||||
|
// write all accepted bytes
|
||||||
|
put_buffer(string_buffer, bytes_after_last_accept);
|
||||||
|
// copy the bytes of the incomplete sequence unchanged
|
||||||
|
put_string(s, s.size() - undumped_chars, s.size());
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
case error_handler_t::replace:
|
case error_handler_t::replace:
|
||||||
{
|
{
|
||||||
// write all accepted bytes
|
// write all accepted bytes
|
||||||
|
|||||||
+2
-43
@@ -1830,51 +1830,10 @@ TEST_CASE("CBOR")
|
|||||||
SECTION("invalid string in map")
|
SECTION("invalid string in map")
|
||||||
{
|
{
|
||||||
json _;
|
json _;
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xa1, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found a break stop code; last byte: 0xFF", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xa1, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF", json::parse_error&);
|
||||||
CHECK(json::from_cbor(std::vector<uint8_t>({0xa1, 0xff, 0x01}), true, false).is_discarded());
|
CHECK(json::from_cbor(std::vector<uint8_t>({0xa1, 0xff, 0x01}), true, false).is_discarded());
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("non-string key (see #2766 and #3381)")
|
|
||||||
{
|
|
||||||
// only text strings map to JSON object keys; any other key is
|
|
||||||
// rejected with a message naming its type
|
|
||||||
const std::vector<std::pair<std::vector<std::uint8_t>, std::string>> cases =
|
|
||||||
{
|
|
||||||
{{0xA1, 0x01, 0x01}, "an unsigned integer; last byte: 0x01"},
|
|
||||||
{{0xA1, 0x20, 0x01}, "a negative integer; last byte: 0x20"},
|
|
||||||
{{0xA1, 0x41, 0x61, 0x01}, "a byte string; last byte: 0x41"},
|
|
||||||
{{0xA1, 0x80, 0x01}, "an array; last byte: 0x80"},
|
|
||||||
{{0xA1, 0xA0, 0x01}, "a map; last byte: 0xA0"},
|
|
||||||
{{0xA1, 0xC0, 0x61, 0x61, 0x01}, "a tag; last byte: 0xC0"},
|
|
||||||
{{0xA1, 0xF4, 0x01}, "a boolean; last byte: 0xF4"},
|
|
||||||
{{0xA1, 0xF5, 0x01}, "a boolean; last byte: 0xF5"},
|
|
||||||
{{0xA1, 0xF6, 0x01}, "null; last byte: 0xF6"},
|
|
||||||
{{0xA1, 0xF7, 0x01}, "undefined; last byte: 0xF7"},
|
|
||||||
{{0xA1, 0xF9, 0x3C, 0x00, 0x01}, "a floating-point number; last byte: 0xF9"},
|
|
||||||
{{0xA1, 0xFA, 0x3F, 0x80, 0x00, 0x00, 0x01}, "a floating-point number; last byte: 0xFA"},
|
|
||||||
{{0xA1, 0xFB, 0x3F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "a floating-point number; last byte: 0xFB"},
|
|
||||||
{{0xA1, 0xE0, 0x01}, "a simple value; last byte: 0xE0"},
|
|
||||||
{{0xA1, 0xF8, 0x20, 0x01}, "a simple value; last byte: 0xF8"},
|
|
||||||
// indefinite-length map
|
|
||||||
{{0xBF, 0x01, 0x01, 0xFF}, "an unsigned integer; last byte: 0x01"},
|
|
||||||
};
|
|
||||||
|
|
||||||
for (const auto& c : cases)
|
|
||||||
{
|
|
||||||
CAPTURE(c.first)
|
|
||||||
const std::string expected = "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found " + c.second;
|
|
||||||
json _;
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(c.first), expected.c_str(), json::parse_error&);
|
|
||||||
CHECK(json::from_cbor(c.first, true, false).is_discarded());
|
|
||||||
}
|
|
||||||
|
|
||||||
// a key of major type 3 with a reserved length is still reported as
|
|
||||||
// a malformed string, and a missing key as the end of input
|
|
||||||
json _;
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing CBOR string: unexpected end of input", json::parse_error&);
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("invalid UTF-8 in string (see #5529)")
|
SECTION("invalid UTF-8 in string (see #5529)")
|
||||||
{
|
{
|
||||||
// a two-character text string (major type 3) whose bytes are not
|
// a two-character text string (major type 3) whose bytes are not
|
||||||
@@ -2325,7 +2284,7 @@ TEST_CASE("CBOR indefinite-length strings do not recurse per chunk")
|
|||||||
SECTION("a break marker outside an indefinite-length string is not a string")
|
SECTION("a break marker outside an indefinite-length string is not a string")
|
||||||
{
|
{
|
||||||
// 0xFF only closes a string that was opened; on its own it is not one
|
// 0xFF only closes a string that was opened; on its own it is not one
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0xFF, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found a break stop code; last byte: 0xFF", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0xFF, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF", json::parse_error&);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1551,69 +1551,10 @@ TEST_CASE("MessagePack")
|
|||||||
SECTION("invalid string in map")
|
SECTION("invalid string in map")
|
||||||
{
|
{
|
||||||
json _;
|
json _;
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found an integer; last byte: 0xFF", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack string: expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0xFF", json::parse_error&);
|
||||||
CHECK(json::from_msgpack(std::vector<uint8_t>({0x81, 0xff, 0x01}), true, false).is_discarded());
|
CHECK(json::from_msgpack(std::vector<uint8_t>({0x81, 0xff, 0x01}), true, false).is_discarded());
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("non-string key (see #3381)")
|
|
||||||
{
|
|
||||||
// only strings map to JSON object keys; any other key is rejected
|
|
||||||
// with a message naming its type
|
|
||||||
const std::vector<std::pair<std::vector<std::uint8_t>, std::string>> cases =
|
|
||||||
{
|
|
||||||
{{0x81, 0xC0, 0x01}, "nil; last byte: 0xC0"},
|
|
||||||
{{0x81, 0xC2, 0x01}, "a boolean; last byte: 0xC2"},
|
|
||||||
{{0x81, 0xC3, 0x01}, "a boolean; last byte: 0xC3"},
|
|
||||||
{{0x81, 0xCA, 0x3F, 0x80, 0x00, 0x00, 0x01}, "a float; last byte: 0xCA"},
|
|
||||||
{{0x81, 0xCB, 0x3F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "a float; last byte: 0xCB"},
|
|
||||||
{{0x81, 0xC4, 0x00, 0x01}, "a bin; last byte: 0xC4"},
|
|
||||||
{{0x81, 0xC5, 0x00, 0x00, 0x01}, "a bin; last byte: 0xC5"},
|
|
||||||
{{0x81, 0xC6, 0x00, 0x00, 0x00, 0x00, 0x01}, "a bin; last byte: 0xC6"},
|
|
||||||
{{0x81, 0xC7, 0x00, 0x01, 0x01}, "an ext; last byte: 0xC7"},
|
|
||||||
{{0x81, 0xC8, 0x00, 0x00, 0x01, 0x01}, "an ext; last byte: 0xC8"},
|
|
||||||
{{0x81, 0xC9, 0x00, 0x00, 0x00, 0x00, 0x01, 0x01}, "an ext; last byte: 0xC9"},
|
|
||||||
{{0x81, 0xD4, 0x01, 0x00, 0x01}, "an ext; last byte: 0xD4"},
|
|
||||||
{{0x81, 0xD5, 0x01, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD5"},
|
|
||||||
{{0x81, 0xD6, 0x01, 0x00, 0x00, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD6"},
|
|
||||||
{{0x81, 0xD7, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD7"},
|
|
||||||
{{0x81, 0xD8, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD8"},
|
|
||||||
{{0x81, 0xCC, 0x01, 0x01}, "an integer; last byte: 0xCC"},
|
|
||||||
{{0x81, 0xCD, 0x00, 0x01, 0x01}, "an integer; last byte: 0xCD"},
|
|
||||||
{{0x81, 0xCE, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xCE"},
|
|
||||||
{{0x81, 0xCF, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xCF"},
|
|
||||||
{{0x81, 0xD0, 0x01, 0x01}, "an integer; last byte: 0xD0"},
|
|
||||||
{{0x81, 0xD1, 0x00, 0x01, 0x01}, "an integer; last byte: 0xD1"},
|
|
||||||
{{0x81, 0xD2, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xD2"},
|
|
||||||
{{0x81, 0xD3, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xD3"},
|
|
||||||
{{0x81, 0x00, 0x01}, "an integer; last byte: 0x00"},
|
|
||||||
{{0x81, 0x7F, 0x01}, "an integer; last byte: 0x7F"},
|
|
||||||
{{0x81, 0xE0, 0x01}, "an integer; last byte: 0xE0"},
|
|
||||||
{{0x81, 0x80, 0x01}, "a map; last byte: 0x80"},
|
|
||||||
{{0x81, 0x8F, 0x01}, "a map; last byte: 0x8F"},
|
|
||||||
{{0x81, 0xDE, 0x00, 0x00, 0x01}, "a map; last byte: 0xDE"},
|
|
||||||
{{0x81, 0xDF, 0x00, 0x00, 0x00, 0x00, 0x01}, "a map; last byte: 0xDF"},
|
|
||||||
{{0x81, 0x90, 0x01}, "an array; last byte: 0x90"},
|
|
||||||
{{0x81, 0x9F, 0x01}, "an array; last byte: 0x9F"},
|
|
||||||
{{0x81, 0xDC, 0x00, 0x00, 0x01}, "an array; last byte: 0xDC"},
|
|
||||||
{{0x81, 0xDD, 0x00, 0x00, 0x00, 0x00, 0x01}, "an array; last byte: 0xDD"},
|
|
||||||
};
|
|
||||||
|
|
||||||
for (const auto& c : cases)
|
|
||||||
{
|
|
||||||
CAPTURE(c.first)
|
|
||||||
const std::string expected = "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found " + c.second;
|
|
||||||
json _;
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(c.first), expected.c_str(), json::parse_error&);
|
|
||||||
CHECK(json::from_msgpack(c.first, true, false).is_discarded());
|
|
||||||
}
|
|
||||||
|
|
||||||
json _;
|
|
||||||
// the unused byte 0xC1 is still reported as a malformed string
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81, 0xC1, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack string: expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0xC1", json::parse_error&);
|
|
||||||
// a missing key is still reported as the end of input
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("invalid UTF-8 in string (see #5529)")
|
SECTION("invalid UTF-8 in string (see #5529)")
|
||||||
{
|
{
|
||||||
// a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8
|
// a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8
|
||||||
|
|||||||
@@ -1018,7 +1018,7 @@ TEST_CASE("regression tests 1")
|
|||||||
};
|
};
|
||||||
|
|
||||||
json _;
|
json _;
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found an array; last byte: 0x98", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x98", json::parse_error&);
|
||||||
|
|
||||||
// related test case: nonempty UTF-8 string (indefinite length)
|
// related test case: nonempty UTF-8 string (indefinite length)
|
||||||
std::vector<uint8_t> const vec1 {0x7f, 0x61, 0x61};
|
std::vector<uint8_t> const vec1 {0x7f, 0x61, 0x61};
|
||||||
@@ -1065,7 +1065,7 @@ TEST_CASE("regression tests 1")
|
|||||||
};
|
};
|
||||||
|
|
||||||
json _;
|
json _;
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec1), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR object key: only string keys are supported, but found a map; last byte: 0xB4", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec1), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xB4", json::parse_error&);
|
||||||
|
|
||||||
// related test case: double-precision
|
// related test case: double-precision
|
||||||
std::vector<uint8_t> const vec2
|
std::vector<uint8_t> const vec2
|
||||||
@@ -1077,7 +1077,7 @@ TEST_CASE("regression tests 1")
|
|||||||
0x96, 0x96, 0xb4, 0xb4, 0xfa, 0x94, 0x94, 0x61,
|
0x96, 0x96, 0xb4, 0xb4, 0xfa, 0x94, 0x94, 0x61,
|
||||||
0x61, 0x61, 0x61, 0x61, 0x61, 0x61, 0x61, 0xfb
|
0x61, 0x61, 0x61, 0x61, 0x61, 0x61, 0x61, 0xfb
|
||||||
};
|
};
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec2), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR object key: only string keys are supported, but found a map; last byte: 0xB4", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec2), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xB4", json::parse_error&);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("issue #452 - Heap-buffer-overflow (OSS-Fuzz issue 585)")
|
SECTION("issue #452 - Heap-buffer-overflow (OSS-Fuzz issue 585)")
|
||||||
|
|||||||
@@ -766,6 +766,15 @@ TEST_CASE("regression tests 2")
|
|||||||
CHECK(j == k);
|
CHECK(j == k);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("issue #4552 - UTF-8 invalid characters are not always ignored when dumping with error_handler_t::ignore")
|
||||||
|
{
|
||||||
|
json node;
|
||||||
|
node["test"] = "test\334\005";
|
||||||
|
CHECK(node.dump(-1, ' ', false, json::error_handler_t::ignore) == "{\"test\":\"test\\u0005\"}");
|
||||||
|
CHECK(node.dump(-1, ' ', false, json::error_handler_t::keep) == "{\"test\":\"test\334\\u0005\"}");
|
||||||
|
CHECK(node.dump(-1, ' ', true, json::error_handler_t::keep) == "{\"test\":\"test\334\\u0005\"}");
|
||||||
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_CASE("regression test - parser callback must not lose a duplicate key's prior value")
|
TEST_CASE("regression test - parser callback must not lose a duplicate key's prior value")
|
||||||
|
|||||||
@@ -92,6 +92,8 @@ TEST_CASE("serialization")
|
|||||||
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"äü\"");
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"äü\"");
|
||||||
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"ä\xEF\xBF\xBDü\"");
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"ä\xEF\xBF\xBDü\"");
|
||||||
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"\\u00e4\\ufffd\\u00fc\"");
|
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"\\u00e4\\ufffd\\u00fc\"");
|
||||||
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"ä\xA9ü\"");
|
||||||
|
CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"\\u00e4\xA9\\u00fc\"");
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("invalid character (regression guard for shared UTF-8 decoder, see #5529)")
|
SECTION("invalid character (regression guard for shared UTF-8 decoder, see #5529)")
|
||||||
@@ -114,6 +116,8 @@ TEST_CASE("serialization")
|
|||||||
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123\"");
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123\"");
|
||||||
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\"");
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\"");
|
||||||
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd\"");
|
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd\"");
|
||||||
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"123\xC2\"");
|
||||||
|
CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"123\xC2\"");
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("unexpected character")
|
SECTION("unexpected character")
|
||||||
@@ -126,6 +130,39 @@ TEST_CASE("serialization")
|
|||||||
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123456\"");
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123456\"");
|
||||||
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\x34\x35\x36\"");
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\x34\x35\x36\"");
|
||||||
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd456\"");
|
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd456\"");
|
||||||
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"123\xF1\xB0\x34\x35\x36\"");
|
||||||
|
CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"123\xF1\xB0\x34\x35\x36\"");
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("keep: valid characters are still escaped")
|
||||||
|
{
|
||||||
|
// an invalid byte followed by characters that must be escaped
|
||||||
|
const json j = "\xC2\"\\\n\xFF\x05";
|
||||||
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xC2\\\"\\\\\\n\xFF\\u0005\"");
|
||||||
|
CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xC2\\\"\\\\\\n\xFF\\u0005\"");
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("keep: truncated multibyte sequences")
|
||||||
|
{
|
||||||
|
CHECK(json("\xF0\x9F\x98").dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xF0\x9F\x98\"");
|
||||||
|
CHECK(json("\xF0\x9F\x98").dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xF0\x9F\x98\"");
|
||||||
|
CHECK(json("\xF0\x9F\x98" "a").dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xF0\x9F\x98" "a\"");
|
||||||
|
CHECK(json("\xF0\x9F\x98" "a").dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xF0\x9F\x98" "a\"");
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("keep: long string with many invalid bytes")
|
||||||
|
{
|
||||||
|
// exceeds the internal string buffer several times
|
||||||
|
std::string input;
|
||||||
|
std::string expected = "\"";
|
||||||
|
for (int i = 0; i < 2000; ++i)
|
||||||
|
{
|
||||||
|
input += "\xFF\xE2\x82\n\xC3\xA4";
|
||||||
|
expected += "\xFF\xE2\x82\\n\xC3\xA4";
|
||||||
|
}
|
||||||
|
expected += "\"";
|
||||||
|
const json j = input;
|
||||||
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == expected);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("U+FFFD Substitution of Maximal Subparts")
|
SECTION("U+FFFD Substitution of Maximal Subparts")
|
||||||
|
|||||||
@@ -14,6 +14,7 @@
|
|||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
using nlohmann::json;
|
using nlohmann::json;
|
||||||
|
|
||||||
|
#include <algorithm>
|
||||||
#include <fstream>
|
#include <fstream>
|
||||||
#include <sstream>
|
#include <sstream>
|
||||||
#include <iostream>
|
#include <iostream>
|
||||||
@@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
static std::string s_replaced2;
|
static std::string s_replaced2;
|
||||||
static std::string s_replaced_ascii;
|
static std::string s_replaced_ascii;
|
||||||
static std::string s_replaced2_ascii;
|
static std::string s_replaced2_ascii;
|
||||||
|
static std::string s_kept;
|
||||||
|
static std::string s_kept2;
|
||||||
|
static std::string s_kept_ascii;
|
||||||
|
|
||||||
// dumping with ignore/replace must not throw in any case
|
// dumping with ignore/replace/keep must not throw in any case
|
||||||
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
|
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
|
||||||
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
|
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
|
||||||
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
|
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
|
||||||
@@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
|
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
|
||||||
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
|
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
|
||||||
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
|
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
|
||||||
|
s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep);
|
||||||
|
s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep);
|
||||||
|
s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep);
|
||||||
|
|
||||||
if (success_expected)
|
if (success_expected)
|
||||||
{
|
{
|
||||||
@@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
// all dumps should agree on the string
|
// all dumps should agree on the string
|
||||||
CHECK(s_strict == s_ignored);
|
CHECK(s_strict == s_ignored);
|
||||||
CHECK(s_strict == s_replaced);
|
CHECK(s_strict == s_replaced);
|
||||||
|
CHECK(s_strict == s_kept);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
|
|
||||||
// check that replace string contains a replacement character
|
// check that replace string contains a replacement character
|
||||||
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
|
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
|
||||||
|
|
||||||
|
// ignore drops the invalid bytes, keep copies them
|
||||||
|
CHECK(s_ignored != s_kept);
|
||||||
|
CHECK(s_ignored_ascii != s_kept_ascii);
|
||||||
|
|
||||||
|
// unless a byte needs escaping, keep copies the input unchanged
|
||||||
|
const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c)
|
||||||
|
{
|
||||||
|
return static_cast<unsigned char>(c) < 0x20 || c == '"' || c == '\\';
|
||||||
|
});
|
||||||
|
if (!needs_escaping)
|
||||||
|
{
|
||||||
|
CHECK(s_kept == "\"" + json_string + "\"");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// check that prefix and suffix are preserved
|
// check that prefix and suffix are preserved
|
||||||
@@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
|
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
|
||||||
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
|
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
|
||||||
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
|
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
|
||||||
|
CHECK(s_kept2.substr(1, 3) == "abc");
|
||||||
|
CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz");
|
||||||
}
|
}
|
||||||
|
|
||||||
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
|
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
|
||||||
|
|||||||
@@ -14,6 +14,7 @@
|
|||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
using nlohmann::json;
|
using nlohmann::json;
|
||||||
|
|
||||||
|
#include <algorithm>
|
||||||
#include <fstream>
|
#include <fstream>
|
||||||
#include <sstream>
|
#include <sstream>
|
||||||
#include <iostream>
|
#include <iostream>
|
||||||
@@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
static std::string s_replaced2;
|
static std::string s_replaced2;
|
||||||
static std::string s_replaced_ascii;
|
static std::string s_replaced_ascii;
|
||||||
static std::string s_replaced2_ascii;
|
static std::string s_replaced2_ascii;
|
||||||
|
static std::string s_kept;
|
||||||
|
static std::string s_kept2;
|
||||||
|
static std::string s_kept_ascii;
|
||||||
|
|
||||||
// dumping with ignore/replace must not throw in any case
|
// dumping with ignore/replace/keep must not throw in any case
|
||||||
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
|
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
|
||||||
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
|
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
|
||||||
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
|
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
|
||||||
@@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
|
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
|
||||||
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
|
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
|
||||||
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
|
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
|
||||||
|
s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep);
|
||||||
|
s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep);
|
||||||
|
s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep);
|
||||||
|
|
||||||
if (success_expected)
|
if (success_expected)
|
||||||
{
|
{
|
||||||
@@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
// all dumps should agree on the string
|
// all dumps should agree on the string
|
||||||
CHECK(s_strict == s_ignored);
|
CHECK(s_strict == s_ignored);
|
||||||
CHECK(s_strict == s_replaced);
|
CHECK(s_strict == s_replaced);
|
||||||
|
CHECK(s_strict == s_kept);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
|
|
||||||
// check that replace string contains a replacement character
|
// check that replace string contains a replacement character
|
||||||
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
|
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
|
||||||
|
|
||||||
|
// ignore drops the invalid bytes, keep copies them
|
||||||
|
CHECK(s_ignored != s_kept);
|
||||||
|
CHECK(s_ignored_ascii != s_kept_ascii);
|
||||||
|
|
||||||
|
// unless a byte needs escaping, keep copies the input unchanged
|
||||||
|
const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c)
|
||||||
|
{
|
||||||
|
return static_cast<unsigned char>(c) < 0x20 || c == '"' || c == '\\';
|
||||||
|
});
|
||||||
|
if (!needs_escaping)
|
||||||
|
{
|
||||||
|
CHECK(s_kept == "\"" + json_string + "\"");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// check that prefix and suffix are preserved
|
// check that prefix and suffix are preserved
|
||||||
@@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
|
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
|
||||||
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
|
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
|
||||||
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
|
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
|
||||||
|
CHECK(s_kept2.substr(1, 3) == "abc");
|
||||||
|
CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz");
|
||||||
}
|
}
|
||||||
|
|
||||||
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
|
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
|
||||||
|
|||||||
@@ -14,6 +14,7 @@
|
|||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
using nlohmann::json;
|
using nlohmann::json;
|
||||||
|
|
||||||
|
#include <algorithm>
|
||||||
#include <fstream>
|
#include <fstream>
|
||||||
#include <sstream>
|
#include <sstream>
|
||||||
#include <iostream>
|
#include <iostream>
|
||||||
@@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
static std::string s_replaced2;
|
static std::string s_replaced2;
|
||||||
static std::string s_replaced_ascii;
|
static std::string s_replaced_ascii;
|
||||||
static std::string s_replaced2_ascii;
|
static std::string s_replaced2_ascii;
|
||||||
|
static std::string s_kept;
|
||||||
|
static std::string s_kept2;
|
||||||
|
static std::string s_kept_ascii;
|
||||||
|
|
||||||
// dumping with ignore/replace must not throw in any case
|
// dumping with ignore/replace/keep must not throw in any case
|
||||||
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
|
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
|
||||||
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
|
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
|
||||||
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
|
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
|
||||||
@@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
|
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
|
||||||
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
|
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
|
||||||
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
|
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
|
||||||
|
s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep);
|
||||||
|
s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep);
|
||||||
|
s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep);
|
||||||
|
|
||||||
if (success_expected)
|
if (success_expected)
|
||||||
{
|
{
|
||||||
@@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
// all dumps should agree on the string
|
// all dumps should agree on the string
|
||||||
CHECK(s_strict == s_ignored);
|
CHECK(s_strict == s_ignored);
|
||||||
CHECK(s_strict == s_replaced);
|
CHECK(s_strict == s_replaced);
|
||||||
|
CHECK(s_strict == s_kept);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
|
|
||||||
// check that replace string contains a replacement character
|
// check that replace string contains a replacement character
|
||||||
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
|
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
|
||||||
|
|
||||||
|
// ignore drops the invalid bytes, keep copies them
|
||||||
|
CHECK(s_ignored != s_kept);
|
||||||
|
CHECK(s_ignored_ascii != s_kept_ascii);
|
||||||
|
|
||||||
|
// unless a byte needs escaping, keep copies the input unchanged
|
||||||
|
const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c)
|
||||||
|
{
|
||||||
|
return static_cast<unsigned char>(c) < 0x20 || c == '"' || c == '\\';
|
||||||
|
});
|
||||||
|
if (!needs_escaping)
|
||||||
|
{
|
||||||
|
CHECK(s_kept == "\"" + json_string + "\"");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// check that prefix and suffix are preserved
|
// check that prefix and suffix are preserved
|
||||||
@@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
|
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
|
||||||
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
|
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
|
||||||
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
|
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
|
||||||
|
CHECK(s_kept2.substr(1, 3) == "abc");
|
||||||
|
CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz");
|
||||||
}
|
}
|
||||||
|
|
||||||
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
|
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
|
||||||
|
|||||||
@@ -14,6 +14,7 @@
|
|||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
using nlohmann::json;
|
using nlohmann::json;
|
||||||
|
|
||||||
|
#include <algorithm>
|
||||||
#include <fstream>
|
#include <fstream>
|
||||||
#include <sstream>
|
#include <sstream>
|
||||||
#include <iostream>
|
#include <iostream>
|
||||||
@@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
static std::string s_replaced2;
|
static std::string s_replaced2;
|
||||||
static std::string s_replaced_ascii;
|
static std::string s_replaced_ascii;
|
||||||
static std::string s_replaced2_ascii;
|
static std::string s_replaced2_ascii;
|
||||||
|
static std::string s_kept;
|
||||||
|
static std::string s_kept2;
|
||||||
|
static std::string s_kept_ascii;
|
||||||
|
|
||||||
// dumping with ignore/replace must not throw in any case
|
// dumping with ignore/replace/keep must not throw in any case
|
||||||
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
|
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
|
||||||
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
|
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
|
||||||
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
|
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
|
||||||
@@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
|
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
|
||||||
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
|
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
|
||||||
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
|
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
|
||||||
|
s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep);
|
||||||
|
s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep);
|
||||||
|
s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep);
|
||||||
|
|
||||||
if (success_expected)
|
if (success_expected)
|
||||||
{
|
{
|
||||||
@@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
// all dumps should agree on the string
|
// all dumps should agree on the string
|
||||||
CHECK(s_strict == s_ignored);
|
CHECK(s_strict == s_ignored);
|
||||||
CHECK(s_strict == s_replaced);
|
CHECK(s_strict == s_replaced);
|
||||||
|
CHECK(s_strict == s_kept);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
|
|
||||||
// check that replace string contains a replacement character
|
// check that replace string contains a replacement character
|
||||||
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
|
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
|
||||||
|
|
||||||
|
// ignore drops the invalid bytes, keep copies them
|
||||||
|
CHECK(s_ignored != s_kept);
|
||||||
|
CHECK(s_ignored_ascii != s_kept_ascii);
|
||||||
|
|
||||||
|
// unless a byte needs escaping, keep copies the input unchanged
|
||||||
|
const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c)
|
||||||
|
{
|
||||||
|
return static_cast<unsigned char>(c) < 0x20 || c == '"' || c == '\\';
|
||||||
|
});
|
||||||
|
if (!needs_escaping)
|
||||||
|
{
|
||||||
|
CHECK(s_kept == "\"" + json_string + "\"");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// check that prefix and suffix are preserved
|
// check that prefix and suffix are preserved
|
||||||
@@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
|
|||||||
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
|
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
|
||||||
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
|
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
|
||||||
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
|
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
|
||||||
|
CHECK(s_kept2.substr(1, 3) == "abc");
|
||||||
|
CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz");
|
||||||
}
|
}
|
||||||
|
|
||||||
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
|
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
|
||||||
|
|||||||
Reference in New Issue
Block a user