mirror of
https://github.com/nlohmann/json.git
synced 2026-09-29 19:20:30 +00:00
Compare commits
11
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
792853d725 | ||
|
|
4bdf1b7e74 | ||
|
|
b1e9d98e41 | ||
|
|
f855d257df | ||
|
|
3e683e9c04 | ||
|
|
d1d84ed9af | ||
|
|
de8529f99b | ||
|
|
677794f076 | ||
|
|
437a95cfdb | ||
|
|
c8735246d0 | ||
|
|
9adb510a0d |
@@ -21,7 +21,6 @@ cc_library(
|
||||
"include/nlohmann/adl_serializer.hpp",
|
||||
"include/nlohmann/byte_container_with_subtype.hpp",
|
||||
"include/nlohmann/detail/abi_macros.hpp",
|
||||
"include/nlohmann/detail/bit_ops.hpp",
|
||||
"include/nlohmann/detail/conversions/from_json.hpp",
|
||||
"include/nlohmann/detail/conversions/to_chars.hpp",
|
||||
"include/nlohmann/detail/conversions/to_json.hpp",
|
||||
@@ -34,7 +33,6 @@ cc_library(
|
||||
"include/nlohmann/detail/input/number_parse.hpp",
|
||||
"include/nlohmann/detail/input/parser.hpp",
|
||||
"include/nlohmann/detail/input/position_t.hpp",
|
||||
"include/nlohmann/detail/input/pow5_table.hpp",
|
||||
"include/nlohmann/detail/input/string_scan.hpp",
|
||||
"include/nlohmann/detail/iterators/internal_iterator.hpp",
|
||||
"include/nlohmann/detail/iterators/iter_impl.hpp",
|
||||
|
||||
@@ -496,7 +496,7 @@ bool key(string_t& val);
|
||||
bool parse_error(std::size_t position, const std::string& last_token, const detail::exception& ex);
|
||||
```
|
||||
|
||||
The return value of each function determines whether parsing should proceed.
|
||||
The return value of each function determines whether parsing should proceed. For `parse_error`, returning `true` [recovers from the error](https://json.nlohmann.me/features/parsing/error_recovery/): the parser repairs the input and continues.
|
||||
|
||||
To implement your own SAX handler, proceed as follows:
|
||||
|
||||
@@ -504,7 +504,7 @@ To implement your own SAX handler, proceed as follows:
|
||||
2. Create an object of your SAX interface class, e.g. `my_sax`.
|
||||
3. Call `bool json::sax_parse(input, &my_sax)`; where the first parameter can be any input like a string or an input stream and the second parameter is a pointer to your SAX interface.
|
||||
|
||||
Note the `sax_parse` function only returns a `bool` indicating the result of the last executed SAX event. It does not return a `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error -- it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file [`json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp).
|
||||
Note the `sax_parse` function only returns a `bool` indicating whether the input was parsed without errors and no SAX event returned `false`. It does not return a `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error -- it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file [`json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp).
|
||||
|
||||
### STL-like access
|
||||
|
||||
@@ -1395,7 +1395,6 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I
|
||||
- The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/)
|
||||
- The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/).
|
||||
- The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0).
|
||||
- The class contains an adapted version of the Eisel-Lemire algorithm and its table of powers of five from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors
|
||||
|
||||
<img align="right" src="https://git.fsfe.org/reuse/reuse-ci/raw/branch/master/reuse-horizontal.png" alt="REUSE Software">
|
||||
|
||||
|
||||
@@ -90,7 +90,9 @@ The SAX event lister must follow the interface of [`json_sax`](../json_sax/index
|
||||
|
||||
## Return value
|
||||
|
||||
return value of the last processed SAX event
|
||||
`#!cpp true` if the input was parsed without errors and no SAX event returned `#!cpp false`; `#!cpp false` otherwise.
|
||||
In particular, the result is `#!cpp false` for input with errors, even if the SAX parser recovered from all of them
|
||||
(see [error recovery](../../features/parsing/error_recovery.md)).
|
||||
|
||||
## Exception safety
|
||||
|
||||
@@ -138,6 +140,7 @@ A UTF-8 byte order mark is silently ignored.
|
||||
- Ignoring comments via `ignore_comments` added in version 3.9.0.
|
||||
- Added `ignore_trailing_commas` in version 3.13.0.
|
||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||
- Recovering from parse errors (see [`parse_error`](../json_sax/parse_error.md)) added in version 3.13.0.
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
- `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave a `#!cpp std::istream` positioned right
|
||||
after the parsed value when `strict` is `#!cpp false`.
|
||||
|
||||
@@ -7,7 +7,8 @@ struct json_sax;
|
||||
|
||||
This class describes the SAX interface used by [sax_parse](../basic_json/sax_parse.md). Each function is called in
|
||||
different situations while the input is parsed. The boolean return value informs the parser whether to continue
|
||||
processing the input.
|
||||
processing the input; for [`parse_error`](parse_error.md), it decides whether to
|
||||
[recover from the error](../../features/parsing/error_recovery.md).
|
||||
|
||||
## Template parameters
|
||||
|
||||
|
||||
@@ -21,7 +21,14 @@ A parse error occurred.
|
||||
|
||||
## Return value
|
||||
|
||||
Whether parsing should proceed (**must return `#!cpp false`**).
|
||||
Whether to recover from the error:
|
||||
|
||||
- `#!cpp false` stops parsing.
|
||||
- `#!cpp true` recovers from the error: the error is repaired and parsing continues. If that is not possible, which
|
||||
happens in the binary formats when the end of the item with the error is unknown, the value read so far is completed
|
||||
and parsing stops. See [error recovery](../../features/parsing/error_recovery.md) for how errors are repaired.
|
||||
|
||||
Either way, [`sax_parse`](../basic_json/sax_parse.md) returns `#!cpp false`.
|
||||
|
||||
## Examples
|
||||
|
||||
@@ -39,6 +46,22 @@ Whether parsing should proceed (**must return `#!cpp false`**).
|
||||
--8<-- "examples/sax_parse.output"
|
||||
```
|
||||
|
||||
??? example
|
||||
|
||||
The example below shows how a SAX parser recovers from errors.
|
||||
|
||||
```cpp
|
||||
--8<-- "examples/sax_parse__error_recovery.cpp"
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```
|
||||
--8<-- "examples/sax_parse__error_recovery.output"
|
||||
```
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.2.0.
|
||||
- Returning `#!cpp true` recovers from the error since version 3.13.0; before, parsing stopped, but the result of
|
||||
[`sax_parse`](../basic_json/sax_parse.md) could be wrong.
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
#include <iostream>
|
||||
#include <iomanip>
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// a SAX parser that creates a JSON value like json::parse does, but that
|
||||
// recovers from parse errors instead of stopping at the first one
|
||||
class recovering_parser : public nlohmann::detail::json_sax_dom_parser<json>
|
||||
{
|
||||
public:
|
||||
explicit recovering_parser(json& result)
|
||||
: nlohmann::detail::json_sax_dom_parser<json>(result, false)
|
||||
{}
|
||||
|
||||
bool parse_error(std::size_t position,
|
||||
const std::string& /*last_token*/,
|
||||
const json::exception& ex)
|
||||
{
|
||||
std::cout << "byte " << position << ": " << ex.what() << '\n';
|
||||
|
||||
// repair the input and continue
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
int main()
|
||||
{
|
||||
// JSON text with several mistakes that ends too early
|
||||
const std::string text = R"({
|
||||
"name": "Hello World",
|
||||
"tags": ["a" "b",],
|
||||
"valid": tru,
|
||||
"size": 1.,
|
||||
"nested": {"x": 1)";
|
||||
|
||||
json result;
|
||||
recovering_parser sax(result);
|
||||
const bool valid = json::sax_parse(text, &sax);
|
||||
|
||||
std::cout << "\nvalid JSON: " << std::boolalpha << valid << '\n'
|
||||
<< std::setw(4) << result << std::endl;
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
byte 49: [json.exception.parse_error.101] parse error at line 3, column 20: syntax error while parsing array - unexpected string literal; expected ']'
|
||||
byte 51: [json.exception.parse_error.101] parse error at line 3, column 22: syntax error while parsing value - unexpected ']'; expected '[', '{', or a literal
|
||||
byte 70: [json.exception.parse_error.101] parse error at line 4, column 17: syntax error while parsing value - invalid literal; last read: '"valid": tru,'
|
||||
byte 86: [json.exception.parse_error.101] parse error at line 5, column 15: syntax error while parsing value - invalid number; expected digit after '.'; last read: '1.,'
|
||||
byte 109: [json.exception.parse_error.101] parse error at line 6, column 22: syntax error while parsing object - unexpected end of input; expected '}'
|
||||
|
||||
valid JSON: false
|
||||
{
|
||||
"name": "Hello World",
|
||||
"nested": {
|
||||
"x": 1
|
||||
},
|
||||
"size": 1,
|
||||
"tags": [
|
||||
"a",
|
||||
"b"
|
||||
],
|
||||
"valid": null
|
||||
}
|
||||
@@ -0,0 +1,121 @@
|
||||
# Error Recovery
|
||||
|
||||
By default, parsing stops at the first error. With the [SAX interface](sax_interface.md), you can instead ask the
|
||||
parser to *recover*: to repair the error and continue, so that you get as much as possible out of malformed input, for
|
||||
instance a file that was cut off, JSON edited by hand, or the output of a language model.
|
||||
|
||||
## Recovering from errors
|
||||
|
||||
The SAX parser's [`parse_error`](../../api/json_sax/parse_error.md) function is called for every error. Its return value
|
||||
decides what happens next:
|
||||
|
||||
- `#!cpp false` stops parsing. This is what the SAX parsers of the library do, so [`parse`](../../api/basic_json/parse.md)
|
||||
and [`accept`](../../api/basic_json/accept.md) never recover.
|
||||
- `#!cpp true` repairs the error and continues parsing.
|
||||
|
||||
When recovering, the SAX parser still receives well-formed events: every `start_object` or `start_array` is followed by
|
||||
the matching `end_object` or `end_array`, and every `key` is followed by exactly one value. A SAX parser that creates a
|
||||
JSON value, such as the one in the example below, therefore gets a complete value. Parsing always ends, and
|
||||
[`sax_parse`](../../api/basic_json/sax_parse.md) returns `#!cpp false` for input that is not valid JSON, even if every
|
||||
error was repaired. Each token is reported at most once, and the SAX parser can stop at any error by returning
|
||||
`#!cpp false`.
|
||||
|
||||
!!! example
|
||||
|
||||
The example below derives a SAX parser from the library's parser for `json` values (`json_sax_dom_parser`),
|
||||
and recovers from all errors.
|
||||
|
||||
```cpp
|
||||
--8<-- "examples/sax_parse__error_recovery.cpp"
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```
|
||||
--8<-- "examples/sax_parse__error_recovery.output"
|
||||
```
|
||||
|
||||
## How errors are repaired
|
||||
|
||||
Each error is repaired with the smallest local edit: a missing separator is inserted, a stray token is removed, what can
|
||||
be read of a broken string or number is kept, and a value that cannot be read at all becomes `#!json null`.
|
||||
|
||||
| Mistake | Repair | Example | Result |
|
||||
|---------------------------|--------------------------------------------------------------------------------|------------------------------------------|----------------------------|
|
||||
| missing `,` or `:` | inserted | `#!json [1 2]`, `#!json {"a" 1}` | `[1,2]`, `{"a":1}` |
|
||||
| missing value | `#!json null` for an object key or between commas in an array | `#!json {"a":}`, `#!json [1,,2]` | `{"a":null}`, `[1,null,2]` |
|
||||
| trailing comma | removed | `#!json [1,2,]` | `[1,2]` |
|
||||
| broken string | invalid escapes and bytes are replaced (see below); a line break ends the string | `#!json ["a\qb"]` | `["aqb"]` |
|
||||
| broken number | the longest valid beginning is kept | `#!json [1., 2e+]` | `[1,2]` |
|
||||
| unreadable value | `#!json null` | `#!json [1, NaN, tru]` | `[1,null,null]` |
|
||||
| number too large | passed as infinity, together with its text | `#!json [1e999]` | infinity (see below) |
|
||||
| stray `:` | removed | `#!json ["a":1]` | `["a",1]` |
|
||||
| member without a key | skipped up to the next `,` or `}` | `#!json {1:2, "b":3}` | `{"b":3}` |
|
||||
| wrong closing bracket | closes the innermost array or object | `#!json {"a":[1,2}, "b":3}` | `{"a":[1,2],"b":3}` |
|
||||
| input ends too early | all open arrays and objects are closed | `#!json {"a":[1,2` | `{"a":[1,2]}` |
|
||||
| text before the value | skipped | `#!json )]}'{"a":1}` | `{"a":1}` |
|
||||
|
||||
In a string, an unknown escape like `\q` stands for the escaped character (`q`), as in JavaScript. An invalid `\u`
|
||||
escape, a lone surrogate, and ill-formed UTF-8 are each replaced by U+FFFD (REPLACEMENT CHARACTER), and control
|
||||
characters are kept. A string without its closing quote ends at the next line break or at the end of the input.
|
||||
|
||||
The input after the top-level value is not repaired: as without recovery, it is reported as an error, and parsing stops.
|
||||
|
||||
## Binary formats
|
||||
|
||||
The binary formats ([BJData](../binary_formats/bjdata.md), [BON8](../binary_formats/bon8.md),
|
||||
[BSON](../binary_formats/bson.md), [CBOR](../binary_formats/cbor.md), [MessagePack](../binary_formats/messagepack.md),
|
||||
and [UBJSON](../binary_formats/ubjson.md)) have no delimiters to find the next value by. So what can be repaired depends
|
||||
on whether the end of the item with the error is known, a distinction that
|
||||
[RFC 8949, Section 5.3](https://www.rfc-editor.org/rfc/rfc8949.html#section-5.3) makes for CBOR, too.
|
||||
|
||||
If the item is complete, but cannot be passed on as it is, it is replaced, and parsing continues after it:
|
||||
|
||||
| Mistake | Formats | Repair |
|
||||
|---------------------------------------------------------------------|-----------------------------------------|-------------------------------------------------------------------------|
|
||||
| tag | CBOR | ignored |
|
||||
| simple value other than `false`, `true`, and `null`, like undefined | CBOR | `#!json null` |
|
||||
| negative integer below the range of `number_integer_t` | CBOR | the nearest floating-point number |
|
||||
| string that is not valid UTF-8 | BJData, BSON, CBOR, MessagePack, UBJSON | each ill-formed sequence becomes U+FFFD |
|
||||
| character (`C`) that is not ASCII | BJData, UBJSON | U+FFFD |
|
||||
| invalid high-precision number (`H`) | BJData, UBJSON | the longest valid beginning is kept, as for JSON text, or `#!json null` |
|
||||
| high-precision number too large | BJData, UBJSON | passed as infinity, together with its text |
|
||||
| object key that is not a string | BON8, CBOR, MessagePack | the member is skipped |
|
||||
| element of a type the library does not read, like ObjectId or date | BSON | `#!json null` |
|
||||
| string without its terminator | BSON | kept |
|
||||
| document whose size does not match its content | BSON | kept |
|
||||
|
||||
CBOR tags and simple values are repaired as [RFC 8949, Section 6.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-6.1)
|
||||
suggests for converting CBOR to JSON. Note that [`sax_parse`](../../api/basic_json/sax_parse.md) has no parameter for
|
||||
CBOR tags, so every tag is an error there; when recovering, tags are ignored like with
|
||||
[`cbor_tag_handler_t::ignore`](../../api/basic_json/cbor_tag_handler_t.md).
|
||||
|
||||
After any other error, the end of the item is unknown: the input ended, a byte is not a valid type marker, or a size
|
||||
cannot be right. Parsing then stops, and the value read so far is completed: a key that waits for its value gets
|
||||
`#!json null`, and all open arrays and objects are closed. This keeps everything before the error of an input that was
|
||||
cut off. The exception is BSON, which stores the size of every document: an element whose end is unknown gets
|
||||
`#!json null`, the rest of its document is skipped, and parsing continues after the document.
|
||||
|
||||
## Limitations
|
||||
|
||||
- A repair is a guess. For example, `#!json {"a" "b": 1}` could be meant as `#!json {"a": "b"}` or as
|
||||
`#!json {"a": null, "b": 1}`; it is repaired to the former. Treat recovered values as a best effort, and check the
|
||||
reported errors.
|
||||
- A closing bracket always closes the innermost array or object. If a bracket is missing rather than wrong, the
|
||||
repair differs from the intention: `#!json {"a": {"b": [1, 2}, "c": 3}` is repaired to
|
||||
`#!json {"a": {"b": [1, 2], "c": 3}}`, although `#!json {"a": {"b": [1, 2]}, "c": 3}` may have been meant.
|
||||
- Keys without quotes, and strings in single quotes, are not supported; such members are skipped.
|
||||
- In the binary formats, a member that is skipped because its key is not a string is lost, and so are the elements of a
|
||||
BSON document after one whose end is unknown.
|
||||
- A number that is too large for `number_float_t` is passed as positive or negative infinity. The SAX parser's
|
||||
`number_float` also gets the number's text, but a JSON value cannot store it, and
|
||||
[`dump`](../../api/basic_json/dump.md) serializes infinity as `#!json null`.
|
||||
- When parsing is not strict (see [`sax_parse`](../../api/basic_json/sax_parse.md)), a repair may read parts of the
|
||||
input after the value, for instance of the next value in a stream of concatenated values.
|
||||
|
||||
## See also
|
||||
|
||||
- [SAX interface](sax_interface.md) - implement a custom SAX handler
|
||||
- [`parse_error`](../../api/json_sax/parse_error.md) - the SAX event for parse errors
|
||||
- [`sax_parse`](../../api/basic_json/sax_parse.md) - generate SAX events
|
||||
- [parsing and exceptions](parse_exceptions.md) - control error handling
|
||||
@@ -65,7 +65,7 @@ You can influence a DOM parse without switching to the SAX interface by passing
|
||||
When the input is not valid JSON, the `parse` function throws an exception by default. If exceptions are undesired or
|
||||
unavailable, the parser can instead return a discarded value, or [`accept`](../../api/basic_json/accept.md) can be used
|
||||
to only check whether an input is valid JSON. See [parsing and exceptions](parse_exceptions.md) for the available
|
||||
options.
|
||||
options. To get as much as possible out of malformed input, a SAX parser can [recover from errors](error_recovery.md).
|
||||
|
||||
## See also
|
||||
|
||||
@@ -76,3 +76,4 @@ options.
|
||||
- [parser callbacks](parser_callbacks.md) - influence the parsing by a callback function
|
||||
- [SAX interface](sax_interface.md) - implement a custom SAX handler
|
||||
- [parsing and exceptions](parse_exceptions.md) - control error handling
|
||||
- [error recovery](error_recovery.md) - get as much as possible out of malformed input
|
||||
|
||||
@@ -64,7 +64,8 @@ bool parse_error(std::size_t position,
|
||||
const json::exception& ex);
|
||||
```
|
||||
|
||||
The return value indicates whether the parsing should continue, so the function should usually return `#!cpp false`.
|
||||
The return value decides whether to stop parsing (`#!cpp false`) or to repair the error and continue
|
||||
(`#!cpp true`); see [error recovery](error_recovery.md) for the latter.
|
||||
|
||||
??? example
|
||||
|
||||
|
||||
@@ -60,7 +60,8 @@ bool key(string_t& val);
|
||||
bool parse_error(std::size_t position, const std::string& last_token, const json::exception& ex);
|
||||
```
|
||||
|
||||
The return value of each function determines whether parsing should proceed.
|
||||
The return value of each function determines whether parsing should proceed. For `parse_error`, returning
|
||||
`#!cpp true` [recovers from the error](error_recovery.md).
|
||||
|
||||
To implement your own SAX handler, proceed as follows:
|
||||
|
||||
@@ -68,7 +69,7 @@ To implement your own SAX handler, proceed as follows:
|
||||
2. Create an object of your SAX interface class, e.g. `my_sax`.
|
||||
3. Call `#!cpp bool json::sax_parse(input, &my_sax);` where the first parameter can be any input like a string or an input stream and the second parameter is a pointer to your SAX interface.
|
||||
|
||||
Note the `sax_parse` function only returns a `#!cpp bool` indicating the result of the last executed SAX event. It does not return `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error - it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file `json_sax.hpp`.
|
||||
Note the `sax_parse` function only returns a `#!cpp bool` indicating whether the input was parsed without errors and no SAX event returned `#!cpp false`. It does not return `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error - it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file `json_sax.hpp`.
|
||||
|
||||
## See also
|
||||
|
||||
|
||||
@@ -19,5 +19,3 @@ The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed und
|
||||
The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/)
|
||||
|
||||
The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/).
|
||||
|
||||
The class contains an adapted version of the Eisel-Lemire algorithm and its table of powers of five from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors
|
||||
|
||||
@@ -87,6 +87,7 @@ nav:
|
||||
- features/object_order.md
|
||||
- Parsing:
|
||||
- features/parsing/index.md
|
||||
- features/parsing/error_recovery.md
|
||||
- features/parsing/json_lines.md
|
||||
- features/parsing/parse_exceptions.md
|
||||
- features/parsing/parser_callbacks.md
|
||||
|
||||
@@ -1,105 +0,0 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint> // uint64_t
|
||||
|
||||
#include <nlohmann/detail/abi_macros.hpp>
|
||||
|
||||
// Portable bit-level helpers for the number and string scanners. They use
|
||||
// compiler builtins where available and plain C++ otherwise, so they need no
|
||||
// platform headers and work regardless of byte order.
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
/// number of leading zero bits of x (x != 0)
|
||||
inline int count_leading_zeros(std::uint64_t x) noexcept
|
||||
{
|
||||
#if defined(__GNUC__) || defined(__clang__)
|
||||
return __builtin_clzll(x);
|
||||
#else
|
||||
int n = 0;
|
||||
for (int shift = 32; shift != 0; shift >>= 1)
|
||||
{
|
||||
if ((x >> (64 - shift)) == 0)
|
||||
{
|
||||
n += shift;
|
||||
x <<= shift;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
#endif
|
||||
}
|
||||
|
||||
/// number of trailing zero bits of x (x != 0)
|
||||
inline int count_trailing_zeros(std::uint64_t x) noexcept
|
||||
{
|
||||
#if defined(__GNUC__) || defined(__clang__)
|
||||
return __builtin_ctzll(x);
|
||||
#else
|
||||
int n = 0;
|
||||
for (int shift = 32; shift != 0; shift >>= 1)
|
||||
{
|
||||
if ((x << (64 - shift)) == 0)
|
||||
{
|
||||
n += shift;
|
||||
x >>= shift;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
#endif
|
||||
}
|
||||
|
||||
/// the 128-bit product of two 64-bit numbers
|
||||
struct uint128_parts
|
||||
{
|
||||
std::uint64_t low;
|
||||
std::uint64_t high;
|
||||
};
|
||||
|
||||
inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexcept
|
||||
{
|
||||
#if defined(__SIZEOF_INT128__)
|
||||
__extension__ using uint128 = unsigned __int128;
|
||||
const uint128 r = static_cast<uint128>(a) * b;
|
||||
return {static_cast<std::uint64_t>(r), static_cast<std::uint64_t>(r >> 64u)};
|
||||
#else
|
||||
const std::uint64_t a_lo = a & 0xFFFFFFFFu;
|
||||
const std::uint64_t a_hi = a >> 32u;
|
||||
const std::uint64_t b_lo = b & 0xFFFFFFFFu;
|
||||
const std::uint64_t b_hi = b >> 32u;
|
||||
const std::uint64_t lo_lo = a_lo * b_lo;
|
||||
const std::uint64_t hi_lo = a_hi * b_lo;
|
||||
const std::uint64_t lo_hi = a_lo * b_hi;
|
||||
const std::uint64_t hi_hi = a_hi * b_hi;
|
||||
const std::uint64_t cross = (lo_lo >> 32u) + (hi_lo & 0xFFFFFFFFu) + lo_hi;
|
||||
return {(cross << 32u) | (lo_lo & 0xFFFFFFFFu), (hi_lo >> 32u) + (cross >> 32u) + hi_hi};
|
||||
#endif
|
||||
}
|
||||
|
||||
/// eight bytes as a little-endian word (compilers fold this into one load on
|
||||
/// little-endian targets)
|
||||
inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
|
||||
{
|
||||
return static_cast<std::uint64_t>(b[0]) | (static_cast<std::uint64_t>(b[1]) << 8u)
|
||||
| (static_cast<std::uint64_t>(b[2]) << 16u) | (static_cast<std::uint64_t>(b[3]) << 24u)
|
||||
| (static_cast<std::uint64_t>(b[4]) << 32u) | (static_cast<std::uint64_t>(b[5]) << 40u)
|
||||
| (static_cast<std::uint64_t>(b[6]) << 48u) | (static_cast<std::uint64_t>(b[7]) << 56u);
|
||||
}
|
||||
|
||||
/// eight bytes as a little-endian word
|
||||
inline std::uint64_t read_eight_bytes(const char* p) noexcept
|
||||
{
|
||||
return read_eight_bytes(reinterpret_cast<const unsigned char*>(p)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
File diff suppressed because it is too large
Load Diff
@@ -131,7 +131,9 @@ struct json_sax
|
||||
@param[in] position the position in the input where the error occurs
|
||||
@param[in] last_token the last read token
|
||||
@param[in] ex an exception object describing the error
|
||||
@return whether parsing should proceed (must return false)
|
||||
@return whether to recover from the error: false stops parsing; true
|
||||
repairs the error and continues, or, if that is not possible,
|
||||
stops after completing the value read so far
|
||||
*/
|
||||
virtual bool parse_error(std::size_t position,
|
||||
const std::string& last_token,
|
||||
@@ -186,9 +188,12 @@ a pointer to the respective array or object for each recursion depth.
|
||||
After successful parsing, the value that is passed by reference to the
|
||||
constructor contains the parsed value.
|
||||
|
||||
@tparam BasicJsonType the JSON type
|
||||
@tparam BasicJsonType the JSON type
|
||||
@tparam InputAdapterType the input adapter of the lexer that can be passed to
|
||||
the constructor to record diagnostic positions; it
|
||||
does not matter if no lexer is passed
|
||||
*/
|
||||
template<typename BasicJsonType, typename InputAdapterType>
|
||||
template<typename BasicJsonType, typename InputAdapterType = string_input_adapter_type>
|
||||
class json_sax_dom_parser
|
||||
{
|
||||
public:
|
||||
@@ -505,7 +510,7 @@ class json_sax_dom_parser
|
||||
lexer_t* m_lexer_ref = nullptr;
|
||||
};
|
||||
|
||||
template<typename BasicJsonType, typename InputAdapterType>
|
||||
template<typename BasicJsonType, typename InputAdapterType = string_input_adapter_type>
|
||||
class json_sax_dom_callback_parser
|
||||
{
|
||||
public:
|
||||
|
||||
@@ -9,8 +9,11 @@
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
#include <utility> // move
|
||||
@@ -215,6 +218,18 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
~lexer() = default;
|
||||
|
||||
private:
|
||||
/////////////////////
|
||||
// locales
|
||||
/////////////////////
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// scan functions
|
||||
/////////////////////
|
||||
@@ -437,8 +452,16 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
if (0xD800 <= codepoint1 && codepoint1 <= 0xDBFF)
|
||||
{
|
||||
// expect next \uxxxx entry
|
||||
if (JSON_HEDLEY_LIKELY(get() == '\\' && get() == 'u'))
|
||||
if (JSON_HEDLEY_LIKELY(get() == '\\'))
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(get() != 'u'))
|
||||
{
|
||||
// current is the character escaped by the backslash
|
||||
error_message = "invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF";
|
||||
string_error_resume = resume_kind::escaped_character;
|
||||
return token_type::parse_error;
|
||||
}
|
||||
|
||||
const int codepoint2 = get_codepoint();
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(codepoint2 == -1))
|
||||
@@ -463,7 +486,11 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
else
|
||||
{
|
||||
// the second escape was read completely and is a
|
||||
// code point of its own
|
||||
error_message = "invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF";
|
||||
string_error_resume = resume_kind::after_escape;
|
||||
string_error_codepoint = codepoint2;
|
||||
return token_type::parse_error;
|
||||
}
|
||||
}
|
||||
@@ -477,7 +504,9 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(0xDC00 <= codepoint1 && codepoint1 <= 0xDFFF))
|
||||
{
|
||||
// the escape was read completely
|
||||
error_message = "invalid string: surrogate U+DC00..U+DFFF must follow U+D800..U+DBFF";
|
||||
string_error_resume = resume_kind::after_escape;
|
||||
return token_type::parse_error;
|
||||
}
|
||||
}
|
||||
@@ -1022,6 +1051,24 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan a number literal
|
||||
|
||||
@@ -1061,7 +1108,7 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
@note The scanner is independent of the current locale: token_buffer
|
||||
always holds `.`. Only the std::strtod fallback of convert_number()
|
||||
depends on the locale, and it looks up the decimal point right
|
||||
before converting (see detail::convert_float_locale_aware()).
|
||||
before converting (see convert_float_locale_aware()).
|
||||
*/
|
||||
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
|
||||
{
|
||||
@@ -1392,6 +1439,59 @@ scan_number_done:
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for this token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] mantissa_end offset just past the last mantissa byte in
|
||||
token_buffer
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
bool mantissa_fits_clinger(std::size_t mantissa_end) const
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. The
|
||||
// fraction is located through decimal_point_position rather than by
|
||||
// searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the number text in token_buffer to its value and token type
|
||||
|
||||
@@ -1405,7 +1505,7 @@ scan_number_done:
|
||||
token_buffer (the index of 'e'/'E', or
|
||||
token_buffer.size() when there is no exponent);
|
||||
used to skip Clinger's fast path when it cannot
|
||||
possibly succeed - see detail::mantissa_fits_clinger()
|
||||
possibly succeed - see mantissa_fits_clinger()
|
||||
*/
|
||||
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||
{
|
||||
@@ -1478,15 +1578,77 @@ scan_number_done:
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod/strtold.
|
||||
if (convert_float_fast(num_begin, num_end, decimal_point_position, mantissa_end, value_float))
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
if (mantissa_fits_clinger(mantissa_end)
|
||||
&& parse_float_fast(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
convert_float_locale_aware(token_buffer, decimal_point_position, value_float);
|
||||
convert_float_locale_aware();
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the float in token_buffer with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
*/
|
||||
void convert_float_locale_aware()
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// get_string() hands the token to the SAX interface with '.'
|
||||
token_buffer[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief contiguous fast path for scanning a number
|
||||
|
||||
@@ -2130,6 +2292,573 @@ scan_number_done:
|
||||
}
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// error recovery
|
||||
/////////////////////
|
||||
|
||||
/*!
|
||||
@brief make the best of the token that scan() rejected
|
||||
|
||||
Called by the parser after scan() returned token_type::parse_error and the
|
||||
SAX parser asked to recover from the error (see #3989). Keeps what can be
|
||||
read of the token and skips the rest:
|
||||
|
||||
- A string keeps its characters. An unknown escape stands for the escaped
|
||||
character itself (as in JavaScript), an invalid `\u` escape and ill-formed
|
||||
UTF-8 become U+FFFD, and a control character is kept. A line break or the
|
||||
end of the input ends a string that lacks its closing quote.
|
||||
- A number keeps its longest valid prefix, e.g. `1` for `1.` or `1e+`.
|
||||
- A block comment that is not closed runs to the end of the input.
|
||||
- Anything else is skipped.
|
||||
|
||||
The rest of an invalid token is skipped up to the next delimiter
|
||||
(whitespace, a structural character, or a quote). A delimiter that the
|
||||
invalid token consumed is returned to the input, so that the next scan()
|
||||
reads it.
|
||||
|
||||
@return token_type::value_string or a number token type if a string or a
|
||||
number could be read, token_type::end_of_input for a block comment
|
||||
that is not closed, token_type::uninitialized otherwise
|
||||
*/
|
||||
token_type recover_token()
|
||||
{
|
||||
const resume_kind resume = string_error_resume;
|
||||
const int codepoint = string_error_codepoint;
|
||||
string_error_resume = resume_kind::character;
|
||||
string_error_codepoint = -1;
|
||||
|
||||
if (error_message_starts_with("invalid string"))
|
||||
{
|
||||
return recover_string(resume, codepoint);
|
||||
}
|
||||
|
||||
if (error_message_starts_with("invalid number"))
|
||||
{
|
||||
return recover_number();
|
||||
}
|
||||
|
||||
if (error_message_starts_with("invalid comment; missing"))
|
||||
{
|
||||
// the comment runs to the end of the input
|
||||
return token_type::end_of_input;
|
||||
}
|
||||
|
||||
skip_to_delimiter();
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief return the token that scan() read last to the input, so that the
|
||||
next scan() reads it again
|
||||
|
||||
Called by the parser when recovering from an error. The token must be a
|
||||
single character (',', ':', '[', ']', '{', or '}') or the end of the
|
||||
input, and scan() must have read it last.
|
||||
*/
|
||||
void unget_token()
|
||||
{
|
||||
JSON_ASSERT(!next_unget);
|
||||
unget();
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief let the token string for the next error begin at the current character
|
||||
|
||||
The token string of an error reaches back to the beginning of the last
|
||||
string or number. After an error, the parser calls this function so that
|
||||
the next error does not report (and, with many errors, copy) everything
|
||||
read since then.
|
||||
*/
|
||||
void restart_token_string()
|
||||
{
|
||||
restart_token_string_impl(std::integral_constant<bool, lazy_token_string> {});
|
||||
}
|
||||
|
||||
private:
|
||||
/// how recover_string() continues after the error scan_string() reported
|
||||
enum class resume_kind : std::uint8_t
|
||||
{
|
||||
/// current is the next character of the string (or the end of input)
|
||||
character,
|
||||
/// current is the character escaped by the preceding backslash
|
||||
escaped_character,
|
||||
/// current is the last character of a complete escape
|
||||
after_escape
|
||||
};
|
||||
|
||||
/// whether error_message begins with @a prefix
|
||||
bool error_message_starts_with(const char* prefix) const noexcept
|
||||
{
|
||||
const char* message = error_message;
|
||||
while (*prefix != '\0')
|
||||
{
|
||||
if (*message++ != *prefix++)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// whether current ends an invalid token (see recover_token())
|
||||
bool current_is_delimiter() const noexcept
|
||||
{
|
||||
switch (current)
|
||||
{
|
||||
case ' ':
|
||||
case '\t':
|
||||
case '\n':
|
||||
case '\r':
|
||||
case '[':
|
||||
case ']':
|
||||
case '{':
|
||||
case '}':
|
||||
case ',':
|
||||
case ':':
|
||||
case '\"':
|
||||
#if !JSON_STRICT_NUL_HANDLING
|
||||
case '\0':
|
||||
#endif
|
||||
case char_traits<char_type>::eof():
|
||||
return true;
|
||||
|
||||
case '/':
|
||||
return ignore_comments;
|
||||
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/// skip the rest of an invalid token and return its delimiter to the input
|
||||
void skip_to_delimiter()
|
||||
{
|
||||
while (!current_is_delimiter())
|
||||
{
|
||||
get();
|
||||
}
|
||||
|
||||
if (current != char_traits<char_type>::eof())
|
||||
{
|
||||
unget();
|
||||
}
|
||||
}
|
||||
|
||||
/// append U+FFFD REPLACEMENT CHARACTER to token_buffer
|
||||
void add_replacement_character()
|
||||
{
|
||||
add(0xEF);
|
||||
add(0xBF);
|
||||
add(0xBD);
|
||||
}
|
||||
|
||||
/// append the UTF-8 encoding of @a codepoint (not a surrogate) to token_buffer
|
||||
void add_codepoint(const int codepoint)
|
||||
{
|
||||
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
|
||||
const auto cp = static_cast<unsigned int>(codepoint);
|
||||
if (cp < 0x80)
|
||||
{
|
||||
add(static_cast<char_int_type>(cp));
|
||||
}
|
||||
else if (cp <= 0x7FF)
|
||||
{
|
||||
add(static_cast<char_int_type>(0xC0u | (cp >> 6u)));
|
||||
add(static_cast<char_int_type>(0x80u | (cp & 0x3Fu)));
|
||||
}
|
||||
else if (cp <= 0xFFFF)
|
||||
{
|
||||
add(static_cast<char_int_type>(0xE0u | (cp >> 12u)));
|
||||
add(static_cast<char_int_type>(0x80u | ((cp >> 6u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | (cp & 0x3Fu)));
|
||||
}
|
||||
else
|
||||
{
|
||||
add(static_cast<char_int_type>(0xF0u | (cp >> 18u)));
|
||||
add(static_cast<char_int_type>(0x80u | ((cp >> 12u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | ((cp >> 6u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | (cp & 0x3Fu)));
|
||||
}
|
||||
}
|
||||
|
||||
/// append a code point read from a `\u` escape; a surrogate becomes U+FFFD
|
||||
void add_escaped_codepoint(const int codepoint)
|
||||
{
|
||||
if (0xD800 <= codepoint && codepoint <= 0xDFFF)
|
||||
{
|
||||
add_replacement_character();
|
||||
}
|
||||
else
|
||||
{
|
||||
add_codepoint(codepoint);
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief remove an incomplete UTF-8 sequence from the end of token_buffer
|
||||
|
||||
next_byte_in_range() adds the bytes of a sequence as it checks them, so
|
||||
when it rejects a byte, the beginning of the sequence is already in
|
||||
token_buffer, which otherwise holds only complete sequences.
|
||||
|
||||
@return whether an incomplete sequence was removed
|
||||
*/
|
||||
bool remove_incomplete_utf8_sequence()
|
||||
{
|
||||
std::size_t lead = token_buffer.size();
|
||||
std::size_t continuation_bytes = 0;
|
||||
while (lead > 0 && continuation_bytes < 3
|
||||
&& (static_cast<unsigned char>(token_buffer[lead - 1]) & 0xC0u) == 0x80u)
|
||||
{
|
||||
--lead;
|
||||
++continuation_bytes;
|
||||
}
|
||||
if (lead == 0)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
const auto lead_byte = static_cast<unsigned char>(token_buffer[lead - 1]);
|
||||
std::size_t expected = 0;
|
||||
if (lead_byte >= 0xF0)
|
||||
{
|
||||
expected = 3;
|
||||
}
|
||||
else if (lead_byte >= 0xE0)
|
||||
{
|
||||
expected = 2;
|
||||
}
|
||||
else if (lead_byte >= 0xC0)
|
||||
{
|
||||
expected = 1;
|
||||
}
|
||||
if (continuation_bytes >= expected)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
token_buffer.resize(lead - 1);
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief read the UTF-8 sequence that begins with current, which is not ASCII
|
||||
@return whether the next character must be read; false if current still
|
||||
needs to be handled, because it does not belong to the sequence
|
||||
*/
|
||||
bool recover_utf8_sequence()
|
||||
{
|
||||
// the number of continuation bytes and the range of the first one;
|
||||
// see the ranges in scan_string()
|
||||
std::size_t count = 0;
|
||||
char_int_type low = 0x80;
|
||||
char_int_type high = 0xBF;
|
||||
if (current >= 0xC2 && current <= 0xDF)
|
||||
{
|
||||
count = 1;
|
||||
}
|
||||
else if (current >= 0xE0 && current <= 0xEF)
|
||||
{
|
||||
count = 2;
|
||||
low = (current == 0xE0) ? 0xA0 : 0x80;
|
||||
high = (current == 0xED) ? 0x9F : 0xBF;
|
||||
}
|
||||
else if (current >= 0xF0 && current <= 0xF4)
|
||||
{
|
||||
count = 3;
|
||||
low = (current == 0xF0) ? 0x90 : 0x80;
|
||||
high = (current == 0xF4) ? 0x8F : 0xBF;
|
||||
}
|
||||
else
|
||||
{
|
||||
// an ill-formed byte
|
||||
add_replacement_character();
|
||||
return true;
|
||||
}
|
||||
|
||||
const std::size_t start = token_buffer.size();
|
||||
add(current);
|
||||
for (std::size_t i = 0; i < count; ++i)
|
||||
{
|
||||
get();
|
||||
if (current < low || current > high)
|
||||
{
|
||||
token_buffer.resize(start);
|
||||
add_replacement_character();
|
||||
return false;
|
||||
}
|
||||
add(current);
|
||||
low = 0x80;
|
||||
high = 0xBF;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief read the low surrogate that must follow the high surrogate @a high
|
||||
@return whether the next character must be read; false if current still
|
||||
needs to be handled
|
||||
*/
|
||||
bool recover_low_surrogate(int high)
|
||||
{
|
||||
while (true)
|
||||
{
|
||||
if (get() != '\\')
|
||||
{
|
||||
add_replacement_character();
|
||||
return false;
|
||||
}
|
||||
if (get() != 'u')
|
||||
{
|
||||
add_replacement_character();
|
||||
// not 'u', so this does not come back here
|
||||
return recover_escape();
|
||||
}
|
||||
|
||||
const int low = get_codepoint();
|
||||
if (low == -1)
|
||||
{
|
||||
add_replacement_character();
|
||||
return false;
|
||||
}
|
||||
if (0xDC00 <= low && low <= 0xDFFF)
|
||||
{
|
||||
add_codepoint(static_cast<int>((static_cast<unsigned int>(high) << 10u)
|
||||
+ static_cast<unsigned int>(low) - 0x35FDC00u));
|
||||
return true;
|
||||
}
|
||||
|
||||
// high has no low surrogate
|
||||
add_replacement_character();
|
||||
if (low < 0xD800 || low > 0xDBFF)
|
||||
{
|
||||
add_codepoint(low);
|
||||
return true;
|
||||
}
|
||||
// another high surrogate
|
||||
high = low;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief read the escape whose backslash was read; current is the escaped character
|
||||
@return whether the next character must be read; false if current still
|
||||
needs to be handled
|
||||
*/
|
||||
bool recover_escape()
|
||||
{
|
||||
switch (current)
|
||||
{
|
||||
case '\"':
|
||||
add('\"');
|
||||
return true;
|
||||
case '\\':
|
||||
add('\\');
|
||||
return true;
|
||||
case '/':
|
||||
add('/');
|
||||
return true;
|
||||
case 'b':
|
||||
add('\b');
|
||||
return true;
|
||||
case 'f':
|
||||
add('\f');
|
||||
return true;
|
||||
case 'n':
|
||||
add('\n');
|
||||
return true;
|
||||
case 'r':
|
||||
add('\r');
|
||||
return true;
|
||||
case 't':
|
||||
add('\t');
|
||||
return true;
|
||||
|
||||
case 'u':
|
||||
{
|
||||
const int codepoint = get_codepoint();
|
||||
if (codepoint == -1)
|
||||
{
|
||||
add_replacement_character();
|
||||
return false;
|
||||
}
|
||||
if (0xD800 <= codepoint && codepoint <= 0xDBFF)
|
||||
{
|
||||
return recover_low_surrogate(codepoint);
|
||||
}
|
||||
add_escaped_codepoint(codepoint);
|
||||
return true;
|
||||
}
|
||||
|
||||
// an unknown escape stands for the escaped character
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief read the rest of a string after scan_string() rejected it
|
||||
|
||||
token_buffer holds what scan_string() read before the error. See
|
||||
recover_token() for how errors are repaired.
|
||||
|
||||
@param[in] resume how to continue, see resume_kind
|
||||
@param[in] codepoint for a high surrogate followed by an escape of another
|
||||
code point: that code point; -1 otherwise
|
||||
*/
|
||||
token_type recover_string(const resume_kind resume, const int codepoint)
|
||||
{
|
||||
// whether the next character must be read before it can be handled
|
||||
bool fetch = false;
|
||||
|
||||
if (error_message_starts_with("invalid string: surrogate")
|
||||
|| error_message_starts_with("invalid string: '\\u'")
|
||||
|| (error_message_starts_with("invalid string: ill-formed UTF-8")
|
||||
&& remove_incomplete_utf8_sequence()))
|
||||
{
|
||||
add_replacement_character();
|
||||
}
|
||||
|
||||
switch (resume)
|
||||
{
|
||||
case resume_kind::escaped_character:
|
||||
fetch = recover_escape();
|
||||
break;
|
||||
case resume_kind::after_escape:
|
||||
if (0xD800 <= codepoint && codepoint <= 0xDBFF)
|
||||
{
|
||||
fetch = recover_low_surrogate(codepoint);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (codepoint != -1)
|
||||
{
|
||||
add_escaped_codepoint(codepoint);
|
||||
}
|
||||
fetch = true;
|
||||
}
|
||||
break;
|
||||
case resume_kind::character:
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
while (true)
|
||||
{
|
||||
if (fetch)
|
||||
{
|
||||
get();
|
||||
}
|
||||
fetch = true;
|
||||
|
||||
switch (current)
|
||||
{
|
||||
case '\"':
|
||||
// a line break or the end of the input ends a string that
|
||||
// lacks its closing quote
|
||||
case '\n':
|
||||
case '\r':
|
||||
case char_traits<char_type>::eof():
|
||||
return token_type::value_string;
|
||||
|
||||
#if !JSON_STRICT_NUL_HANDLING
|
||||
case '\0':
|
||||
// the end of the input, see scan()
|
||||
unget();
|
||||
return token_type::value_string;
|
||||
#endif
|
||||
|
||||
case '\\':
|
||||
get();
|
||||
fetch = recover_escape();
|
||||
break;
|
||||
|
||||
default:
|
||||
if (current < 0x80)
|
||||
{
|
||||
// including control characters
|
||||
add(current);
|
||||
}
|
||||
else
|
||||
{
|
||||
fetch = recover_utf8_sequence();
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief keep the longest valid prefix of a number that scan_number() rejected
|
||||
|
||||
token_buffer holds the characters scan_number() accepted before the error,
|
||||
so the prefix ends at its last digit.
|
||||
*/
|
||||
token_type recover_number()
|
||||
{
|
||||
// only size(), operator[], and resize() are used, which every string
|
||||
// type the library supports provides
|
||||
std::size_t length = token_buffer.size();
|
||||
while (length != 0 && (token_buffer[length - 1] < '0' || token_buffer[length - 1] > '9'))
|
||||
{
|
||||
--length;
|
||||
}
|
||||
token_buffer.resize(length);
|
||||
|
||||
if (length == 0)
|
||||
{
|
||||
skip_to_delimiter();
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
|
||||
if (decimal_point_position >= length)
|
||||
{
|
||||
decimal_point_position = std::string::npos;
|
||||
}
|
||||
|
||||
std::size_t exponent = std::string::npos;
|
||||
for (std::size_t i = 0; i < length; ++i)
|
||||
{
|
||||
if (token_buffer[i] == 'e' || token_buffer[i] == 'E')
|
||||
{
|
||||
exponent = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
const std::size_t mantissa_end = (exponent == std::string::npos) ? length : exponent;
|
||||
token_type number_type = token_type::value_unsigned;
|
||||
if (decimal_point_position != std::string::npos || exponent != std::string::npos)
|
||||
{
|
||||
number_type = token_type::value_float;
|
||||
}
|
||||
else if (token_buffer[0] == '-')
|
||||
{
|
||||
number_type = token_type::value_integer;
|
||||
}
|
||||
|
||||
const token_type result = convert_number(number_type, mantissa_end);
|
||||
skip_to_delimiter();
|
||||
return result;
|
||||
}
|
||||
|
||||
/// seekable adapter: the token string begins at current, which was consumed
|
||||
void restart_token_string_impl(std::true_type /*lazy*/) noexcept
|
||||
{
|
||||
const std::size_t consumed = ia.get_consumed_count();
|
||||
token_string_start = (consumed > 0 && current != char_traits<char_type>::eof()) ? consumed - 1 : consumed;
|
||||
}
|
||||
|
||||
/// streaming adapter: the token string begins at current; a character
|
||||
/// that was put back is copied again when it is read again
|
||||
void restart_token_string_impl(std::false_type /*lazy*/)
|
||||
{
|
||||
token_string.clear();
|
||||
if (!next_unget && current != char_traits<char_type>::eof())
|
||||
{
|
||||
token_string.push_back(char_traits<char_type>::to_char_type(current));
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
/// input adapter
|
||||
InputAdapterType ia;
|
||||
@@ -2170,6 +2899,13 @@ scan_number_done:
|
||||
/// a description of occurred lexer errors
|
||||
const char* error_message = "";
|
||||
|
||||
/// how recover_token() continues a string that scan_string() rejected;
|
||||
/// set only on the error paths that need more than error_message
|
||||
resume_kind string_error_resume = resume_kind::character;
|
||||
/// the code point of the second escape when a high surrogate is followed
|
||||
/// by an escape that is not a low surrogate; -1 otherwise
|
||||
int string_error_codepoint = -1;
|
||||
|
||||
// number values
|
||||
number_integer_t value_integer = 0;
|
||||
number_unsigned_t value_unsigned = 0;
|
||||
|
||||
@@ -3,7 +3,6 @@
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2021 The fast_float authors <https://github.com/fastfloat/fast_float>
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
@@ -11,16 +10,10 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <cstdlib> // strtof, strtod, strtold
|
||||
#include <cstring> // memcpy
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
|
||||
#include <nlohmann/detail/bit_ops.hpp>
|
||||
#include <nlohmann/detail/input/pow5_table.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
// std::from_chars lives in <charconv>, but being in C++17 mode does not
|
||||
@@ -36,9 +29,8 @@
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod where possible. They are free functions
|
||||
// so the lexer stays focused on scanning (see lexer::convert_number()) and so
|
||||
// that other parsers of JSON text can convert tokens exactly like it does.
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -301,415 +293,5 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out)
|
||||
#endif
|
||||
}
|
||||
|
||||
/// whether the eight bytes of @a v (see read_eight_bytes()) are ASCII digits
|
||||
/// (after fast_float's is_made_of_eight_digits_fast)
|
||||
inline bool is_eight_digits(std::uint64_t v) noexcept
|
||||
{
|
||||
return ((v & 0xF0F0F0F0F0F0F0F0u) | (((v + 0x0606060606060606u) & 0xF0F0F0F0F0F0F0F0u) >> 4u)) == 0x3333333333333333u;
|
||||
}
|
||||
|
||||
/// the value of the eight ASCII digits in @a v (see read_eight_bytes()), three
|
||||
/// multiplications instead of eight (after simdjson and fast_float)
|
||||
inline std::uint32_t parse_eight_digits(std::uint64_t v) noexcept
|
||||
{
|
||||
v = ((v & 0x0F0F0F0F0F0F0F0Fu) * 2561u) >> 8u;
|
||||
v = ((v & 0x00FF00FF00FF00FFu) * 6553601u) >> 16u;
|
||||
return static_cast<std::uint32_t>(((v & 0x0000FFFF0000FFFFu) * 42949672960001u) >> 32u);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the double nearest to w * 10^q (Eisel-Lemire)
|
||||
|
||||
The algorithm of Daniel Lemire, "Number Parsing at a Gigabyte per Second"
|
||||
(Software: Practice and Experience, 2021), after fast_float's compute_float
|
||||
(used under the MIT license). With a 128-bit approximation of 5^q, the product
|
||||
is always sufficient to round correctly for w with at most 19 digits (Noble
|
||||
Mushtak and Daniel Lemire, "Fast number parsing without fallback", Software:
|
||||
Practice and Experience, 2023). Only integer arithmetic is used, so the result
|
||||
does not depend on the floating-point environment.
|
||||
|
||||
@param[in] q decimal exponent
|
||||
@param[in] w significand, w != 0
|
||||
@return the IEEE-754 bits of the positive result (0 for underflow, infinity
|
||||
for overflow)
|
||||
*/
|
||||
inline std::uint64_t eisel_lemire(std::int64_t q, std::uint64_t w) noexcept
|
||||
{
|
||||
constexpr int mantissa_bits = 52;
|
||||
constexpr std::uint64_t infinity = std::uint64_t{0x7FF} << mantissa_bits;
|
||||
if (q < pow5_128_smallest_power)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
if (q > pow5_128_largest_power)
|
||||
{
|
||||
return infinity;
|
||||
}
|
||||
|
||||
const int lz = count_leading_zeros(w);
|
||||
w <<= static_cast<unsigned>(lz);
|
||||
const auto index = static_cast<std::size_t>(2 * (q - pow5_128_smallest_power));
|
||||
uint128_parts product = full_multiplication(w, pow5_128()[index]);
|
||||
constexpr std::uint64_t precision_mask = 0xFFFFFFFFFFFFFFFFu >> (mantissa_bits + 3);
|
||||
if ((product.high & precision_mask) == precision_mask)
|
||||
{
|
||||
// the lower bits may carry into the result: use the next 64 bits of 5^q
|
||||
const uint128_parts second = full_multiplication(w, pow5_128()[index + 1]);
|
||||
product.low += second.high;
|
||||
if (second.high > product.low)
|
||||
{
|
||||
++product.high;
|
||||
}
|
||||
}
|
||||
|
||||
const auto upperbit = static_cast<int>(product.high >> 63u);
|
||||
const int shift = upperbit + 64 - mantissa_bits - 3;
|
||||
std::uint64_t mantissa = product.high >> static_cast<unsigned>(shift);
|
||||
// floor(log2(10^q)) + 63 + 1023, with log2(10) ~ 217706 / 2^16
|
||||
std::int64_t power2 = (((152170 + 65536) * q) >> 16) + 63 + upperbit - lz + 1023;
|
||||
|
||||
if (power2 <= 0) // subnormal
|
||||
{
|
||||
if (-power2 + 1 >= 64)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
mantissa >>= static_cast<unsigned>(-power2 + 1);
|
||||
mantissa += (mantissa & 1u);
|
||||
mantissa >>= 1u;
|
||||
// rounding up may produce the smallest normal number
|
||||
power2 = (mantissa < (std::uint64_t{1} << mantissa_bits)) ? 0 : 1;
|
||||
return mantissa | (static_cast<std::uint64_t>(power2) << mantissa_bits);
|
||||
}
|
||||
|
||||
// a value exactly between two doubles rounds to even; this can only
|
||||
// happen for small |q|, where 5^q is exact
|
||||
if (product.low <= 1 && q >= -4 && q <= 23 && (mantissa & 3u) == 1
|
||||
&& (mantissa << static_cast<unsigned>(shift)) == product.high)
|
||||
{
|
||||
mantissa &= ~std::uint64_t{1};
|
||||
}
|
||||
mantissa += (mantissa & 1u);
|
||||
mantissa >>= 1u;
|
||||
if (mantissa >= (std::uint64_t{2} << mantissa_bits))
|
||||
{
|
||||
mantissa = std::uint64_t{1} << mantissa_bits;
|
||||
++power2;
|
||||
}
|
||||
mantissa &= ~(std::uint64_t{1} << mantissa_bits);
|
||||
if (power2 >= 0x7FF)
|
||||
{
|
||||
return infinity;
|
||||
}
|
||||
return mantissa | (static_cast<std::uint64_t>(power2) << mantissa_bits);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse a validated float token with the Eisel-Lemire algorithm
|
||||
|
||||
The significand is accumulated eight digits at a time where possible. A token
|
||||
with more than 19 significant digits is truncated to w; the value then lies
|
||||
in [w, w + 1) * 10^q, and it is only returned if both ends round to the same
|
||||
double, which covers all but a few such tokens.
|
||||
|
||||
@param[in] first pointer to the first character of the token
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] out the correctly rounded value on success (±infinity if it
|
||||
overflows, like strtod)
|
||||
@return true on success; false if strtod must decide
|
||||
*/
|
||||
inline bool parse_float_eisel_lemire(const char* first, const char* last, double& out) noexcept
|
||||
{
|
||||
const char* p = first;
|
||||
const bool negative = (p != last && *p == '-');
|
||||
if (negative)
|
||||
{
|
||||
++p;
|
||||
}
|
||||
|
||||
std::uint64_t w = 0;
|
||||
int digits = 0; // significant digits in w
|
||||
std::int64_t exponent = 0;
|
||||
bool truncated = false;
|
||||
bool in_fraction = false;
|
||||
for (;;)
|
||||
{
|
||||
// eight digits at a time, as long as they fit into w
|
||||
while (w != 0 && digits <= 19 - 8 && last - p >= 8)
|
||||
{
|
||||
const std::uint64_t v = read_eight_bytes(p);
|
||||
if (!is_eight_digits(v))
|
||||
{
|
||||
break;
|
||||
}
|
||||
w = (w * 100000000u) + parse_eight_digits(v);
|
||||
digits += 8;
|
||||
exponent -= in_fraction ? 8 : 0;
|
||||
p += 8;
|
||||
}
|
||||
if (p == last)
|
||||
{
|
||||
break;
|
||||
}
|
||||
const char c = *p;
|
||||
if (c >= '0' && c <= '9')
|
||||
{
|
||||
if (w == 0 && c == '0')
|
||||
{
|
||||
// leading zeros are not significant, but scale a fraction
|
||||
exponent -= in_fraction ? 1 : 0;
|
||||
}
|
||||
else if (digits < 19)
|
||||
{
|
||||
w = (w * 10u) + static_cast<std::uint64_t>(c - '0');
|
||||
++digits;
|
||||
exponent -= in_fraction ? 1 : 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
// dropped: the value lies between w and w + 1 (in units of
|
||||
// the last kept digit) unless all dropped digits are zero
|
||||
truncated = truncated || c != '0';
|
||||
exponent += in_fraction ? 0 : 1;
|
||||
}
|
||||
++p;
|
||||
}
|
||||
else if (c == '.')
|
||||
{
|
||||
in_fraction = true;
|
||||
++p;
|
||||
}
|
||||
else
|
||||
{
|
||||
break; // 'e' or 'E'
|
||||
}
|
||||
}
|
||||
|
||||
if (p != last)
|
||||
{
|
||||
++p; // 'e' or 'E'
|
||||
bool exp_negative = false;
|
||||
if (p != last && (*p == '-' || *p == '+'))
|
||||
{
|
||||
exp_negative = (*p == '-');
|
||||
++p;
|
||||
}
|
||||
std::int64_t exp_value = 0;
|
||||
for (; p != last; ++p)
|
||||
{
|
||||
// saturate: any exponent beyond this under- or overflows anyway
|
||||
if (exp_value < 100000)
|
||||
{
|
||||
exp_value = (exp_value * 10) + (*p - '0');
|
||||
}
|
||||
}
|
||||
exponent += exp_negative ? -exp_value : exp_value;
|
||||
}
|
||||
|
||||
std::uint64_t bits = 0;
|
||||
if (w != 0)
|
||||
{
|
||||
bits = eisel_lemire(exponent, w);
|
||||
if (truncated && (w + 1 == 0 || eisel_lemire(exponent, w + 1) != bits))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
bits |= negative ? (std::uint64_t{1} << 63u) : 0u;
|
||||
static_assert(sizeof(double) == sizeof(std::uint64_t), "double must have 64 bits");
|
||||
std::memcpy(&out, &bits, sizeof(out));
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Eisel-Lemire is only implemented for `double`
|
||||
template<typename FloatType>
|
||||
bool parse_float_eisel_lemire(const char* /*first*/, const char* /*last*/, FloatType& /*out*/) noexcept
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for a float token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] token the validated number token ('.' as decimal point)
|
||||
@param[in] decimal_point_position index of the '.' in @a token, or
|
||||
std::string::npos if there is none
|
||||
@param[in] mantissa_end offset just past the last mantissa byte
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_position, std::size_t mantissa_end) noexcept
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (token[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. The
|
||||
// fraction is located through decimal_point_position rather than by
|
||||
// searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert a validated float token without the C library, if possible
|
||||
|
||||
Tries std::from_chars (when available), Clinger's exact fast path (double
|
||||
only, skipped when it cannot succeed), and the Eisel-Lemire algorithm (double
|
||||
only).
|
||||
|
||||
@param[in] first pointer to the first character of the token
|
||||
@param[in] last pointer past the last character
|
||||
@param[in] decimal_point_position index of the '.' in the token, or
|
||||
std::string::npos if there is none
|
||||
@param[in] mantissa_end offset just past the last mantissa byte (the
|
||||
index of 'e'/'E', or the token length)
|
||||
@param[out] value the converted value on success
|
||||
@return true if the value was converted; false if convert_float_locale_aware()
|
||||
must convert it
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool convert_float_fast(const char* first, const char* last, std::size_t decimal_point_position,
|
||||
std::size_t mantissa_end, FloatType& value) noexcept
|
||||
{
|
||||
if (parse_float_from_chars(first, last, value))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
if (mantissa_fits_clinger(first, decimal_point_position, mantissa_end)
|
||||
&& parse_float_fast(first, last, value))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
return parse_float_eisel_lemire(first, last, value);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
inline void strtof_by_type(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/// return the decimal point of the current locale
|
||||
inline char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert a validated float token with strtof/strtod/strtold
|
||||
|
||||
These functions expect the decimal point of the *current* locale, so it is
|
||||
looked up right before the conversion instead of once when the lexer is
|
||||
constructed: a locale change in between (by a parser callback, a SAX
|
||||
handler, or another thread) must not truncate the value (#5198). The
|
||||
token has been validated before, so if the conversion stops early and the
|
||||
decimal point changed in the meantime, the locale changed between the
|
||||
lookup and the call, and the conversion is repeated with the new decimal
|
||||
point. If the decimal point did not change, a retry cannot succeed: the
|
||||
locale's decimal point is not a single character (e.g., the two-byte
|
||||
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
|
||||
The value strtod parsed up to that point is kept, as before this change.
|
||||
|
||||
Note that changing the locale in another thread *while* strtod runs is
|
||||
undefined behavior of the C library, which this function cannot prevent.
|
||||
|
||||
@param[in,out] token the token with '.' as decimal point; its
|
||||
decimal point is replaced during the
|
||||
conversion and restored afterwards
|
||||
(data() must be NUL-terminated)
|
||||
@param[in] decimal_point_position index of the '.' in @a token, or
|
||||
std::string::npos if there is none
|
||||
@param[out] value the converted value
|
||||
*/
|
||||
template<typename StringType, typename FloatType>
|
||||
void convert_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& value)
|
||||
{
|
||||
const bool has_dot = decimal_point_position != std::string::npos;
|
||||
char decimal_point = get_decimal_point();
|
||||
for (;;)
|
||||
{
|
||||
const bool substitute = has_dot && decimal_point != '.';
|
||||
if (substitute)
|
||||
{
|
||||
token[decimal_point_position] = static_cast<typename StringType::value_type>(decimal_point);
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof_by_type(value, token.data(), &endptr);
|
||||
|
||||
if (substitute)
|
||||
{
|
||||
// the caller hands the token on (e.g. to the SAX interface) with '.'
|
||||
token[decimal_point_position] = '.';
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(endptr == token.data() + token.size()))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// retry only if the locale changed; otherwise, this would loop forever
|
||||
const char current_decimal_point = get_decimal_point();
|
||||
if (current_decimal_point == decimal_point)
|
||||
{
|
||||
return;
|
||||
}
|
||||
decimal_point = current_decimal_point;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -98,7 +98,7 @@ class parser
|
||||
if (callback)
|
||||
{
|
||||
json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer);
|
||||
sax_parse_internal(&sdp);
|
||||
sax_parse_internal<false>(&sdp);
|
||||
|
||||
if (strict)
|
||||
{
|
||||
@@ -135,7 +135,7 @@ class parser
|
||||
else
|
||||
{
|
||||
json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer);
|
||||
sax_parse_internal(&sdp);
|
||||
sax_parse_internal<false>(&sdp);
|
||||
|
||||
if (strict)
|
||||
{
|
||||
@@ -173,26 +173,59 @@ class parser
|
||||
bool accept(const bool strict = true)
|
||||
{
|
||||
json_sax_acceptor<BasicJsonType> sax_acceptor;
|
||||
return sax_parse(&sax_acceptor, strict);
|
||||
return sax_parse_impl<false>(&sax_acceptor, strict);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief public SAX interface
|
||||
|
||||
If the SAX parser's parse_error() returns true, the parser recovers from
|
||||
the error: it repairs the input and continues (see #3989).
|
||||
|
||||
@param[in] sax the SAX parser
|
||||
@param[in] strict whether to expect the last token to be EOF
|
||||
@return whether the input was parsed without errors and no SAX event
|
||||
returned false
|
||||
*/
|
||||
template<typename SAX>
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
bool sax_parse(SAX* sax, const bool strict = true)
|
||||
{
|
||||
return sax_parse_impl<true>(sax, strict);
|
||||
}
|
||||
|
||||
private:
|
||||
/// what sax_parse_internal() does after an object key was expected
|
||||
enum class next_step : std::uint8_t
|
||||
{
|
||||
/// stop parsing
|
||||
stop,
|
||||
/// parse a value that begins with last_token
|
||||
parse_value,
|
||||
/// evaluate the state of the innermost container, which reads
|
||||
/// last_token again
|
||||
evaluate_state
|
||||
};
|
||||
|
||||
template<bool AllowRecovery, typename SAX>
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
bool sax_parse_impl(SAX* sax, const bool strict)
|
||||
{
|
||||
(void)detail::is_sax_static_asserts<SAX, BasicJsonType> {};
|
||||
const bool result = sax_parse_internal(sax);
|
||||
const bool result = sax_parse_internal<AllowRecovery>(sax);
|
||||
|
||||
if (result)
|
||||
{
|
||||
if (strict)
|
||||
{
|
||||
// strict mode: next byte must be EOF
|
||||
if (get_token() != token_type::end_of_input)
|
||||
// strict mode: next byte must be EOF; after recovering from an
|
||||
// error, the end of the input may already have been read
|
||||
if (last_token != token_type::end_of_input && get_token() != token_type::end_of_input)
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
|
||||
// the value is complete, so there is nothing to recover
|
||||
static_cast<void>(report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr),
|
||||
std::integral_constant<bool, AllowRecovery> {}));
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -203,14 +236,23 @@ class parser
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
return result && !error_reported;
|
||||
}
|
||||
|
||||
private:
|
||||
template<typename SAX>
|
||||
/*!
|
||||
@brief parse a JSON value and pass it to a SAX parser
|
||||
|
||||
@tparam AllowRecovery whether to recover from an error if the SAX parser's
|
||||
parse_error() returns true; false for the SAX parsers
|
||||
of parse() and accept(), which never do, so that no
|
||||
code for recovering is generated for them
|
||||
*/
|
||||
template<bool AllowRecovery, typename SAX>
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
bool sax_parse_internal(SAX* sax)
|
||||
{
|
||||
const std::integral_constant<bool, AllowRecovery> allow_recovery{};
|
||||
|
||||
// stack to remember the hierarchy of structured values we are parsing
|
||||
// true = array; false = object
|
||||
std::vector<bool> states;
|
||||
@@ -241,12 +283,18 @@ class parser
|
||||
break;
|
||||
}
|
||||
|
||||
// parse key
|
||||
// remember we are now inside an object
|
||||
states.push_back(false);
|
||||
|
||||
// parse key (the steps of parse_key(), which are
|
||||
// repeated here and below for speed)
|
||||
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr));
|
||||
if (!continue_after(key_error(sax, allow_recovery, false), skip_to_state_evaluation))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
||||
{
|
||||
@@ -256,14 +304,13 @@ class parser
|
||||
// parse separator (:)
|
||||
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr));
|
||||
if (!continue_after(key_error(sax, allow_recovery, true), skip_to_state_evaluation))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// remember we are now inside an object
|
||||
states.push_back(false);
|
||||
|
||||
// parse values
|
||||
get_token();
|
||||
continue;
|
||||
@@ -299,9 +346,11 @@ class parser
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!std::isfinite(res)))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr));
|
||||
if (!overflow_error(sax, res, allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->number_float(res, m_lexer.get_string())))
|
||||
@@ -369,23 +418,63 @@ class parser
|
||||
case token_type::parse_error:
|
||||
{
|
||||
// using "uninitialized" to avoid an "expected" message
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::uninitialized, "value"), nullptr));
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::uninitialized, "value"), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// recover: keep what can be read of the token
|
||||
recover_token();
|
||||
if (last_token != token_type::uninitialized)
|
||||
{
|
||||
// a string or a number
|
||||
continue;
|
||||
}
|
||||
if (states.empty())
|
||||
{
|
||||
// look for the value after the garbage
|
||||
if (!skip_to_value())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
// nothing could be read
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->null()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case token_type::end_of_input:
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(m_lexer.get_position().chars_read_total == 1))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(),
|
||||
"attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr));
|
||||
// there is nothing to recover
|
||||
static_cast<void>(report_error(sax, parse_error::create(101, m_lexer.get_position(),
|
||||
"attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr), allow_recovery));
|
||||
return false;
|
||||
}
|
||||
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr));
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// recover: the input ends where a value is missing
|
||||
if (states.empty())
|
||||
{
|
||||
// there is no value
|
||||
return false;
|
||||
}
|
||||
if (!recover_missing_value(sax, states))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
// the state evaluation reads the token again
|
||||
m_lexer.unget_token();
|
||||
skip_to_state_evaluation = true;
|
||||
continue;
|
||||
}
|
||||
case token_type::uninitialized:
|
||||
case token_type::end_array:
|
||||
@@ -395,9 +484,35 @@ class parser
|
||||
case token_type::literal_or_value:
|
||||
default: // the last token was unexpected
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr));
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// recover
|
||||
if (states.empty())
|
||||
{
|
||||
// look for the value after the garbage
|
||||
if (!skip_to_value())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (last_token == token_type::name_separator)
|
||||
{
|
||||
// a stray ':'; the value may follow
|
||||
get_token();
|
||||
continue;
|
||||
}
|
||||
if (!recover_missing_value(sax, states))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
// the state evaluation reads the token again
|
||||
m_lexer.unget_token();
|
||||
skip_to_state_evaluation = true;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -447,9 +562,30 @@ class parser
|
||||
continue;
|
||||
}
|
||||
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_array, "array"), nullptr));
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_array, "array"), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// recover
|
||||
if (last_token == token_type::end_of_input)
|
||||
{
|
||||
// the input ends inside the array
|
||||
return close_containers(sax, states);
|
||||
}
|
||||
if (last_token == token_type::end_object)
|
||||
{
|
||||
// a wrong closing bracket closes the innermost container
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->end_array()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
states.pop_back();
|
||||
skip_to_state_evaluation = true;
|
||||
}
|
||||
// otherwise, a missing ',' (or a stray ':', which value
|
||||
// parsing drops): the next value begins here
|
||||
continue;
|
||||
}
|
||||
|
||||
// states.back() is false -> object
|
||||
@@ -466,11 +602,12 @@ class parser
|
||||
// parse key
|
||||
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr));
|
||||
if (!continue_after(key_error(sax, allow_recovery, false), skip_to_state_evaluation))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
||||
{
|
||||
return false;
|
||||
@@ -479,9 +616,11 @@ class parser
|
||||
// parse separator (:)
|
||||
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr));
|
||||
if (!continue_after(key_error(sax, allow_recovery, true), skip_to_state_evaluation))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// parse values
|
||||
@@ -508,12 +647,479 @@ class parser
|
||||
continue;
|
||||
}
|
||||
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_object, "object"), nullptr));
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_object, "object"), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// recover
|
||||
if (last_token == token_type::end_of_input)
|
||||
{
|
||||
// the input ends inside the object
|
||||
return close_containers(sax, states);
|
||||
}
|
||||
if (last_token == token_type::end_array)
|
||||
{
|
||||
// a wrong closing bracket closes the innermost container
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->end_object()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
states.pop_back();
|
||||
skip_to_state_evaluation = true;
|
||||
continue;
|
||||
}
|
||||
if (!continue_after(recover_member(sax, allow_recovery), skip_to_state_evaluation))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief continue sax_parse_internal() after a recovery
|
||||
@return whether to continue parsing
|
||||
*/
|
||||
bool continue_after(const next_step step, bool& skip_to_state_evaluation)
|
||||
{
|
||||
if (step == next_step::evaluate_state)
|
||||
{
|
||||
// the state evaluation reads the token again
|
||||
m_lexer.unget_token();
|
||||
skip_to_state_evaluation = true;
|
||||
}
|
||||
return step != next_step::stop;
|
||||
}
|
||||
|
||||
/// the parser for parse() and accept() never recovers: stop parsing
|
||||
static std::false_type continue_after(std::false_type /*step*/, bool& /*skip_to_state_evaluation*/) noexcept
|
||||
{
|
||||
return {};
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse an object key and the name separator (:) after it
|
||||
|
||||
last_token is the token where the key is expected. sax_parse_internal()
|
||||
repeats these steps rather than calling this function, which is used
|
||||
when recovering from an error.
|
||||
|
||||
@return next_step::parse_value if the value follows, with last_token its
|
||||
first token; next_step::evaluate_state if the object's state is
|
||||
to be evaluated after recovering from an error; next_step::stop
|
||||
to stop parsing
|
||||
*/
|
||||
template<typename SAX>
|
||||
next_step parse_key(SAX* sax)
|
||||
{
|
||||
const std::true_type allow_recovery{};
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
||||
{
|
||||
return key_error(sax, allow_recovery, false);
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
||||
{
|
||||
return next_step::stop;
|
||||
}
|
||||
|
||||
// parse separator (:)
|
||||
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
||||
{
|
||||
return key_error(sax, allow_recovery, true);
|
||||
}
|
||||
|
||||
// the value begins with the next token
|
||||
get_token();
|
||||
return next_step::parse_value;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief report a number that is too large for number_float_t, and recover
|
||||
from the error by passing the value on; the SAX parser gets the
|
||||
number's text as well
|
||||
|
||||
This is a separate function, as reading other numbers is measurably
|
||||
slower if the error is handled where they are read.
|
||||
|
||||
@param[in] sax the SAX parser
|
||||
@param[in] value the value that is not finite
|
||||
@return whether to continue parsing
|
||||
*/
|
||||
template<typename SAX, typename AllowRecovery>
|
||||
bool overflow_error(SAX* sax, const number_float_t value, AllowRecovery allow_recovery)
|
||||
{
|
||||
if (!report_error(sax, out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
return sax->number_float(value, m_lexer.get_string());
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief report a missing key, or a missing name separator (:) after the
|
||||
key; the parser for parse() and accept() never recovers
|
||||
|
||||
@param[in] key_read whether the key was read, so that the name separator
|
||||
is missing
|
||||
@return std::false_type, see report_error()
|
||||
*/
|
||||
template<typename SAX>
|
||||
std::false_type key_error(SAX* sax, std::false_type allow_recovery, const bool key_read)
|
||||
{
|
||||
return report_error(sax, parse_error::create(101, m_lexer.get_position(), key_read
|
||||
? exception_message(token_type::name_separator, "object separator")
|
||||
: exception_message(token_type::value_string, "object key"), nullptr), allow_recovery);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief report a missing key, or a missing name separator (:) after the
|
||||
key, and recover from it
|
||||
|
||||
@param[in] key_read whether the key was read, so that the name separator
|
||||
is missing
|
||||
*/
|
||||
template<typename SAX>
|
||||
next_step key_error(SAX* sax, std::true_type allow_recovery, const bool key_read)
|
||||
{
|
||||
if (!key_read)
|
||||
{
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr), allow_recovery))
|
||||
{
|
||||
return next_step::stop;
|
||||
}
|
||||
return recover_key(sax);
|
||||
}
|
||||
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr), allow_recovery))
|
||||
{
|
||||
return next_step::stop;
|
||||
}
|
||||
return recover_name_separator(sax);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// error recovery
|
||||
/////////////////////
|
||||
|
||||
/*
|
||||
The functions below repair an error after the SAX parser's parse_error()
|
||||
returned true (see #3989). Each mistake is repaired by the smallest local
|
||||
edit: a missing ',' or ':' is inserted, a stray token is removed, what can
|
||||
be read of an invalid string or number is kept (see
|
||||
lexer::recover_token()), a missing value becomes null, a wrong closing
|
||||
bracket closes the innermost container, and the end of the input closes
|
||||
all of them. The events stay balanced, and every key() is followed by
|
||||
exactly one value.
|
||||
|
||||
A repair hands a token to the state evaluation, by returning it to the
|
||||
lexer (lexer::unget_token()) so that the state evaluation reads it again,
|
||||
only if it is ',', ']', '}', or the end of the input. The state evaluation
|
||||
hands a token to value or key parsing only if it is none of them, so a
|
||||
token is never handed back and forth. Every other step reads a token or
|
||||
closes a container, so parsing always ends.
|
||||
*/
|
||||
|
||||
/*!
|
||||
@brief report an error to the SAX parser; the parser for parse() and
|
||||
accept() never recovers
|
||||
|
||||
@return std::false_type rather than false: its value is known where the
|
||||
function is called even if the call is not inlined, so the code
|
||||
for recovering is not generated
|
||||
*/
|
||||
template<typename SAX, typename Exception>
|
||||
std::false_type report_error(SAX* sax, const Exception& ex, std::false_type /*allow_recovery*/)
|
||||
{
|
||||
error_reported = true;
|
||||
static_cast<void>(sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), ex));
|
||||
return {};
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief report an error to the SAX parser
|
||||
@return whether to recover from the error
|
||||
*/
|
||||
template<typename SAX, typename Exception>
|
||||
bool report_error(SAX* sax, const Exception& ex, std::true_type /*allow_recovery*/)
|
||||
{
|
||||
const std::size_t position = m_lexer.get_position().chars_read_total;
|
||||
if (error_reported && position == last_error_position && last_token == last_error_token)
|
||||
{
|
||||
// a repair handed on the token of the error it repaired; the
|
||||
// token was reported already, and the SAX parser asked to recover
|
||||
return true;
|
||||
}
|
||||
|
||||
error_reported = true;
|
||||
last_error_position = position;
|
||||
last_error_token = last_token;
|
||||
|
||||
if (!sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), ex))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// the token string of the next error begins here
|
||||
m_lexer.restart_token_string();
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief keep what can be read of the token that the lexer rejected
|
||||
|
||||
The error was reported for the rejected token, so it is not reported again
|
||||
for the token it is repaired to (see lexer::recover_token()).
|
||||
*/
|
||||
token_type recover_token()
|
||||
{
|
||||
last_token = m_lexer.recover_token();
|
||||
last_error_position = m_lexer.get_position().chars_read_total;
|
||||
last_error_token = last_token;
|
||||
return last_token;
|
||||
}
|
||||
|
||||
/// pass the end events of all open containers
|
||||
template<typename SAX>
|
||||
bool close_containers(SAX* sax, std::vector<bool>& states)
|
||||
{
|
||||
while (!states.empty())
|
||||
{
|
||||
const bool is_array = states.back();
|
||||
states.pop_back();
|
||||
if (JSON_HEDLEY_UNLIKELY(is_array ? !sax->end_array() : !sax->end_object()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief read tokens until one begins a value, skipping everything before
|
||||
the top-level value
|
||||
@return whether a value begins with last_token
|
||||
*/
|
||||
bool skip_to_value()
|
||||
{
|
||||
while (true)
|
||||
{
|
||||
switch (get_token())
|
||||
{
|
||||
case token_type::begin_array:
|
||||
case token_type::begin_object:
|
||||
case token_type::literal_false:
|
||||
case token_type::literal_null:
|
||||
case token_type::literal_true:
|
||||
case token_type::value_float:
|
||||
case token_type::value_integer:
|
||||
case token_type::value_string:
|
||||
case token_type::value_unsigned:
|
||||
return true;
|
||||
|
||||
case token_type::end_of_input:
|
||||
return false;
|
||||
|
||||
case token_type::parse_error:
|
||||
recover_token();
|
||||
if (last_token != token_type::uninitialized)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
break;
|
||||
|
||||
case token_type::uninitialized:
|
||||
case token_type::end_array:
|
||||
case token_type::end_object:
|
||||
case token_type::name_separator:
|
||||
case token_type::value_separator:
|
||||
case token_type::literal_or_value:
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief skip the rest of an object member that cannot be read
|
||||
|
||||
Reads tokens, beginning with last_token, until a ',', '}', or ']' that is
|
||||
not inside a container that begins in the skipped tokens, or the end of
|
||||
the input.
|
||||
*/
|
||||
void skip_member()
|
||||
{
|
||||
std::size_t depth = 0;
|
||||
while (true)
|
||||
{
|
||||
switch (last_token)
|
||||
{
|
||||
case token_type::begin_array:
|
||||
case token_type::begin_object:
|
||||
++depth;
|
||||
break;
|
||||
|
||||
case token_type::end_array:
|
||||
case token_type::end_object:
|
||||
if (depth == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
--depth;
|
||||
break;
|
||||
|
||||
case token_type::value_separator:
|
||||
if (depth == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
break;
|
||||
|
||||
case token_type::end_of_input:
|
||||
return;
|
||||
|
||||
case token_type::parse_error:
|
||||
recover_token();
|
||||
break;
|
||||
|
||||
case token_type::uninitialized:
|
||||
case token_type::literal_true:
|
||||
case token_type::literal_false:
|
||||
case token_type::literal_null:
|
||||
case token_type::value_string:
|
||||
case token_type::value_unsigned:
|
||||
case token_type::value_integer:
|
||||
case token_type::value_float:
|
||||
case token_type::name_separator:
|
||||
case token_type::literal_or_value:
|
||||
default:
|
||||
break;
|
||||
}
|
||||
get_token();
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a value where it is missing
|
||||
|
||||
last_token is ',', ']', '}', or the end of the input, where a value was
|
||||
expected. In an object, the key gets null; in an array, a ',' where a
|
||||
value is missing stands for null (as in JavaScript), while an array that
|
||||
ends there just ends.
|
||||
*/
|
||||
template<typename SAX>
|
||||
bool recover_missing_value(SAX* sax, const std::vector<bool>& states)
|
||||
{
|
||||
JSON_ASSERT(!states.empty());
|
||||
if (!states.back() || last_token == token_type::value_separator)
|
||||
{
|
||||
return sax->null();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// recover from a missing key; last_token is where it was expected
|
||||
template<typename SAX>
|
||||
next_step recover_key(SAX* sax)
|
||||
{
|
||||
switch (last_token)
|
||||
{
|
||||
case token_type::value_separator:
|
||||
case token_type::end_object:
|
||||
case token_type::end_array:
|
||||
case token_type::end_of_input:
|
||||
// no member: the object's state handles the token
|
||||
return next_step::evaluate_state;
|
||||
|
||||
case token_type::parse_error:
|
||||
recover_token();
|
||||
if (last_token == token_type::value_string)
|
||||
{
|
||||
// a key that could be repaired
|
||||
return parse_key(sax);
|
||||
}
|
||||
skip_member();
|
||||
return next_step::evaluate_state;
|
||||
|
||||
case token_type::uninitialized:
|
||||
case token_type::literal_true:
|
||||
case token_type::literal_false:
|
||||
case token_type::literal_null:
|
||||
case token_type::value_string:
|
||||
case token_type::value_unsigned:
|
||||
case token_type::value_integer:
|
||||
case token_type::value_float:
|
||||
case token_type::begin_array:
|
||||
case token_type::begin_object:
|
||||
case token_type::name_separator:
|
||||
case token_type::literal_or_value:
|
||||
default:
|
||||
// a member without a key
|
||||
skip_member();
|
||||
return next_step::evaluate_state;
|
||||
}
|
||||
}
|
||||
|
||||
/// recover from a missing name separator (:) after the key; last_token
|
||||
/// is where it was expected
|
||||
template<typename SAX>
|
||||
next_step recover_name_separator(SAX* sax)
|
||||
{
|
||||
switch (last_token)
|
||||
{
|
||||
case token_type::value_separator:
|
||||
case token_type::end_object:
|
||||
case token_type::end_array:
|
||||
case token_type::end_of_input:
|
||||
// the value is missing as well
|
||||
return sax->null() ? next_step::evaluate_state : next_step::stop;
|
||||
|
||||
case token_type::uninitialized:
|
||||
case token_type::literal_true:
|
||||
case token_type::literal_false:
|
||||
case token_type::literal_null:
|
||||
case token_type::value_string:
|
||||
case token_type::value_unsigned:
|
||||
case token_type::value_integer:
|
||||
case token_type::value_float:
|
||||
case token_type::begin_array:
|
||||
case token_type::begin_object:
|
||||
case token_type::name_separator:
|
||||
case token_type::parse_error:
|
||||
case token_type::literal_or_value:
|
||||
default:
|
||||
// a missing ':'; the value begins here
|
||||
return next_step::parse_value;
|
||||
}
|
||||
}
|
||||
|
||||
/// recover from a token after an object member that is neither ',' nor
|
||||
/// '}' (nor ']' or the end of the input, which the caller handles)
|
||||
template<typename SAX>
|
||||
next_step recover_member(SAX* sax, std::true_type /*allow_recovery*/)
|
||||
{
|
||||
if (last_token == token_type::parse_error)
|
||||
{
|
||||
recover_token();
|
||||
}
|
||||
if (last_token == token_type::value_string)
|
||||
{
|
||||
// a missing ','; the next key begins here
|
||||
return parse_key(sax);
|
||||
}
|
||||
skip_member();
|
||||
return next_step::evaluate_state;
|
||||
}
|
||||
|
||||
/// the parser for parse() and accept() never recovers (and does not come
|
||||
/// here, as report_error() returned false)
|
||||
template<typename SAX>
|
||||
std::false_type recover_member(SAX* /*sax*/, std::false_type /*allow_recovery*/) const noexcept
|
||||
{
|
||||
return {};
|
||||
}
|
||||
|
||||
/// get next token from lexer
|
||||
token_type get_token()
|
||||
{
|
||||
@@ -560,6 +1166,12 @@ class parser
|
||||
const bool allow_exceptions = true;
|
||||
/// whether trailing commas in objects and arrays should be ignored (true) or signaled as errors (false)
|
||||
const bool ignore_trailing_commas = false;
|
||||
/// whether an error was reported to the SAX parser
|
||||
bool error_reported = false;
|
||||
/// the position of the last reported error
|
||||
std::size_t last_error_position = 0;
|
||||
/// the token of the last reported error
|
||||
token_type last_error_token = token_type::uninitialized;
|
||||
};
|
||||
|
||||
} // namespace detail
|
||||
|
||||
@@ -1,371 +0,0 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2021 The fast_float authors <https://github.com/fastfloat/fast_float>
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
|
||||
#include <nlohmann/detail/abi_macros.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
/// the range of decimal exponents covered by pow5_128()
|
||||
constexpr std::int64_t pow5_128_smallest_power = -342;
|
||||
constexpr std::int64_t pow5_128_largest_power = 308;
|
||||
|
||||
/*!
|
||||
@brief 128-bit approximations of 5^q for q in [-342, 308]
|
||||
|
||||
Entry q (at index 2 * (q + 342)) holds the most significant 128 bits of 5^q,
|
||||
normalized so that the highest bit is set: for q >= 0 the truncated value, for
|
||||
q < 0 the value rounded up. This is the table of fast_float (Daniel Lemire and
|
||||
contributors, used under the MIT license), generated like its
|
||||
script/table_generation.py; unit-class_lexer.cpp recomputes every entry.
|
||||
*/
|
||||
inline const std::array<std::uint64_t, 1302>& pow5_128() noexcept
|
||||
{
|
||||
static const std::array<std::uint64_t, 1302> table =
|
||||
{
|
||||
{
|
||||
0xeef453d6923bd65au, 0x113faa2906a13b3fu, 0x9558b4661b6565f8u, 0x4ac7ca59a424c507u,
|
||||
0xbaaee17fa23ebf76u, 0x5d79bcf00d2df649u, 0xe95a99df8ace6f53u, 0xf4d82c2c107973dcu,
|
||||
0x91d8a02bb6c10594u, 0x79071b9b8a4be869u, 0xb64ec836a47146f9u, 0x9748e2826cdee284u,
|
||||
0xe3e27a444d8d98b7u, 0xfd1b1b2308169b25u, 0x8e6d8c6ab0787f72u, 0xfe30f0f5e50e20f7u,
|
||||
0xb208ef855c969f4fu, 0xbdbd2d335e51a935u, 0xde8b2b66b3bc4723u, 0xad2c788035e61382u,
|
||||
0x8b16fb203055ac76u, 0x4c3bcb5021afcc31u, 0xaddcb9e83c6b1793u, 0xdf4abe242a1bbf3du,
|
||||
0xd953e8624b85dd78u, 0xd71d6dad34a2af0du, 0x87d4713d6f33aa6bu, 0x8672648c40e5ad68u,
|
||||
0xa9c98d8ccb009506u, 0x680efdaf511f18c2u, 0xd43bf0effdc0ba48u, 0x0212bd1b2566def2u,
|
||||
0x84a57695fe98746du, 0x014bb630f7604b57u, 0xa5ced43b7e3e9188u, 0x419ea3bd35385e2du,
|
||||
0xcf42894a5dce35eau, 0x52064cac828675b9u, 0x818995ce7aa0e1b2u, 0x7343efebd1940993u,
|
||||
0xa1ebfb4219491a1fu, 0x1014ebe6c5f90bf8u, 0xca66fa129f9b60a6u, 0xd41a26e077774ef6u,
|
||||
0xfd00b897478238d0u, 0x8920b098955522b4u, 0x9e20735e8cb16382u, 0x55b46e5f5d5535b0u,
|
||||
0xc5a890362fddbc62u, 0xeb2189f734aa831du, 0xf712b443bbd52b7bu, 0xa5e9ec7501d523e4u,
|
||||
0x9a6bb0aa55653b2du, 0x47b233c92125366eu, 0xc1069cd4eabe89f8u, 0x999ec0bb696e840au,
|
||||
0xf148440a256e2c76u, 0xc00670ea43ca250du, 0x96cd2a865764dbcau, 0x380406926a5e5728u,
|
||||
0xbc807527ed3e12bcu, 0xc605083704f5ecf2u, 0xeba09271e88d976bu, 0xf7864a44c633682eu,
|
||||
0x93445b8731587ea3u, 0x7ab3ee6afbe0211du, 0xb8157268fdae9e4cu, 0x5960ea05bad82964u,
|
||||
0xe61acf033d1a45dfu, 0x6fb92487298e33bdu, 0x8fd0c16206306babu, 0xa5d3b6d479f8e056u,
|
||||
0xb3c4f1ba87bc8696u, 0x8f48a4899877186cu, 0xe0b62e2929aba83cu, 0x331acdabfe94de87u,
|
||||
0x8c71dcd9ba0b4925u, 0x9ff0c08b7f1d0b14u, 0xaf8e5410288e1b6fu, 0x07ecf0ae5ee44dd9u,
|
||||
0xdb71e91432b1a24au, 0xc9e82cd9f69d6150u, 0x892731ac9faf056eu, 0xbe311c083a225cd2u,
|
||||
0xab70fe17c79ac6cau, 0x6dbd630a48aaf406u, 0xd64d3d9db981787du, 0x092cbbccdad5b108u,
|
||||
0x85f0468293f0eb4eu, 0x25bbf56008c58ea5u, 0xa76c582338ed2621u, 0xaf2af2b80af6f24eu,
|
||||
0xd1476e2c07286faau, 0x1af5af660db4aee1u, 0x82cca4db847945cau, 0x50d98d9fc890ed4du,
|
||||
0xa37fce126597973cu, 0xe50ff107bab528a0u, 0xcc5fc196fefd7d0cu, 0x1e53ed49a96272c8u,
|
||||
0xff77b1fcbebcdc4fu, 0x25e8e89c13bb0f7au, 0x9faacf3df73609b1u, 0x77b191618c54e9acu,
|
||||
0xc795830d75038c1du, 0xd59df5b9ef6a2417u, 0xf97ae3d0d2446f25u, 0x4b0573286b44ad1du,
|
||||
0x9becce62836ac577u, 0x4ee367f9430aec32u, 0xc2e801fb244576d5u, 0x229c41f793cda73fu,
|
||||
0xf3a20279ed56d48au, 0x6b43527578c1110fu, 0x9845418c345644d6u, 0x830a13896b78aaa9u,
|
||||
0xbe5691ef416bd60cu, 0x23cc986bc656d553u, 0xedec366b11c6cb8fu, 0x2cbfbe86b7ec8aa8u,
|
||||
0x94b3a202eb1c3f39u, 0x7bf7d71432f3d6a9u, 0xb9e08a83a5e34f07u, 0xdaf5ccd93fb0cc53u,
|
||||
0xe858ad248f5c22c9u, 0xd1b3400f8f9cff68u, 0x91376c36d99995beu, 0x23100809b9c21fa1u,
|
||||
0xb58547448ffffb2du, 0xabd40a0c2832a78au, 0xe2e69915b3fff9f9u, 0x16c90c8f323f516cu,
|
||||
0x8dd01fad907ffc3bu, 0xae3da7d97f6792e3u, 0xb1442798f49ffb4au, 0x99cd11cfdf41779cu,
|
||||
0xdd95317f31c7fa1du, 0x40405643d711d583u, 0x8a7d3eef7f1cfc52u, 0x482835ea666b2572u,
|
||||
0xad1c8eab5ee43b66u, 0xda3243650005eecfu, 0xd863b256369d4a40u, 0x90bed43e40076a82u,
|
||||
0x873e4f75e2224e68u, 0x5a7744a6e804a291u, 0xa90de3535aaae202u, 0x711515d0a205cb36u,
|
||||
0xd3515c2831559a83u, 0x0d5a5b44ca873e03u, 0x8412d9991ed58091u, 0xe858790afe9486c2u,
|
||||
0xa5178fff668ae0b6u, 0x626e974dbe39a872u, 0xce5d73ff402d98e3u, 0xfb0a3d212dc8128fu,
|
||||
0x80fa687f881c7f8eu, 0x7ce66634bc9d0b99u, 0xa139029f6a239f72u, 0x1c1fffc1ebc44e80u,
|
||||
0xc987434744ac874eu, 0xa327ffb266b56220u, 0xfbe9141915d7a922u, 0x4bf1ff9f0062baa8u,
|
||||
0x9d71ac8fada6c9b5u, 0x6f773fc3603db4a9u, 0xc4ce17b399107c22u, 0xcb550fb4384d21d3u,
|
||||
0xf6019da07f549b2bu, 0x7e2a53a146606a48u, 0x99c102844f94e0fbu, 0x2eda7444cbfc426du,
|
||||
0xc0314325637a1939u, 0xfa911155fefb5308u, 0xf03d93eebc589f88u, 0x793555ab7eba27cau,
|
||||
0x96267c7535b763b5u, 0x4bc1558b2f3458deu, 0xbbb01b9283253ca2u, 0x9eb1aaedfb016f16u,
|
||||
0xea9c227723ee8bcbu, 0x465e15a979c1cadcu, 0x92a1958a7675175fu, 0x0bfacd89ec191ec9u,
|
||||
0xb749faed14125d36u, 0xcef980ec671f667bu, 0xe51c79a85916f484u, 0x82b7e12780e7401au,
|
||||
0x8f31cc0937ae58d2u, 0xd1b2ecb8b0908810u, 0xb2fe3f0b8599ef07u, 0x861fa7e6dcb4aa15u,
|
||||
0xdfbdcece67006ac9u, 0x67a791e093e1d49au, 0x8bd6a141006042bdu, 0xe0c8bb2c5c6d24e0u,
|
||||
0xaecc49914078536du, 0x58fae9f773886e18u, 0xda7f5bf590966848u, 0xaf39a475506a899eu,
|
||||
0x888f99797a5e012du, 0x6d8406c952429603u, 0xaab37fd7d8f58178u, 0xc8e5087ba6d33b83u,
|
||||
0xd5605fcdcf32e1d6u, 0xfb1e4a9a90880a64u, 0x855c3be0a17fcd26u, 0x5cf2eea09a55067fu,
|
||||
0xa6b34ad8c9dfc06fu, 0xf42faa48c0ea481eu, 0xd0601d8efc57b08bu, 0xf13b94daf124da26u,
|
||||
0x823c12795db6ce57u, 0x76c53d08d6b70858u, 0xa2cb1717b52481edu, 0x54768c4b0c64ca6eu,
|
||||
0xcb7ddcdda26da268u, 0xa9942f5dcf7dfd09u, 0xfe5d54150b090b02u, 0xd3f93b35435d7c4cu,
|
||||
0x9efa548d26e5a6e1u, 0xc47bc5014a1a6dafu, 0xc6b8e9b0709f109au, 0x359ab6419ca1091bu,
|
||||
0xf867241c8cc6d4c0u, 0xc30163d203c94b62u, 0x9b407691d7fc44f8u, 0x79e0de63425dcf1du,
|
||||
0xc21094364dfb5636u, 0x985915fc12f542e4u, 0xf294b943e17a2bc4u, 0x3e6f5b7b17b2939du,
|
||||
0x979cf3ca6cec5b5au, 0xa705992ceecf9c42u, 0xbd8430bd08277231u, 0x50c6ff782a838353u,
|
||||
0xece53cec4a314ebdu, 0xa4f8bf5635246428u, 0x940f4613ae5ed136u, 0x871b7795e136be99u,
|
||||
0xb913179899f68584u, 0x28e2557b59846e3fu, 0xe757dd7ec07426e5u, 0x331aeada2fe589cfu,
|
||||
0x9096ea6f3848984fu, 0x3ff0d2c85def7621u, 0xb4bca50b065abe63u, 0x0fed077a756b53a9u,
|
||||
0xe1ebce4dc7f16dfbu, 0xd3e8495912c62894u, 0x8d3360f09cf6e4bdu, 0x64712dd7abbbd95cu,
|
||||
0xb080392cc4349decu, 0xbd8d794d96aacfb3u, 0xdca04777f541c567u, 0xecf0d7a0fc5583a0u,
|
||||
0x89e42caaf9491b60u, 0xf41686c49db57244u, 0xac5d37d5b79b6239u, 0x311c2875c522ced5u,
|
||||
0xd77485cb25823ac7u, 0x7d633293366b828bu, 0x86a8d39ef77164bcu, 0xae5dff9c02033197u,
|
||||
0xa8530886b54dbdebu, 0xd9f57f830283fdfcu, 0xd267caa862a12d66u, 0xd072df63c324fd7bu,
|
||||
0x8380dea93da4bc60u, 0x4247cb9e59f71e6du, 0xa46116538d0deb78u, 0x52d9be85f074e608u,
|
||||
0xcd795be870516656u, 0x67902e276c921f8bu, 0x806bd9714632dff6u, 0x00ba1cd8a3db53b6u,
|
||||
0xa086cfcd97bf97f3u, 0x80e8a40eccd228a4u, 0xc8a883c0fdaf7df0u, 0x6122cd128006b2cdu,
|
||||
0xfad2a4b13d1b5d6cu, 0x796b805720085f81u, 0x9cc3a6eec6311a63u, 0xcbe3303674053bb0u,
|
||||
0xc3f490aa77bd60fcu, 0xbedbfc4411068a9cu, 0xf4f1b4d515acb93bu, 0xee92fb5515482d44u,
|
||||
0x991711052d8bf3c5u, 0x751bdd152d4d1c4au, 0xbf5cd54678eef0b6u, 0xd262d45a78a0635du,
|
||||
0xef340a98172aace4u, 0x86fb897116c87c34u, 0x9580869f0e7aac0eu, 0xd45d35e6ae3d4da0u,
|
||||
0xbae0a846d2195712u, 0x8974836059cca109u, 0xe998d258869facd7u, 0x2bd1a438703fc94bu,
|
||||
0x91ff83775423cc06u, 0x7b6306a34627ddcfu, 0xb67f6455292cbf08u, 0x1a3bc84c17b1d542u,
|
||||
0xe41f3d6a7377eecau, 0x20caba5f1d9e4a93u, 0x8e938662882af53eu, 0x547eb47b7282ee9cu,
|
||||
0xb23867fb2a35b28du, 0xe99e619a4f23aa43u, 0xdec681f9f4c31f31u, 0x6405fa00e2ec94d4u,
|
||||
0x8b3c113c38f9f37eu, 0xde83bc408dd3dd04u, 0xae0b158b4738705eu, 0x9624ab50b148d445u,
|
||||
0xd98ddaee19068c76u, 0x3badd624dd9b0957u, 0x87f8a8d4cfa417c9u, 0xe54ca5d70a80e5d6u,
|
||||
0xa9f6d30a038d1dbcu, 0x5e9fcf4ccd211f4cu, 0xd47487cc8470652bu, 0x7647c3200069671fu,
|
||||
0x84c8d4dfd2c63f3bu, 0x29ecd9f40041e073u, 0xa5fb0a17c777cf09u, 0xf468107100525890u,
|
||||
0xcf79cc9db955c2ccu, 0x7182148d4066eeb4u, 0x81ac1fe293d599bfu, 0xc6f14cd848405530u,
|
||||
0xa21727db38cb002fu, 0xb8ada00e5a506a7cu, 0xca9cf1d206fdc03bu, 0xa6d90811f0e4851cu,
|
||||
0xfd442e4688bd304au, 0x908f4a166d1da663u, 0x9e4a9cec15763e2eu, 0x9a598e4e043287feu,
|
||||
0xc5dd44271ad3cdbau, 0x40eff1e1853f29fdu, 0xf7549530e188c128u, 0xd12bee59e68ef47cu,
|
||||
0x9a94dd3e8cf578b9u, 0x82bb74f8301958ceu, 0xc13a148e3032d6e7u, 0xe36a52363c1faf01u,
|
||||
0xf18899b1bc3f8ca1u, 0xdc44e6c3cb279ac1u, 0x96f5600f15a7b7e5u, 0x29ab103a5ef8c0b9u,
|
||||
0xbcb2b812db11a5deu, 0x7415d448f6b6f0e7u, 0xebdf661791d60f56u, 0x111b495b3464ad21u,
|
||||
0x936b9fcebb25c995u, 0xcab10dd900beec34u, 0xb84687c269ef3bfbu, 0x3d5d514f40eea742u,
|
||||
0xe65829b3046b0afau, 0x0cb4a5a3112a5112u, 0x8ff71a0fe2c2e6dcu, 0x47f0e785eaba72abu,
|
||||
0xb3f4e093db73a093u, 0x59ed216765690f56u, 0xe0f218b8d25088b8u, 0x306869c13ec3532cu,
|
||||
0x8c974f7383725573u, 0x1e414218c73a13fbu, 0xafbd2350644eeacfu, 0xe5d1929ef90898fau,
|
||||
0xdbac6c247d62a583u, 0xdf45f746b74abf39u, 0x894bc396ce5da772u, 0x6b8bba8c328eb783u,
|
||||
0xab9eb47c81f5114fu, 0x066ea92f3f326564u, 0xd686619ba27255a2u, 0xc80a537b0efefebdu,
|
||||
0x8613fd0145877585u, 0xbd06742ce95f5f36u, 0xa798fc4196e952e7u, 0x2c48113823b73704u,
|
||||
0xd17f3b51fca3a7a0u, 0xf75a15862ca504c5u, 0x82ef85133de648c4u, 0x9a984d73dbe722fbu,
|
||||
0xa3ab66580d5fdaf5u, 0xc13e60d0d2e0ebbau, 0xcc963fee10b7d1b3u, 0x318df905079926a8u,
|
||||
0xffbbcfe994e5c61fu, 0xfdf17746497f7052u, 0x9fd561f1fd0f9bd3u, 0xfeb6ea8bedefa633u,
|
||||
0xc7caba6e7c5382c8u, 0xfe64a52ee96b8fc0u, 0xf9bd690a1b68637bu, 0x3dfdce7aa3c673b0u,
|
||||
0x9c1661a651213e2du, 0x06bea10ca65c084eu, 0xc31bfa0fe5698db8u, 0x486e494fcff30a62u,
|
||||
0xf3e2f893dec3f126u, 0x5a89dba3c3efccfau, 0x986ddb5c6b3a76b7u, 0xf89629465a75e01cu,
|
||||
0xbe89523386091465u, 0xf6bbb397f1135823u, 0xee2ba6c0678b597fu, 0x746aa07ded582e2cu,
|
||||
0x94db483840b717efu, 0xa8c2a44eb4571cdcu, 0xba121a4650e4ddebu, 0x92f34d62616ce413u,
|
||||
0xe896a0d7e51e1566u, 0x77b020baf9c81d17u, 0x915e2486ef32cd60u, 0x0ace1474dc1d122eu,
|
||||
0xb5b5ada8aaff80b8u, 0x0d819992132456bau, 0xe3231912d5bf60e6u, 0x10e1fff697ed6c69u,
|
||||
0x8df5efabc5979c8fu, 0xca8d3ffa1ef463c1u, 0xb1736b96b6fd83b3u, 0xbd308ff8a6b17cb2u,
|
||||
0xddd0467c64bce4a0u, 0xac7cb3f6d05ddbdeu, 0x8aa22c0dbef60ee4u, 0x6bcdf07a423aa96bu,
|
||||
0xad4ab7112eb3929du, 0x86c16c98d2c953c6u, 0xd89d64d57a607744u, 0xe871c7bf077ba8b7u,
|
||||
0x87625f056c7c4a8bu, 0x11471cd764ad4972u, 0xa93af6c6c79b5d2du, 0xd598e40d3dd89bcfu,
|
||||
0xd389b47879823479u, 0x4aff1d108d4ec2c3u, 0x843610cb4bf160cbu, 0xcedf722a585139bau,
|
||||
0xa54394fe1eedb8feu, 0xc2974eb4ee658828u, 0xce947a3da6a9273eu, 0x733d226229feea32u,
|
||||
0x811ccc668829b887u, 0x0806357d5a3f525fu, 0xa163ff802a3426a8u, 0xca07c2dcb0cf26f7u,
|
||||
0xc9bcff6034c13052u, 0xfc89b393dd02f0b5u, 0xfc2c3f3841f17c67u, 0xbbac2078d443ace2u,
|
||||
0x9d9ba7832936edc0u, 0xd54b944b84aa4c0du, 0xc5029163f384a931u, 0x0a9e795e65d4df11u,
|
||||
0xf64335bcf065d37du, 0x4d4617b5ff4a16d5u, 0x99ea0196163fa42eu, 0x504bced1bf8e4e45u,
|
||||
0xc06481fb9bcf8d39u, 0xe45ec2862f71e1d6u, 0xf07da27a82c37088u, 0x5d767327bb4e5a4cu,
|
||||
0x964e858c91ba2655u, 0x3a6a07f8d510f86fu, 0xbbe226efb628afeau, 0x890489f70a55368bu,
|
||||
0xeadab0aba3b2dbe5u, 0x2b45ac74ccea842eu, 0x92c8ae6b464fc96fu, 0x3b0b8bc90012929du,
|
||||
0xb77ada0617e3bbcbu, 0x09ce6ebb40173744u, 0xe55990879ddcaabdu, 0xcc420a6a101d0515u,
|
||||
0x8f57fa54c2a9eab6u, 0x9fa946824a12232du, 0xb32df8e9f3546564u, 0x47939822dc96abf9u,
|
||||
0xdff9772470297ebdu, 0x59787e2b93bc56f7u, 0x8bfbea76c619ef36u, 0x57eb4edb3c55b65au,
|
||||
0xaefae51477a06b03u, 0xede622920b6b23f1u, 0xdab99e59958885c4u, 0xe95fab368e45ecedu,
|
||||
0x88b402f7fd75539bu, 0x11dbcb0218ebb414u, 0xaae103b5fcd2a881u, 0xd652bdc29f26a119u,
|
||||
0xd59944a37c0752a2u, 0x4be76d3346f0495fu, 0x857fcae62d8493a5u, 0x6f70a4400c562ddbu,
|
||||
0xa6dfbd9fb8e5b88eu, 0xcb4ccd500f6bb952u, 0xd097ad07a71f26b2u, 0x7e2000a41346a7a7u,
|
||||
0x825ecc24c873782fu, 0x8ed400668c0c28c8u, 0xa2f67f2dfa90563bu, 0x728900802f0f32fau,
|
||||
0xcbb41ef979346bcau, 0x4f2b40a03ad2ffb9u, 0xfea126b7d78186bcu, 0xe2f610c84987bfa8u,
|
||||
0x9f24b832e6b0f436u, 0x0dd9ca7d2df4d7c9u, 0xc6ede63fa05d3143u, 0x91503d1c79720dbbu,
|
||||
0xf8a95fcf88747d94u, 0x75a44c6397ce912au, 0x9b69dbe1b548ce7cu, 0xc986afbe3ee11abau,
|
||||
0xc24452da229b021bu, 0xfbe85badce996168u, 0xf2d56790ab41c2a2u, 0xfae27299423fb9c3u,
|
||||
0x97c560ba6b0919a5u, 0xdccd879fc967d41au, 0xbdb6b8e905cb600fu, 0x5400e987bbc1c920u,
|
||||
0xed246723473e3813u, 0x290123e9aab23b68u, 0x9436c0760c86e30bu, 0xf9a0b6720aaf6521u,
|
||||
0xb94470938fa89bceu, 0xf808e40e8d5b3e69u, 0xe7958cb87392c2c2u, 0xb60b1d1230b20e04u,
|
||||
0x90bd77f3483bb9b9u, 0xb1c6f22b5e6f48c2u, 0xb4ecd5f01a4aa828u, 0x1e38aeb6360b1af3u,
|
||||
0xe2280b6c20dd5232u, 0x25c6da63c38de1b0u, 0x8d590723948a535fu, 0x579c487e5a38ad0eu,
|
||||
0xb0af48ec79ace837u, 0x2d835a9df0c6d851u, 0xdcdb1b2798182244u, 0xf8e431456cf88e65u,
|
||||
0x8a08f0f8bf0f156bu, 0x1b8e9ecb641b58ffu, 0xac8b2d36eed2dac5u, 0xe272467e3d222f3fu,
|
||||
0xd7adf884aa879177u, 0x5b0ed81dcc6abb0fu, 0x86ccbb52ea94baeau, 0x98e947129fc2b4e9u,
|
||||
0xa87fea27a539e9a5u, 0x3f2398d747b36224u, 0xd29fe4b18e88640eu, 0x8eec7f0d19a03aadu,
|
||||
0x83a3eeeef9153e89u, 0x1953cf68300424acu, 0xa48ceaaab75a8e2bu, 0x5fa8c3423c052dd7u,
|
||||
0xcdb02555653131b6u, 0x3792f412cb06794du, 0x808e17555f3ebf11u, 0xe2bbd88bbee40bd0u,
|
||||
0xa0b19d2ab70e6ed6u, 0x5b6aceaeae9d0ec4u, 0xc8de047564d20a8bu, 0xf245825a5a445275u,
|
||||
0xfb158592be068d2eu, 0xeed6e2f0f0d56712u, 0x9ced737bb6c4183du, 0x55464dd69685606bu,
|
||||
0xc428d05aa4751e4cu, 0xaa97e14c3c26b886u, 0xf53304714d9265dfu, 0xd53dd99f4b3066a8u,
|
||||
0x993fe2c6d07b7fabu, 0xe546a8038efe4029u, 0xbf8fdb78849a5f96u, 0xde98520472bdd033u,
|
||||
0xef73d256a5c0f77cu, 0x963e66858f6d4440u, 0x95a8637627989aadu, 0xdde7001379a44aa8u,
|
||||
0xbb127c53b17ec159u, 0x5560c018580d5d52u, 0xe9d71b689dde71afu, 0xaab8f01e6e10b4a6u,
|
||||
0x9226712162ab070du, 0xcab3961304ca70e8u, 0xb6b00d69bb55c8d1u, 0x3d607b97c5fd0d22u,
|
||||
0xe45c10c42a2b3b05u, 0x8cb89a7db77c506au, 0x8eb98a7a9a5b04e3u, 0x77f3608e92adb242u,
|
||||
0xb267ed1940f1c61cu, 0x55f038b237591ed3u, 0xdf01e85f912e37a3u, 0x6b6c46dec52f6688u,
|
||||
0x8b61313bbabce2c6u, 0x2323ac4b3b3da015u, 0xae397d8aa96c1b77u, 0xabec975e0a0d081au,
|
||||
0xd9c7dced53c72255u, 0x96e7bd358c904a21u, 0x881cea14545c7575u, 0x7e50d64177da2e54u,
|
||||
0xaa242499697392d2u, 0xdde50bd1d5d0b9e9u, 0xd4ad2dbfc3d07787u, 0x955e4ec64b44e864u,
|
||||
0x84ec3c97da624ab4u, 0xbd5af13bef0b113eu, 0xa6274bbdd0fadd61u, 0xecb1ad8aeacdd58eu,
|
||||
0xcfb11ead453994bau, 0x67de18eda5814af2u, 0x81ceb32c4b43fcf4u, 0x80eacf948770ced7u,
|
||||
0xa2425ff75e14fc31u, 0xa1258379a94d028du, 0xcad2f7f5359a3b3eu, 0x096ee45813a04330u,
|
||||
0xfd87b5f28300ca0du, 0x8bca9d6e188853fcu, 0x9e74d1b791e07e48u, 0x775ea264cf55347eu,
|
||||
0xc612062576589ddau, 0x95364afe032a819eu, 0xf79687aed3eec551u, 0x3a83ddbd83f52205u,
|
||||
0x9abe14cd44753b52u, 0xc4926a9672793543u, 0xc16d9a0095928a27u, 0x75b7053c0f178294u,
|
||||
0xf1c90080baf72cb1u, 0x5324c68b12dd6339u, 0x971da05074da7beeu, 0xd3f6fc16ebca5e04u,
|
||||
0xbce5086492111aeau, 0x88f4bb1ca6bcf585u, 0xec1e4a7db69561a5u, 0x2b31e9e3d06c32e6u,
|
||||
0x9392ee8e921d5d07u, 0x3aff322e62439fd0u, 0xb877aa3236a4b449u, 0x09befeb9fad487c3u,
|
||||
0xe69594bec44de15bu, 0x4c2ebe687989a9b4u, 0x901d7cf73ab0acd9u, 0x0f9d37014bf60a11u,
|
||||
0xb424dc35095cd80fu, 0x538484c19ef38c95u, 0xe12e13424bb40e13u, 0x2865a5f206b06fbau,
|
||||
0x8cbccc096f5088cbu, 0xf93f87b7442e45d4u, 0xafebff0bcb24aafeu, 0xf78f69a51539d749u,
|
||||
0xdbe6fecebdedd5beu, 0xb573440e5a884d1cu, 0x89705f4136b4a597u, 0x31680a88f8953031u,
|
||||
0xabcc77118461cefcu, 0xfdc20d2b36ba7c3eu, 0xd6bf94d5e57a42bcu, 0x3d32907604691b4du,
|
||||
0x8637bd05af6c69b5u, 0xa63f9a49c2c1b110u, 0xa7c5ac471b478423u, 0x0fcf80dc33721d54u,
|
||||
0xd1b71758e219652bu, 0xd3c36113404ea4a9u, 0x83126e978d4fdf3bu, 0x645a1cac083126eau,
|
||||
0xa3d70a3d70a3d70au, 0x3d70a3d70a3d70a4u, 0xccccccccccccccccu, 0xcccccccccccccccdu,
|
||||
0x8000000000000000u, 0x0000000000000000u, 0xa000000000000000u, 0x0000000000000000u,
|
||||
0xc800000000000000u, 0x0000000000000000u, 0xfa00000000000000u, 0x0000000000000000u,
|
||||
0x9c40000000000000u, 0x0000000000000000u, 0xc350000000000000u, 0x0000000000000000u,
|
||||
0xf424000000000000u, 0x0000000000000000u, 0x9896800000000000u, 0x0000000000000000u,
|
||||
0xbebc200000000000u, 0x0000000000000000u, 0xee6b280000000000u, 0x0000000000000000u,
|
||||
0x9502f90000000000u, 0x0000000000000000u, 0xba43b74000000000u, 0x0000000000000000u,
|
||||
0xe8d4a51000000000u, 0x0000000000000000u, 0x9184e72a00000000u, 0x0000000000000000u,
|
||||
0xb5e620f480000000u, 0x0000000000000000u, 0xe35fa931a0000000u, 0x0000000000000000u,
|
||||
0x8e1bc9bf04000000u, 0x0000000000000000u, 0xb1a2bc2ec5000000u, 0x0000000000000000u,
|
||||
0xde0b6b3a76400000u, 0x0000000000000000u, 0x8ac7230489e80000u, 0x0000000000000000u,
|
||||
0xad78ebc5ac620000u, 0x0000000000000000u, 0xd8d726b7177a8000u, 0x0000000000000000u,
|
||||
0x878678326eac9000u, 0x0000000000000000u, 0xa968163f0a57b400u, 0x0000000000000000u,
|
||||
0xd3c21bcecceda100u, 0x0000000000000000u, 0x84595161401484a0u, 0x0000000000000000u,
|
||||
0xa56fa5b99019a5c8u, 0x0000000000000000u, 0xcecb8f27f4200f3au, 0x0000000000000000u,
|
||||
0x813f3978f8940984u, 0x4000000000000000u, 0xa18f07d736b90be5u, 0x5000000000000000u,
|
||||
0xc9f2c9cd04674edeu, 0xa400000000000000u, 0xfc6f7c4045812296u, 0x4d00000000000000u,
|
||||
0x9dc5ada82b70b59du, 0xf020000000000000u, 0xc5371912364ce305u, 0x6c28000000000000u,
|
||||
0xf684df56c3e01bc6u, 0xc732000000000000u, 0x9a130b963a6c115cu, 0x3c7f400000000000u,
|
||||
0xc097ce7bc90715b3u, 0x4b9f100000000000u, 0xf0bdc21abb48db20u, 0x1e86d40000000000u,
|
||||
0x96769950b50d88f4u, 0x1314448000000000u, 0xbc143fa4e250eb31u, 0x17d955a000000000u,
|
||||
0xeb194f8e1ae525fdu, 0x5dcfab0800000000u, 0x92efd1b8d0cf37beu, 0x5aa1cae500000000u,
|
||||
0xb7abc627050305adu, 0xf14a3d9e40000000u, 0xe596b7b0c643c719u, 0x6d9ccd05d0000000u,
|
||||
0x8f7e32ce7bea5c6fu, 0xe4820023a2000000u, 0xb35dbf821ae4f38bu, 0xdda2802c8a800000u,
|
||||
0xe0352f62a19e306eu, 0xd50b2037ad200000u, 0x8c213d9da502de45u, 0x4526f422cc340000u,
|
||||
0xaf298d050e4395d6u, 0x9670b12b7f410000u, 0xdaf3f04651d47b4cu, 0x3c0cdd765f114000u,
|
||||
0x88d8762bf324cd0fu, 0xa5880a69fb6ac800u, 0xab0e93b6efee0053u, 0x8eea0d047a457a00u,
|
||||
0xd5d238a4abe98068u, 0x72a4904598d6d880u, 0x85a36366eb71f041u, 0x47a6da2b7f864750u,
|
||||
0xa70c3c40a64e6c51u, 0x999090b65f67d924u, 0xd0cf4b50cfe20765u, 0xfff4b4e3f741cf6du,
|
||||
0x82818f1281ed449fu, 0xbff8f10e7a8921a4u, 0xa321f2d7226895c7u, 0xaff72d52192b6a0du,
|
||||
0xcbea6f8ceb02bb39u, 0x9bf4f8a69f764490u, 0xfee50b7025c36a08u, 0x02f236d04753d5b4u,
|
||||
0x9f4f2726179a2245u, 0x01d762422c946590u, 0xc722f0ef9d80aad6u, 0x424d3ad2b7b97ef5u,
|
||||
0xf8ebad2b84e0d58bu, 0xd2e0898765a7deb2u, 0x9b934c3b330c8577u, 0x63cc55f49f88eb2fu,
|
||||
0xc2781f49ffcfa6d5u, 0x3cbf6b71c76b25fbu, 0xf316271c7fc3908au, 0x8bef464e3945ef7au,
|
||||
0x97edd871cfda3a56u, 0x97758bf0e3cbb5acu, 0xbde94e8e43d0c8ecu, 0x3d52eeed1cbea317u,
|
||||
0xed63a231d4c4fb27u, 0x4ca7aaa863ee4bddu, 0x945e455f24fb1cf8u, 0x8fe8caa93e74ef6au,
|
||||
0xb975d6b6ee39e436u, 0xb3e2fd538e122b44u, 0xe7d34c64a9c85d44u, 0x60dbbca87196b616u,
|
||||
0x90e40fbeea1d3a4au, 0xbc8955e946fe31cdu, 0xb51d13aea4a488ddu, 0x6babab6398bdbe41u,
|
||||
0xe264589a4dcdab14u, 0xc696963c7eed2dd1u, 0x8d7eb76070a08aecu, 0xfc1e1de5cf543ca2u,
|
||||
0xb0de65388cc8ada8u, 0x3b25a55f43294bcbu, 0xdd15fe86affad912u, 0x49ef0eb713f39ebeu,
|
||||
0x8a2dbf142dfcc7abu, 0x6e3569326c784337u, 0xacb92ed9397bf996u, 0x49c2c37f07965404u,
|
||||
0xd7e77a8f87daf7fbu, 0xdc33745ec97be906u, 0x86f0ac99b4e8dafdu, 0x69a028bb3ded71a3u,
|
||||
0xa8acd7c0222311bcu, 0xc40832ea0d68ce0cu, 0xd2d80db02aabd62bu, 0xf50a3fa490c30190u,
|
||||
0x83c7088e1aab65dbu, 0x792667c6da79e0fau, 0xa4b8cab1a1563f52u, 0x577001b891185938u,
|
||||
0xcde6fd5e09abcf26u, 0xed4c0226b55e6f86u, 0x80b05e5ac60b6178u, 0x544f8158315b05b4u,
|
||||
0xa0dc75f1778e39d6u, 0x696361ae3db1c721u, 0xc913936dd571c84cu, 0x03bc3a19cd1e38e9u,
|
||||
0xfb5878494ace3a5fu, 0x04ab48a04065c723u, 0x9d174b2dcec0e47bu, 0x62eb0d64283f9c76u,
|
||||
0xc45d1df942711d9au, 0x3ba5d0bd324f8394u, 0xf5746577930d6500u, 0xca8f44ec7ee36479u,
|
||||
0x9968bf6abbe85f20u, 0x7e998b13cf4e1ecbu, 0xbfc2ef456ae276e8u, 0x9e3fedd8c321a67eu,
|
||||
0xefb3ab16c59b14a2u, 0xc5cfe94ef3ea101eu, 0x95d04aee3b80ece5u, 0xbba1f1d158724a12u,
|
||||
0xbb445da9ca61281fu, 0x2a8a6e45ae8edc97u, 0xea1575143cf97226u, 0xf52d09d71a3293bdu,
|
||||
0x924d692ca61be758u, 0x593c2626705f9c56u, 0xb6e0c377cfa2e12eu, 0x6f8b2fb00c77836cu,
|
||||
0xe498f455c38b997au, 0x0b6dfb9c0f956447u, 0x8edf98b59a373fecu, 0x4724bd4189bd5eacu,
|
||||
0xb2977ee300c50fe7u, 0x58edec91ec2cb657u, 0xdf3d5e9bc0f653e1u, 0x2f2967b66737e3edu,
|
||||
0x8b865b215899f46cu, 0xbd79e0d20082ee74u, 0xae67f1e9aec07187u, 0xecd8590680a3aa11u,
|
||||
0xda01ee641a708de9u, 0xe80e6f4820cc9495u, 0x884134fe908658b2u, 0x3109058d147fdcddu,
|
||||
0xaa51823e34a7eedeu, 0xbd4b46f0599fd415u, 0xd4e5e2cdc1d1ea96u, 0x6c9e18ac7007c91au,
|
||||
0x850fadc09923329eu, 0x03e2cf6bc604ddb0u, 0xa6539930bf6bff45u, 0x84db8346b786151cu,
|
||||
0xcfe87f7cef46ff16u, 0xe612641865679a63u, 0x81f14fae158c5f6eu, 0x4fcb7e8f3f60c07eu,
|
||||
0xa26da3999aef7749u, 0xe3be5e330f38f09du, 0xcb090c8001ab551cu, 0x5cadf5bfd3072cc5u,
|
||||
0xfdcb4fa002162a63u, 0x73d9732fc7c8f7f6u, 0x9e9f11c4014dda7eu, 0x2867e7fddcdd9afau,
|
||||
0xc646d63501a1511du, 0xb281e1fd541501b8u, 0xf7d88bc24209a565u, 0x1f225a7ca91a4226u,
|
||||
0x9ae757596946075fu, 0x3375788de9b06958u, 0xc1a12d2fc3978937u, 0x0052d6b1641c83aeu,
|
||||
0xf209787bb47d6b84u, 0xc0678c5dbd23a49au, 0x9745eb4d50ce6332u, 0xf840b7ba963646e0u,
|
||||
0xbd176620a501fbffu, 0xb650e5a93bc3d898u, 0xec5d3fa8ce427affu, 0xa3e51f138ab4cebeu,
|
||||
0x93ba47c980e98cdfu, 0xc66f336c36b10137u, 0xb8a8d9bbe123f017u, 0xb80b0047445d4184u,
|
||||
0xe6d3102ad96cec1du, 0xa60dc059157491e5u, 0x9043ea1ac7e41392u, 0x87c89837ad68db2fu,
|
||||
0xb454e4a179dd1877u, 0x29babe4598c311fbu, 0xe16a1dc9d8545e94u, 0xf4296dd6fef3d67au,
|
||||
0x8ce2529e2734bb1du, 0x1899e4a65f58660cu, 0xb01ae745b101e9e4u, 0x5ec05dcff72e7f8fu,
|
||||
0xdc21a1171d42645du, 0x76707543f4fa1f73u, 0x899504ae72497ebau, 0x6a06494a791c53a8u,
|
||||
0xabfa45da0edbde69u, 0x0487db9d17636892u, 0xd6f8d7509292d603u, 0x45a9d2845d3c42b6u,
|
||||
0x865b86925b9bc5c2u, 0x0b8a2392ba45a9b2u, 0xa7f26836f282b732u, 0x8e6cac7768d7141eu,
|
||||
0xd1ef0244af2364ffu, 0x3207d795430cd926u, 0x8335616aed761f1fu, 0x7f44e6bd49e807b8u,
|
||||
0xa402b9c5a8d3a6e7u, 0x5f16206c9c6209a6u, 0xcd036837130890a1u, 0x36dba887c37a8c0fu,
|
||||
0x802221226be55a64u, 0xc2494954da2c9789u, 0xa02aa96b06deb0fdu, 0xf2db9baa10b7bd6cu,
|
||||
0xc83553c5c8965d3du, 0x6f92829494e5acc7u, 0xfa42a8b73abbf48cu, 0xcb772339ba1f17f9u,
|
||||
0x9c69a97284b578d7u, 0xff2a760414536efbu, 0xc38413cf25e2d70du, 0xfef5138519684abau,
|
||||
0xf46518c2ef5b8cd1u, 0x7eb258665fc25d69u, 0x98bf2f79d5993802u, 0xef2f773ffbd97a61u,
|
||||
0xbeeefb584aff8603u, 0xaafb550ffacfd8fau, 0xeeaaba2e5dbf6784u, 0x95ba2a53f983cf38u,
|
||||
0x952ab45cfa97a0b2u, 0xdd945a747bf26183u, 0xba756174393d88dfu, 0x94f971119aeef9e4u,
|
||||
0xe912b9d1478ceb17u, 0x7a37cd5601aab85du, 0x91abb422ccb812eeu, 0xac62e055c10ab33au,
|
||||
0xb616a12b7fe617aau, 0x577b986b314d6009u, 0xe39c49765fdf9d94u, 0xed5a7e85fda0b80bu,
|
||||
0x8e41ade9fbebc27du, 0x14588f13be847307u, 0xb1d219647ae6b31cu, 0x596eb2d8ae258fc8u,
|
||||
0xde469fbd99a05fe3u, 0x6fca5f8ed9aef3bbu, 0x8aec23d680043beeu, 0x25de7bb9480d5854u,
|
||||
0xada72ccc20054ae9u, 0xaf561aa79a10ae6au, 0xd910f7ff28069da4u, 0x1b2ba1518094da04u,
|
||||
0x87aa9aff79042286u, 0x90fb44d2f05d0842u, 0xa99541bf57452b28u, 0x353a1607ac744a53u,
|
||||
0xd3fa922f2d1675f2u, 0x42889b8997915ce8u, 0x847c9b5d7c2e09b7u, 0x69956135febada11u,
|
||||
0xa59bc234db398c25u, 0x43fab9837e699095u, 0xcf02b2c21207ef2eu, 0x94f967e45e03f4bbu,
|
||||
0x8161afb94b44f57du, 0x1d1be0eebac278f5u, 0xa1ba1ba79e1632dcu, 0x6462d92a69731732u,
|
||||
0xca28a291859bbf93u, 0x7d7b8f7503cfdcfeu, 0xfcb2cb35e702af78u, 0x5cda735244c3d43eu,
|
||||
0x9defbf01b061adabu, 0x3a0888136afa64a7u, 0xc56baec21c7a1916u, 0x088aaa1845b8fdd0u,
|
||||
0xf6c69a72a3989f5bu, 0x8aad549e57273d45u, 0x9a3c2087a63f6399u, 0x36ac54e2f678864bu,
|
||||
0xc0cb28a98fcf3c7fu, 0x84576a1bb416a7ddu, 0xf0fdf2d3f3c30b9fu, 0x656d44a2a11c51d5u,
|
||||
0x969eb7c47859e743u, 0x9f644ae5a4b1b325u, 0xbc4665b596706114u, 0x873d5d9f0dde1feeu,
|
||||
0xeb57ff22fc0c7959u, 0xa90cb506d155a7eau, 0x9316ff75dd87cbd8u, 0x09a7f12442d588f2u,
|
||||
0xb7dcbf5354e9beceu, 0x0c11ed6d538aeb2fu, 0xe5d3ef282a242e81u, 0x8f1668c8a86da5fau,
|
||||
0x8fa475791a569d10u, 0xf96e017d694487bcu, 0xb38d92d760ec4455u, 0x37c981dcc395a9acu,
|
||||
0xe070f78d3927556au, 0x85bbe253f47b1417u, 0x8c469ab843b89562u, 0x93956d7478ccec8eu,
|
||||
0xaf58416654a6babbu, 0x387ac8d1970027b2u, 0xdb2e51bfe9d0696au, 0x06997b05fcc0319eu,
|
||||
0x88fcf317f22241e2u, 0x441fece3bdf81f03u, 0xab3c2fddeeaad25au, 0xd527e81cad7626c3u,
|
||||
0xd60b3bd56a5586f1u, 0x8a71e223d8d3b074u, 0x85c7056562757456u, 0xf6872d5667844e49u,
|
||||
0xa738c6bebb12d16cu, 0xb428f8ac016561dbu, 0xd106f86e69d785c7u, 0xe13336d701beba52u,
|
||||
0x82a45b450226b39cu, 0xecc0024661173473u, 0xa34d721642b06084u, 0x27f002d7f95d0190u,
|
||||
0xcc20ce9bd35c78a5u, 0x31ec038df7b441f4u, 0xff290242c83396ceu, 0x7e67047175a15271u,
|
||||
0x9f79a169bd203e41u, 0x0f0062c6e984d386u, 0xc75809c42c684dd1u, 0x52c07b78a3e60868u,
|
||||
0xf92e0c3537826145u, 0xa7709a56ccdf8a82u, 0x9bbcc7a142b17ccbu, 0x88a66076400bb691u,
|
||||
0xc2abf989935ddbfeu, 0x6acff893d00ea435u, 0xf356f7ebf83552feu, 0x0583f6b8c4124d43u,
|
||||
0x98165af37b2153deu, 0xc3727a337a8b704au, 0xbe1bf1b059e9a8d6u, 0x744f18c0592e4c5cu,
|
||||
0xeda2ee1c7064130cu, 0x1162def06f79df73u, 0x9485d4d1c63e8be7u, 0x8addcb5645ac2ba8u,
|
||||
0xb9a74a0637ce2ee1u, 0x6d953e2bd7173692u, 0xe8111c87c5c1ba99u, 0xc8fa8db6ccdd0437u,
|
||||
0x910ab1d4db9914a0u, 0x1d9c9892400a22a2u, 0xb54d5e4a127f59c8u, 0x2503beb6d00cab4bu,
|
||||
0xe2a0b5dc971f303au, 0x2e44ae64840fd61du, 0x8da471a9de737e24u, 0x5ceaecfed289e5d2u,
|
||||
0xb10d8e1456105dadu, 0x7425a83e872c5f47u, 0xdd50f1996b947518u, 0xd12f124e28f77719u,
|
||||
0x8a5296ffe33cc92fu, 0x82bd6b70d99aaa6fu, 0xace73cbfdc0bfb7bu, 0x636cc64d1001550bu,
|
||||
0xd8210befd30efa5au, 0x3c47f7e05401aa4eu, 0x8714a775e3e95c78u, 0x65acfaec34810a71u,
|
||||
0xa8d9d1535ce3b396u, 0x7f1839a741a14d0du, 0xd31045a8341ca07cu, 0x1ede48111209a050u,
|
||||
0x83ea2b892091e44du, 0x934aed0aab460432u, 0xa4e4b66b68b65d60u, 0xf81da84d5617853fu,
|
||||
0xce1de40642e3f4b9u, 0x36251260ab9d668eu, 0x80d2ae83e9ce78f3u, 0xc1d72b7c6b426019u,
|
||||
0xa1075a24e4421730u, 0xb24cf65b8612f81fu, 0xc94930ae1d529cfcu, 0xdee033f26797b627u,
|
||||
0xfb9b7cd9a4a7443cu, 0x169840ef017da3b1u, 0x9d412e0806e88aa5u, 0x8e1f289560ee864eu,
|
||||
0xc491798a08a2ad4eu, 0xf1a6f2bab92a27e2u, 0xf5b5d7ec8acb58a2u, 0xae10af696774b1dbu,
|
||||
0x9991a6f3d6bf1765u, 0xacca6da1e0a8ef29u, 0xbff610b0cc6edd3fu, 0x17fd090a58d32af3u,
|
||||
0xeff394dcff8a948eu, 0xddfc4b4cef07f5b0u, 0x95f83d0a1fb69cd9u, 0x4abdaf101564f98eu,
|
||||
0xbb764c4ca7a4440fu, 0x9d6d1ad41abe37f1u, 0xea53df5fd18d5513u, 0x84c86189216dc5edu,
|
||||
0x92746b9be2f8552cu, 0x32fd3cf5b4e49bb4u, 0xb7118682dbb66a77u, 0x3fbc8c33221dc2a1u,
|
||||
0xe4d5e82392a40515u, 0x0fabaf3feaa5334au, 0x8f05b1163ba6832du, 0x29cb4d87f2a7400eu,
|
||||
0xb2c71d5bca9023f8u, 0x743e20e9ef511012u, 0xdf78e4b2bd342cf6u, 0x914da9246b255416u,
|
||||
0x8bab8eefb6409c1au, 0x1ad089b6c2f7548eu, 0xae9672aba3d0c320u, 0xa184ac2473b529b1u,
|
||||
0xda3c0f568cc4f3e8u, 0xc9e5d72d90a2741eu, 0x8865899617fb1871u, 0x7e2fa67c7a658892u,
|
||||
0xaa7eebfb9df9de8du, 0xddbb901b98feeab7u, 0xd51ea6fa85785631u, 0x552a74227f3ea565u,
|
||||
0x8533285c936b35deu, 0xd53a88958f87275fu, 0xa67ff273b8460356u, 0x8a892abaf368f137u,
|
||||
0xd01fef10a657842cu, 0x2d2b7569b0432d85u, 0x8213f56a67f6b29bu, 0x9c3b29620e29fc73u,
|
||||
0xa298f2c501f45f42u, 0x8349f3ba91b47b8fu, 0xcb3f2f7642717713u, 0x241c70a936219a73u,
|
||||
0xfe0efb53d30dd4d7u, 0xed238cd383aa0110u, 0x9ec95d1463e8a506u, 0xf4363804324a40aau,
|
||||
0xc67bb4597ce2ce48u, 0xb143c6053edcd0d5u, 0xf81aa16fdc1b81dau, 0xdd94b7868e94050au,
|
||||
0x9b10a4e5e9913128u, 0xca7cf2b4191c8326u, 0xc1d4ce1f63f57d72u, 0xfd1c2f611f63a3f0u,
|
||||
0xf24a01a73cf2dccfu, 0xbc633b39673c8cecu, 0x976e41088617ca01u, 0xd5be0503e085d813u,
|
||||
0xbd49d14aa79dbc82u, 0x4b2d8644d8a74e18u, 0xec9c459d51852ba2u, 0xddf8e7d60ed1219eu,
|
||||
0x93e1ab8252f33b45u, 0xcabb90e5c942b503u, 0xb8da1662e7b00a17u, 0x3d6a751f3b936243u,
|
||||
0xe7109bfba19c0c9du, 0x0cc512670a783ad4u, 0x906a617d450187e2u, 0x27fb2b80668b24c5u,
|
||||
0xb484f9dc9641e9dau, 0xb1f9f660802dedf6u, 0xe1a63853bbd26451u, 0x5e7873f8a0396973u,
|
||||
0x8d07e33455637eb2u, 0xdb0b487b6423e1e8u, 0xb049dc016abc5e5fu, 0x91ce1a9a3d2cda62u,
|
||||
0xdc5c5301c56b75f7u, 0x7641a140cc7810fbu, 0x89b9b3e11b6329bau, 0xa9e904c87fcb0a9du,
|
||||
0xac2820d9623bf429u, 0x546345fa9fbdcd44u, 0xd732290fbacaf133u, 0xa97c177947ad4095u,
|
||||
0x867f59a9d4bed6c0u, 0x49ed8eabcccc485du, 0xa81f301449ee8c70u, 0x5c68f256bfff5a74u,
|
||||
0xd226fc195c6a2f8cu, 0x73832eec6fff3111u, 0x83585d8fd9c25db7u, 0xc831fd53c5ff7eabu,
|
||||
0xa42e74f3d032f525u, 0xba3e7ca8b77f5e55u, 0xcd3a1230c43fb26fu, 0x28ce1bd2e55f35ebu,
|
||||
0x80444b5e7aa7cf85u, 0x7980d163cf5b81b3u, 0xa0555e361951c366u, 0xd7e105bcc332621fu,
|
||||
0xc86ab5c39fa63440u, 0x8dd9472bf3fefaa7u, 0xfa856334878fc150u, 0xb14f98f6f0feb951u,
|
||||
0x9c935e00d4b9d8d2u, 0x6ed1bf9a569f33d3u, 0xc3b8358109e84f07u, 0x0a862f80ec4700c8u,
|
||||
0xf4a642e14c6262c8u, 0xcd27bb612758c0fau, 0x98e7e9cccfbd7dbdu, 0x8038d51cb897789cu,
|
||||
0xbf21e44003acdd2cu, 0xe0470a63e6bd56c3u, 0xeeea5d5004981478u, 0x1858ccfce06cac74u,
|
||||
0x95527a5202df0ccbu, 0x0f37801e0c43ebc8u, 0xbaa718e68396cffdu, 0xd30560258f54e6bau,
|
||||
0xe950df20247c83fdu, 0x47c6b82ef32a2069u, 0x91d28b7416cdd27eu, 0x4cdc331d57fa5441u,
|
||||
0xb6472e511c81471du, 0xe0133fe4adf8e952u, 0xe3d8f9e563a198e5u, 0x58180fddd97723a6u,
|
||||
0x8e679c2f5e44ff8fu, 0x570f09eaa7ea7648u,
|
||||
}
|
||||
};
|
||||
return table;
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -12,7 +12,6 @@
|
||||
#include <cstdint> // uint64_t
|
||||
#include <cstring> // memcpy
|
||||
|
||||
#include <nlohmann/detail/bit_ops.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in external
|
||||
@@ -70,12 +69,18 @@ inline std::size_t find_string_special(const unsigned char* data, std::size_t n)
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
const std::uint64_t special = swar_string_special(read_eight_bytes(data + i));
|
||||
if (special != 0)
|
||||
std::uint64_t word = 0;
|
||||
std::memcpy(&word, data + i, sizeof(word));
|
||||
if (swar_string_special(word) != 0)
|
||||
{
|
||||
// the lowest flagged byte is the first special one: the borrows of
|
||||
// the subtractions can only flag bytes above a true hit
|
||||
return i + (static_cast<std::size_t>(count_trailing_zeros(special)) / 8);
|
||||
// a special byte is in this word; locate it (endian-agnostic)
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
if (is_string_special(data[i + j]))
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
@@ -109,7 +114,8 @@ inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
const std::uint64_t v = read_eight_bytes(data + i);
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, data + i, sizeof(v));
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
||||
const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F)
|
||||
@@ -120,9 +126,7 @@ inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_
|
||||
| (v & high); // >= 0x80
|
||||
if (stop != 0)
|
||||
{
|
||||
// the lowest flagged byte is the first one to stop at (see
|
||||
// find_string_special())
|
||||
return i + (static_cast<std::size_t>(count_trailing_zeros(stop)) / 8);
|
||||
break;
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
@@ -249,18 +253,12 @@ inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t
|
||||
{
|
||||
break; // end of buffer, or a quote/escape/control byte
|
||||
}
|
||||
// a run of multi-byte sequences (e.g. CJK text) is validated sequence
|
||||
// by sequence without searching for the next special byte in between
|
||||
do
|
||||
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
||||
if (seq == 0)
|
||||
{
|
||||
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
||||
if (seq == 0)
|
||||
{
|
||||
return pos; // ill-formed or truncated: let the byte path diagnose it
|
||||
}
|
||||
pos += seq;
|
||||
break; // ill-formed or truncated: let the byte path diagnose it
|
||||
}
|
||||
while (pos < n && data[pos] >= 0x80u);
|
||||
pos += seq;
|
||||
}
|
||||
return pos;
|
||||
}
|
||||
@@ -275,7 +273,8 @@ inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
const std::uint64_t v = read_eight_bytes(data + i);
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, data + i, sizeof(v));
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull;
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull;
|
||||
const std::uint64_t hit = ((q - ones) & ~q & high)
|
||||
@@ -283,8 +282,14 @@ inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t
|
||||
| ((v - 0x2020202020202020ull) & ~v & high);
|
||||
if (hit != 0)
|
||||
{
|
||||
// the lowest flagged byte is the first delimiter (see find_string_special())
|
||||
return i + (static_cast<std::size_t>(count_trailing_zeros(hit)) / 8);
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
const unsigned char c = data[i + j];
|
||||
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t, uint32_t
|
||||
#include <string> // string, to_string
|
||||
#include <utility> // move
|
||||
|
||||
#include <nlohmann/detail/abi_macros.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
@@ -133,5 +134,78 @@ inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noex
|
||||
return state == UTF8_ACCEPT;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief append U+FFFD REPLACEMENT CHARACTER, encoded in UTF-8
|
||||
@param[in,out] s the string to append to
|
||||
*/
|
||||
template<typename StringType>
|
||||
inline void append_replacement_character(StringType& s)
|
||||
{
|
||||
s.push_back(static_cast<typename StringType::value_type>(0xEFu));
|
||||
s.push_back(static_cast<typename StringType::value_type>(0xBFu));
|
||||
s.push_back(static_cast<typename StringType::value_type>(0xBDu));
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief replace ill-formed UTF-8 with U+FFFD REPLACEMENT CHARACTER
|
||||
|
||||
Each maximal subpart of an ill-formed sequence becomes one U+FFFD, as the
|
||||
Unicode Standard recommends (Section 3.9, "U+FFFD Substitution of Maximal
|
||||
Subparts"), and as the parser for JSON text does when it recovers from errors.
|
||||
|
||||
@param[in,out] s the string to repair
|
||||
@param[in] first index of the first byte to repair; the bytes before it are
|
||||
assumed to be valid UTF-8 that ends on a code point boundary
|
||||
*/
|
||||
template<typename StringType>
|
||||
inline void replace_invalid_utf8(StringType& s, const std::size_t first = 0)
|
||||
{
|
||||
StringType result = s;
|
||||
result.resize(first);
|
||||
|
||||
std::uint8_t state = UTF8_ACCEPT;
|
||||
std::uint32_t codepoint = 0;
|
||||
// the first byte of the sequence being decoded
|
||||
std::size_t sequence_start = first;
|
||||
|
||||
std::size_t i = first;
|
||||
while (i < s.size())
|
||||
{
|
||||
switch (decode(state, codepoint, static_cast<std::uint8_t>(s[i])))
|
||||
{
|
||||
case UTF8_ACCEPT:
|
||||
for (++i; sequence_start < i; ++sequence_start)
|
||||
{
|
||||
result.push_back(s[sequence_start]);
|
||||
}
|
||||
break;
|
||||
|
||||
case UTF8_REJECT:
|
||||
append_replacement_character(result);
|
||||
// the byte that made the sequence ill-formed begins the next
|
||||
// one, unless it began this one
|
||||
if (i == sequence_start)
|
||||
{
|
||||
++i;
|
||||
}
|
||||
state = UTF8_ACCEPT;
|
||||
sequence_start = i;
|
||||
break;
|
||||
|
||||
default: // in the middle of a sequence
|
||||
++i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// a sequence that the string ends in the middle of
|
||||
if (state != UTF8_ACCEPT)
|
||||
{
|
||||
append_replacement_character(result);
|
||||
}
|
||||
|
||||
s = std::move(result);
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -143,7 +143,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
friend class ::nlohmann::detail::iter_impl;
|
||||
template<typename BasicJsonType, typename CharType, typename OutputSinkType>
|
||||
friend class ::nlohmann::detail::binary_writer;
|
||||
template<typename BasicJsonType, typename InputType, typename SAX>
|
||||
template<typename BasicJsonType, typename InputType, typename SAX, bool AllowRecovery>
|
||||
friend class ::nlohmann::detail::binary_reader;
|
||||
template<typename BasicJsonType, typename InputAdapterType>
|
||||
friend class ::nlohmann::detail::json_sax_dom_parser;
|
||||
@@ -4992,7 +4992,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
auto ia = detail::input_adapter(std::forward<InputType>(i));
|
||||
return format == input_format_t::json
|
||||
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX>(std::move(ia), format).sax_parse(format, sax, strict);
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX, true>(std::move(ia), format).sax_parse(format, sax, strict);
|
||||
}
|
||||
|
||||
/// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
@@ -5009,7 +5009,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
auto ia = detail::input_adapter(std::move(first), std::move(last));
|
||||
return format == input_format_t::json
|
||||
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX>(std::move(ia), format).sax_parse(format, sax, strict);
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX, true>(std::move(ia), format).sax_parse(format, sax, strict);
|
||||
}
|
||||
|
||||
/// @brief generate SAX events
|
||||
@@ -5031,7 +5031,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
// NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg)
|
||||
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
||||
// NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg)
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX>(std::move(ia), format).sax_parse(format, sax, strict);
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX, true>(std::move(ia), format).sax_parse(format, sax, strict);
|
||||
}
|
||||
#ifndef JSON_NO_IO
|
||||
/// @brief deserialize from stream
|
||||
|
||||
+2675
-1097
File diff suppressed because it is too large
Load Diff
@@ -46,6 +46,7 @@ inline namespace json_literals
|
||||
namespace detail
|
||||
{
|
||||
using NLOHMANN_JSON_NAMESPACE::detail::json_sax_dom_callback_parser;
|
||||
using NLOHMANN_JSON_NAMESPACE::detail::json_sax_dom_parser;
|
||||
using NLOHMANN_JSON_NAMESPACE::detail::unknown_size;
|
||||
} // namespace detail
|
||||
|
||||
|
||||
@@ -45,6 +45,10 @@ dumps is stable under exactly the same values that break operator==.
|
||||
The unit tests run the same checks on a fixed corpus (see the "BJData round-trip
|
||||
invariants" test case), so keep both in sync.
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_bjdata() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -59,6 +63,8 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// value-stable comparison for the round-trip checks below; see the note
|
||||
@@ -71,11 +77,15 @@ static bool is_value_stable(const json& lhs, const json& rhs)
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// step 0: recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bjdata).errors == 0;
|
||||
|
||||
try
|
||||
{
|
||||
// step 1: parse input
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
json const j1 = json::from_bjdata(vec1);
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
@@ -109,6 +119,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::parse_error&)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -117,6 +128,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::out_of_range&)
|
||||
{
|
||||
// out of range errors may happen if provided sizes are excessive
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -19,6 +19,10 @@ It also checks that reading the data from a stream, which reads strings byte by
|
||||
byte, gives the same value or error as reading it from contiguous memory, which
|
||||
copies strings in bulk.
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_bon8() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -33,6 +37,8 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
namespace
|
||||
@@ -56,6 +62,9 @@ std::string read_bon8(InputType&& input)
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// step 0: recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bon8).errors == 0;
|
||||
|
||||
// contiguous and stream input must be read alike
|
||||
{
|
||||
std::istringstream stream(std::string(reinterpret_cast<const char*>(data), size));
|
||||
@@ -67,6 +76,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
// step 1: parse input
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
json const j1 = json::from_bon8(vec1);
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
@@ -88,6 +98,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::parse_error&)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -96,6 +107,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::out_of_range&)
|
||||
{
|
||||
// out of range errors may happen if provided sizes are excessive
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -15,6 +15,10 @@ array data, it performs the following steps:
|
||||
- j2 = from_bson(vec)
|
||||
- assert(j1 == j2)
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_bson() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -29,16 +33,22 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// step 0: recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bson).errors == 0;
|
||||
|
||||
try
|
||||
{
|
||||
// step 1: parse input
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
json const j1 = json::from_bson(vec1);
|
||||
assert(recovered_without_errors);
|
||||
|
||||
if (j1.is_discarded())
|
||||
{
|
||||
@@ -65,6 +75,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::parse_error&)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -73,6 +84,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::out_of_range&)
|
||||
{
|
||||
// out of range errors can occur during parsing, too
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -15,6 +15,10 @@ array data, it performs the following steps:
|
||||
- j2 = from_cbor(vec)
|
||||
- assert(j1 == j2)
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_cbor() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -29,16 +33,22 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// step 0: recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::cbor).errors == 0;
|
||||
|
||||
try
|
||||
{
|
||||
// step 1: parse input
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
json const j1 = json::from_cbor(vec1);
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
@@ -60,6 +70,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::parse_error&)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -68,6 +79,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::out_of_range&)
|
||||
{
|
||||
// out of range errors can occur during parsing, too
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -16,6 +16,10 @@ array data, it performs the following steps:
|
||||
- s2 = serialize(j2)
|
||||
- assert(s1 == s2)
|
||||
|
||||
Furthermore, it parses data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that parsing ends, and that valid
|
||||
input is parsed without errors (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -23,6 +27,7 @@ drivers.
|
||||
#include <cassert>
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
// the round-trip checks below are assertions; NDEBUG would compile them away
|
||||
@@ -30,11 +35,20 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// step 0: recover from all errors, reading from memory and from a stream
|
||||
{
|
||||
const auto checker = check_recovering_parse(data, size, json::input_format_t::json);
|
||||
assert(checker.events <= (4 * size) + 4);
|
||||
assert((checker.errors == 0) == json::accept(data, data + size));
|
||||
}
|
||||
|
||||
try
|
||||
{
|
||||
// step 1: parse input
|
||||
|
||||
@@ -15,6 +15,10 @@ array data, it performs the following steps:
|
||||
- j2 = from_msgpack(vec)
|
||||
- assert(j1 == j2)
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_msgpack() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -29,16 +33,22 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// step 0: recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::msgpack).errors == 0;
|
||||
|
||||
try
|
||||
{
|
||||
// step 1: parse input
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
json const j1 = json::from_msgpack(vec1);
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
@@ -60,6 +70,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::parse_error&)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -68,6 +79,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::out_of_range&)
|
||||
{
|
||||
// out of range errors may happen if provided sizes are excessive
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -24,6 +24,10 @@ array data, it performs the following steps:
|
||||
The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip
|
||||
invariants" test case), so keep both in sync.
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_ubjson() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -38,16 +42,22 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// step 0: recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::ubjson).errors == 0;
|
||||
|
||||
try
|
||||
{
|
||||
// step 1: parse input
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
json const j1 = json::from_ubjson(vec1);
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
@@ -79,6 +89,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::parse_error&)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -87,6 +98,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
catch (const json::out_of_range&)
|
||||
{
|
||||
// out of range errors may happen if provided sizes are excessive
|
||||
assert(!recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cassert>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
namespace
|
||||
{
|
||||
// a SAX parser that recovers from every error and checks that the events are
|
||||
// balanced and that every key is followed by exactly one value
|
||||
class recovering_checker : public nlohmann::json_sax<nlohmann::json>
|
||||
{
|
||||
public:
|
||||
bool null() override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool boolean(bool /*val*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool number_integer(number_integer_t /*val*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool number_unsigned(number_unsigned_t /*val*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool number_float(number_float_t /*val*/, const string_t& /*s*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool string(string_t& /*val*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool binary(binary_t& /*val*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool start_object(std::size_t /*elements*/) override
|
||||
{
|
||||
value();
|
||||
stack.push_back('o');
|
||||
return true;
|
||||
}
|
||||
|
||||
bool key(string_t& /*val*/) override
|
||||
{
|
||||
++events;
|
||||
assert(!stack.empty() && stack.back() == 'o');
|
||||
stack.back() = 'v';
|
||||
return true;
|
||||
}
|
||||
|
||||
bool end_object() override
|
||||
{
|
||||
++events;
|
||||
assert(!stack.empty() && stack.back() == 'o');
|
||||
stack.pop_back();
|
||||
return true;
|
||||
}
|
||||
|
||||
bool start_array(std::size_t /*elements*/) override
|
||||
{
|
||||
value();
|
||||
stack.push_back('a');
|
||||
return true;
|
||||
}
|
||||
|
||||
bool end_array() override
|
||||
{
|
||||
++events;
|
||||
assert(!stack.empty() && stack.back() == 'a');
|
||||
stack.pop_back();
|
||||
return true;
|
||||
}
|
||||
|
||||
bool parse_error(std::size_t /*position*/, const std::string& /*last_token*/, const nlohmann::detail::exception& /*ex*/) override
|
||||
{
|
||||
++errors;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool complete() const
|
||||
{
|
||||
return stack.empty();
|
||||
}
|
||||
|
||||
std::size_t events = 0;
|
||||
std::size_t errors = 0;
|
||||
|
||||
private:
|
||||
bool value()
|
||||
{
|
||||
++events;
|
||||
if (!stack.empty())
|
||||
{
|
||||
// an array element, or the value of a key
|
||||
assert(stack.back() != 'o');
|
||||
if (stack.back() == 'v')
|
||||
{
|
||||
stack.back() = 'o';
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// 'a' for an array, 'o' for an object that expects a key, 'v' for an
|
||||
// object that expects the value of a key
|
||||
std::vector<char> stack {}; // NOLINT(readability-redundant-member-init)
|
||||
};
|
||||
|
||||
/// parses @a data with a recovering_checker from memory and from a stream,
|
||||
/// checks that both see the same, that the events are balanced, and that the
|
||||
/// number of errors is bounded, and returns the checker (see #3989)
|
||||
inline recovering_checker check_recovering_parse(const std::uint8_t* data, const std::size_t size, const nlohmann::json::input_format_t format)
|
||||
{
|
||||
recovering_checker checker;
|
||||
const bool ok = nlohmann::json::sax_parse(data, data + size, &checker, format);
|
||||
assert(checker.complete());
|
||||
assert(checker.errors <= size + 1);
|
||||
assert(ok == (checker.errors == 0));
|
||||
|
||||
std::istringstream stream(std::string(reinterpret_cast<const char*>(data), size));
|
||||
recovering_checker stream_checker;
|
||||
assert(nlohmann::json::sax_parse(stream, &stream_checker, format) == ok);
|
||||
assert(stream_checker.complete());
|
||||
assert(stream_checker.events == checker.events);
|
||||
assert(stream_checker.errors == checker.errors);
|
||||
|
||||
return checker;
|
||||
}
|
||||
} // namespace
|
||||
@@ -374,4 +374,41 @@ TEST_CASE("alternative string type")
|
||||
const auto j2 = j.flatten();
|
||||
CHECK(j2.dump() == R"({"/foo/0":"bar","/foo/1":"baz"})");
|
||||
}
|
||||
|
||||
SECTION("error recovery")
|
||||
{
|
||||
// a SAX parser that recovers from every error (see #3989)
|
||||
struct recovering_parser : nlohmann::detail::json_sax_dom_parser<alt_json>
|
||||
{
|
||||
explicit recovering_parser(alt_json& j)
|
||||
: nlohmann::detail::json_sax_dom_parser<alt_json>(j, false)
|
||||
{}
|
||||
|
||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const nlohmann::detail::exception& /*unused*/)
|
||||
{
|
||||
++errors;
|
||||
return true;
|
||||
}
|
||||
|
||||
std::size_t errors = 0;
|
||||
};
|
||||
|
||||
alt_json j;
|
||||
recovering_parser sax(j);
|
||||
// not inside CHECK(): MSVC reads the escape in a stringized raw string
|
||||
const std::string input = R"([1., "a\qb", tru, {"k" 2}])";
|
||||
CHECK(!alt_json::sax_parse(input, &sax));
|
||||
CHECK(sax.errors == 4);
|
||||
CHECK(j.dump() == R"([1,"aqb",null,{"k":2}])");
|
||||
|
||||
// a UBJSON high-precision number, a CBOR key that is not a string
|
||||
alt_json u;
|
||||
recovering_parser ubjson_sax(u);
|
||||
CHECK(!alt_json::sax_parse(std::vector<std::uint8_t> {'[', 'H', 'i', 2, '1', '.', ']'}, &ubjson_sax, alt_json::input_format_t::ubjson));
|
||||
CHECK(u.dump() == "[1]");
|
||||
alt_json c;
|
||||
recovering_parser cbor_sax(c);
|
||||
CHECK(!alt_json::sax_parse(std::vector<std::uint8_t> {0xA2, 0x01, 0x02, 0x61, 'a', 0x03}, &cbor_sax, alt_json::input_format_t::cbor));
|
||||
CHECK(c.dump() == R"({"a":3})");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,14 +12,10 @@
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <cstdint> // uint32_t, uint64_t
|
||||
#include <cstdlib> // strtod
|
||||
#include <cstring> // memcpy
|
||||
#include <sstream> // stringstream
|
||||
#include <string> // string
|
||||
#include <utility> // pair
|
||||
#include <vector> // vector
|
||||
|
||||
namespace
|
||||
@@ -704,706 +700,3 @@ TEST_CASE("parse_float_fast declines what it cannot convert exactly")
|
||||
CHECK_FALSE(fast("1e23", out));
|
||||
CHECK_FALSE(fast("1e-23", out));
|
||||
}
|
||||
|
||||
namespace
|
||||
{
|
||||
// arbitrary-precision unsigned integers, just enough to recompute the table of
|
||||
// powers of five (little-endian 32-bit limbs)
|
||||
using big_uint = std::vector<std::uint32_t>;
|
||||
|
||||
void big_trim(big_uint& a)
|
||||
{
|
||||
while (!a.empty() && a.back() == 0)
|
||||
{
|
||||
a.pop_back();
|
||||
}
|
||||
}
|
||||
|
||||
big_uint big_from(std::uint64_t high, std::uint64_t low)
|
||||
{
|
||||
big_uint a = {static_cast<std::uint32_t>(low), static_cast<std::uint32_t>(low >> 32u),
|
||||
static_cast<std::uint32_t>(high), static_cast<std::uint32_t>(high >> 32u)
|
||||
};
|
||||
big_trim(a);
|
||||
return a;
|
||||
}
|
||||
|
||||
big_uint big_mul(const big_uint& a, const big_uint& b)
|
||||
{
|
||||
big_uint r(a.size() + b.size(), 0);
|
||||
for (std::size_t i = 0; i < a.size(); ++i)
|
||||
{
|
||||
std::uint64_t carry = 0;
|
||||
for (std::size_t j = 0; j < b.size(); ++j)
|
||||
{
|
||||
const std::uint64_t t = (static_cast<std::uint64_t>(a[i]) * b[j]) + r[i + j] + carry;
|
||||
r[i + j] = static_cast<std::uint32_t>(t);
|
||||
carry = t >> 32u;
|
||||
}
|
||||
r[i + b.size()] = static_cast<std::uint32_t>(carry);
|
||||
}
|
||||
big_trim(r);
|
||||
return r;
|
||||
}
|
||||
|
||||
big_uint big_shl(const big_uint& a, std::size_t s)
|
||||
{
|
||||
big_uint r(s / 32, 0);
|
||||
std::uint32_t carry = 0;
|
||||
for (const std::uint32_t x : a)
|
||||
{
|
||||
const std::uint64_t t = static_cast<std::uint64_t>(x) << (s % 32);
|
||||
r.push_back(static_cast<std::uint32_t>(t) | carry);
|
||||
carry = static_cast<std::uint32_t>(t >> 32u);
|
||||
}
|
||||
r.push_back(carry);
|
||||
big_trim(r);
|
||||
return r;
|
||||
}
|
||||
|
||||
// a + 1 (add) or a - 1 (!add, a > 0)
|
||||
big_uint big_step(big_uint a, bool add)
|
||||
{
|
||||
for (auto& x : a)
|
||||
{
|
||||
const std::uint32_t old = x;
|
||||
x = add ? x + 1 : x - 1;
|
||||
if ((add && x > old) || (!add && x < old))
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (add && (a.empty() || a.back() == 0))
|
||||
{
|
||||
a.push_back(1);
|
||||
}
|
||||
big_trim(a);
|
||||
return a;
|
||||
}
|
||||
|
||||
bool big_less_equal(const big_uint& a, const big_uint& b)
|
||||
{
|
||||
if (a.size() != b.size())
|
||||
{
|
||||
return a.size() < b.size();
|
||||
}
|
||||
for (std::size_t i = a.size(); i-- > 0;)
|
||||
{
|
||||
if (a[i] != b[i])
|
||||
{
|
||||
return a[i] < b[i];
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
std::size_t big_bit_length(const big_uint& a)
|
||||
{
|
||||
std::size_t n = a.size() * 32;
|
||||
for (std::uint32_t top = a.back(); (top & 0x80000000u) == 0; top <<= 1u)
|
||||
{
|
||||
--n;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
std::uint64_t bits_of(double d)
|
||||
{
|
||||
std::uint64_t b = 0;
|
||||
std::memcpy(&b, &d, sizeof(b));
|
||||
return b;
|
||||
}
|
||||
|
||||
bool eisel_lemire(const std::string& s, double& out)
|
||||
{
|
||||
return nlohmann::detail::parse_float_eisel_lemire(s.data(), s.data() + s.size(), out);
|
||||
}
|
||||
|
||||
// significant digits of a token, without trailing zeros
|
||||
std::size_t significant_digits(const std::string& s)
|
||||
{
|
||||
std::string digits;
|
||||
for (const char c : s)
|
||||
{
|
||||
if (c == 'e' || c == 'E')
|
||||
{
|
||||
break;
|
||||
}
|
||||
if (c >= '0' && c <= '9' && !(digits.empty() && c == '0'))
|
||||
{
|
||||
digits += c;
|
||||
}
|
||||
}
|
||||
while (!digits.empty() && digits.back() == '0')
|
||||
{
|
||||
digits.pop_back();
|
||||
}
|
||||
return digits.size();
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("Eisel-Lemire float conversion")
|
||||
{
|
||||
SECTION("the table of powers of five")
|
||||
{
|
||||
// Recompute every entry the way fast_float's table_generation.py
|
||||
// defines it, using only multiplications and comparisons: for q >= 0
|
||||
// the most significant 128 bits of 5^q; for q < 0 floor(2^b / 5^-q) + 1
|
||||
// for b = z + 127 (q >= -27), or that value for b = 2z + 128 cut to
|
||||
// its most significant 128 bits (q < -27), where z is the bit length
|
||||
// of 5^-q.
|
||||
const auto& table = nlohmann::detail::pow5_128();
|
||||
big_uint power5 = {1};
|
||||
for (std::int64_t q = 0; q <= nlohmann::detail::pow5_128_largest_power; ++q)
|
||||
{
|
||||
const auto index = static_cast<std::size_t>(2 * (q - nlohmann::detail::pow5_128_smallest_power));
|
||||
const big_uint entry = big_from(table[index], table[index + 1]);
|
||||
const std::size_t bits = big_bit_length(power5);
|
||||
if (bits <= 128)
|
||||
{
|
||||
CHECK(entry == big_shl(power5, 128 - bits));
|
||||
}
|
||||
else
|
||||
{
|
||||
// floor(5^q / 2^(bits - 128))
|
||||
CHECK(big_less_equal(big_shl(entry, bits - 128), power5));
|
||||
CHECK_FALSE(big_less_equal(big_shl(big_step(entry, true), bits - 128), power5));
|
||||
}
|
||||
power5 = big_mul(power5, {5});
|
||||
}
|
||||
|
||||
power5 = {5};
|
||||
for (std::int64_t q = -1; q >= nlohmann::detail::pow5_128_smallest_power; --q)
|
||||
{
|
||||
const auto index = static_cast<std::size_t>(2 * (q - nlohmann::detail::pow5_128_smallest_power));
|
||||
const big_uint entry = big_from(table[index], table[index + 1]);
|
||||
CHECK(big_bit_length(entry) == 128);
|
||||
const std::size_t z = big_bit_length(power5);
|
||||
const big_uint two_b = big_shl({1}, q >= -27 ? z + 127 : (2 * z) + 128);
|
||||
// c = floor(2^b / p) + 1, stored as floor(c / 2^s):
|
||||
// (entry * 2^s - 1) * p <= 2^b < ((entry + 1) * 2^s - 1) * p
|
||||
const std::size_t s = q >= -27 ? 0 : z + 1;
|
||||
CHECK(big_less_equal(big_mul(big_step(big_shl(entry, s), false), power5), two_b));
|
||||
CHECK_FALSE(big_less_equal(big_mul(big_step(big_shl(big_step(entry, true), s), false), power5), two_b));
|
||||
power5 = big_mul(power5, {5});
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("128-bit products and leading zeros")
|
||||
{
|
||||
// whichever implementation the compiler gets (with or without a
|
||||
// 128-bit integer type or a builtin)
|
||||
std::uint64_t state = 42;
|
||||
for (int i = 0; i < 10000; ++i)
|
||||
{
|
||||
state ^= state << 13u;
|
||||
state ^= state >> 7u;
|
||||
state ^= state << 17u;
|
||||
const std::uint64_t a = state;
|
||||
const std::uint64_t b = (state * 0x9E3779B97F4A7C15u) >> (i % 64);
|
||||
const auto product = nlohmann::detail::full_multiplication(a, b);
|
||||
CHECK(big_from(product.high, product.low) == big_mul(big_from(0, a), big_from(0, b)));
|
||||
|
||||
const int k = i % 64;
|
||||
const std::uint64_t x = (std::uint64_t{1} << k) | (a & ((std::uint64_t{1} << k) - 1));
|
||||
CHECK(nlohmann::detail::count_leading_zeros(x) == 63 - k);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("known values")
|
||||
{
|
||||
// Generated with Python, whose float() is correctly rounded:
|
||||
// cases = [<hard cases>, 2**53 + 2k + 1, 2**54 + 4k + 2, and exact midpoints
|
||||
// between neighbouring doubles, also 1e-60 above and below them]
|
||||
// print('{"%s", 0x%016xu},' % (s, struct.unpack('<Q', struct.pack('<d', float(s)))[0]))
|
||||
const std::vector<std::pair<std::string, std::uint64_t>> known =
|
||||
{
|
||||
{"0", 0x0000000000000000u},
|
||||
{"-0", 0x8000000000000000u},
|
||||
{"0.0", 0x0000000000000000u},
|
||||
{"-0.0", 0x8000000000000000u},
|
||||
{"0e5", 0x0000000000000000u},
|
||||
{"0.000e-9", 0x0000000000000000u},
|
||||
{"1", 0x3ff0000000000000u},
|
||||
{"-1", 0xbff0000000000000u},
|
||||
{"0.1", 0x3fb999999999999au},
|
||||
{"0.3", 0x3fd3333333333333u},
|
||||
{"1.5", 0x3ff8000000000000u},
|
||||
{"-2.5e-3", 0xbf647ae147ae147bu},
|
||||
{"1e23", 0x44b52d02c7e14af6u},
|
||||
{"1e22", 0x4480f0cf064dd592u},
|
||||
{"8.98846567431158e307", 0x7fe0000000000000u},
|
||||
{"2.2250738585072011e-308", 0x000fffffffffffffu},
|
||||
{"2.2250738585072012e-308", 0x0010000000000000u},
|
||||
{"2.2250738585072014e-308", 0x0010000000000000u},
|
||||
{"4.9406564584124654e-324", 0x0000000000000001u},
|
||||
{"2.4703282292062327e-324", 0x0000000000000000u},
|
||||
{"2.4703282292062328e-324", 0x0000000000000001u},
|
||||
{"1e-324", 0x0000000000000000u},
|
||||
{"3e-324", 0x0000000000000001u},
|
||||
{"1.7976931348623157e308", 0x7fefffffffffffffu},
|
||||
{"1.7976931348623158e308", 0x7fefffffffffffffu},
|
||||
{"1.7976931348623159e308", 0x7ff0000000000000u},
|
||||
{"1e308", 0x7fe1ccf385ebc8a0u},
|
||||
{"1e309", 0x7ff0000000000000u},
|
||||
{"-1e400", 0xfff0000000000000u},
|
||||
{"1e-400", 0x0000000000000000u},
|
||||
{"9007199254740991", 0x433fffffffffffffu},
|
||||
{"9007199254740992", 0x4340000000000000u},
|
||||
{"9007199254740993", 0x4340000000000000u},
|
||||
{"9007199254740995", 0x4340000000000002u},
|
||||
{"18014398509481986", 0x4350000000000000u},
|
||||
{"18014398509481990", 0x4350000000000002u},
|
||||
{"7.2057594037927933e16", 0x4370000000000000u},
|
||||
{"123456789012345678901234567890", 0x45f8ee90ff6c373eu},
|
||||
{"1.000000000000000111", 0x3ff0000000000000u},
|
||||
{"1.0000000000000001110223", 0x3ff0000000000000u},
|
||||
{"1.00000000000000011102230246251565404236316680908203125", 0x3ff0000000000000u},
|
||||
{"1.00000000000000011102230246251565404236316680908203126", 0x3ff0000000000001u},
|
||||
{"0.00000000000000000000000000000000000000000000000000000000000001", 0x3310747ddddf22a8u},
|
||||
{"100000000000000000000000000000000000000000000", 0x4911efc659cf7d4cu},
|
||||
{"1234567890123456789", 0x43b12210f47de981u},
|
||||
{"12345678901234567890", 0x43e56a95319d63e1u},
|
||||
{"1234567890123456789.5", 0x43b12210f47de981u},
|
||||
{"0.1234567890123456789012345", 0x3fbf9add3746f65fu},
|
||||
{"4.4501477170144023e-308", 0x001fffffffffffffu},
|
||||
{"2.4406961166466664e-309", 0x0001c14ae5310a48u},
|
||||
{"5e-324", 0x0000000000000001u},
|
||||
{"1.0e-307", 0x0031fa182c40c60du},
|
||||
{"179769313486231570814527423731704356798070567525844996598917476803157260780028538760589558632766878171540458953514382464234321326889464182768467546703537516986049910576551282076245490090389328944075868508455133942304583236903222948165808559332123348274797826204144723168738177180919299881250404026184124858368", 0x7fefffffffffffffu},
|
||||
{"4.9e-324", 0x0000000000000001u},
|
||||
{"9007199254740993", 0x4340000000000000u},
|
||||
{"18014398509481986", 0x4350000000000000u},
|
||||
{"9007199254740995", 0x4340000000000002u},
|
||||
{"18014398509481990", 0x4350000000000002u},
|
||||
{"9007199254740997", 0x4340000000000002u},
|
||||
{"18014398509481994", 0x4350000000000002u},
|
||||
{"9007199254740999", 0x4340000000000004u},
|
||||
{"18014398509481998", 0x4350000000000004u},
|
||||
{"9007199254741001", 0x4340000000000004u},
|
||||
{"18014398509482002", 0x4350000000000004u},
|
||||
{"9007199254741003", 0x4340000000000006u},
|
||||
{"18014398509482006", 0x4350000000000006u},
|
||||
{"9007199254741005", 0x4340000000000006u},
|
||||
{"18014398509482010", 0x4350000000000006u},
|
||||
{"9007199254741007", 0x4340000000000008u},
|
||||
{"18014398509482014", 0x4350000000000008u},
|
||||
{"9007199254741009", 0x4340000000000008u},
|
||||
{"18014398509482018", 0x4350000000000008u},
|
||||
{"9007199254741011", 0x434000000000000au},
|
||||
{"18014398509482022", 0x435000000000000au},
|
||||
{"9007199254741013", 0x434000000000000au},
|
||||
{"18014398509482026", 0x435000000000000au},
|
||||
{"9007199254741015", 0x434000000000000cu},
|
||||
{"18014398509482030", 0x435000000000000cu},
|
||||
{"9007199254741017", 0x434000000000000cu},
|
||||
{"18014398509482034", 0x435000000000000cu},
|
||||
{"9007199254741019", 0x434000000000000eu},
|
||||
{"18014398509482038", 0x435000000000000eu},
|
||||
{"9007199254741021", 0x434000000000000eu},
|
||||
{"18014398509482042", 0x435000000000000eu},
|
||||
{"9007199254741023", 0x4340000000000010u},
|
||||
{"18014398509482046", 0x4350000000000010u},
|
||||
{"9007199254741025", 0x4340000000000010u},
|
||||
{"18014398509482050", 0x4350000000000010u},
|
||||
{"9007199254741027", 0x4340000000000012u},
|
||||
{"18014398509482054", 0x4350000000000012u},
|
||||
{"9007199254741029", 0x4340000000000012u},
|
||||
{"18014398509482058", 0x4350000000000012u},
|
||||
{"9007199254741031", 0x4340000000000014u},
|
||||
{"18014398509482062", 0x4350000000000014u},
|
||||
{"9007199254741033", 0x4340000000000014u},
|
||||
{"18014398509482066", 0x4350000000000014u},
|
||||
{"9007199254741035", 0x4340000000000016u},
|
||||
{"18014398509482070", 0x4350000000000016u},
|
||||
{"9007199254741037", 0x4340000000000016u},
|
||||
{"18014398509482074", 0x4350000000000016u},
|
||||
{"9007199254741039", 0x4340000000000018u},
|
||||
{"18014398509482078", 0x4350000000000018u},
|
||||
{"9007199254741041", 0x4340000000000018u},
|
||||
{"18014398509482082", 0x4350000000000018u},
|
||||
{"9007199254741043", 0x434000000000001au},
|
||||
{"18014398509482086", 0x435000000000001au},
|
||||
{"9007199254741045", 0x434000000000001au},
|
||||
{"18014398509482090", 0x435000000000001au},
|
||||
{"9007199254741047", 0x434000000000001cu},
|
||||
{"18014398509482094", 0x435000000000001cu},
|
||||
{"9007199254741049", 0x434000000000001cu},
|
||||
{"18014398509482098", 0x435000000000001cu},
|
||||
{"9007199254741051", 0x434000000000001eu},
|
||||
{"18014398509482102", 0x435000000000001eu},
|
||||
{"9007199254741053", 0x434000000000001eu},
|
||||
{"18014398509482106", 0x435000000000001eu},
|
||||
{"9007199254741055", 0x4340000000000020u},
|
||||
{"18014398509482110", 0x4350000000000020u},
|
||||
{"9007199254741057", 0x4340000000000020u},
|
||||
{"18014398509482114", 0x4350000000000020u},
|
||||
{"9007199254741059", 0x4340000000000022u},
|
||||
{"18014398509482118", 0x4350000000000022u},
|
||||
{"9007199254741061", 0x4340000000000022u},
|
||||
{"18014398509482122", 0x4350000000000022u},
|
||||
{"9007199254741063", 0x4340000000000024u},
|
||||
{"18014398509482126", 0x4350000000000024u},
|
||||
{"9007199254741065", 0x4340000000000024u},
|
||||
{"18014398509482130", 0x4350000000000024u},
|
||||
{"9007199254741067", 0x4340000000000026u},
|
||||
{"18014398509482134", 0x4350000000000026u},
|
||||
{"9007199254741069", 0x4340000000000026u},
|
||||
{"18014398509482138", 0x4350000000000026u},
|
||||
{"9007199254741071", 0x4340000000000028u},
|
||||
{"18014398509482142", 0x4350000000000028u},
|
||||
{"0.00000000000000000142055942108419951085063380808124102279024543292671443374397544090470546507276594638824462890625", 0x3c3a3466f662d406u},
|
||||
{"0.000000000000000001420559421084199510850633808081241022790245432926714433743975440904705465072765946388244628906251", 0x3c3a3466f662d407u},
|
||||
{"0.00000000000000000142055942108419951085063380808124102279024543292671443374397444090470546507276594638824462890625", 0x3c3a3466f662d406u},
|
||||
{"8656.5250079159513916238211095333099365234375", 0x40c0e84333759a94u},
|
||||
{"8656.52500791595139162382110953330993652343751", 0x40c0e84333759a94u},
|
||||
{"8656.525007915951391623821109533309936523437499999999999999999", 0x40c0e84333759a93u},
|
||||
{"13.07696731650454946560557800694368779659271240234375", 0x402a276842967ef0u},
|
||||
{"13.076967316504549465605578006943687796592712402343751", 0x402a276842967ef0u},
|
||||
{"13.07696731650454946560557800694368779659271240234374999999999", 0x402a276842967eefu},
|
||||
{"74708253715391928", 0x437096ac2cc7ee5cu},
|
||||
{"747082537153919281", 0x43a4bc5737f9e9f2u},
|
||||
{"74708253715391927.99999999999999999999999999999999999999999999", 0x437096ac2cc7ee5bu},
|
||||
{"1809802988.27203977108001708984375", 0x41daf7d9bb11691au},
|
||||
{"1809802988.272039771080017089843751", 0x41daf7d9bb11691au},
|
||||
{"1809802988.272039771080017089843749999999999999999999999999999", 0x41daf7d9bb116919u},
|
||||
{"51.390809684186766759239617385901510715484619140625", 0x4049b2060d3e4568u},
|
||||
{"51.3908096841867667592396173859015107154846191406251", 0x4049b2060d3e4569u},
|
||||
{"51.39080968418676675923961738590151071548461914062499999999999", 0x4049b2060d3e4568u},
|
||||
{"9999807412.59738445281982421875", 0x4202a0479da4c772u},
|
||||
{"9999807412.597384452819824218751", 0x4202a0479da4c772u},
|
||||
{"9999807412.597384452819824218749999999999999999999999999999999", 0x4202a0479da4c771u},
|
||||
{"0.00000000023260971767101600534534272272645127367651785021962496102787554264068603515625", 0x3deff83a135dec10u},
|
||||
{"0.000000000232609717671016005345342722726451273676517850219624961027875542640686035156251", 0x3deff83a135dec11u},
|
||||
{"0.00000000023260971767101600534534272272645127367651785021962496102787544264068603515625", 0x3deff83a135dec10u},
|
||||
{"0.000000000497610482021202508216234789619066523902457532813059515319764614105224609375", 0x3e01190730d1ec48u},
|
||||
{"0.0000000004976104820212025082162347896190665239024575328130595153197646141052246093751", 0x3e01190730d1ec48u},
|
||||
{"0.000000000497610482021202508216234789619066523902457532813059515319764514105224609375", 0x3e01190730d1ec47u},
|
||||
{"0.0000000000291336422596533830691676779231223432149733287843673679162748157978057861328125", 0x3dc004321559736eu},
|
||||
{"0.00000000002913364225965338306916767792312234321497332878436736791627481579780578613281251", 0x3dc004321559736fu},
|
||||
{"0.0000000000291336422596533830691676779231223432149733287843673679162748057978057861328125", 0x3dc004321559736eu},
|
||||
{"0.000000000000000039237155154865396441907405399546892080260012902422940561653064150959835387766361236572265625", 0x3c869e61cfa3b8a4u},
|
||||
{"0.0000000000000000392371551548653964419074053995468920802600129024229405616530641509598353877663612365722656251", 0x3c869e61cfa3b8a5u},
|
||||
{"0.000000000000000039237155154865396441907405399546892080260012902422940561653054150959835387766361236572265625", 0x3c869e61cfa3b8a4u},
|
||||
{"0.00000000000000006496592863767266414092845207857846208631635335985395063307379359685000963509082794189453125", 0x3c92b9a3b219ee84u},
|
||||
{"0.000000000000000064965928637672664140928452078578462086316353359853950633073793596850009635090827941894531251", 0x3c92b9a3b219ee85u},
|
||||
{"0.00000000000000006496592863767266414092845207857846208631635335985395063307378359685000963509082794189453125", 0x3c92b9a3b219ee84u},
|
||||
{"0.0000000000448169607439179753541822628688597626549217078917308754171244800090789794921875", 0x3dc8a36d2e8094dau},
|
||||
{"0.00000000004481696074391797535418226286885976265492170789173087541712448000907897949218751", 0x3dc8a36d2e8094dau},
|
||||
{"0.0000000000448169607439179753541822628688597626549217078917308754171244700090789794921875", 0x3dc8a36d2e8094d9u},
|
||||
{"1800873890234250.875", 0x4319978a820cfe2cu},
|
||||
{"1800873890234250.8751", 0x4319978a820cfe2cu},
|
||||
{"1800873890234250.874999999999999999999999999999999999999999999", 0x4319978a820cfe2bu},
|
||||
{"0.000023941132739153309216405436654628857695570331998169422149658203125", 0x3ef91aa61d42e0e0u},
|
||||
{"0.0000239411327391533092164054366546288576955703319981694221496582031251", 0x3ef91aa61d42e0e1u},
|
||||
{"0.000023941132739153309216405436654628857695570331998169422149658193125", 0x3ef91aa61d42e0e0u},
|
||||
{"8339818978785937.5", 0x433da1056bb8ba92u},
|
||||
{"8339818978785937.51", 0x433da1056bb8ba92u},
|
||||
{"8339818978785937.499999999999999999999999999999999999999999999", 0x433da1056bb8ba91u},
|
||||
{"283649145986385328", 0x438f7dcb29dae42eu},
|
||||
{"2836491459863853281", 0x43c3ae9efa28ce9cu},
|
||||
{"283649145986385327.9999999999999999999999999999999999999999999", 0x438f7dcb29dae42du},
|
||||
{"0.0000000000000000004203729478287971874304476341685497759528753070014375965192388040492232903488911688327789306640625", 0x3c1f049ed78cb8a2u},
|
||||
{"0.00000000000000000042037294782879718743044763416854977595287530700143759651923880404922329034889116883277893066406251", 0x3c1f049ed78cb8a3u},
|
||||
{"0.0000000000000000004203729478287971874304476341685497759528753070014375965192387040492232903488911688327789306640625", 0x3c1f049ed78cb8a2u},
|
||||
{"340031.78635183119331486523151397705078125", 0x4114c0ff25396a18u},
|
||||
{"340031.786351831193314865231513977050781251", 0x4114c0ff25396a19u},
|
||||
{"340031.7863518311933148652315139770507812499999999999999999999", 0x4114c0ff25396a18u},
|
||||
{"0.0000000004543709717853330820053712055931242029538363880192264332436025142669677734375", 0x3dff3960f070bf76u},
|
||||
{"0.00000000045437097178533308200537120559312420295383638801922643324360251426696777343751", 0x3dff3960f070bf76u},
|
||||
{"0.0000000004543709717853330820053712055931242029538363880192264332436024142669677734375", 0x3dff3960f070bf75u},
|
||||
{"0.0000000000007937958898451257082591244100968761704503924570008877026339177973568439483642578125", 0x3d6bede0b40e37c2u},
|
||||
{"0.00000000000079379588984512570825912441009687617045039245700088770263391779735684394836425781251", 0x3d6bede0b40e37c3u},
|
||||
{"0.0000000000007937958898451257082591244100968761704503924570008877026339176973568439483642578125", 0x3d6bede0b40e37c2u},
|
||||
{"0.00000000000000068304843500200357021582520000166071595321259251644419041582523277611471712589263916015625", 0x3cc89c02848f8654u},
|
||||
{"0.000000000000000683048435002003570215825200001660715953212592516444190415825232776114717125892639160156251", 0x3cc89c02848f8655u},
|
||||
{"0.00000000000000068304843500200357021582520000166071595321259251644419041582513277611471712589263916015625", 0x3cc89c02848f8654u},
|
||||
{"0.00000000147705080029041786940146712084494760863773166192913777194917201995849609375", 0x3e1960235bc34d06u},
|
||||
{"0.000000001477050800290417869401467120844947608637731661929137771949172019958496093751", 0x3e1960235bc34d06u},
|
||||
{"0.00000000147705080029041786940146712084494760863773166192913777194917101995849609375", 0x3e1960235bc34d05u},
|
||||
{"2164972979236447104", 0x43be0b87743fb524u},
|
||||
{"21649729792364471041", 0x43f2c734a8a7d136u},
|
||||
{"2164972979236447103.999999999999999999999999999999999999999999", 0x43be0b87743fb523u},
|
||||
{"13124633159586767", 0x4347506464a469e8u},
|
||||
{"131246331595867671", 0x437d247d7dcd8461u},
|
||||
{"13124633159586766.99999999999999999999999999999999999999999999", 0x4347506464a469e7u},
|
||||
{"0.000000000000027605165650764659887597286048864477738779108113853499872902830247767269611358642578125", 0x3d1f14a5b417cb08u},
|
||||
{"0.0000000000000276051656507646598875972860488644777387791081138534998729028302477672696113586425781251", 0x3d1f14a5b417cb09u},
|
||||
{"0.000000000000027605165650764659887597286048864477738779108113853499872902820247767269611358642578125", 0x3d1f14a5b417cb08u},
|
||||
{"0.01727792833395616796388072344825559412129223346710205078125", 0x3f91b14e248c42c8u},
|
||||
{"0.017277928333956167963880723448255594121292233467102050781251", 0x3f91b14e248c42c9u},
|
||||
{"0.01727792833395616796388072344825559412129223346710205078124999", 0x3f91b14e248c42c8u},
|
||||
{"0.00000000000003096211381890547702500862566064822950373937142376501441276559489779174327850341796875", 0x3d216e1c61130b22u},
|
||||
{"0.000000000000030962113818905477025008625660648229503739371423765014412765594897791743278503417968751", 0x3d216e1c61130b22u},
|
||||
{"0.00000000000003096211381890547702500862566064822950373937142376501441276558489779174327850341796875", 0x3d216e1c61130b21u},
|
||||
{"0.0000000000000683414954481284270259359974108983284531919182025472281338807079009711742401123046875", 0x3d333c86137e5170u},
|
||||
{"0.00000000000006834149544812842702593599741089832845319191820254722813388070790097117424011230468751", 0x3d333c86137e5170u},
|
||||
{"0.0000000000000683414954481284270259359974108983284531919182025472281338806979009711742401123046875", 0x3d333c86137e516fu},
|
||||
{"3237539054990129.75", 0x4327010c9aa36664u},
|
||||
{"3237539054990129.751", 0x4327010c9aa36664u},
|
||||
{"3237539054990129.749999999999999999999999999999999999999999999", 0x4327010c9aa36663u},
|
||||
{"0.0000000000000105429301486355911969397173440055240907798901443814809653076736140064895153045654296875", 0x3d07bd95dfb8a1eeu},
|
||||
{"0.00000000000001054293014863559119693971734400552409077989014438148096530767361400648951530456542968751", 0x3d07bd95dfb8a1eeu},
|
||||
{"0.0000000000000105429301486355911969397173440055240907798901443814809653076636140064895153045654296875", 0x3d07bd95dfb8a1edu},
|
||||
{"9460.2061893763575426419265568256378173828125", 0x40c27a1a6469da1eu},
|
||||
{"9460.20618937635754264192655682563781738281251", 0x40c27a1a6469da1fu},
|
||||
{"9460.206189376357542641926556825637817382812499999999999999999", 0x40c27a1a6469da1eu},
|
||||
{"465000373656610.53125", 0x42fa6ea56179c228u},
|
||||
{"465000373656610.531251", 0x42fa6ea56179c229u},
|
||||
{"465000373656610.5312499999999999999999999999999999999999999999", 0x42fa6ea56179c228u},
|
||||
{"0.000000000107709773707743713401260815383950956818093214195641849073581397533416748046875", 0x3ddd9b66c974bb14u},
|
||||
{"0.0000000001077097737077437134012608153839509568180932141956418490735813975334167480468751", 0x3ddd9b66c974bb14u},
|
||||
{"0.000000000107709773707743713401260815383950956818093214195641849073581297533416748046875", 0x3ddd9b66c974bb13u},
|
||||
{"0.012083347821554271152300064073870089487172663211822509765625", 0x3f88bf277dc215f4u},
|
||||
{"0.0120833478215542711523000640738700894871726632118225097656251", 0x3f88bf277dc215f5u},
|
||||
{"0.01208334782155427115230006407387008948717266321182250976562499", 0x3f88bf277dc215f4u},
|
||||
{"2309804058391724800", 0x43c0070946d098e2u},
|
||||
{"23098040583917248001", 0x43f408cb9884bf1au},
|
||||
{"2309804058391724799.999999999999999999999999999999999999999999", 0x43c0070946d098e1u},
|
||||
{"0.000000000078286047058220682019445698291671103911937290575906445155851542949676513671875", 0x3dd584e40ca80638u},
|
||||
{"0.0000000000782860470582206820194456982916711039119372905759064451558515429496765136718751", 0x3dd584e40ca80638u},
|
||||
{"0.000000000078286047058220682019445698291671103911937290575906445155851532949676513671875", 0x3dd584e40ca80637u},
|
||||
{"2940024994425.709228515625", 0x4285643929d3cdacu},
|
||||
{"2940024994425.7092285156251", 0x4285643929d3cdadu},
|
||||
{"2940024994425.709228515624999999999999999999999999999999999999", 0x4285643929d3cdacu},
|
||||
{"9503358.427352792583405971527099609375", 0x4162204fcdacdfc4u},
|
||||
{"9503358.4273527925834059715270996093751", 0x4162204fcdacdfc4u},
|
||||
{"9503358.427352792583405971527099609374999999999999999999999999", 0x4162204fcdacdfc3u},
|
||||
{"0.00005589147834433423614399101542193903924271580763161182403564453125", 0x3f0d4da092aa5d6au},
|
||||
{"0.000055891478344334236143991015421939039242715807631611824035644531251", 0x3f0d4da092aa5d6bu},
|
||||
{"0.00005589147834433423614399101542193903924271580763161182403564452125", 0x3f0d4da092aa5d6au},
|
||||
{"0.0000000164797038506435664295103900420409737126448135313694365322589874267578125", 0x3e51b1e8107bd640u},
|
||||
{"0.00000001647970385064356642951039004204097371264481353136943653225898742675781251", 0x3e51b1e8107bd641u},
|
||||
{"0.0000000164797038506435664295103900420409737126448135313694365322589774267578125", 0x3e51b1e8107bd640u},
|
||||
{"73055.7873927834807545877993106842041015625", 0x40f1d5fc99292ce2u},
|
||||
{"73055.78739278348075458779931068420410156251", 0x40f1d5fc99292ce3u},
|
||||
{"73055.78739278348075458779931068420410156249999999999999999999", 0x40f1d5fc99292ce2u},
|
||||
{"0.00000000002558172364455232122427880575336067736115508441940846751094795763492584228515625", 0x3dbc209d7509115au},
|
||||
{"0.000000000025581723644552321224278805753360677361155084419408467510947957634925842285156251", 0x3dbc209d7509115bu},
|
||||
{"0.00000000002558172364455232122427880575336067736115508441940846751094794763492584228515625", 0x3dbc209d7509115au},
|
||||
{"0.000000000854497507560657937804791333430312443020238077906469698064029216766357421875", 0x3e0d5c3d540cc0d2u},
|
||||
{"0.0000000008544975075606579378047913334303124430202380779064696980640292167663574218751", 0x3e0d5c3d540cc0d2u},
|
||||
{"0.000000000854497507560657937804791333430312443020238077906469698064029116766357421875", 0x3e0d5c3d540cc0d1u},
|
||||
{"26671499731071461376", 0x43f722433b19970eu},
|
||||
{"266714997310714613761", 0x442cead409dffcd2u},
|
||||
{"26671499731071461375.99999999999999999999999999999999999999999", 0x43f722433b19970eu},
|
||||
{"0.0000000002725195963972150049060368088050545186395989816219298518262803554534912109375", 0x3df2ba37271cf5f2u},
|
||||
{"0.00000000027251959639721500490603680880505451863959898162192985182628035545349121093751", 0x3df2ba37271cf5f2u},
|
||||
{"0.0000000002725195963972150049060368088050545186395989816219298518262802554534912109375", 0x3df2ba37271cf5f1u},
|
||||
{"0.0000000000377224663702246439557904325637937886957218314165629635681398212909698486328125", 0x3dc4bcf7157af68eu},
|
||||
{"0.00000000003772246637022464395579043256379378869572183141656296356813982129096984863281251", 0x3dc4bcf7157af68fu},
|
||||
{"0.0000000000377224663702246439557904325637937886957218314165629635681398112909698486328125", 0x3dc4bcf7157af68eu},
|
||||
{"0.00000000000000009977342762593364154535842258994910996905560208471673566688053824691451154649257659912109375", 0x3c9cc1fac312e2a6u},
|
||||
{"0.000000000000000099773427625933641545358422589949109969055602084716735666880538246914511546492576599121093751", 0x3c9cc1fac312e2a6u},
|
||||
{"0.00000000000000009977342762593364154535842258994910996905560208471673566688052824691451154649257659912109375", 0x3c9cc1fac312e2a5u},
|
||||
{"0.0000000003146248171139977444431948290159907662133509376189977047033607959747314453125", 0x3df59ef03588a228u},
|
||||
{"0.00000000031462481711399774444319482901599076621335093761899770470336079597473144531251", 0x3df59ef03588a229u},
|
||||
{"0.0000000003146248171139977444431948290159907662133509376189977047033606959747314453125", 0x3df59ef03588a228u},
|
||||
{"488899209263030304", 0x439b23acf64c5e80u},
|
||||
{"4888992092630303041", 0x43d0f64c19efbb10u},
|
||||
{"488899209263030303.9999999999999999999999999999999999999999999", 0x439b23acf64c5e80u},
|
||||
{"1.88357157350592807620870416940306313335895538330078125", 0x3ffe231bf23e21acu},
|
||||
{"1.883571573505928076208704169403063133358955383300781251", 0x3ffe231bf23e21adu},
|
||||
{"1.883571573505928076208704169403063133358955383300781249999999", 0x3ffe231bf23e21acu},
|
||||
{"0.0000000216458400594294836451424756990254139044083103726734407246112823486328125", 0x3e573df694e72fb8u},
|
||||
{"0.00000002164584005942948364514247569902541390440831037267344072461128234863281251", 0x3e573df694e72fb9u},
|
||||
{"0.0000000216458400594294836451424756990254139044083103726734407246112723486328125", 0x3e573df694e72fb8u},
|
||||
{"5107.79271041116453488939441740512847900390625", 0x40b3f3caef11cb26u},
|
||||
{"5107.792710411164534889394417405128479003906251", 0x40b3f3caef11cb27u},
|
||||
{"5107.792710411164534889394417405128479003906249999999999999999", 0x40b3f3caef11cb26u},
|
||||
{"734059.8035226609208621084690093994140625", 0x412666d79b67527cu},
|
||||
{"734059.80352266092086210846900939941406251", 0x412666d79b67527du},
|
||||
{"734059.8035226609208621084690093994140624999999999999999999999", 0x412666d79b67527cu},
|
||||
{"61431562016722684", 0x436b47f5c40021e0u},
|
||||
{"614315620167226841", 0x43a10cf99a80152cu},
|
||||
{"61431562016722683.99999999999999999999999999999999999999999999", 0x436b47f5c40021dfu},
|
||||
{"2.0060840449445034305853141631814651191234588623046875", 0x40000c75cab08326u},
|
||||
{"2.00608404494450343058531416318146511912345886230468751", 0x40000c75cab08326u},
|
||||
{"2.006084044944503430585314163181465119123458862304687499999999", 0x40000c75cab08325u},
|
||||
{"0.0000001760623599453036952716420489480075861621344301966018974781036376953125", 0x3e87a174e55262cau},
|
||||
{"0.00000017606235994530369527164204894800758616213443019660189747810363769531251", 0x3e87a174e55262cbu},
|
||||
{"0.0000001760623599453036952716420489480075861621344301966018974781035376953125", 0x3e87a174e55262cau},
|
||||
{"0.833085849636964581588216560703585855662822723388671875", 0x3feaa8a3a7de6fb6u},
|
||||
{"0.8330858496369645815882165607035858556628227233886718751", 0x3feaa8a3a7de6fb6u},
|
||||
{"0.8330858496369645815882165607035858556628227233886718749999999", 0x3feaa8a3a7de6fb5u},
|
||||
{"45031428.4182307310402393341064453125", 0x418579002358895au},
|
||||
{"45031428.41823073104023933410644531251", 0x418579002358895bu},
|
||||
{"45031428.41823073104023933410644531249999999999999999999999999", 0x418579002358895au},
|
||||
{"5003361733758455296", 0x43d15be0bf39dd24u},
|
||||
{"50033617337584552961", 0x4405b2d8ef08546cu},
|
||||
{"5003361733758455295.999999999999999999999999999999999999999999", 0x43d15be0bf39dd23u},
|
||||
};
|
||||
|
||||
for (const auto& c : known)
|
||||
{
|
||||
CAPTURE(c.first);
|
||||
double out = 0;
|
||||
if (eisel_lemire(c.first, out))
|
||||
{
|
||||
CHECK(bits_of(out) == c.second);
|
||||
}
|
||||
else
|
||||
{
|
||||
// only tokens with more than 19 significant digits are left to
|
||||
// strtod: those whose value lies too close to a tie
|
||||
CHECK(significant_digits(c.first) > 19);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("round trip")
|
||||
{
|
||||
// every double written by to_chars and read back, also with trailing
|
||||
// digits that make the token longer than 19 digits
|
||||
std::uint64_t state = 5295;
|
||||
std::size_t declined = 0;
|
||||
for (int i = 0; i < 200000; ++i)
|
||||
{
|
||||
state ^= state << 13u;
|
||||
state ^= state >> 7u;
|
||||
state ^= state << 17u;
|
||||
std::uint64_t b = state;
|
||||
if ((b & 0x7FF0000000000000u) == 0x7FF0000000000000u)
|
||||
{
|
||||
continue; // infinity or NaN
|
||||
}
|
||||
if (i % 4 == 0)
|
||||
{
|
||||
b &= 0x800FFFFFFFFFFFFFu; // subnormals
|
||||
}
|
||||
double d = 0;
|
||||
std::memcpy(&d, &b, sizeof(d));
|
||||
|
||||
std::array<char, 64> buffer{};
|
||||
const char* end = nlohmann::detail::to_chars(buffer.data(), buffer.data() + buffer.size(), d);
|
||||
const std::string token(buffer.data(), static_cast<std::size_t>(end - buffer.data()));
|
||||
CAPTURE(token);
|
||||
double out = 0;
|
||||
REQUIRE(eisel_lemire(token, out));
|
||||
CHECK(bits_of(out) == b);
|
||||
|
||||
// insert digits before the exponent: the value moves by far less
|
||||
// than the distance to the rounding boundary, so it must not change
|
||||
std::string longer = token;
|
||||
const std::size_t e = longer.find('e');
|
||||
const std::string extra = longer.find('.') == std::string::npos ? ".000000000000000000001" : "000000000000000000001";
|
||||
longer.insert(e == std::string::npos ? longer.size() : e, extra);
|
||||
CAPTURE(longer);
|
||||
if (eisel_lemire(longer, out))
|
||||
{
|
||||
CHECK(bits_of(out) == b);
|
||||
}
|
||||
else
|
||||
{
|
||||
// w and w + 1 round differently: only when the value is very
|
||||
// close to a rounding boundary
|
||||
++declined;
|
||||
}
|
||||
}
|
||||
CHECK(declined < 1000); // 107 of the 200,000
|
||||
}
|
||||
|
||||
SECTION("used by the lexer")
|
||||
{
|
||||
// 17 significant digits: beyond Clinger's fast path
|
||||
CHECK(bits_of(json::parse("-65.613616999999977").get<double>()) == bits_of(-65.613616999999977));
|
||||
CHECK(bits_of(json::parse("2.2250738585072011e-308").get<double>()) == 0x000FFFFFFFFFFFFFu);
|
||||
CHECK(bits_of(json::parse("4.9406564584124654e-324").get<double>()) == 1u);
|
||||
CHECK_THROWS_WITH_AS(json::parse("1.7976931348623159e308"),
|
||||
"[json.exception.out_of_range.406] number overflow parsing '1.7976931348623159e308'", json::out_of_range&);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("string scanning kernels")
|
||||
{
|
||||
// the word-at-a-time kernels must stop exactly where a byte-by-byte scan
|
||||
// stops, for any content, length, and alignment
|
||||
const auto reference_special = [](const unsigned char* data, std::size_t n)
|
||||
{
|
||||
std::size_t i = 0;
|
||||
while (i < n && !nlohmann::detail::is_string_special(data[i]))
|
||||
{
|
||||
++i;
|
||||
}
|
||||
return i;
|
||||
};
|
||||
const auto reference_copyable = [](const unsigned char* data, std::size_t n)
|
||||
{
|
||||
std::size_t i = 0;
|
||||
while (i < n && nlohmann::detail::is_ascii_copyable(data[i]))
|
||||
{
|
||||
++i;
|
||||
}
|
||||
return i;
|
||||
};
|
||||
const auto reference_bulk_run = [](const unsigned char* data, std::size_t n)
|
||||
{
|
||||
std::size_t i = 0;
|
||||
while (i < n)
|
||||
{
|
||||
if (data[i] < 0x80u)
|
||||
{
|
||||
if (nlohmann::detail::is_string_special(data[i]))
|
||||
{
|
||||
break;
|
||||
}
|
||||
++i;
|
||||
continue;
|
||||
}
|
||||
const std::size_t seq = nlohmann::detail::validate_one_utf8(data + i, n - i);
|
||||
if (seq == 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
i += seq;
|
||||
}
|
||||
return i;
|
||||
};
|
||||
|
||||
// pieces: ordinary ASCII, stops, DEL, well-formed sequences of every
|
||||
// length, and ill-formed or truncated ones
|
||||
const std::vector<std::string> pieces =
|
||||
{
|
||||
"a", "Z", " ", "~", "0123456789", "\"", "\\", std::string(1, '\0'), "\n", "\x1F", "\x7F",
|
||||
"\xC3\xA4", "\xE2\x82\xAC", "\xE6\x97\xA5\xE6\x9C\xAC", "\xF0\x9F\x98\x80", "\xED\x9F\xBF",
|
||||
"\x80", "\xC0\x80", "\xC3", "\xE2\x82", "\xED\xA0\x80", "\xF4\x90\x80\x80", "\xFF",
|
||||
};
|
||||
std::uint64_t state = 5295;
|
||||
const auto next = [&state]()
|
||||
{
|
||||
state ^= state << 13u;
|
||||
state ^= state >> 7u;
|
||||
state ^= state << 17u;
|
||||
return state;
|
||||
};
|
||||
for (int round = 0; round < 100000; ++round)
|
||||
{
|
||||
// mostly ordinary text, so that runs span several words
|
||||
std::string text(static_cast<std::size_t>(next() % 8), '.');
|
||||
const auto count = static_cast<std::size_t>(next() % 12);
|
||||
for (std::size_t k = 0; k < count; ++k)
|
||||
{
|
||||
const std::size_t p = (next() % 4 == 0) ? static_cast<std::size_t>(next() % pieces.size()) : 0;
|
||||
text += pieces[p];
|
||||
text += std::string(static_cast<std::size_t>(next() % 10), 'x');
|
||||
}
|
||||
const auto* data = reinterpret_cast<const unsigned char*>(text.data()); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
for (std::size_t offset = 0; offset < 3 && offset <= text.size(); ++offset)
|
||||
{
|
||||
const std::size_t n = text.size() - offset;
|
||||
CAPTURE(text);
|
||||
CAPTURE(offset);
|
||||
CHECK(nlohmann::detail::find_string_special(data + offset, n) == reference_special(data + offset, n));
|
||||
CHECK(nlohmann::detail::find_ascii_copyable_run(data + offset, n) == reference_copyable(data + offset, n));
|
||||
CHECK(nlohmann::detail::scalar_string_bulk_run(data + offset, n) == reference_bulk_run(data + offset, n));
|
||||
}
|
||||
}
|
||||
|
||||
// the trailing-zero count, whichever implementation the compiler gets
|
||||
for (int k = 0; k < 64; ++k)
|
||||
{
|
||||
const std::uint64_t bit = std::uint64_t{1} << k;
|
||||
CHECK(nlohmann::detail::count_trailing_zeros(bit) == k);
|
||||
CHECK(nlohmann::detail::count_trailing_zeros(bit | (bit << 1u) | 0x8000000000000000u) == k);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2817,3 +2817,586 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX")
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
namespace
|
||||
{
|
||||
/// builds a value like json::parse(), but asks the parser to recover from
|
||||
/// errors (see #3989), and checks that the events it receives are balanced
|
||||
class RecoveringDomParser : public nlohmann::detail::json_sax_dom_parser<json>
|
||||
{
|
||||
using base = nlohmann::detail::json_sax_dom_parser<json>;
|
||||
|
||||
public:
|
||||
explicit RecoveringDomParser(json& j, std::size_t max_errors_ = static_cast<std::size_t>(-1))
|
||||
: base(j, false)
|
||||
, max_errors(max_errors_)
|
||||
{}
|
||||
|
||||
bool null()
|
||||
{
|
||||
value();
|
||||
return base::null();
|
||||
}
|
||||
|
||||
bool boolean(bool val)
|
||||
{
|
||||
value();
|
||||
return base::boolean(val);
|
||||
}
|
||||
|
||||
bool number_integer(json::number_integer_t val)
|
||||
{
|
||||
value();
|
||||
return base::number_integer(val);
|
||||
}
|
||||
|
||||
bool number_unsigned(json::number_unsigned_t val)
|
||||
{
|
||||
value();
|
||||
return base::number_unsigned(val);
|
||||
}
|
||||
|
||||
bool number_float(json::number_float_t val, const std::string& s)
|
||||
{
|
||||
value();
|
||||
return base::number_float(val, s);
|
||||
}
|
||||
|
||||
bool string(std::string& val)
|
||||
{
|
||||
value();
|
||||
return base::string(val);
|
||||
}
|
||||
|
||||
bool start_object(std::size_t elements)
|
||||
{
|
||||
value();
|
||||
stack.push_back('o');
|
||||
return base::start_object(elements);
|
||||
}
|
||||
|
||||
bool key(std::string& val)
|
||||
{
|
||||
++events;
|
||||
if (stack.empty() || stack.back() != 'o')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.back() = 'v';
|
||||
return base::key(val);
|
||||
}
|
||||
|
||||
bool end_object()
|
||||
{
|
||||
++events;
|
||||
if (stack.empty() || stack.back() != 'o')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.pop_back();
|
||||
return base::end_object();
|
||||
}
|
||||
|
||||
bool start_array(std::size_t elements)
|
||||
{
|
||||
value();
|
||||
stack.push_back('a');
|
||||
return base::start_array(elements);
|
||||
}
|
||||
|
||||
bool end_array()
|
||||
{
|
||||
++events;
|
||||
if (stack.empty() || stack.back() != 'a')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.pop_back();
|
||||
return base::end_array();
|
||||
}
|
||||
|
||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& ex)
|
||||
{
|
||||
errors.emplace_back(ex.what());
|
||||
return errors.size() < max_errors;
|
||||
}
|
||||
|
||||
/// whether the events were balanced and every key was followed by a value
|
||||
bool balanced() const
|
||||
{
|
||||
return well_formed && stack.empty();
|
||||
}
|
||||
|
||||
std::vector<std::string> errors {}; // NOLINT(readability-redundant-member-init)
|
||||
std::size_t events = 0;
|
||||
/// the open containers: 'a' for an array, 'o' for an object that expects
|
||||
/// a key, 'v' for an object that expects the value of a key
|
||||
std::vector<char> stack {}; // NOLINT(readability-redundant-member-init)
|
||||
bool well_formed = true;
|
||||
std::size_t max_errors;
|
||||
|
||||
private:
|
||||
/// a value is passed: it is an array element, or the value of a key
|
||||
void value()
|
||||
{
|
||||
++events;
|
||||
if (!stack.empty())
|
||||
{
|
||||
if (stack.back() == 'v')
|
||||
{
|
||||
stack.back() = 'o';
|
||||
}
|
||||
else if (stack.back() == 'o')
|
||||
{
|
||||
// a value without a key
|
||||
well_formed = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
struct RecoveryResult
|
||||
{
|
||||
json value;
|
||||
std::vector<std::string> errors;
|
||||
std::size_t events;
|
||||
bool ok;
|
||||
bool balanced;
|
||||
};
|
||||
|
||||
template<typename InputType>
|
||||
RecoveryResult parse_recovering(InputType&& input, const bool strict = true,
|
||||
const bool ignore_comments = false, const bool ignore_trailing_commas = false)
|
||||
{
|
||||
json j;
|
||||
RecoveringDomParser sax(j);
|
||||
const bool ok = json::sax_parse(std::forward<InputType>(input), &sax, json::input_format_t::json,
|
||||
strict, ignore_comments, ignore_trailing_commas);
|
||||
return {j, sax.errors, sax.events, ok, sax.balanced()};
|
||||
}
|
||||
|
||||
/// logs the events as strings and recovers from errors
|
||||
class RecoveringEventLogger : public SaxEventLogger
|
||||
{
|
||||
public:
|
||||
bool parse_error(std::size_t position, const std::string& /*unused*/, const json::exception& /*unused*/)
|
||||
{
|
||||
events.push_back("parse_error(" + std::to_string(position) + ")");
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
/// stops after a number of events, but recovers from errors
|
||||
class RecoveringCountdown : public SaxCountdown
|
||||
{
|
||||
public:
|
||||
using SaxCountdown::SaxCountdown;
|
||||
|
||||
bool parse_error(std::size_t /*position*/, const std::string& /*last_token*/, const json::exception& /*ex*/) override
|
||||
{
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
/// a repaired input: the value it is repaired to, and the number of errors
|
||||
struct Repair
|
||||
{
|
||||
const char* input;
|
||||
const char* expected;
|
||||
std::size_t errors;
|
||||
};
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("parser error recovery (#3989)")
|
||||
{
|
||||
SECTION("repairs")
|
||||
{
|
||||
const std::vector<Repair> repairs =
|
||||
{
|
||||
// a missing separator is inserted
|
||||
{"[1 2]", "[1,2]", 1},
|
||||
{R"({"a":1 "b":2})", R"({"a":1,"b":2})", 1},
|
||||
{R"({"a" 1})", R"({"a":1})", 1},
|
||||
{"[1 tru 2]", "[1,null,2]", 2},
|
||||
{R"({"a" "b": 1})", R"({"a":"b"})", 2},
|
||||
|
||||
// a missing value is null in an object; in an array, a ',' stands
|
||||
// for null, while an array that ends there just ends
|
||||
{R"({"a":})", R"({"a":null})", 1},
|
||||
{R"({"a"})", R"({"a":null})", 1},
|
||||
{R"({"a","b":1})", R"({"a":null,"b":1})", 1},
|
||||
{"[1,,2]", "[1,null,2]", 1},
|
||||
{"[,1]", "[null,1]", 1},
|
||||
{"[1,]", "[1]", 1},
|
||||
{"[1,2,3,]", "[1,2,3]", 1},
|
||||
{R"({"a":1,})", R"({"a":1})", 1},
|
||||
|
||||
// a broken string keeps what can be read
|
||||
{R"(["a\qb"])", R"(["aqb"])", 1},
|
||||
{R"({"na\me":1})", R"({"name":1})", 1},
|
||||
{"[\"\xFF\"]", R"(["\uFFFD"])", 1},
|
||||
{"[\"a\xC3(\"]", R"(["a\uFFFD("])", 1},
|
||||
{"[\"\xE2\x82\"]", R"(["\uFFFD"])", 1},
|
||||
{"[\"\xC3\\\\\", 1]", R"(["\uFFFD\\",1])", 1},
|
||||
{R"(["\u12"])", R"(["\uFFFD"])", 1},
|
||||
{R"(["\u12G4"])", R"(["\uFFFDG4"])", 1},
|
||||
{R"(["\uDC00x"])", R"(["\uFFFDx"])", 1},
|
||||
{R"(["\uD800x"])", R"(["\uFFFDx"])", 1},
|
||||
{R"(["\uD800\u0041"])", R"(["\uFFFDA"])", 1},
|
||||
{R"(["\uD800\uD800\uDC00"])", R"(["\uFFFD\uD800\uDC00"])", 1},
|
||||
{R"(["\uD800\uD800\uD800x"])", R"(["\uFFFD\uFFFD\uFFFDx"])", 1},
|
||||
{
|
||||
R"(["\uD800\"x", 1])", R"(["\uFFFD\"x",1])", 1
|
||||
},
|
||||
{R"(["\uD800\q"])", R"(["\uFFFDq"])", 1},
|
||||
{"[\"a\tb\"]", R"(["a\tb"])", 1},
|
||||
{R"(["a\qb\u0041\x"])", R"(["aqbAx"])", 1},
|
||||
|
||||
// a broken number keeps its longest valid prefix
|
||||
{"[1.]", "[1]", 1},
|
||||
{"[-2.]", "[-2]", 1},
|
||||
{"[1.5e]", "[1.5]", 1},
|
||||
{"[1e+]", "[1]", 1},
|
||||
{"[1.x2, 3]", "[1,3]", 1},
|
||||
|
||||
// what cannot be read at all is null
|
||||
{"[1,NaN,3]", "[1,null,3]", 1},
|
||||
{"[tru]", "[null]", 1},
|
||||
{"[-]", "[null]", 1},
|
||||
{R"({"a":Infinity})", R"({"a":null})", 1},
|
||||
|
||||
// a stray token is dropped
|
||||
{"[:1]", "[1]", 1},
|
||||
{R"(["a":1])", R"(["a",1])", 1},
|
||||
{R"({"a"::1})", R"({"a":1})", 1},
|
||||
|
||||
// a member that cannot be read is skipped
|
||||
{R"({1:2,"b":3})", R"({"b":3})", 1},
|
||||
{R"({"a":1 2})", R"({"a":1})", 1},
|
||||
{R"({,"a":1})", R"({"a":1})", 1},
|
||||
{R"({"a":1,,"b":2})", R"({"a":1,"b":2})", 1},
|
||||
{"{a:1}", "{}", 1},
|
||||
{R"({"a":1 [1,{"b":2}], "c":3})", R"({"a":1,"c":3})", 1},
|
||||
{R"([{1}, "a"])", R"([{},"a"])", 1},
|
||||
|
||||
// a wrong closing bracket closes the innermost container
|
||||
{R"({"a":[1,2}, "b":3})", R"({"a":[1,2],"b":3})", 1},
|
||||
{R"([{"a":1], 2])", R"([{"a":1},2])", 1},
|
||||
{"{]", "{}", 1},
|
||||
{"[}", "[]", 1},
|
||||
|
||||
// the end of the input closes all containers
|
||||
{R"({"a":[1,2)", R"({"a":[1,2]})", 1},
|
||||
{"[", "[]", 1},
|
||||
{"{", "{}", 1},
|
||||
{R"({"a")", R"({"a":null})", 1},
|
||||
{R"({"a":)", R"({"a":null})", 1},
|
||||
{"[1,", "[1]", 1},
|
||||
{"[[[1", "[[[1]]]", 1},
|
||||
{
|
||||
R"(["abc)", R"(["abc"])", 2
|
||||
},
|
||||
{"[1,tr", "[1,null]", 2},
|
||||
{"\"abc", "\"abc\"", 1},
|
||||
{"[\"ab\ncd\"]", R"(["ab",null,"]"])", 4},
|
||||
|
||||
// what comes before the top-level value is skipped
|
||||
{")]}'\n{\"a\":1}", R"({"a":1})", 1},
|
||||
{R"(data: {"a":1})", R"({"a":1})", 1},
|
||||
{"\xEF\xBB[1]", "[1]", 1},
|
||||
|
||||
// what comes after it is an error that ends parsing
|
||||
{R"({"a":1}})", R"({"a":1})", 1},
|
||||
{"[1}]", "[1]", 2},
|
||||
{"[1] [2]", "[1]", 1},
|
||||
};
|
||||
|
||||
for (const auto& repair : repairs)
|
||||
{
|
||||
CAPTURE(repair.input);
|
||||
const auto result = parse_recovering(std::string(repair.input));
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.value == json::parse(repair.expected));
|
||||
CHECK(result.errors.size() == repair.errors);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("number overflow")
|
||||
{
|
||||
const auto result = parse_recovering(std::string("[1e999,-1e999]"));
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.errors.size() == 2);
|
||||
CHECK(result.errors[0] == "[json.exception.out_of_range.406] number overflow parsing '1e999'");
|
||||
REQUIRE(result.value.size() == 2);
|
||||
CHECK(result.value[0].is_number_float());
|
||||
CHECK(result.value[0].get<double>() == std::numeric_limits<double>::infinity());
|
||||
CHECK(result.value[1].get<double>() == -std::numeric_limits<double>::infinity());
|
||||
|
||||
// the SAX parser gets the number's text
|
||||
RecoveringEventLogger logger;
|
||||
CHECK(!json::sax_parse("1e999", &logger));
|
||||
CHECK(logger.events == std::vector<std::string>({"parse_error(5)", "number_float(1e999)"}));
|
||||
}
|
||||
|
||||
SECTION("nothing to recover")
|
||||
{
|
||||
for (const std::string s :
|
||||
{
|
||||
"", " ", "]", "tru", "NaN", ",:", "/* comment"
|
||||
})
|
||||
{
|
||||
CAPTURE(s);
|
||||
const auto result = parse_recovering(s, true, true);
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.events == 0);
|
||||
CHECK(result.value == nullptr);
|
||||
CHECK(result.errors.size() == 1);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("error messages")
|
||||
{
|
||||
// the first error is reported as without recovery
|
||||
for (const std::string s :
|
||||
{
|
||||
"[1 2]", R"({"a":1 "b":2})", R"({"a" 1})", R"({"a":})", "[1,]", "[1.]",
|
||||
R"(["a\qb"])", "[1e999]", "{1:2}", R"({"a":[1,2}})", "[1,", "[1] [2]", "{a:1}"
|
||||
})
|
||||
{
|
||||
CAPTURE(s);
|
||||
const auto result = parse_recovering(s);
|
||||
REQUIRE(!result.errors.empty());
|
||||
json _;
|
||||
CHECK_THROWS_WITH_STD_STR(_ = json::parse(s), result.errors.front());
|
||||
}
|
||||
|
||||
// the token of an error begins where the previous error was
|
||||
const auto result = parse_recovering(std::string("[tru, fals, nul]"));
|
||||
CHECK(result.errors == std::vector<std::string>(
|
||||
{
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: '[tru,'",
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 11: syntax error while parsing value - invalid literal; last read: ', fals,'",
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 16: syntax error while parsing value - invalid literal; last read: ', nul]'"
|
||||
}));
|
||||
CHECK(result.value == json::parse("[null,null,null]"));
|
||||
}
|
||||
|
||||
SECTION("events")
|
||||
{
|
||||
// see #4522
|
||||
RecoveringEventLogger logger;
|
||||
CHECK(!json::sax_parse(R"([{1}, "a"])", &logger));
|
||||
CHECK(logger.events == std::vector<std::string>(
|
||||
{
|
||||
"start_array()", "start_object()", "parse_error(3)", "end_object()", "string(a)", "end_array()"
|
||||
}));
|
||||
}
|
||||
|
||||
SECTION("options")
|
||||
{
|
||||
SECTION("strict")
|
||||
{
|
||||
const auto result = parse_recovering(std::string("[1 2] [3]"), false);
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.value == json::parse("[1,2]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
}
|
||||
|
||||
SECTION("ignore_trailing_commas")
|
||||
{
|
||||
for (const std::string s :
|
||||
{
|
||||
"[1,]", R"({"a":1,})", "[[1,],]"
|
||||
})
|
||||
{
|
||||
CAPTURE(s);
|
||||
const auto result = parse_recovering(s, true, false, true);
|
||||
CHECK(result.ok);
|
||||
CHECK(result.errors.empty());
|
||||
}
|
||||
|
||||
auto result = parse_recovering(std::string("[1,,]"), true, false, true);
|
||||
CHECK(result.value == json::parse("[1,null]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
|
||||
result = parse_recovering(std::string(R"({"a":1,,})"), true, false, true);
|
||||
CHECK(result.value == json::parse(R"({"a":1})"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
}
|
||||
|
||||
SECTION("ignore_comments")
|
||||
{
|
||||
auto result = parse_recovering(std::string("[1 /* one */ 2]"), true, true);
|
||||
CHECK(result.value == json::parse("[1,2]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
|
||||
// a comment that is not closed runs to the end of the input, which
|
||||
// is not reported again
|
||||
result = parse_recovering(std::string("[1, 2 /* unterminated"), true, true);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.value == json::parse("[1,2]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
|
||||
// a '/' that does not begin a comment is garbage
|
||||
result = parse_recovering(std::string("[1, /x, 2]"), true, true);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.value == json::parse("[1,null,2]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("null bytes")
|
||||
{
|
||||
// a null byte ends the input, unless JSON_STRICT_NUL_HANDLING is set
|
||||
const auto result = parse_recovering(std::string("[1,\0x", 5));
|
||||
CHECK(result.balanced);
|
||||
CHECK(!result.ok);
|
||||
#ifdef JSON_TEST_STRICT_NUL_HANDLING_ENABLED
|
||||
CHECK(result.value == json::parse("[1,null]"));
|
||||
#else
|
||||
CHECK(result.value == json::parse("[1]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
#endif
|
||||
|
||||
const auto in_string = parse_recovering(std::string("[\"a\0b\"]", 7));
|
||||
CHECK(in_string.balanced);
|
||||
#ifdef JSON_TEST_STRICT_NUL_HANDLING_ENABLED
|
||||
CHECK(in_string.value == json::array({std::string("a\0b", 3)}));
|
||||
#else
|
||||
CHECK(in_string.value == json::parse(R"(["a"])"));
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("the SAX parser stops recovering")
|
||||
{
|
||||
json j;
|
||||
RecoveringDomParser sax(j, 2);
|
||||
CHECK(!json::sax_parse("[1 2 3 4 5]", &sax));
|
||||
CHECK(sax.errors.size() == 2);
|
||||
|
||||
// an error at a delimiter that an invalid token consumed is reported
|
||||
// to the SAX parser, too
|
||||
json j2;
|
||||
RecoveringDomParser sax2(j2, 2);
|
||||
CHECK(!json::sax_parse("[tru}, 1]", &sax2));
|
||||
CHECK(sax2.errors.size() == 2);
|
||||
}
|
||||
|
||||
SECTION("an event stops parsing during a repair")
|
||||
{
|
||||
// start_object() and key() are passed, then null() for the missing
|
||||
// value returns false
|
||||
RecoveringCountdown countdown(2);
|
||||
CHECK(!json::sax_parse(R"({"a":})", &countdown));
|
||||
|
||||
// the end of the input: end_array() for the second array returns false
|
||||
RecoveringCountdown countdown2(4);
|
||||
CHECK(!json::sax_parse("[[1", &countdown2));
|
||||
}
|
||||
|
||||
SECTION("input adapters")
|
||||
{
|
||||
// the lexer reads contiguous and streaming input differently, and it
|
||||
// puts back a character that ended an invalid token
|
||||
for (const std::string s :
|
||||
{
|
||||
"[1 2]", "[tru}, 1]", R"({"a" "b\q", "c":[1.x, 2}})", "[\"\xFF\xC3(\", -, 1e+]", "{a:1,\"b\":2", ")]}' [1]"
|
||||
})
|
||||
{
|
||||
CAPTURE(s);
|
||||
const auto reference = parse_recovering(s);
|
||||
CHECK(reference.balanced);
|
||||
|
||||
const auto from_c_string = parse_recovering(s.c_str());
|
||||
CHECK(from_c_string.value == reference.value);
|
||||
CHECK(from_c_string.errors == reference.errors);
|
||||
|
||||
const std::list<char> l(s.begin(), s.end());
|
||||
json j;
|
||||
RecoveringDomParser sax(j);
|
||||
CHECK(!json::sax_parse(l.begin(), l.end(), &sax));
|
||||
CHECK(j == reference.value);
|
||||
CHECK(sax.errors == reference.errors);
|
||||
|
||||
std::istringstream ss(s);
|
||||
const auto from_stream = parse_recovering(ss);
|
||||
CHECK(from_stream.value == reference.value);
|
||||
CHECK(from_stream.errors == reference.errors);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("long runs of errors")
|
||||
{
|
||||
// no error may copy all the input read before it
|
||||
const auto closing = parse_recovering("[" + std::string(100000, '}'));
|
||||
CHECK(closing.balanced);
|
||||
CHECK(closing.value == json::array());
|
||||
|
||||
const auto garbage = parse_recovering("[" + std::string(100000, 'x') + "]");
|
||||
CHECK(garbage.balanced);
|
||||
CHECK(garbage.errors.size() == 1);
|
||||
|
||||
const auto commas = parse_recovering("{" + std::string(100000, ',') + "}");
|
||||
CHECK(commas.balanced);
|
||||
CHECK(commas.value == json::object());
|
||||
}
|
||||
|
||||
SECTION("mutations of valid input")
|
||||
{
|
||||
// whatever the input, the events are balanced, every error is reported
|
||||
// at most once, and valid input is parsed as usual
|
||||
const std::vector<std::string> documents =
|
||||
{
|
||||
R"({"name": "value", "list": [1, -2.5, true, null, {"x": [[]]}], "e": "\u00e9"})",
|
||||
R"([{"a": [1, 2, {"b": "c"}]}, [], {}, "\ud83d\ude00", 1e10])",
|
||||
"{\"\xC3\xA9\": \"\xF0\x9F\x98\x80\"}",
|
||||
R"( {"k" : [ "v" , 0 ] } )",
|
||||
};
|
||||
// each character that can be inserted, including a null byte
|
||||
const std::string insertions("[]{},:\"x\\\0\xFF", 11);
|
||||
|
||||
std::vector<std::string> inputs;
|
||||
for (const auto& doc : documents)
|
||||
{
|
||||
for (std::size_t i = 0; i <= doc.size(); ++i)
|
||||
{
|
||||
inputs.push_back(doc.substr(0, i));
|
||||
if (i < doc.size())
|
||||
{
|
||||
inputs.push_back(doc.substr(0, i) + doc.substr(i + 1));
|
||||
}
|
||||
for (const char c : insertions)
|
||||
{
|
||||
inputs.push_back(doc.substr(0, i) + c + doc.substr(i));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (const auto& s : inputs)
|
||||
{
|
||||
CAPTURE(s);
|
||||
const auto result = parse_recovering(s);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.errors.size() <= s.size() + 1);
|
||||
CHECK(result.events <= (4 * s.size()) + 4);
|
||||
if (json::accept(s))
|
||||
{
|
||||
CHECK(result.ok);
|
||||
CHECK(result.errors.empty());
|
||||
CHECK(result.value == json::parse(s));
|
||||
}
|
||||
else
|
||||
{
|
||||
CHECK(!result.ok);
|
||||
CHECK(!result.errors.empty());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -870,4 +870,585 @@ TEST_CASE("regression test - excessive binary container size honors allow_except
|
||||
CHECK(json::from_cbor(std::vector<std::uint8_t> {0x9b, 0, 0, 0, 0, 0, 0, 0, 0x02}, true, false).is_discarded());
|
||||
}
|
||||
|
||||
namespace
|
||||
{
|
||||
/// builds a value from SAX events, asks the parser to recover from its first
|
||||
/// 100 errors, and checks that the events are balanced (see #3989)
|
||||
class RecoveringParser : public nlohmann::detail::json_sax_dom_parser<json>
|
||||
{
|
||||
using base = nlohmann::detail::json_sax_dom_parser<json>;
|
||||
|
||||
public:
|
||||
explicit RecoveringParser(json& j)
|
||||
: base(j, false)
|
||||
{}
|
||||
|
||||
bool null()
|
||||
{
|
||||
value();
|
||||
return base::null();
|
||||
}
|
||||
|
||||
bool boolean(bool val)
|
||||
{
|
||||
value();
|
||||
return base::boolean(val);
|
||||
}
|
||||
|
||||
bool number_integer(json::number_integer_t val)
|
||||
{
|
||||
value();
|
||||
return base::number_integer(val);
|
||||
}
|
||||
|
||||
bool number_unsigned(json::number_unsigned_t val)
|
||||
{
|
||||
value();
|
||||
return base::number_unsigned(val);
|
||||
}
|
||||
|
||||
bool number_float(json::number_float_t val, const std::string& s)
|
||||
{
|
||||
value();
|
||||
return base::number_float(val, s);
|
||||
}
|
||||
|
||||
bool string(std::string& val)
|
||||
{
|
||||
value();
|
||||
return base::string(val);
|
||||
}
|
||||
|
||||
bool binary(json::binary_t& val)
|
||||
{
|
||||
value();
|
||||
return base::binary(val);
|
||||
}
|
||||
|
||||
bool start_object(std::size_t elements)
|
||||
{
|
||||
value();
|
||||
stack.push_back('o');
|
||||
return base::start_object(elements);
|
||||
}
|
||||
|
||||
bool key(std::string& val)
|
||||
{
|
||||
if (stack.empty() || stack.back() != 'o')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.back() = 'v';
|
||||
return base::key(val);
|
||||
}
|
||||
|
||||
bool end_object()
|
||||
{
|
||||
if (stack.empty() || stack.back() != 'o')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.pop_back();
|
||||
return base::end_object();
|
||||
}
|
||||
|
||||
bool start_array(std::size_t elements)
|
||||
{
|
||||
value();
|
||||
stack.push_back('a');
|
||||
return base::start_array(elements);
|
||||
}
|
||||
|
||||
bool end_array()
|
||||
{
|
||||
if (stack.empty() || stack.back() != 'a')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.pop_back();
|
||||
return base::end_array();
|
||||
}
|
||||
|
||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& ex)
|
||||
{
|
||||
messages.emplace_back(ex.what());
|
||||
// a limit, so that a reader that does not stop fails the test
|
||||
// instead of making it hang
|
||||
return ++errors < 100;
|
||||
}
|
||||
|
||||
/// whether the events were balanced and every key was followed by a value
|
||||
bool balanced() const
|
||||
{
|
||||
return well_formed && stack.empty();
|
||||
}
|
||||
|
||||
std::size_t errors = 0;
|
||||
std::vector<std::string> messages {}; // NOLINT(readability-redundant-member-init)
|
||||
std::vector<char> stack {}; // NOLINT(readability-redundant-member-init)
|
||||
bool well_formed = true;
|
||||
|
||||
private:
|
||||
void value()
|
||||
{
|
||||
if (!stack.empty())
|
||||
{
|
||||
if (stack.back() == 'v')
|
||||
{
|
||||
stack.back() = 'o';
|
||||
}
|
||||
else if (stack.back() == 'o')
|
||||
{
|
||||
well_formed = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
struct BinaryParseResult
|
||||
{
|
||||
json value;
|
||||
std::size_t errors;
|
||||
std::vector<std::string> messages;
|
||||
bool ok;
|
||||
bool balanced;
|
||||
};
|
||||
|
||||
BinaryParseResult parse_binary_recovering(const std::vector<std::uint8_t>& input, const json::input_format_t format)
|
||||
{
|
||||
json j;
|
||||
RecoveringParser sax(j);
|
||||
const bool ok = json::sax_parse(input, &sax, format);
|
||||
return {j, sax.errors, sax.messages, ok, sax.balanced()};
|
||||
}
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
/// the message of the exception that reading @a input into a JSON value
|
||||
/// throws, or an empty string if reading succeeds
|
||||
std::string binary_error_message(const std::vector<std::uint8_t>& input, const json::input_format_t format)
|
||||
{
|
||||
try
|
||||
{
|
||||
json _;
|
||||
switch (format)
|
||||
{
|
||||
case json::input_format_t::cbor:
|
||||
_ = json::from_cbor(input);
|
||||
break;
|
||||
case json::input_format_t::msgpack:
|
||||
_ = json::from_msgpack(input);
|
||||
break;
|
||||
case json::input_format_t::ubjson:
|
||||
_ = json::from_ubjson(input);
|
||||
break;
|
||||
case json::input_format_t::bjdata:
|
||||
_ = json::from_bjdata(input);
|
||||
break;
|
||||
case json::input_format_t::bson:
|
||||
_ = json::from_bson(input);
|
||||
break;
|
||||
case json::input_format_t::bon8:
|
||||
_ = json::from_bon8(input);
|
||||
break;
|
||||
case json::input_format_t::json:
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
catch (const json::exception& e)
|
||||
{
|
||||
return e.what();
|
||||
}
|
||||
return "";
|
||||
}
|
||||
#endif
|
||||
|
||||
/// a BSON element: its type, its name, and its value
|
||||
std::vector<std::uint8_t> bson_element(const std::uint8_t type, const std::string& name, const std::vector<std::uint8_t>& value)
|
||||
{
|
||||
std::vector<std::uint8_t> result = {type};
|
||||
result.insert(result.end(), name.begin(), name.end());
|
||||
result.push_back(0x00);
|
||||
result.insert(result.end(), value.begin(), value.end());
|
||||
return result;
|
||||
}
|
||||
|
||||
/// a BSON document of the given elements; @a size_offset is added to the
|
||||
/// size it declares
|
||||
std::vector<std::uint8_t> bson_document(const std::vector<std::vector<std::uint8_t>>& elements, const int size_offset = 0)
|
||||
{
|
||||
std::vector<std::uint8_t> body;
|
||||
for (const auto& element : elements)
|
||||
{
|
||||
body.insert(body.end(), element.begin(), element.end());
|
||||
}
|
||||
const auto size = static_cast<std::uint32_t>(static_cast<int>(body.size()) + 5 + size_offset);
|
||||
std::vector<std::uint8_t> result = {static_cast<std::uint8_t>(size & 0xFFu), static_cast<std::uint8_t>((size >> 8u) & 0xFFu),
|
||||
static_cast<std::uint8_t>((size >> 16u) & 0xFFu), static_cast<std::uint8_t>((size >> 24u) & 0xFFu)
|
||||
};
|
||||
result.insert(result.end(), body.begin(), body.end());
|
||||
result.push_back(0x00);
|
||||
return result;
|
||||
}
|
||||
|
||||
/// a BSON int32 value
|
||||
std::vector<std::uint8_t> bson_int32(const std::int32_t value)
|
||||
{
|
||||
const auto u = static_cast<std::uint32_t>(value);
|
||||
return {static_cast<std::uint8_t>(u & 0xFFu), static_cast<std::uint8_t>((u >> 8u) & 0xFFu),
|
||||
static_cast<std::uint8_t>((u >> 16u) & 0xFFu), static_cast<std::uint8_t>((u >> 24u) & 0xFFu)};
|
||||
}
|
||||
|
||||
/// a BSON string value, whose length is @a length_offset off
|
||||
std::vector<std::uint8_t> bson_string(const std::string& value, const std::int32_t length_offset = 0)
|
||||
{
|
||||
auto result = bson_int32(static_cast<std::int32_t>(value.size() + 1) + length_offset);
|
||||
result.insert(result.end(), value.begin(), value.end());
|
||||
result.push_back(0x00);
|
||||
return result;
|
||||
}
|
||||
|
||||
/// @a count bytes of value 0xAB
|
||||
std::vector<std::uint8_t> bytes(const std::size_t count)
|
||||
{
|
||||
return std::vector<std::uint8_t>(count, 0xAB);
|
||||
}
|
||||
|
||||
template<typename... Parts>
|
||||
std::vector<std::uint8_t> concatenated(const std::vector<std::uint8_t>& first, const Parts& ... rest)
|
||||
{
|
||||
std::vector<std::uint8_t> result = first;
|
||||
for (const auto& part : std::initializer_list<std::vector<std::uint8_t>> {rest...})
|
||||
{
|
||||
result.insert(result.end(), part.begin(), part.end());
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/// U+FFFD REPLACEMENT CHARACTER
|
||||
std::string replacement_character()
|
||||
{
|
||||
return "\xEF\xBF\xBD";
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("regression test - #3989 SAX parse_error() returning true")
|
||||
{
|
||||
SECTION("binary formats complete what was read before the input ends")
|
||||
{
|
||||
const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}};
|
||||
|
||||
const std::vector<std::pair<json::input_format_t, std::vector<std::uint8_t>>> encodings =
|
||||
{
|
||||
{json::input_format_t::cbor, json::to_cbor(j)},
|
||||
{json::input_format_t::msgpack, json::to_msgpack(j)},
|
||||
{json::input_format_t::ubjson, json::to_ubjson(j)},
|
||||
{json::input_format_t::ubjson, json::to_ubjson(j, true, true)},
|
||||
{json::input_format_t::bjdata, json::to_bjdata(j)},
|
||||
{json::input_format_t::bjdata, json::to_bjdata(j, true, true)},
|
||||
{json::input_format_t::bson, json::to_bson(j)},
|
||||
{json::input_format_t::bon8, json::to_bon8(j)},
|
||||
};
|
||||
|
||||
for (const auto& encoding : encodings)
|
||||
{
|
||||
const auto format = encoding.first;
|
||||
const auto& bytes = encoding.second;
|
||||
CAPTURE(format);
|
||||
|
||||
// every prefix is truncated input
|
||||
for (std::size_t length = 0; length < bytes.size(); ++length)
|
||||
{
|
||||
CAPTURE(length);
|
||||
const auto result = parse_binary_recovering(std::vector<std::uint8_t>(bytes.begin(), bytes.begin() + static_cast<std::ptrdiff_t>(length)), format);
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.errors == 1);
|
||||
CHECK(result.balanced);
|
||||
}
|
||||
|
||||
// the complete input is read as usual (binary values do not
|
||||
// round-trip through every format, so compare with a plain parse)
|
||||
json expected;
|
||||
nlohmann::detail::json_sax_dom_parser<json> dom(expected);
|
||||
CHECK(json::sax_parse(bytes, &dom, format));
|
||||
const auto complete = parse_binary_recovering(bytes, format);
|
||||
CHECK(complete.ok);
|
||||
CHECK(complete.errors == 0);
|
||||
CHECK(complete.value == expected);
|
||||
|
||||
// a byte after the value
|
||||
auto trailing_bytes = bytes;
|
||||
trailing_bytes.push_back(0x01);
|
||||
const auto trailing = parse_binary_recovering(trailing_bytes, format);
|
||||
CHECK(!trailing.ok);
|
||||
CHECK(trailing.errors == 1);
|
||||
CHECK(trailing.value == expected);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("containers without an end")
|
||||
{
|
||||
// these made the readers loop, or read on, after the error
|
||||
const auto cbor_array = parse_binary_recovering({0x9F}, json::input_format_t::cbor);
|
||||
CHECK(cbor_array.errors == 1);
|
||||
CHECK(cbor_array.value == json::array());
|
||||
|
||||
const auto cbor_map = parse_binary_recovering({0xBF, 0x61, 'a'}, json::input_format_t::cbor);
|
||||
CHECK(cbor_map.errors == 1);
|
||||
CHECK(cbor_map.value == json({{"a", nullptr}}));
|
||||
|
||||
const auto msgpack_array = parse_binary_recovering({0xDD, 0xFF, 0xFF, 0xFF, 0xFF}, json::input_format_t::msgpack);
|
||||
CHECK(msgpack_array.errors == 1);
|
||||
CHECK(msgpack_array.value == json::array());
|
||||
|
||||
const auto msgpack_map = parse_binary_recovering({0x81, 0xA1, 'a', 0x92, 0x01}, json::input_format_t::msgpack);
|
||||
CHECK(msgpack_map.errors == 1);
|
||||
CHECK(msgpack_map.value == json({{"a", {1}}}));
|
||||
}
|
||||
|
||||
SECTION("BJData ndarray")
|
||||
{
|
||||
// a 2x3 int8 array with two of its six elements; the annotated array
|
||||
// format opens an object and two arrays of its own
|
||||
const auto result = parse_binary_recovering({'[', '$', 'i', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2}, json::input_format_t::bjdata);
|
||||
CHECK(result.errors == 1);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.value == json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2}}}));
|
||||
}
|
||||
|
||||
SECTION("binary formats repair items whose end is known")
|
||||
{
|
||||
struct Repair
|
||||
{
|
||||
json::input_format_t format;
|
||||
std::vector<std::uint8_t> input;
|
||||
json expected;
|
||||
std::size_t errors;
|
||||
};
|
||||
|
||||
const std::vector<Repair> repairs =
|
||||
{
|
||||
// CBOR: tags are ignored (here tag 1 and the self-describe tag 55799)
|
||||
{json::input_format_t::cbor, {0x82, 0xC1, 0x05, 0xD9, 0xD9, 0xF7, 0x06}, {5, 6}, 2},
|
||||
// CBOR: undefined and other simple values become null
|
||||
{json::input_format_t::cbor, {0x84, 0xF7, 0xE0, 0xF8, 0x20, 0x01}, {nullptr, nullptr, nullptr, 1}, 3},
|
||||
// CBOR: ill-formed UTF-8 becomes U+FFFD, also in keys
|
||||
{json::input_format_t::cbor, {0xA1, 0x61, 0xFF, 0x62, 0xC3, 0x28}, {{replacement_character(), replacement_character() + "("}}, 2},
|
||||
// CBOR: members whose key is not a string are skipped, whatever their key and value
|
||||
{json::input_format_t::cbor, {0xA4, 0x01, 0x02, 0x82, 0x01, 0x02, 0xA1, 0x61, 'x', 0x9F, 0xFF, 0xC1, 0x01, 0x5F, 0x41, 0x00, 0xFF, 0x61, 'a', 0x03}, {{"a", 3}}, 3},
|
||||
{json::input_format_t::cbor, {0xBF, 0xF5, 0xBF, 0x61, 'x', 0x7F, 0x61, 'y', 0xFF, 0xFF, 0x61, 'a', 0x03, 0xFF}, {{"a", 3}}, 1},
|
||||
// MessagePack: members whose key is not a string are skipped
|
||||
{json::input_format_t::msgpack, {0x84, 0x01, 0x02, 0x81, 0xA1, 'x', 0x01, 0x92, 0x01, 0x02, 0xD4, 0x01, 0x02, 0xC0, 0xA1, 'a', 0x04}, {{"a", 4}}, 3},
|
||||
// MessagePack: ill-formed UTF-8 becomes U+FFFD
|
||||
{json::input_format_t::msgpack, {0x92, 0xA2, 0xC3, 0x28, 0xA3, 0xE2, 0x82, 'x'}, {replacement_character() + "(", replacement_character() + "x"}, 2},
|
||||
// UBJSON: a char that is not ASCII becomes U+FFFD
|
||||
{json::input_format_t::ubjson, {'[', 'C', 0x80, 'C', 'A', ']'}, {replacement_character(), "A"}, 1},
|
||||
// UBJSON: the longest beginning of a high-precision number is kept
|
||||
{json::input_format_t::ubjson, {'[', 'H', 'i', 5, '1', '2', 'a', 'b', 'c', 'H', 'i', 2, '1', '.', 'H', 'i', 3, 'a', 'b', 'c', 'H', 'i', 3, '4', '.', '5', ']'}, {12, 1, nullptr, 4.5}, 3},
|
||||
// BJData, too
|
||||
{json::input_format_t::bjdata, {'[', 'C', 0xFF, 'H', 'i', 2, '-', '1', 'H', 'i', 2, '-', 'x', ']'}, {replacement_character(), -1, nullptr}, 2},
|
||||
// BON8: members whose key is not a string are skipped
|
||||
{json::input_format_t::bon8, {0x89, 0x91, 0x92, 0xC9, 0x40, 0x82, 0x91, 0x92, 0x61, 0x93}, {{"a", 3}}, 2},
|
||||
{json::input_format_t::bon8, {0x8B, 0x91, 0x85, 0x91, 0xFE, 0xFA, 0x8B, 'x', 0x91, 0xFE, 0x61, 0x93, 0xFE}, {{"a", 3}}, 2},
|
||||
// BSON: elements of types the library does not read become null
|
||||
{
|
||||
json::input_format_t::bson, bson_document(
|
||||
{
|
||||
bson_element(0x07, "_id", bytes(12)), // ObjectId
|
||||
bson_element(0x09, "date", bytes(8)), // UTC datetime
|
||||
bson_element(0x13, "decimal", bytes(16)), // 128-bit decimal
|
||||
bson_element(0x0B, "regex", {'a', '+', 0, 'i', 0}), // regular expression
|
||||
bson_element(0x0D, "code", bson_string("f()")), // JavaScript code
|
||||
bson_element(0x0E, "symbol", bson_string("s")), // symbol
|
||||
bson_element(0x0C, "pointer", concatenated(bson_string("c"), bytes(12))), // DBPointer
|
||||
bson_element(0x0F, "scope", concatenated(bson_int32(15), bson_string("g"), bson_document({}))), // code with scope
|
||||
bson_element(0x06, "undefined", {}), // undefined
|
||||
bson_element(0xFF, "min", {}), // min key
|
||||
bson_element(0x7F, "max", {}), // max key
|
||||
bson_element(0x10, "z", bson_int32(7)),
|
||||
}),
|
||||
{{"_id", nullptr}, {"date", nullptr}, {"decimal", nullptr}, {"regex", nullptr}, {"code", nullptr}, {"symbol", nullptr}, {"pointer", nullptr}, {"scope", nullptr}, {"undefined", nullptr}, {"min", nullptr}, {"max", nullptr}, {"z", 7}},
|
||||
11
|
||||
},
|
||||
// BSON: an element of an unknown type becomes null, and the rest of its document is skipped
|
||||
{
|
||||
json::input_format_t::bson, bson_document(
|
||||
{
|
||||
bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3)), bson_element(0x10, "b", bson_int32(2))})),
|
||||
bson_element(0x04, "array", bson_document({bson_element(0x10, "0", bson_int32(1)), bson_element(0x42, "1", bytes(3))})),
|
||||
bson_element(0x10, "after", bson_int32(3)),
|
||||
}),
|
||||
{{"inner", {{"a", 1}, {"x", nullptr}}}, {"array", {1, nullptr}}, {"after", 3}},
|
||||
2
|
||||
},
|
||||
// BSON: so does a string or byte array whose length cannot be right
|
||||
{
|
||||
json::input_format_t::bson, bson_document(
|
||||
{
|
||||
bson_element(0x03, "inner", bson_document({bson_element(0x02, "s", bson_string("abc", -10)), bson_element(0x10, "b", bson_int32(2))})),
|
||||
bson_element(0x03, "bin", bson_document({bson_element(0x05, "b", concatenated(bson_int32(-1), bytes(1))), bson_element(0x10, "b", bson_int32(2))})),
|
||||
bson_element(0x10, "after", bson_int32(3)),
|
||||
}),
|
||||
{{"inner", {{"s", nullptr}}}, {"bin", {{"b", nullptr}}}, {"after", 3}},
|
||||
2
|
||||
},
|
||||
// BSON: a string without its terminator, and a document whose size does not match, are kept
|
||||
{
|
||||
json::input_format_t::bson, bson_document(
|
||||
{
|
||||
bson_element(0x02, "s", {2, 0, 0, 0, 'a', 'X'}),
|
||||
bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1))}, 1)),
|
||||
}),
|
||||
{{"s", "a"}, {"inner", {{"a", 1}}}},
|
||||
2
|
||||
},
|
||||
};
|
||||
|
||||
for (const auto& repair : repairs)
|
||||
{
|
||||
CAPTURE(repair.format);
|
||||
CAPTURE(repair.input);
|
||||
const auto result = parse_binary_recovering(repair.input, repair.format);
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.errors == repair.errors);
|
||||
CHECK(result.value == repair.expected);
|
||||
REQUIRE(!result.messages.empty());
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
// the first error is the one reported without recovering; under
|
||||
// JSON_NOEXCEPTION, reading without recovering aborts instead of
|
||||
// throwing, so there is no message to compare with
|
||||
CHECK(result.messages.front() == binary_error_message(repair.input, repair.format));
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("binary formats repair numbers that are out of range")
|
||||
{
|
||||
// CBOR: a negative integer below the range of number_integer_t
|
||||
const auto cbor = parse_binary_recovering({0x3B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}, json::input_format_t::cbor);
|
||||
CHECK(cbor.errors == 1);
|
||||
CHECK(cbor.value.is_number_float());
|
||||
CHECK(cbor.value.get<double>() == -18446744073709551616.0);
|
||||
|
||||
// UBJSON: a high-precision number too large for number_float_t
|
||||
const auto ubjson = parse_binary_recovering({'H', 'i', 5, '1', 'e', '9', '9', '9'}, json::input_format_t::ubjson);
|
||||
CHECK(ubjson.errors == 1);
|
||||
CHECK(ubjson.value.is_number_float());
|
||||
CHECK(std::isinf(ubjson.value.get<double>()));
|
||||
}
|
||||
|
||||
SECTION("binary formats stop where the end of an item is not known")
|
||||
{
|
||||
// a byte that begins no item
|
||||
const auto cbor = parse_binary_recovering({0x82, 0x01, 0x1C, 0x02}, json::input_format_t::cbor);
|
||||
CHECK(cbor.errors == 1);
|
||||
CHECK(cbor.value == json({1}));
|
||||
|
||||
// a key that is no item: the unused MessagePack byte, a CBOR break
|
||||
// in a map of known size, and the end of a BON8 container
|
||||
const auto msgpack = parse_binary_recovering({0x82, 0xA1, 'a', 0x01, 0xC1, 0x02}, json::input_format_t::msgpack);
|
||||
CHECK(msgpack.errors == 1);
|
||||
CHECK(msgpack.value == json({{"a", 1}}));
|
||||
const auto cbor_break = parse_binary_recovering({0xA2, 0x61, 'a', 0x01, 0xFF, 0x02}, json::input_format_t::cbor);
|
||||
CHECK(cbor_break.errors == 1);
|
||||
CHECK(cbor_break.value == json({{"a", 1}}));
|
||||
const auto bon8 = parse_binary_recovering({0x88, 0x61, 0x91, 0xFE}, json::input_format_t::bon8);
|
||||
CHECK(bon8.errors == 1);
|
||||
CHECK(bon8.value == json({{"a", 1}}));
|
||||
|
||||
// a skipped member that the input ends in
|
||||
const auto truncated = parse_binary_recovering({0xA2, 0x01, 0x82, 0x01}, json::input_format_t::cbor);
|
||||
CHECK(truncated.errors == 2);
|
||||
CHECK(truncated.balanced);
|
||||
CHECK(truncated.value == json::object());
|
||||
|
||||
// a BSON element of an unknown type in a document whose size cannot be right
|
||||
const auto bson = parse_binary_recovering(bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3))}, -10), json::input_format_t::bson);
|
||||
CHECK(bson.errors == 1);
|
||||
CHECK(bson.value == json({{"a", 1}, {"x", nullptr}}));
|
||||
}
|
||||
|
||||
SECTION("changed bytes in binary input")
|
||||
{
|
||||
const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}, {"i", "\xC3\xA4"}};
|
||||
|
||||
const std::vector<std::pair<json::input_format_t, std::vector<std::uint8_t>>> encodings =
|
||||
{
|
||||
{json::input_format_t::cbor, json::to_cbor(j)},
|
||||
{json::input_format_t::msgpack, json::to_msgpack(j)},
|
||||
{json::input_format_t::ubjson, json::to_ubjson(j)},
|
||||
{json::input_format_t::ubjson, json::to_ubjson(j, true, true)},
|
||||
{json::input_format_t::bjdata, json::to_bjdata(j)},
|
||||
{json::input_format_t::bjdata, json::to_bjdata(j, true, true)},
|
||||
{json::input_format_t::bson, json::to_bson(j)},
|
||||
{json::input_format_t::bon8, json::to_bon8(j)},
|
||||
};
|
||||
const std::vector<std::uint8_t> replacements = {0x00, 0x01, 0x7F, 0x80, 0xC1, 0xD9, 0xE0, 0xF7, 0xFE, 0xFF};
|
||||
|
||||
for (const auto& encoding : encodings)
|
||||
{
|
||||
const auto format = encoding.first;
|
||||
const auto& original = encoding.second;
|
||||
CAPTURE(format);
|
||||
|
||||
std::vector<std::vector<std::uint8_t>> inputs;
|
||||
for (std::size_t position = 0; position < original.size(); ++position)
|
||||
{
|
||||
for (const auto replacement : replacements)
|
||||
{
|
||||
auto changed = original;
|
||||
changed[position] = replacement;
|
||||
inputs.push_back(changed);
|
||||
}
|
||||
auto removed = original;
|
||||
removed.erase(removed.begin() + static_cast<std::ptrdiff_t>(position));
|
||||
inputs.push_back(removed);
|
||||
}
|
||||
|
||||
for (const auto& input : inputs)
|
||||
{
|
||||
CAPTURE(input);
|
||||
const auto result = parse_binary_recovering(input, format);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.errors <= input.size() + 1);
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
// an error is reported exactly if reading into a JSON value
|
||||
// fails, and the first one is the same (under JSON_NOEXCEPTION,
|
||||
// that reading aborts instead of throwing)
|
||||
const auto message = binary_error_message(input, format);
|
||||
CHECK(result.ok == message.empty());
|
||||
if (!result.ok && result.errors < 100)
|
||||
{
|
||||
CHECK(result.messages.front() == message);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("JSON text")
|
||||
{
|
||||
// the parser stopped, but reported success
|
||||
json j;
|
||||
RecoveringParser sax(j);
|
||||
CHECK(!json::sax_parse("[1,2,3,]", &sax));
|
||||
CHECK(sax.errors == 1);
|
||||
CHECK(j == json({1, 2, 3}));
|
||||
}
|
||||
|
||||
SECTION("the SAX parsers of the library stop")
|
||||
{
|
||||
json _;
|
||||
CHECK(json::from_cbor(std::vector<std::uint8_t> {0x9F}, true, false).is_discarded());
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<std::uint8_t> {0x9F}), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&);
|
||||
CHECK(json::parse("[1,2,3,]", nullptr, false).is_discarded());
|
||||
CHECK(!json::accept("[1,2,3,]"));
|
||||
}
|
||||
}
|
||||
|
||||
DOCTEST_CLANG_SUPPRESS_WARNING_POP
|
||||
|
||||
@@ -8,6 +8,3 @@ The following changes have been made to the code with respect to <https://github
|
||||
- membership check
|
||||
- made function from `_is_within`
|
||||
- removed unused variable `actual_path`
|
||||
- Added the optional config key `external`: include paths listed there are kept as
|
||||
`#include` directives instead of being inlined (the first directive per path; the
|
||||
repeated ones are commented out).
|
||||
|
||||
@@ -57,11 +57,6 @@ Python v.2.7.0 or higher is required.
|
||||
amalgamation. Have a look at `test/source.c.json` and `test/include.h.json`
|
||||
to see two examples.
|
||||
|
||||
The optional `external` list names include paths that are kept as `#include`
|
||||
directives instead of being inlined, e.g. `["nlohmann/json.hpp"]` for a header
|
||||
that includes another amalgamated header. Only the first directive for each
|
||||
of these paths is kept; the repeated ones are commented out.
|
||||
|
||||
* The `-s, --source` option should specify the path to the source directory.
|
||||
This is useful for supporting separate source and build directories.
|
||||
|
||||
|
||||
@@ -62,10 +62,6 @@ class Amalgamation(object):
|
||||
return None
|
||||
|
||||
def __init__(self, args):
|
||||
# include paths that are kept as #include directives instead of
|
||||
# being inlined (e.g. a header amalgamated on its own)
|
||||
self.external = []
|
||||
self.included_external = []
|
||||
with open(args.config, 'r') as f:
|
||||
config = json.loads(f.read())
|
||||
for key in config:
|
||||
@@ -224,14 +220,11 @@ class TranslationUnit(object):
|
||||
while include_match:
|
||||
if not _is_within(include_match, skippable_contexts):
|
||||
include_path = include_match.group("path")
|
||||
if include_path in self.amalgamation.external:
|
||||
includes.append((include_match, None))
|
||||
else:
|
||||
search_same_dir = include_match.group(1) == '"'
|
||||
found_included_path = self.amalgamation.find_included_file(
|
||||
include_path, self.file_dir if search_same_dir else None)
|
||||
if found_included_path:
|
||||
includes.append((include_match, found_included_path))
|
||||
search_same_dir = include_match.group(1) == '"'
|
||||
found_included_path = self.amalgamation.find_included_file(
|
||||
include_path, self.file_dir if search_same_dir else None)
|
||||
if found_included_path:
|
||||
includes.append((include_match, found_included_path))
|
||||
|
||||
include_match = self.include_pattern.search(self.content,
|
||||
include_match.end())
|
||||
@@ -242,17 +235,6 @@ class TranslationUnit(object):
|
||||
for include in includes:
|
||||
include_match, found_included_path = include
|
||||
tmp_content += self.content[prev_end:include_match.start()]
|
||||
if found_included_path is None:
|
||||
# an external header: keep the first directive and comment
|
||||
# out the repeated ones
|
||||
include_path = include_match.group("path")
|
||||
if include_path in self.amalgamation.included_external:
|
||||
tmp_content += "// {0}".format(include_match.group(0))
|
||||
else:
|
||||
self.amalgamation.included_external.append(include_path)
|
||||
tmp_content += include_match.group(0)
|
||||
prev_end = include_match.end()
|
||||
continue
|
||||
tmp_content += "// {0}\n".format(include_match.group(0))
|
||||
if found_included_path not in self.amalgamation.included_files:
|
||||
t = TranslationUnit(found_included_path, self.amalgamation, False)
|
||||
|
||||
@@ -1,9 +0,0 @@
|
||||
{
|
||||
"project": "JSON for Modern C++",
|
||||
"target": "single_include/nlohmann/json_view.hpp",
|
||||
"sources": [
|
||||
"include/nlohmann/json_view.hpp"
|
||||
],
|
||||
"include_paths": ["include"],
|
||||
"external": ["nlohmann/json.hpp"]
|
||||
}
|
||||
Reference in New Issue
Block a user