Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5e93415d91 | ||
|
|
b54ed188e6 | ||
|
|
cff0a61369 | ||
|
|
3926fcaac3 | ||
|
|
18dd5663b0 | ||
|
|
7e8e8e219b | ||
|
|
5f1727cef2 | ||
|
|
c16dd7e4f5 | ||
|
|
792853d725 | ||
|
|
4bdf1b7e74 | ||
|
|
b1e9d98e41 | ||
|
|
f855d257df | ||
|
|
3e683e9c04 | ||
|
|
d1d84ed9af | ||
|
|
de8529f99b | ||
|
|
677794f076 | ||
|
|
437a95cfdb | ||
|
|
c8735246d0 | ||
|
|
9adb510a0d |
@@ -38,14 +38,14 @@ jobs:
|
|||||||
|
|
||||||
# Initializes the CodeQL tools for scanning.
|
# Initializes the CodeQL tools for scanning.
|
||||||
- name: Initialize CodeQL
|
- name: Initialize CodeQL
|
||||||
uses: github/codeql-action/init@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4.38.2
|
uses: github/codeql-action/init@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1
|
||||||
with:
|
with:
|
||||||
languages: c-cpp
|
languages: c-cpp
|
||||||
|
|
||||||
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
||||||
# If this step fails, then you should remove it and run the build manually (see below)
|
# If this step fails, then you should remove it and run the build manually (see below)
|
||||||
- name: Autobuild
|
- name: Autobuild
|
||||||
uses: github/codeql-action/autobuild@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4.38.2
|
uses: github/codeql-action/autobuild@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1
|
||||||
|
|
||||||
- name: Perform CodeQL Analysis
|
- name: Perform CodeQL Analysis
|
||||||
uses: github/codeql-action/analyze@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4.38.2
|
uses: github/codeql-action/analyze@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1
|
||||||
|
|||||||
@@ -47,6 +47,6 @@ jobs:
|
|||||||
output: 'flawfinder_results.sarif'
|
output: 'flawfinder_results.sarif'
|
||||||
|
|
||||||
- name: Upload analysis results to GitHub Security tab
|
- name: Upload analysis results to GitHub Security tab
|
||||||
uses: github/codeql-action/upload-sarif@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4.38.2
|
uses: github/codeql-action/upload-sarif@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1
|
||||||
with:
|
with:
|
||||||
sarif_file: ${{github.workspace}}/flawfinder_results.sarif
|
sarif_file: ${{github.workspace}}/flawfinder_results.sarif
|
||||||
|
|||||||
@@ -80,6 +80,6 @@ jobs:
|
|||||||
|
|
||||||
# Upload the results to GitHub's code scanning dashboard.
|
# Upload the results to GitHub's code scanning dashboard.
|
||||||
- name: "Upload to code-scanning"
|
- name: "Upload to code-scanning"
|
||||||
uses: github/codeql-action/upload-sarif@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4.38.2
|
uses: github/codeql-action/upload-sarif@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1
|
||||||
with:
|
with:
|
||||||
sarif_file: results.sarif
|
sarif_file: results.sarif
|
||||||
|
|||||||
@@ -65,7 +65,7 @@ jobs:
|
|||||||
|
|
||||||
# Upload SARIF file generated in previous step
|
# Upload SARIF file generated in previous step
|
||||||
- name: Upload SARIF file
|
- name: Upload SARIF file
|
||||||
uses: github/codeql-action/upload-sarif@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4.38.2
|
uses: github/codeql-action/upload-sarif@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1
|
||||||
with:
|
with:
|
||||||
sarif_file: semgrep.sarif
|
sarif_file: semgrep.sarif
|
||||||
if: always()
|
if: always()
|
||||||
|
|||||||
@@ -23,10 +23,6 @@ Files: tests/thirdparty/fifo_map/*
|
|||||||
Copyright: 2015-2017 Niels Lohmann
|
Copyright: 2015-2017 Niels Lohmann
|
||||||
License: MIT
|
License: MIT
|
||||||
|
|
||||||
Files: tests/thirdparty/Fuzzer/*
|
|
||||||
Copyright: 2003-2022 LLVM Project.
|
|
||||||
License: Apache-2.0
|
|
||||||
|
|
||||||
Files: tests/thirdparty/imapdl/*
|
Files: tests/thirdparty/imapdl/*
|
||||||
Copyright: 2017 Georg Sauthoff <mail@gms.tf>
|
Copyright: 2017 Georg Sauthoff <mail@gms.tf>
|
||||||
License: GPL-3.0-only
|
License: GPL-3.0-only
|
||||||
|
|||||||
@@ -495,7 +495,7 @@ bool key(string_t& val);
|
|||||||
bool parse_error(std::size_t position, const std::string& last_token, const detail::exception& ex);
|
bool parse_error(std::size_t position, const std::string& last_token, const detail::exception& ex);
|
||||||
```
|
```
|
||||||
|
|
||||||
The return value of each function determines whether parsing should proceed.
|
The return value of each function determines whether parsing should proceed. For `parse_error`, returning `true` [recovers from the error](https://json.nlohmann.me/features/parsing/error_recovery/): the parser repairs the input and continues.
|
||||||
|
|
||||||
To implement your own SAX handler, proceed as follows:
|
To implement your own SAX handler, proceed as follows:
|
||||||
|
|
||||||
@@ -503,7 +503,7 @@ To implement your own SAX handler, proceed as follows:
|
|||||||
2. Create an object of your SAX interface class, e.g. `my_sax`.
|
2. Create an object of your SAX interface class, e.g. `my_sax`.
|
||||||
3. Call `bool json::sax_parse(input, &my_sax)`; where the first parameter can be any input like a string or an input stream and the second parameter is a pointer to your SAX interface.
|
3. Call `bool json::sax_parse(input, &my_sax)`; where the first parameter can be any input like a string or an input stream and the second parameter is a pointer to your SAX interface.
|
||||||
|
|
||||||
Note the `sax_parse` function only returns a `bool` indicating the result of the last executed SAX event. It does not return a `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error -- it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file [`json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp).
|
Note the `sax_parse` function only returns a `bool` indicating whether the input was parsed without errors and no SAX event returned `false`. It does not return a `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error -- it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file [`json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp).
|
||||||
|
|
||||||
### STL-like access
|
### STL-like access
|
||||||
|
|
||||||
|
|||||||
@@ -90,7 +90,9 @@ The SAX event lister must follow the interface of [`json_sax`](../json_sax/index
|
|||||||
|
|
||||||
## Return value
|
## Return value
|
||||||
|
|
||||||
return value of the last processed SAX event
|
`#!cpp true` if the input was parsed without errors and no SAX event returned `#!cpp false`; `#!cpp false` otherwise.
|
||||||
|
In particular, the result is `#!cpp false` for input with errors, even if the SAX parser recovered from all of them
|
||||||
|
(see [error recovery](../../features/parsing/error_recovery.md)).
|
||||||
|
|
||||||
## Exception safety
|
## Exception safety
|
||||||
|
|
||||||
@@ -138,6 +140,7 @@ A UTF-8 byte order mark is silently ignored.
|
|||||||
- Ignoring comments via `ignore_comments` added in version 3.9.0.
|
- Ignoring comments via `ignore_comments` added in version 3.9.0.
|
||||||
- Added `ignore_trailing_commas` in version 3.13.0.
|
- Added `ignore_trailing_commas` in version 3.13.0.
|
||||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||||
|
- Recovering from parse errors (see [`parse_error`](../json_sax/parse_error.md)) added in version 3.13.0.
|
||||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||||
- `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave a `#!cpp std::istream` positioned right
|
- `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave a `#!cpp std::istream` positioned right
|
||||||
after the parsed value when `strict` is `#!cpp false`.
|
after the parsed value when `strict` is `#!cpp false`.
|
||||||
|
|||||||
@@ -7,7 +7,8 @@ struct json_sax;
|
|||||||
|
|
||||||
This class describes the SAX interface used by [sax_parse](../basic_json/sax_parse.md). Each function is called in
|
This class describes the SAX interface used by [sax_parse](../basic_json/sax_parse.md). Each function is called in
|
||||||
different situations while the input is parsed. The boolean return value informs the parser whether to continue
|
different situations while the input is parsed. The boolean return value informs the parser whether to continue
|
||||||
processing the input.
|
processing the input; for [`parse_error`](parse_error.md), it decides whether to
|
||||||
|
[recover from the error](../../features/parsing/error_recovery.md).
|
||||||
|
|
||||||
## Template parameters
|
## Template parameters
|
||||||
|
|
||||||
|
|||||||
@@ -21,7 +21,14 @@ A parse error occurred.
|
|||||||
|
|
||||||
## Return value
|
## Return value
|
||||||
|
|
||||||
Whether parsing should proceed (**must return `#!cpp false`**).
|
Whether to recover from the error:
|
||||||
|
|
||||||
|
- `#!cpp false` stops parsing.
|
||||||
|
- `#!cpp true` recovers from the error: the error is repaired and parsing continues. If that is not possible, which
|
||||||
|
happens in the binary formats when the end of the item with the error is unknown, the value read so far is completed
|
||||||
|
and parsing stops. See [error recovery](../../features/parsing/error_recovery.md) for how errors are repaired.
|
||||||
|
|
||||||
|
Either way, [`sax_parse`](../basic_json/sax_parse.md) returns `#!cpp false`.
|
||||||
|
|
||||||
## Examples
|
## Examples
|
||||||
|
|
||||||
@@ -39,6 +46,22 @@ Whether parsing should proceed (**must return `#!cpp false`**).
|
|||||||
--8<-- "examples/sax_parse.output"
|
--8<-- "examples/sax_parse.output"
|
||||||
```
|
```
|
||||||
|
|
||||||
|
??? example
|
||||||
|
|
||||||
|
The example below shows how a SAX parser recovers from errors.
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
--8<-- "examples/sax_parse__error_recovery.cpp"
|
||||||
|
```
|
||||||
|
|
||||||
|
Output:
|
||||||
|
|
||||||
|
```
|
||||||
|
--8<-- "examples/sax_parse__error_recovery.output"
|
||||||
|
```
|
||||||
|
|
||||||
## Version history
|
## Version history
|
||||||
|
|
||||||
- Added in version 3.2.0.
|
- Added in version 3.2.0.
|
||||||
|
- Returning `#!cpp true` recovers from the error since version 3.13.0; before, parsing stopped, but the result of
|
||||||
|
[`sax_parse`](../basic_json/sax_parse.md) could be wrong.
|
||||||
|
|||||||
@@ -0,0 +1,43 @@
|
|||||||
|
#include <iostream>
|
||||||
|
#include <iomanip>
|
||||||
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
|
using json = nlohmann::json;
|
||||||
|
|
||||||
|
// a SAX parser that creates a JSON value like json::parse does, but that
|
||||||
|
// recovers from parse errors instead of stopping at the first one
|
||||||
|
class recovering_parser : public nlohmann::detail::json_sax_dom_parser<json>
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
explicit recovering_parser(json& result)
|
||||||
|
: nlohmann::detail::json_sax_dom_parser<json>(result, false)
|
||||||
|
{}
|
||||||
|
|
||||||
|
bool parse_error(std::size_t position,
|
||||||
|
const std::string& /*last_token*/,
|
||||||
|
const json::exception& ex)
|
||||||
|
{
|
||||||
|
std::cout << "byte " << position << ": " << ex.what() << '\n';
|
||||||
|
|
||||||
|
// repair the input and continue
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
int main()
|
||||||
|
{
|
||||||
|
// JSON text with several mistakes that ends too early
|
||||||
|
const std::string text = R"({
|
||||||
|
"name": "Hello World",
|
||||||
|
"tags": ["a" "b",],
|
||||||
|
"valid": tru,
|
||||||
|
"size": 1.,
|
||||||
|
"nested": {"x": 1)";
|
||||||
|
|
||||||
|
json result;
|
||||||
|
recovering_parser sax(result);
|
||||||
|
const bool valid = json::sax_parse(text, &sax);
|
||||||
|
|
||||||
|
std::cout << "\nvalid JSON: " << std::boolalpha << valid << '\n'
|
||||||
|
<< std::setw(4) << result << std::endl;
|
||||||
|
}
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
byte 49: [json.exception.parse_error.101] parse error at line 3, column 20: syntax error while parsing array - unexpected string literal; expected ']'
|
||||||
|
byte 51: [json.exception.parse_error.101] parse error at line 3, column 22: syntax error while parsing value - unexpected ']'; expected '[', '{', or a literal
|
||||||
|
byte 70: [json.exception.parse_error.101] parse error at line 4, column 17: syntax error while parsing value - invalid literal; last read: '"valid": tru,'
|
||||||
|
byte 86: [json.exception.parse_error.101] parse error at line 5, column 15: syntax error while parsing value - invalid number; expected digit after '.'; last read: '1.,'
|
||||||
|
byte 109: [json.exception.parse_error.101] parse error at line 6, column 22: syntax error while parsing object - unexpected end of input; expected '}'
|
||||||
|
|
||||||
|
valid JSON: false
|
||||||
|
{
|
||||||
|
"name": "Hello World",
|
||||||
|
"nested": {
|
||||||
|
"x": 1
|
||||||
|
},
|
||||||
|
"size": 1,
|
||||||
|
"tags": [
|
||||||
|
"a",
|
||||||
|
"b"
|
||||||
|
],
|
||||||
|
"valid": null
|
||||||
|
}
|
||||||
@@ -0,0 +1,121 @@
|
|||||||
|
# Error Recovery
|
||||||
|
|
||||||
|
By default, parsing stops at the first error. With the [SAX interface](sax_interface.md), you can instead ask the
|
||||||
|
parser to *recover*: to repair the error and continue, so that you get as much as possible out of malformed input, for
|
||||||
|
instance a file that was cut off, JSON edited by hand, or the output of a language model.
|
||||||
|
|
||||||
|
## Recovering from errors
|
||||||
|
|
||||||
|
The SAX parser's [`parse_error`](../../api/json_sax/parse_error.md) function is called for every error. Its return value
|
||||||
|
decides what happens next:
|
||||||
|
|
||||||
|
- `#!cpp false` stops parsing. This is what the SAX parsers of the library do, so [`parse`](../../api/basic_json/parse.md)
|
||||||
|
and [`accept`](../../api/basic_json/accept.md) never recover.
|
||||||
|
- `#!cpp true` repairs the error and continues parsing.
|
||||||
|
|
||||||
|
When recovering, the SAX parser still receives well-formed events: every `start_object` or `start_array` is followed by
|
||||||
|
the matching `end_object` or `end_array`, and every `key` is followed by exactly one value. A SAX parser that creates a
|
||||||
|
JSON value, such as the one in the example below, therefore gets a complete value. Parsing always ends, and
|
||||||
|
[`sax_parse`](../../api/basic_json/sax_parse.md) returns `#!cpp false` for input that is not valid JSON, even if every
|
||||||
|
error was repaired. Each token is reported at most once, and the SAX parser can stop at any error by returning
|
||||||
|
`#!cpp false`.
|
||||||
|
|
||||||
|
!!! example
|
||||||
|
|
||||||
|
The example below derives a SAX parser from the library's parser for `json` values (`json_sax_dom_parser`),
|
||||||
|
and recovers from all errors.
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
--8<-- "examples/sax_parse__error_recovery.cpp"
|
||||||
|
```
|
||||||
|
|
||||||
|
Output:
|
||||||
|
|
||||||
|
```
|
||||||
|
--8<-- "examples/sax_parse__error_recovery.output"
|
||||||
|
```
|
||||||
|
|
||||||
|
## How errors are repaired
|
||||||
|
|
||||||
|
Each error is repaired with the smallest local edit: a missing separator is inserted, a stray token is removed, what can
|
||||||
|
be read of a broken string or number is kept, and a value that cannot be read at all becomes `#!json null`.
|
||||||
|
|
||||||
|
| Mistake | Repair | Example | Result |
|
||||||
|
|---------------------------|--------------------------------------------------------------------------------|------------------------------------------|----------------------------|
|
||||||
|
| missing `,` or `:` | inserted | `#!json [1 2]`, `#!json {"a" 1}` | `[1,2]`, `{"a":1}` |
|
||||||
|
| missing value | `#!json null` for an object key or between commas in an array | `#!json {"a":}`, `#!json [1,,2]` | `{"a":null}`, `[1,null,2]` |
|
||||||
|
| trailing comma | removed | `#!json [1,2,]` | `[1,2]` |
|
||||||
|
| broken string | invalid escapes and bytes are replaced (see below); a line break ends the string | `#!json ["a\qb"]` | `["aqb"]` |
|
||||||
|
| broken number | the longest valid beginning is kept | `#!json [1., 2e+]` | `[1,2]` |
|
||||||
|
| unreadable value | `#!json null` | `#!json [1, NaN, tru]` | `[1,null,null]` |
|
||||||
|
| number too large | passed as infinity, together with its text | `#!json [1e999]` | infinity (see below) |
|
||||||
|
| stray `:` | removed | `#!json ["a":1]` | `["a",1]` |
|
||||||
|
| member without a key | skipped up to the next `,` or `}` | `#!json {1:2, "b":3}` | `{"b":3}` |
|
||||||
|
| wrong closing bracket | closes the innermost array or object | `#!json {"a":[1,2}, "b":3}` | `{"a":[1,2],"b":3}` |
|
||||||
|
| input ends too early | all open arrays and objects are closed | `#!json {"a":[1,2` | `{"a":[1,2]}` |
|
||||||
|
| text before the value | skipped | `#!json )]}'{"a":1}` | `{"a":1}` |
|
||||||
|
|
||||||
|
In a string, an unknown escape like `\q` stands for the escaped character (`q`), as in JavaScript. An invalid `\u`
|
||||||
|
escape, a lone surrogate, and ill-formed UTF-8 are each replaced by U+FFFD (REPLACEMENT CHARACTER), and control
|
||||||
|
characters are kept. A string without its closing quote ends at the next line break or at the end of the input.
|
||||||
|
|
||||||
|
The input after the top-level value is not repaired: as without recovery, it is reported as an error, and parsing stops.
|
||||||
|
|
||||||
|
## Binary formats
|
||||||
|
|
||||||
|
The binary formats ([BJData](../binary_formats/bjdata.md), [BON8](../binary_formats/bon8.md),
|
||||||
|
[BSON](../binary_formats/bson.md), [CBOR](../binary_formats/cbor.md), [MessagePack](../binary_formats/messagepack.md),
|
||||||
|
and [UBJSON](../binary_formats/ubjson.md)) have no delimiters to find the next value by. So what can be repaired depends
|
||||||
|
on whether the end of the item with the error is known, a distinction that
|
||||||
|
[RFC 8949, Section 5.3](https://www.rfc-editor.org/rfc/rfc8949.html#section-5.3) makes for CBOR, too.
|
||||||
|
|
||||||
|
If the item is complete, but cannot be passed on as it is, it is replaced, and parsing continues after it:
|
||||||
|
|
||||||
|
| Mistake | Formats | Repair |
|
||||||
|
|---------------------------------------------------------------------|-----------------------------------------|-------------------------------------------------------------------------|
|
||||||
|
| tag | CBOR | ignored |
|
||||||
|
| simple value other than `false`, `true`, and `null`, like undefined | CBOR | `#!json null` |
|
||||||
|
| negative integer below the range of `number_integer_t` | CBOR | the nearest floating-point number |
|
||||||
|
| string that is not valid UTF-8 | BJData, BSON, CBOR, MessagePack, UBJSON | each ill-formed sequence becomes U+FFFD |
|
||||||
|
| character (`C`) that is not ASCII | BJData, UBJSON | U+FFFD |
|
||||||
|
| invalid high-precision number (`H`) | BJData, UBJSON | the longest valid beginning is kept, as for JSON text, or `#!json null` |
|
||||||
|
| high-precision number too large | BJData, UBJSON | passed as infinity, together with its text |
|
||||||
|
| object key that is not a string | BON8, CBOR, MessagePack | the member is skipped |
|
||||||
|
| element of a type the library does not read, like ObjectId or date | BSON | `#!json null` |
|
||||||
|
| string without its terminator | BSON | kept |
|
||||||
|
| document whose size does not match its content | BSON | kept |
|
||||||
|
|
||||||
|
CBOR tags and simple values are repaired as [RFC 8949, Section 6.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-6.1)
|
||||||
|
suggests for converting CBOR to JSON. Note that [`sax_parse`](../../api/basic_json/sax_parse.md) has no parameter for
|
||||||
|
CBOR tags, so every tag is an error there; when recovering, tags are ignored like with
|
||||||
|
[`cbor_tag_handler_t::ignore`](../../api/basic_json/cbor_tag_handler_t.md).
|
||||||
|
|
||||||
|
After any other error, the end of the item is unknown: the input ended, a byte is not a valid type marker, or a size
|
||||||
|
cannot be right. Parsing then stops, and the value read so far is completed: a key that waits for its value gets
|
||||||
|
`#!json null`, and all open arrays and objects are closed. This keeps everything before the error of an input that was
|
||||||
|
cut off. The exception is BSON, which stores the size of every document: an element whose end is unknown gets
|
||||||
|
`#!json null`, the rest of its document is skipped, and parsing continues after the document.
|
||||||
|
|
||||||
|
## Limitations
|
||||||
|
|
||||||
|
- A repair is a guess. For example, `#!json {"a" "b": 1}` could be meant as `#!json {"a": "b"}` or as
|
||||||
|
`#!json {"a": null, "b": 1}`; it is repaired to the former. Treat recovered values as a best effort, and check the
|
||||||
|
reported errors.
|
||||||
|
- A closing bracket always closes the innermost array or object. If a bracket is missing rather than wrong, the
|
||||||
|
repair differs from the intention: `#!json {"a": {"b": [1, 2}, "c": 3}` is repaired to
|
||||||
|
`#!json {"a": {"b": [1, 2], "c": 3}}`, although `#!json {"a": {"b": [1, 2]}, "c": 3}` may have been meant.
|
||||||
|
- Keys without quotes, and strings in single quotes, are not supported; such members are skipped.
|
||||||
|
- In the binary formats, a member that is skipped because its key is not a string is lost, and so are the elements of a
|
||||||
|
BSON document after one whose end is unknown.
|
||||||
|
- A number that is too large for `number_float_t` is passed as positive or negative infinity. The SAX parser's
|
||||||
|
`number_float` also gets the number's text, but a JSON value cannot store it, and
|
||||||
|
[`dump`](../../api/basic_json/dump.md) serializes infinity as `#!json null`.
|
||||||
|
- When parsing is not strict (see [`sax_parse`](../../api/basic_json/sax_parse.md)), a repair may read parts of the
|
||||||
|
input after the value, for instance of the next value in a stream of concatenated values.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [SAX interface](sax_interface.md) - implement a custom SAX handler
|
||||||
|
- [`parse_error`](../../api/json_sax/parse_error.md) - the SAX event for parse errors
|
||||||
|
- [`sax_parse`](../../api/basic_json/sax_parse.md) - generate SAX events
|
||||||
|
- [parsing and exceptions](parse_exceptions.md) - control error handling
|
||||||
@@ -65,7 +65,7 @@ You can influence a DOM parse without switching to the SAX interface by passing
|
|||||||
When the input is not valid JSON, the `parse` function throws an exception by default. If exceptions are undesired or
|
When the input is not valid JSON, the `parse` function throws an exception by default. If exceptions are undesired or
|
||||||
unavailable, the parser can instead return a discarded value, or [`accept`](../../api/basic_json/accept.md) can be used
|
unavailable, the parser can instead return a discarded value, or [`accept`](../../api/basic_json/accept.md) can be used
|
||||||
to only check whether an input is valid JSON. See [parsing and exceptions](parse_exceptions.md) for the available
|
to only check whether an input is valid JSON. See [parsing and exceptions](parse_exceptions.md) for the available
|
||||||
options.
|
options. To get as much as possible out of malformed input, a SAX parser can [recover from errors](error_recovery.md).
|
||||||
|
|
||||||
## See also
|
## See also
|
||||||
|
|
||||||
@@ -76,3 +76,4 @@ options.
|
|||||||
- [parser callbacks](parser_callbacks.md) - influence the parsing by a callback function
|
- [parser callbacks](parser_callbacks.md) - influence the parsing by a callback function
|
||||||
- [SAX interface](sax_interface.md) - implement a custom SAX handler
|
- [SAX interface](sax_interface.md) - implement a custom SAX handler
|
||||||
- [parsing and exceptions](parse_exceptions.md) - control error handling
|
- [parsing and exceptions](parse_exceptions.md) - control error handling
|
||||||
|
- [error recovery](error_recovery.md) - get as much as possible out of malformed input
|
||||||
|
|||||||
@@ -64,7 +64,8 @@ bool parse_error(std::size_t position,
|
|||||||
const json::exception& ex);
|
const json::exception& ex);
|
||||||
```
|
```
|
||||||
|
|
||||||
The return value indicates whether the parsing should continue, so the function should usually return `#!cpp false`.
|
The return value decides whether to stop parsing (`#!cpp false`) or to repair the error and continue
|
||||||
|
(`#!cpp true`); see [error recovery](error_recovery.md) for the latter.
|
||||||
|
|
||||||
??? example
|
??? example
|
||||||
|
|
||||||
|
|||||||
@@ -60,7 +60,8 @@ bool key(string_t& val);
|
|||||||
bool parse_error(std::size_t position, const std::string& last_token, const json::exception& ex);
|
bool parse_error(std::size_t position, const std::string& last_token, const json::exception& ex);
|
||||||
```
|
```
|
||||||
|
|
||||||
The return value of each function determines whether parsing should proceed.
|
The return value of each function determines whether parsing should proceed. For `parse_error`, returning
|
||||||
|
`#!cpp true` [recovers from the error](error_recovery.md).
|
||||||
|
|
||||||
To implement your own SAX handler, proceed as follows:
|
To implement your own SAX handler, proceed as follows:
|
||||||
|
|
||||||
@@ -68,7 +69,7 @@ To implement your own SAX handler, proceed as follows:
|
|||||||
2. Create an object of your SAX interface class, e.g. `my_sax`.
|
2. Create an object of your SAX interface class, e.g. `my_sax`.
|
||||||
3. Call `#!cpp bool json::sax_parse(input, &my_sax);` where the first parameter can be any input like a string or an input stream and the second parameter is a pointer to your SAX interface.
|
3. Call `#!cpp bool json::sax_parse(input, &my_sax);` where the first parameter can be any input like a string or an input stream and the second parameter is a pointer to your SAX interface.
|
||||||
|
|
||||||
Note the `sax_parse` function only returns a `#!cpp bool` indicating the result of the last executed SAX event. It does not return `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error - it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file `json_sax.hpp`.
|
Note the `sax_parse` function only returns a `#!cpp bool` indicating whether the input was parsed without errors and no SAX event returned `#!cpp false`. It does not return `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error - it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file `json_sax.hpp`.
|
||||||
|
|
||||||
## See also
|
## See also
|
||||||
|
|
||||||
|
|||||||
@@ -87,6 +87,7 @@ nav:
|
|||||||
- features/object_order.md
|
- features/object_order.md
|
||||||
- Parsing:
|
- Parsing:
|
||||||
- features/parsing/index.md
|
- features/parsing/index.md
|
||||||
|
- features/parsing/error_recovery.md
|
||||||
- features/parsing/json_lines.md
|
- features/parsing/json_lines.md
|
||||||
- features/parsing/parse_exceptions.md
|
- features/parsing/parse_exceptions.md
|
||||||
- features/parsing/parser_callbacks.md
|
- features/parsing/parser_callbacks.md
|
||||||
|
|||||||
@@ -8,12 +8,12 @@
|
|||||||
|
|
||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
|
#include <algorithm> // min
|
||||||
#include <array> // array
|
#include <array> // array
|
||||||
#include <cstddef> // size_t
|
#include <cstddef> // size_t
|
||||||
|
#include <cstdint> // uint32_t
|
||||||
#include <cstring> // strlen
|
#include <cstring> // strlen
|
||||||
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
|
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
|
||||||
#include <memory> // shared_ptr, make_shared, addressof
|
|
||||||
#include <numeric> // accumulate
|
|
||||||
#include <streambuf> // streambuf
|
#include <streambuf> // streambuf
|
||||||
#include <string> // string, char_traits
|
#include <string> // string, char_traits
|
||||||
#include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer
|
#include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer
|
||||||
@@ -28,6 +28,7 @@
|
|||||||
#include <nlohmann/detail/iterators/iterator_traits.hpp>
|
#include <nlohmann/detail/iterators/iterator_traits.hpp>
|
||||||
#include <nlohmann/detail/macro_scope.hpp>
|
#include <nlohmann/detail/macro_scope.hpp>
|
||||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||||
|
#include <nlohmann/detail/string_utils.hpp>
|
||||||
|
|
||||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||||
namespace detail
|
namespace detail
|
||||||
@@ -82,8 +83,9 @@ class file_input_adapter
|
|||||||
};
|
};
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
Input adapter for a (caching) istream. Ignores a UFT Byte Order Mark at
|
Input adapter for a (caching) istream. Does not skip a UTF Byte Order Mark
|
||||||
beginning of input. Does not support changing the underlying std::streambuf
|
itself; that is done by the lexer's skip_bom(). Does not support changing
|
||||||
|
the underlying std::streambuf
|
||||||
in mid-input. Maintains underlying std::istream and std::streambuf to support
|
in mid-input. Maintains underlying std::istream and std::streambuf to support
|
||||||
subsequent use of standard std::istream operations to process any input
|
subsequent use of standard std::istream operations to process any input
|
||||||
characters following those used in parsing the JSON input. Clears the
|
characters following those used in parsing the JSON input. Clears the
|
||||||
@@ -454,32 +456,14 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
|
|||||||
// get the current character
|
// get the current character
|
||||||
const auto wc = input.get_character();
|
const auto wc = input.get_character();
|
||||||
|
|
||||||
// UTF-32 to UTF-8 encoding
|
if (wc <= 0x10FFFF)
|
||||||
if (wc < 0x80)
|
|
||||||
{
|
{
|
||||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
// UTF-32 to UTF-8 encoding
|
||||||
utf8_bytes_filled = 1;
|
utf8_bytes_filled = 0;
|
||||||
}
|
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
|
||||||
else if (wc <= 0x7FF)
|
{
|
||||||
{
|
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
|
||||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u) & 0x1Fu));
|
});
|
||||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
|
||||||
utf8_bytes_filled = 2;
|
|
||||||
}
|
|
||||||
else if (wc <= 0xFFFF)
|
|
||||||
{
|
|
||||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u) & 0x0Fu));
|
|
||||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
|
|
||||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
|
||||||
utf8_bytes_filled = 3;
|
|
||||||
}
|
|
||||||
else if (wc <= 0x10FFFF)
|
|
||||||
{
|
|
||||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | ((static_cast<unsigned int>(wc) >> 18u) & 0x07u));
|
|
||||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 12u) & 0x3Fu));
|
|
||||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
|
|
||||||
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
|
||||||
utf8_bytes_filled = 4;
|
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -516,24 +500,15 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
|
|||||||
// get the current character
|
// get the current character
|
||||||
const auto wc = input.get_character();
|
const auto wc = input.get_character();
|
||||||
|
|
||||||
// UTF-16 to UTF-8 encoding
|
if (0xD800 > wc || wc >= 0xE000)
|
||||||
if (wc < 0x80)
|
|
||||||
{
|
{
|
||||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
// a UTF-16 code unit outside the surrogate range is a valid
|
||||||
utf8_bytes_filled = 1;
|
// code point (at most U+FFFF) on its own
|
||||||
}
|
utf8_bytes_filled = 0;
|
||||||
else if (wc <= 0x7FF)
|
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
|
||||||
{
|
{
|
||||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u)));
|
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
|
||||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
});
|
||||||
utf8_bytes_filled = 2;
|
|
||||||
}
|
|
||||||
else if (0xD800 > wc || wc >= 0xE000)
|
|
||||||
{
|
|
||||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u)));
|
|
||||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
|
|
||||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
|
||||||
utf8_bytes_filled = 3;
|
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -551,11 +526,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
|
|||||||
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
|
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
|
||||||
{
|
{
|
||||||
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
|
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
|
||||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
|
utf8_bytes_filled = 0;
|
||||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
|
encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
|
||||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu));
|
{
|
||||||
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu));
|
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
|
||||||
utf8_bytes_filled = 4;
|
});
|
||||||
valid_pair = true;
|
valid_pair = true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -884,9 +859,9 @@ auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) /
|
|||||||
return input_adapter(array, array + N);
|
return input_adapter(array, array + N);
|
||||||
}
|
}
|
||||||
|
|
||||||
// This class only handles inputs of input_buffer_adapter type.
|
// This class only handles inputs that construct a contiguous_bytes_input_adapter
|
||||||
// It's required so that expressions like {ptr, len} can be implicitly cast
|
// (e.g. span_input_adapter). It's required so that expressions like {ptr, len}
|
||||||
// to the correct adapter.
|
// can be implicitly cast to the correct adapter.
|
||||||
class span_input_adapter
|
class span_input_adapter
|
||||||
{
|
{
|
||||||
public:
|
public:
|
||||||
|
|||||||
@@ -10,6 +10,7 @@
|
|||||||
|
|
||||||
#include <algorithm> // find_if, min
|
#include <algorithm> // find_if, min
|
||||||
#include <cstddef>
|
#include <cstddef>
|
||||||
|
#include <limits> // numeric_limits
|
||||||
#include <string> // string
|
#include <string> // string
|
||||||
#include <type_traits> // enable_if_t
|
#include <type_traits> // enable_if_t
|
||||||
#include <utility> // move, pair
|
#include <utility> // move, pair
|
||||||
@@ -131,7 +132,9 @@ struct json_sax
|
|||||||
@param[in] position the position in the input where the error occurs
|
@param[in] position the position in the input where the error occurs
|
||||||
@param[in] last_token the last read token
|
@param[in] last_token the last read token
|
||||||
@param[in] ex an exception object describing the error
|
@param[in] ex an exception object describing the error
|
||||||
@return whether parsing should proceed (must return false)
|
@return whether to recover from the error: false stops parsing; true
|
||||||
|
repairs the error and continues, or, if that is not possible,
|
||||||
|
stops after completing the value read so far
|
||||||
*/
|
*/
|
||||||
virtual bool parse_error(std::size_t position,
|
virtual bool parse_error(std::size_t position,
|
||||||
const std::string& last_token,
|
const std::string& last_token,
|
||||||
@@ -175,6 +178,88 @@ template<typename ArrayType>
|
|||||||
inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/)
|
inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/)
|
||||||
{}
|
{}
|
||||||
|
|
||||||
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
|
/*!
|
||||||
|
@brief set the diagnostic positions of a value the DOM SAX parsers just stored
|
||||||
|
|
||||||
|
Shared by json_sax_dom_parser and json_sax_dom_callback_parser. basic_json
|
||||||
|
befriends this struct, as the position members are private.
|
||||||
|
*/
|
||||||
|
struct diagnostic_positions
|
||||||
|
{
|
||||||
|
/*!
|
||||||
|
@param[in,out] v the value that was just parsed
|
||||||
|
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
|
||||||
|
*/
|
||||||
|
template<typename BasicJsonType, typename LexerType>
|
||||||
|
static void set_from_lexer(BasicJsonType& v, LexerType* lexer)
|
||||||
|
{
|
||||||
|
if (lexer)
|
||||||
|
{
|
||||||
|
// Lexer has read past the current field value, so set the end position to the current position.
|
||||||
|
// The start position will be set below based on the length of the string representation
|
||||||
|
// of the value.
|
||||||
|
v.end_position = lexer->get_position();
|
||||||
|
|
||||||
|
switch (v.type())
|
||||||
|
{
|
||||||
|
case value_t::boolean:
|
||||||
|
{
|
||||||
|
// 4 and 5 are the string length of "true" and "false"
|
||||||
|
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
case value_t::null:
|
||||||
|
{
|
||||||
|
// 4 is the string length of "null"
|
||||||
|
v.start_position = v.end_position - 4;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
case value_t::string:
|
||||||
|
{
|
||||||
|
// escape sequences make the token longer than the value it
|
||||||
|
// parses to, so the start position cannot be derived from
|
||||||
|
// the value; use the offset the lexer recorded instead
|
||||||
|
v.start_position = lexer->get_token_start_position();
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
case value_t::discarded:
|
||||||
|
{
|
||||||
|
// an object or array the callback of
|
||||||
|
// json_sax_dom_callback_parser rejected has no position
|
||||||
|
v.end_position = std::string::npos;
|
||||||
|
v.start_position = v.end_position;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
case value_t::binary:
|
||||||
|
case value_t::number_integer:
|
||||||
|
case value_t::number_unsigned:
|
||||||
|
case value_t::number_float:
|
||||||
|
{
|
||||||
|
v.start_position = v.end_position - lexer->get_string().size();
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
case value_t::object:
|
||||||
|
case value_t::array:
|
||||||
|
{
|
||||||
|
// object and array are handled in start_object() and start_array() handlers
|
||||||
|
// skip setting the values here.
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
default: // LCOV_EXCL_LINE
|
||||||
|
// Handle all possible types discretely, default handler should never be reached.
|
||||||
|
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
#endif
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief SAX implementation to create a JSON value from SAX events
|
@brief SAX implementation to create a JSON value from SAX events
|
||||||
|
|
||||||
@@ -186,9 +271,12 @@ a pointer to the respective array or object for each recursion depth.
|
|||||||
After successful parsing, the value that is passed by reference to the
|
After successful parsing, the value that is passed by reference to the
|
||||||
constructor contains the parsed value.
|
constructor contains the parsed value.
|
||||||
|
|
||||||
@tparam BasicJsonType the JSON type
|
@tparam BasicJsonType the JSON type
|
||||||
|
@tparam InputAdapterType the input adapter of the lexer that can be passed to
|
||||||
|
the constructor to record diagnostic positions; it
|
||||||
|
does not matter if no lexer is passed
|
||||||
*/
|
*/
|
||||||
template<typename BasicJsonType, typename InputAdapterType>
|
template<typename BasicJsonType, typename InputAdapterType = string_input_adapter_type>
|
||||||
class json_sax_dom_parser
|
class json_sax_dom_parser
|
||||||
{
|
{
|
||||||
public:
|
public:
|
||||||
@@ -376,76 +464,6 @@ class json_sax_dom_parser
|
|||||||
|
|
||||||
private:
|
private:
|
||||||
|
|
||||||
#if JSON_DIAGNOSTIC_POSITIONS
|
|
||||||
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
|
|
||||||
{
|
|
||||||
if (m_lexer_ref)
|
|
||||||
{
|
|
||||||
// Lexer has read past the current field value, so set the end position to the current position.
|
|
||||||
// The start position will be set below based on the length of the string representation
|
|
||||||
// of the value.
|
|
||||||
v.end_position = m_lexer_ref->get_position();
|
|
||||||
|
|
||||||
switch (v.type())
|
|
||||||
{
|
|
||||||
case value_t::boolean:
|
|
||||||
{
|
|
||||||
// 4 and 5 are the string length of "true" and "false"
|
|
||||||
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::null:
|
|
||||||
{
|
|
||||||
// 4 is the string length of "null"
|
|
||||||
v.start_position = v.end_position - 4;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::string:
|
|
||||||
{
|
|
||||||
// escape sequences make the token longer than the value it
|
|
||||||
// parses to, so the start position cannot be derived from
|
|
||||||
// the value; use the offset the lexer recorded instead
|
|
||||||
v.start_position = m_lexer_ref->get_token_start_position();
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
// As we handle the start and end positions for values created during parsing,
|
|
||||||
// we do not expect the following value type to be called. Regardless, set the positions
|
|
||||||
// in case this is created manually or through a different constructor. Exclude from lcov
|
|
||||||
// since the exact condition of this switch is esoteric.
|
|
||||||
// LCOV_EXCL_START
|
|
||||||
case value_t::discarded:
|
|
||||||
{
|
|
||||||
v.end_position = std::string::npos;
|
|
||||||
v.start_position = v.end_position;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
// LCOV_EXCL_STOP
|
|
||||||
case value_t::binary:
|
|
||||||
case value_t::number_integer:
|
|
||||||
case value_t::number_unsigned:
|
|
||||||
case value_t::number_float:
|
|
||||||
{
|
|
||||||
v.start_position = v.end_position - m_lexer_ref->get_string().size();
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
case value_t::object:
|
|
||||||
case value_t::array:
|
|
||||||
{
|
|
||||||
// object and array are handled in start_object() and start_array() handlers
|
|
||||||
// skip setting the values here.
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
default: // LCOV_EXCL_LINE
|
|
||||||
// Handle all possible types discretely, default handler should never be reached.
|
|
||||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@invariant If the ref stack is empty, then the passed value will be the new
|
@invariant If the ref stack is empty, then the passed value will be the new
|
||||||
root.
|
root.
|
||||||
@@ -461,7 +479,7 @@ class json_sax_dom_parser
|
|||||||
root = BasicJsonType(std::forward<Value>(v));
|
root = BasicJsonType(std::forward<Value>(v));
|
||||||
|
|
||||||
#if JSON_DIAGNOSTIC_POSITIONS
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
handle_diagnostic_positions_for_json_value(root);
|
diagnostic_positions::set_from_lexer(root, m_lexer_ref);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
return &root;
|
return &root;
|
||||||
@@ -474,7 +492,7 @@ class json_sax_dom_parser
|
|||||||
ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v));
|
ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v));
|
||||||
|
|
||||||
#if JSON_DIAGNOSTIC_POSITIONS
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
handle_diagnostic_positions_for_json_value(ref_stack.back()->m_data.m_value.array->back());
|
diagnostic_positions::set_from_lexer(ref_stack.back()->m_data.m_value.array->back(), m_lexer_ref);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
return &(ref_stack.back()->m_data.m_value.array->back());
|
return &(ref_stack.back()->m_data.m_value.array->back());
|
||||||
@@ -485,7 +503,7 @@ class json_sax_dom_parser
|
|||||||
*object_element = BasicJsonType(std::forward<Value>(v));
|
*object_element = BasicJsonType(std::forward<Value>(v));
|
||||||
|
|
||||||
#if JSON_DIAGNOSTIC_POSITIONS
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
handle_diagnostic_positions_for_json_value(*object_element);
|
diagnostic_positions::set_from_lexer(*object_element, m_lexer_ref);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
return object_element;
|
return object_element;
|
||||||
@@ -505,7 +523,7 @@ class json_sax_dom_parser
|
|||||||
lexer_t* m_lexer_ref = nullptr;
|
lexer_t* m_lexer_ref = nullptr;
|
||||||
};
|
};
|
||||||
|
|
||||||
template<typename BasicJsonType, typename InputAdapterType>
|
template<typename BasicJsonType, typename InputAdapterType = string_input_adapter_type>
|
||||||
class json_sax_dom_callback_parser
|
class json_sax_dom_callback_parser
|
||||||
{
|
{
|
||||||
public:
|
public:
|
||||||
@@ -674,7 +692,7 @@ class json_sax_dom_callback_parser
|
|||||||
|
|
||||||
#if JSON_DIAGNOSTIC_POSITIONS
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
// Set start/end positions for discarded object.
|
// Set start/end positions for discarded object.
|
||||||
handle_diagnostic_positions_for_json_value(*ref_stack.back());
|
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -790,7 +808,7 @@ class json_sax_dom_callback_parser
|
|||||||
|
|
||||||
#if JSON_DIAGNOSTIC_POSITIONS
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
// Set start/end positions for discarded array.
|
// Set start/end positions for discarded array.
|
||||||
handle_diagnostic_positions_for_json_value(*ref_stack.back());
|
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -843,72 +861,6 @@ class json_sax_dom_callback_parser
|
|||||||
|
|
||||||
private:
|
private:
|
||||||
|
|
||||||
#if JSON_DIAGNOSTIC_POSITIONS
|
|
||||||
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
|
|
||||||
{
|
|
||||||
if (m_lexer_ref)
|
|
||||||
{
|
|
||||||
// Lexer has read past the current field value, so set the end position to the current position.
|
|
||||||
// The start position will be set below based on the length of the string representation
|
|
||||||
// of the value.
|
|
||||||
v.end_position = m_lexer_ref->get_position();
|
|
||||||
|
|
||||||
switch (v.type())
|
|
||||||
{
|
|
||||||
case value_t::boolean:
|
|
||||||
{
|
|
||||||
// 4 and 5 are the string length of "true" and "false"
|
|
||||||
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::null:
|
|
||||||
{
|
|
||||||
// 4 is the string length of "null"
|
|
||||||
v.start_position = v.end_position - 4;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::string:
|
|
||||||
{
|
|
||||||
// escape sequences make the token longer than the value it
|
|
||||||
// parses to, so the start position cannot be derived from
|
|
||||||
// the value; use the offset the lexer recorded instead
|
|
||||||
v.start_position = m_lexer_ref->get_token_start_position();
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::discarded:
|
|
||||||
{
|
|
||||||
v.end_position = std::string::npos;
|
|
||||||
v.start_position = v.end_position;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::binary:
|
|
||||||
case value_t::number_integer:
|
|
||||||
case value_t::number_unsigned:
|
|
||||||
case value_t::number_float:
|
|
||||||
{
|
|
||||||
v.start_position = v.end_position - m_lexer_ref->get_string().size();
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::object:
|
|
||||||
case value_t::array:
|
|
||||||
{
|
|
||||||
// object and array are handled in start_object() and start_array() handlers
|
|
||||||
// skip setting the values here.
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
default: // LCOV_EXCL_LINE
|
|
||||||
// Handle all possible types discretely, default handler should never be reached.
|
|
||||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/// if there is a pending duplicate-key stash entry for this exact slot,
|
/// if there is a pending duplicate-key stash entry for this exact slot,
|
||||||
/// remove it from the stash; if restore_value is true, the stashed
|
/// remove it from the stash; if restore_value is true, the stashed
|
||||||
/// previous value is moved back into the slot first (use this when the
|
/// previous value is moved back into the slot first (use this when the
|
||||||
@@ -1030,7 +982,7 @@ class json_sax_dom_callback_parser
|
|||||||
auto value = BasicJsonType(std::forward<Value>(v));
|
auto value = BasicJsonType(std::forward<Value>(v));
|
||||||
|
|
||||||
#if JSON_DIAGNOSTIC_POSITIONS
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
handle_diagnostic_positions_for_json_value(value);
|
diagnostic_positions::set_from_lexer(value, m_lexer_ref);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
// check callback
|
// check callback
|
||||||
|
|||||||
@@ -10,6 +10,7 @@
|
|||||||
|
|
||||||
#include <array> // array
|
#include <array> // array
|
||||||
#include <cstddef> // size_t
|
#include <cstddef> // size_t
|
||||||
|
#include <cstdint> // uint8_t, uint32_t
|
||||||
#include <cstdio> // snprintf
|
#include <cstdio> // snprintf
|
||||||
#include <initializer_list> // initializer_list
|
#include <initializer_list> // initializer_list
|
||||||
#include <string> // char_traits, string
|
#include <string> // char_traits, string
|
||||||
@@ -22,6 +23,7 @@
|
|||||||
#include <nlohmann/detail/input/string_scan.hpp>
|
#include <nlohmann/detail/input/string_scan.hpp>
|
||||||
#include <nlohmann/detail/macro_scope.hpp>
|
#include <nlohmann/detail/macro_scope.hpp>
|
||||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||||
|
#include <nlohmann/detail/string_utils.hpp>
|
||||||
|
|
||||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||||
namespace detail
|
namespace detail
|
||||||
@@ -437,8 +439,16 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
if (0xD800 <= codepoint1 && codepoint1 <= 0xDBFF)
|
if (0xD800 <= codepoint1 && codepoint1 <= 0xDBFF)
|
||||||
{
|
{
|
||||||
// expect next \uxxxx entry
|
// expect next \uxxxx entry
|
||||||
if (JSON_HEDLEY_LIKELY(get() == '\\' && get() == 'u'))
|
if (JSON_HEDLEY_LIKELY(get() == '\\'))
|
||||||
{
|
{
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(get() != 'u'))
|
||||||
|
{
|
||||||
|
// current is the character escaped by the backslash
|
||||||
|
error_message = "invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF";
|
||||||
|
string_error_resume = resume_kind::escaped_character;
|
||||||
|
return token_type::parse_error;
|
||||||
|
}
|
||||||
|
|
||||||
const int codepoint2 = get_codepoint();
|
const int codepoint2 = get_codepoint();
|
||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(codepoint2 == -1))
|
if (JSON_HEDLEY_UNLIKELY(codepoint2 == -1))
|
||||||
@@ -463,7 +473,11 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
|
// the second escape was read completely and is a
|
||||||
|
// code point of its own
|
||||||
error_message = "invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF";
|
error_message = "invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF";
|
||||||
|
string_error_resume = resume_kind::after_escape;
|
||||||
|
string_error_codepoint = codepoint2;
|
||||||
return token_type::parse_error;
|
return token_type::parse_error;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -477,7 +491,9 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
{
|
{
|
||||||
if (JSON_HEDLEY_UNLIKELY(0xDC00 <= codepoint1 && codepoint1 <= 0xDFFF))
|
if (JSON_HEDLEY_UNLIKELY(0xDC00 <= codepoint1 && codepoint1 <= 0xDFFF))
|
||||||
{
|
{
|
||||||
|
// the escape was read completely
|
||||||
error_message = "invalid string: surrogate U+DC00..U+DFFF must follow U+D800..U+DBFF";
|
error_message = "invalid string: surrogate U+DC00..U+DFFF must follow U+D800..U+DBFF";
|
||||||
|
string_error_resume = resume_kind::after_escape;
|
||||||
return token_type::parse_error;
|
return token_type::parse_error;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -486,32 +502,10 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
|
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
|
||||||
|
|
||||||
// translate codepoint into bytes
|
// translate codepoint into bytes
|
||||||
if (codepoint < 0x80)
|
encode_utf8(static_cast<std::uint32_t>(codepoint), [this](std::uint32_t byte)
|
||||||
{
|
{
|
||||||
// 1-byte characters: 0xxxxxxx (ASCII)
|
add(static_cast<char_int_type>(byte));
|
||||||
add(static_cast<char_int_type>(codepoint));
|
});
|
||||||
}
|
|
||||||
else if (codepoint <= 0x7FF)
|
|
||||||
{
|
|
||||||
// 2-byte characters: 110xxxxx 10xxxxxx
|
|
||||||
add(static_cast<char_int_type>(0xC0u | (static_cast<unsigned int>(codepoint) >> 6u)));
|
|
||||||
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
|
|
||||||
}
|
|
||||||
else if (codepoint <= 0xFFFF)
|
|
||||||
{
|
|
||||||
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
|
|
||||||
add(static_cast<char_int_type>(0xE0u | (static_cast<unsigned int>(codepoint) >> 12u)));
|
|
||||||
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
|
|
||||||
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
|
|
||||||
add(static_cast<char_int_type>(0xF0u | (static_cast<unsigned int>(codepoint) >> 18u)));
|
|
||||||
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 12u) & 0x3Fu)));
|
|
||||||
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
|
|
||||||
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
|
|
||||||
}
|
|
||||||
|
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
@@ -1409,45 +1403,30 @@ scan_number_done:
|
|||||||
*/
|
*/
|
||||||
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||||
{
|
{
|
||||||
// If the caller does not need the converted value (only whether the
|
// accept() only needs to know whether the input is valid, so it sets
|
||||||
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
// discard_number_values (see json.hpp), and an integer token whose
|
||||||
// unsigned/integer token can be reported without calling
|
// digit count shows that it fits is reported without calling
|
||||||
// strtoull()/strtoll() at all, *provided* we can already tell from
|
// convert_integer(). A number with up to 18 digits always fits into
|
||||||
// the digit count alone that the conversion cannot overflow 64 bits.
|
// both std::uint64_t and std::int64_t (18 nines is about 1e18, below
|
||||||
// Such tokens are always finite and are accepted unconditionally by
|
// INT64_MAX, which is about 9.2e18). Longer tokens take the exact path
|
||||||
// the parser regardless of their actual value (parser::sax_parse_internal()
|
// below, including the fallback to floating point when the value does
|
||||||
// never checks finiteness for value_unsigned/value_integer), so the
|
// not fit.
|
||||||
// classification below is all that is needed.
|
|
||||||
//
|
//
|
||||||
// A decimal number with up to 18 digits is always representable in
|
// With a narrower number_unsigned_t/number_integer_t (e.g.
|
||||||
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
// std::uint32_t), the exact path would reclassify some of these tokens
|
||||||
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
// as (finite) floats, while this check reports integers. That does not
|
||||||
// could not have set errno to ERANGE for it. Numbers with more digits
|
// change the result of accept(): it always parses through
|
||||||
// (rare in practice) fall through to the exact code below, unchanged,
|
// json_sax_acceptor, whose number callbacks discard their argument and
|
||||||
// so their handling -- including reclassification to value_float when
|
// return true, and the parser rejects neither integers nor finite
|
||||||
// the value overflows 64 bits, and rejection when it is not even
|
// floats. value_unsigned/value_integer are left unset here, so a caller
|
||||||
// finite as a double -- is bit-for-bit identical to before this
|
// that reads the converted value must not set discard_number_values.
|
||||||
// optimization.
|
|
||||||
//
|
//
|
||||||
// Note this reasons about std::uint64_t/std::int64_t, not about
|
// On contiguous input, scan_number_bulk_contiguous() converts integer
|
||||||
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
// tokens itself and does not pass them to this function, unless
|
||||||
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
// JSON_DIAGNOSTIC_POSITIONS is enabled. This check is therefore only
|
||||||
// *only* because discard_number_values is exclusively set by
|
// reached for input without bulk access (e.g. streams), with
|
||||||
// accept() (see json.hpp), and accept() always parses through the
|
// JSON_DIAGNOSTIC_POSITIONS, or when scan_number_bulk_contiguous()
|
||||||
// library's own json_sax_acceptor -- never a user-supplied SAX
|
// falls back to scan_number().
|
||||||
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
|
||||||
// callbacks unconditionally discard their argument and return true.
|
|
||||||
// So for every caller that can reach this branch, neither the token
|
|
||||||
// classification below nor the eventual (possibly narrowed, and on
|
|
||||||
// this fast path left stale/unset) value_unsigned/value_integer is
|
|
||||||
// ever consulted -- an unsigned/integer token is accepted outright,
|
|
||||||
// and even a >18-digit token that this fast path deliberately falls
|
|
||||||
// through for is, once reclassified to value_float, still finite
|
|
||||||
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
|
||||||
// or number_integer_t regardless of that type's width. If this
|
|
||||||
// function is ever taught to run with discard_number_values true for
|
|
||||||
// a caller that *does* read the converted value, this reasoning (and
|
|
||||||
// the fast path below) would need to be revisited.
|
|
||||||
if (discard_number_values)
|
if (discard_number_values)
|
||||||
{
|
{
|
||||||
constexpr std::size_t safe_digit_count = 18;
|
constexpr std::size_t safe_digit_count = 18;
|
||||||
@@ -1876,7 +1855,7 @@ scan_number_done:
|
|||||||
return value_float;
|
return value_float;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// return current string value (implicitly resets the token; useful only once)
|
/// return current string value
|
||||||
string_t& get_string()
|
string_t& get_string()
|
||||||
{
|
{
|
||||||
// a number token holds '.' regardless of the locale (#4084)
|
// a number token holds '.' regardless of the locale (#4084)
|
||||||
@@ -2130,6 +2109,573 @@ scan_number_done:
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/////////////////////
|
||||||
|
// error recovery
|
||||||
|
/////////////////////
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief make the best of the token that scan() rejected
|
||||||
|
|
||||||
|
Called by the parser after scan() returned token_type::parse_error and the
|
||||||
|
SAX parser asked to recover from the error (see #3989). Keeps what can be
|
||||||
|
read of the token and skips the rest:
|
||||||
|
|
||||||
|
- A string keeps its characters. An unknown escape stands for the escaped
|
||||||
|
character itself (as in JavaScript), an invalid `\u` escape and ill-formed
|
||||||
|
UTF-8 become U+FFFD, and a control character is kept. A line break or the
|
||||||
|
end of the input ends a string that lacks its closing quote.
|
||||||
|
- A number keeps its longest valid prefix, e.g. `1` for `1.` or `1e+`.
|
||||||
|
- A block comment that is not closed runs to the end of the input.
|
||||||
|
- Anything else is skipped.
|
||||||
|
|
||||||
|
The rest of an invalid token is skipped up to the next delimiter
|
||||||
|
(whitespace, a structural character, or a quote). A delimiter that the
|
||||||
|
invalid token consumed is returned to the input, so that the next scan()
|
||||||
|
reads it.
|
||||||
|
|
||||||
|
@return token_type::value_string or a number token type if a string or a
|
||||||
|
number could be read, token_type::end_of_input for a block comment
|
||||||
|
that is not closed, token_type::uninitialized otherwise
|
||||||
|
*/
|
||||||
|
token_type recover_token()
|
||||||
|
{
|
||||||
|
const resume_kind resume = string_error_resume;
|
||||||
|
const int codepoint = string_error_codepoint;
|
||||||
|
string_error_resume = resume_kind::character;
|
||||||
|
string_error_codepoint = -1;
|
||||||
|
|
||||||
|
if (error_message_starts_with("invalid string"))
|
||||||
|
{
|
||||||
|
return recover_string(resume, codepoint);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (error_message_starts_with("invalid number"))
|
||||||
|
{
|
||||||
|
return recover_number();
|
||||||
|
}
|
||||||
|
|
||||||
|
if (error_message_starts_with("invalid comment; missing"))
|
||||||
|
{
|
||||||
|
// the comment runs to the end of the input
|
||||||
|
return token_type::end_of_input;
|
||||||
|
}
|
||||||
|
|
||||||
|
skip_to_delimiter();
|
||||||
|
return token_type::uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief return the token that scan() read last to the input, so that the
|
||||||
|
next scan() reads it again
|
||||||
|
|
||||||
|
Called by the parser when recovering from an error. The token must be a
|
||||||
|
single character (',', ':', '[', ']', '{', or '}') or the end of the
|
||||||
|
input, and scan() must have read it last.
|
||||||
|
*/
|
||||||
|
void unget_token()
|
||||||
|
{
|
||||||
|
JSON_ASSERT(!next_unget);
|
||||||
|
unget();
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief let the token string for the next error begin at the current character
|
||||||
|
|
||||||
|
The token string of an error reaches back to the beginning of the last
|
||||||
|
string or number. After an error, the parser calls this function so that
|
||||||
|
the next error does not report (and, with many errors, copy) everything
|
||||||
|
read since then.
|
||||||
|
*/
|
||||||
|
void restart_token_string()
|
||||||
|
{
|
||||||
|
restart_token_string_impl(std::integral_constant<bool, lazy_token_string> {});
|
||||||
|
}
|
||||||
|
|
||||||
|
private:
|
||||||
|
/// how recover_string() continues after the error scan_string() reported
|
||||||
|
enum class resume_kind : std::uint8_t
|
||||||
|
{
|
||||||
|
/// current is the next character of the string (or the end of input)
|
||||||
|
character,
|
||||||
|
/// current is the character escaped by the preceding backslash
|
||||||
|
escaped_character,
|
||||||
|
/// current is the last character of a complete escape
|
||||||
|
after_escape
|
||||||
|
};
|
||||||
|
|
||||||
|
/// whether error_message begins with @a prefix
|
||||||
|
bool error_message_starts_with(const char* prefix) const noexcept
|
||||||
|
{
|
||||||
|
const char* message = error_message;
|
||||||
|
while (*prefix != '\0')
|
||||||
|
{
|
||||||
|
if (*message++ != *prefix++)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// whether current ends an invalid token (see recover_token())
|
||||||
|
bool current_is_delimiter() const noexcept
|
||||||
|
{
|
||||||
|
switch (current)
|
||||||
|
{
|
||||||
|
case ' ':
|
||||||
|
case '\t':
|
||||||
|
case '\n':
|
||||||
|
case '\r':
|
||||||
|
case '[':
|
||||||
|
case ']':
|
||||||
|
case '{':
|
||||||
|
case '}':
|
||||||
|
case ',':
|
||||||
|
case ':':
|
||||||
|
case '\"':
|
||||||
|
#if !JSON_STRICT_NUL_HANDLING
|
||||||
|
case '\0':
|
||||||
|
#endif
|
||||||
|
case char_traits<char_type>::eof():
|
||||||
|
return true;
|
||||||
|
|
||||||
|
case '/':
|
||||||
|
return ignore_comments;
|
||||||
|
|
||||||
|
default:
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// skip the rest of an invalid token and return its delimiter to the input
|
||||||
|
void skip_to_delimiter()
|
||||||
|
{
|
||||||
|
while (!current_is_delimiter())
|
||||||
|
{
|
||||||
|
get();
|
||||||
|
}
|
||||||
|
|
||||||
|
if (current != char_traits<char_type>::eof())
|
||||||
|
{
|
||||||
|
unget();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// append U+FFFD REPLACEMENT CHARACTER to token_buffer
|
||||||
|
void add_replacement_character()
|
||||||
|
{
|
||||||
|
add(0xEF);
|
||||||
|
add(0xBF);
|
||||||
|
add(0xBD);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// append the UTF-8 encoding of @a codepoint (not a surrogate) to token_buffer
|
||||||
|
void add_codepoint(const int codepoint)
|
||||||
|
{
|
||||||
|
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
|
||||||
|
const auto cp = static_cast<unsigned int>(codepoint);
|
||||||
|
if (cp < 0x80)
|
||||||
|
{
|
||||||
|
add(static_cast<char_int_type>(cp));
|
||||||
|
}
|
||||||
|
else if (cp <= 0x7FF)
|
||||||
|
{
|
||||||
|
add(static_cast<char_int_type>(0xC0u | (cp >> 6u)));
|
||||||
|
add(static_cast<char_int_type>(0x80u | (cp & 0x3Fu)));
|
||||||
|
}
|
||||||
|
else if (cp <= 0xFFFF)
|
||||||
|
{
|
||||||
|
add(static_cast<char_int_type>(0xE0u | (cp >> 12u)));
|
||||||
|
add(static_cast<char_int_type>(0x80u | ((cp >> 6u) & 0x3Fu)));
|
||||||
|
add(static_cast<char_int_type>(0x80u | (cp & 0x3Fu)));
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
add(static_cast<char_int_type>(0xF0u | (cp >> 18u)));
|
||||||
|
add(static_cast<char_int_type>(0x80u | ((cp >> 12u) & 0x3Fu)));
|
||||||
|
add(static_cast<char_int_type>(0x80u | ((cp >> 6u) & 0x3Fu)));
|
||||||
|
add(static_cast<char_int_type>(0x80u | (cp & 0x3Fu)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// append a code point read from a `\u` escape; a surrogate becomes U+FFFD
|
||||||
|
void add_escaped_codepoint(const int codepoint)
|
||||||
|
{
|
||||||
|
if (0xD800 <= codepoint && codepoint <= 0xDFFF)
|
||||||
|
{
|
||||||
|
add_replacement_character();
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
add_codepoint(codepoint);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief remove an incomplete UTF-8 sequence from the end of token_buffer
|
||||||
|
|
||||||
|
next_byte_in_range() adds the bytes of a sequence as it checks them, so
|
||||||
|
when it rejects a byte, the beginning of the sequence is already in
|
||||||
|
token_buffer, which otherwise holds only complete sequences.
|
||||||
|
|
||||||
|
@return whether an incomplete sequence was removed
|
||||||
|
*/
|
||||||
|
bool remove_incomplete_utf8_sequence()
|
||||||
|
{
|
||||||
|
std::size_t lead = token_buffer.size();
|
||||||
|
std::size_t continuation_bytes = 0;
|
||||||
|
while (lead > 0 && continuation_bytes < 3
|
||||||
|
&& (static_cast<unsigned char>(token_buffer[lead - 1]) & 0xC0u) == 0x80u)
|
||||||
|
{
|
||||||
|
--lead;
|
||||||
|
++continuation_bytes;
|
||||||
|
}
|
||||||
|
if (lead == 0)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
const auto lead_byte = static_cast<unsigned char>(token_buffer[lead - 1]);
|
||||||
|
std::size_t expected = 0;
|
||||||
|
if (lead_byte >= 0xF0)
|
||||||
|
{
|
||||||
|
expected = 3;
|
||||||
|
}
|
||||||
|
else if (lead_byte >= 0xE0)
|
||||||
|
{
|
||||||
|
expected = 2;
|
||||||
|
}
|
||||||
|
else if (lead_byte >= 0xC0)
|
||||||
|
{
|
||||||
|
expected = 1;
|
||||||
|
}
|
||||||
|
if (continuation_bytes >= expected)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
token_buffer.resize(lead - 1);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief read the UTF-8 sequence that begins with current, which is not ASCII
|
||||||
|
@return whether the next character must be read; false if current still
|
||||||
|
needs to be handled, because it does not belong to the sequence
|
||||||
|
*/
|
||||||
|
bool recover_utf8_sequence()
|
||||||
|
{
|
||||||
|
// the number of continuation bytes and the range of the first one;
|
||||||
|
// see the ranges in scan_string()
|
||||||
|
std::size_t count = 0;
|
||||||
|
char_int_type low = 0x80;
|
||||||
|
char_int_type high = 0xBF;
|
||||||
|
if (current >= 0xC2 && current <= 0xDF)
|
||||||
|
{
|
||||||
|
count = 1;
|
||||||
|
}
|
||||||
|
else if (current >= 0xE0 && current <= 0xEF)
|
||||||
|
{
|
||||||
|
count = 2;
|
||||||
|
low = (current == 0xE0) ? 0xA0 : 0x80;
|
||||||
|
high = (current == 0xED) ? 0x9F : 0xBF;
|
||||||
|
}
|
||||||
|
else if (current >= 0xF0 && current <= 0xF4)
|
||||||
|
{
|
||||||
|
count = 3;
|
||||||
|
low = (current == 0xF0) ? 0x90 : 0x80;
|
||||||
|
high = (current == 0xF4) ? 0x8F : 0xBF;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// an ill-formed byte
|
||||||
|
add_replacement_character();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
const std::size_t start = token_buffer.size();
|
||||||
|
add(current);
|
||||||
|
for (std::size_t i = 0; i < count; ++i)
|
||||||
|
{
|
||||||
|
get();
|
||||||
|
if (current < low || current > high)
|
||||||
|
{
|
||||||
|
token_buffer.resize(start);
|
||||||
|
add_replacement_character();
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
add(current);
|
||||||
|
low = 0x80;
|
||||||
|
high = 0xBF;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief read the low surrogate that must follow the high surrogate @a high
|
||||||
|
@return whether the next character must be read; false if current still
|
||||||
|
needs to be handled
|
||||||
|
*/
|
||||||
|
bool recover_low_surrogate(int high)
|
||||||
|
{
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
if (get() != '\\')
|
||||||
|
{
|
||||||
|
add_replacement_character();
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (get() != 'u')
|
||||||
|
{
|
||||||
|
add_replacement_character();
|
||||||
|
// not 'u', so this does not come back here
|
||||||
|
return recover_escape();
|
||||||
|
}
|
||||||
|
|
||||||
|
const int low = get_codepoint();
|
||||||
|
if (low == -1)
|
||||||
|
{
|
||||||
|
add_replacement_character();
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (0xDC00 <= low && low <= 0xDFFF)
|
||||||
|
{
|
||||||
|
add_codepoint(static_cast<int>((static_cast<unsigned int>(high) << 10u)
|
||||||
|
+ static_cast<unsigned int>(low) - 0x35FDC00u));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// high has no low surrogate
|
||||||
|
add_replacement_character();
|
||||||
|
if (low < 0xD800 || low > 0xDBFF)
|
||||||
|
{
|
||||||
|
add_codepoint(low);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
// another high surrogate
|
||||||
|
high = low;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief read the escape whose backslash was read; current is the escaped character
|
||||||
|
@return whether the next character must be read; false if current still
|
||||||
|
needs to be handled
|
||||||
|
*/
|
||||||
|
bool recover_escape()
|
||||||
|
{
|
||||||
|
switch (current)
|
||||||
|
{
|
||||||
|
case '\"':
|
||||||
|
add('\"');
|
||||||
|
return true;
|
||||||
|
case '\\':
|
||||||
|
add('\\');
|
||||||
|
return true;
|
||||||
|
case '/':
|
||||||
|
add('/');
|
||||||
|
return true;
|
||||||
|
case 'b':
|
||||||
|
add('\b');
|
||||||
|
return true;
|
||||||
|
case 'f':
|
||||||
|
add('\f');
|
||||||
|
return true;
|
||||||
|
case 'n':
|
||||||
|
add('\n');
|
||||||
|
return true;
|
||||||
|
case 'r':
|
||||||
|
add('\r');
|
||||||
|
return true;
|
||||||
|
case 't':
|
||||||
|
add('\t');
|
||||||
|
return true;
|
||||||
|
|
||||||
|
case 'u':
|
||||||
|
{
|
||||||
|
const int codepoint = get_codepoint();
|
||||||
|
if (codepoint == -1)
|
||||||
|
{
|
||||||
|
add_replacement_character();
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (0xD800 <= codepoint && codepoint <= 0xDBFF)
|
||||||
|
{
|
||||||
|
return recover_low_surrogate(codepoint);
|
||||||
|
}
|
||||||
|
add_escaped_codepoint(codepoint);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// an unknown escape stands for the escaped character
|
||||||
|
default:
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief read the rest of a string after scan_string() rejected it
|
||||||
|
|
||||||
|
token_buffer holds what scan_string() read before the error. See
|
||||||
|
recover_token() for how errors are repaired.
|
||||||
|
|
||||||
|
@param[in] resume how to continue, see resume_kind
|
||||||
|
@param[in] codepoint for a high surrogate followed by an escape of another
|
||||||
|
code point: that code point; -1 otherwise
|
||||||
|
*/
|
||||||
|
token_type recover_string(const resume_kind resume, const int codepoint)
|
||||||
|
{
|
||||||
|
// whether the next character must be read before it can be handled
|
||||||
|
bool fetch = false;
|
||||||
|
|
||||||
|
if (error_message_starts_with("invalid string: surrogate")
|
||||||
|
|| error_message_starts_with("invalid string: '\\u'")
|
||||||
|
|| (error_message_starts_with("invalid string: ill-formed UTF-8")
|
||||||
|
&& remove_incomplete_utf8_sequence()))
|
||||||
|
{
|
||||||
|
add_replacement_character();
|
||||||
|
}
|
||||||
|
|
||||||
|
switch (resume)
|
||||||
|
{
|
||||||
|
case resume_kind::escaped_character:
|
||||||
|
fetch = recover_escape();
|
||||||
|
break;
|
||||||
|
case resume_kind::after_escape:
|
||||||
|
if (0xD800 <= codepoint && codepoint <= 0xDBFF)
|
||||||
|
{
|
||||||
|
fetch = recover_low_surrogate(codepoint);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
if (codepoint != -1)
|
||||||
|
{
|
||||||
|
add_escaped_codepoint(codepoint);
|
||||||
|
}
|
||||||
|
fetch = true;
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
case resume_kind::character:
|
||||||
|
default:
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
if (fetch)
|
||||||
|
{
|
||||||
|
get();
|
||||||
|
}
|
||||||
|
fetch = true;
|
||||||
|
|
||||||
|
switch (current)
|
||||||
|
{
|
||||||
|
case '\"':
|
||||||
|
// a line break or the end of the input ends a string that
|
||||||
|
// lacks its closing quote
|
||||||
|
case '\n':
|
||||||
|
case '\r':
|
||||||
|
case char_traits<char_type>::eof():
|
||||||
|
return token_type::value_string;
|
||||||
|
|
||||||
|
#if !JSON_STRICT_NUL_HANDLING
|
||||||
|
case '\0':
|
||||||
|
// the end of the input, see scan()
|
||||||
|
unget();
|
||||||
|
return token_type::value_string;
|
||||||
|
#endif
|
||||||
|
|
||||||
|
case '\\':
|
||||||
|
get();
|
||||||
|
fetch = recover_escape();
|
||||||
|
break;
|
||||||
|
|
||||||
|
default:
|
||||||
|
if (current < 0x80)
|
||||||
|
{
|
||||||
|
// including control characters
|
||||||
|
add(current);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
fetch = recover_utf8_sequence();
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief keep the longest valid prefix of a number that scan_number() rejected
|
||||||
|
|
||||||
|
token_buffer holds the characters scan_number() accepted before the error,
|
||||||
|
so the prefix ends at its last digit.
|
||||||
|
*/
|
||||||
|
token_type recover_number()
|
||||||
|
{
|
||||||
|
// only size(), operator[], and resize() are used, which every string
|
||||||
|
// type the library supports provides
|
||||||
|
std::size_t length = token_buffer.size();
|
||||||
|
while (length != 0 && (token_buffer[length - 1] < '0' || token_buffer[length - 1] > '9'))
|
||||||
|
{
|
||||||
|
--length;
|
||||||
|
}
|
||||||
|
token_buffer.resize(length);
|
||||||
|
|
||||||
|
if (length == 0)
|
||||||
|
{
|
||||||
|
skip_to_delimiter();
|
||||||
|
return token_type::uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (decimal_point_position >= length)
|
||||||
|
{
|
||||||
|
decimal_point_position = std::string::npos;
|
||||||
|
}
|
||||||
|
|
||||||
|
std::size_t exponent = std::string::npos;
|
||||||
|
for (std::size_t i = 0; i < length; ++i)
|
||||||
|
{
|
||||||
|
if (token_buffer[i] == 'e' || token_buffer[i] == 'E')
|
||||||
|
{
|
||||||
|
exponent = i;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const std::size_t mantissa_end = (exponent == std::string::npos) ? length : exponent;
|
||||||
|
token_type number_type = token_type::value_unsigned;
|
||||||
|
if (decimal_point_position != std::string::npos || exponent != std::string::npos)
|
||||||
|
{
|
||||||
|
number_type = token_type::value_float;
|
||||||
|
}
|
||||||
|
else if (token_buffer[0] == '-')
|
||||||
|
{
|
||||||
|
number_type = token_type::value_integer;
|
||||||
|
}
|
||||||
|
|
||||||
|
const token_type result = convert_number(number_type, mantissa_end);
|
||||||
|
skip_to_delimiter();
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// seekable adapter: the token string begins at current, which was consumed
|
||||||
|
void restart_token_string_impl(std::true_type /*lazy*/) noexcept
|
||||||
|
{
|
||||||
|
const std::size_t consumed = ia.get_consumed_count();
|
||||||
|
token_string_start = (consumed > 0 && current != char_traits<char_type>::eof()) ? consumed - 1 : consumed;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// streaming adapter: the token string begins at current; a character
|
||||||
|
/// that was put back is copied again when it is read again
|
||||||
|
void restart_token_string_impl(std::false_type /*lazy*/)
|
||||||
|
{
|
||||||
|
token_string.clear();
|
||||||
|
if (!next_unget && current != char_traits<char_type>::eof())
|
||||||
|
{
|
||||||
|
token_string.push_back(char_traits<char_type>::to_char_type(current));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
/// input adapter
|
/// input adapter
|
||||||
InputAdapterType ia;
|
InputAdapterType ia;
|
||||||
@@ -2170,6 +2716,13 @@ scan_number_done:
|
|||||||
/// a description of occurred lexer errors
|
/// a description of occurred lexer errors
|
||||||
const char* error_message = "";
|
const char* error_message = "";
|
||||||
|
|
||||||
|
/// how recover_token() continues a string that scan_string() rejected;
|
||||||
|
/// set only on the error paths that need more than error_message
|
||||||
|
resume_kind string_error_resume = resume_kind::character;
|
||||||
|
/// the code point of the second escape when a high surrogate is followed
|
||||||
|
/// by an escape that is not a low surrogate; -1 otherwise
|
||||||
|
int string_error_codepoint = -1;
|
||||||
|
|
||||||
// number values
|
// number values
|
||||||
number_integer_t value_integer = 0;
|
number_integer_t value_integer = 0;
|
||||||
number_unsigned_t value_unsigned = 0;
|
number_unsigned_t value_unsigned = 0;
|
||||||
@@ -2178,11 +2731,11 @@ scan_number_done:
|
|||||||
/// the position of the decimal point in token_buffer
|
/// the position of the decimal point in token_buffer
|
||||||
std::size_t decimal_point_position = std::string::npos;
|
std::size_t decimal_point_position = std::string::npos;
|
||||||
|
|
||||||
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
/// whether the caller only needs the token types and never looks at the
|
||||||
/// token classification and never looks at the converted numeric value;
|
/// converted numeric values; set only by accept(), which parses through
|
||||||
/// when set, scan_number() may skip strtoull()/strtoll() for
|
/// json_sax_acceptor. When set, convert_number() skips converting integer
|
||||||
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
/// tokens whose digit count guarantees that they fit into 64 bits (see
|
||||||
/// fit into 64 bits (see scan_number())
|
/// there)
|
||||||
const bool discard_number_values = false;
|
const bool discard_number_values = false;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
@@ -54,7 +54,8 @@ using parser_callback_t =
|
|||||||
/*!
|
/*!
|
||||||
@brief syntax analysis
|
@brief syntax analysis
|
||||||
|
|
||||||
This class implements a recursive descent parser.
|
This class implements an iterative parser that keeps the open containers on
|
||||||
|
an explicit stack and reports what it reads as SAX events.
|
||||||
*/
|
*/
|
||||||
template<typename BasicJsonType, typename InputAdapterType>
|
template<typename BasicJsonType, typename InputAdapterType>
|
||||||
class parser
|
class parser
|
||||||
@@ -98,28 +99,9 @@ class parser
|
|||||||
if (callback)
|
if (callback)
|
||||||
{
|
{
|
||||||
json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer);
|
json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer);
|
||||||
sax_parse_internal(&sdp);
|
|
||||||
|
|
||||||
if (strict)
|
|
||||||
{
|
|
||||||
// in strict mode, input must be completely read
|
|
||||||
if (get_token() != token_type::end_of_input)
|
|
||||||
{
|
|
||||||
sdp.parse_error(m_lexer.get_position(),
|
|
||||||
m_lexer.get_token_string(),
|
|
||||||
parse_error::create(101, m_lexer.get_position(),
|
|
||||||
exception_message(token_type::end_of_input, "value"), nullptr));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// the caller keeps using the input: position it right after
|
|
||||||
// the value by leaving the character that terminated it
|
|
||||||
m_lexer.release_lookahead();
|
|
||||||
}
|
|
||||||
|
|
||||||
// in case of an error, return a discarded value
|
// in case of an error, return a discarded value
|
||||||
if (sdp.is_errored())
|
if (!parse_dom(sdp, strict))
|
||||||
{
|
{
|
||||||
result = value_t::discarded;
|
result = value_t::discarded;
|
||||||
return;
|
return;
|
||||||
@@ -135,26 +117,9 @@ class parser
|
|||||||
else
|
else
|
||||||
{
|
{
|
||||||
json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer);
|
json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer);
|
||||||
sax_parse_internal(&sdp);
|
|
||||||
|
|
||||||
if (strict)
|
|
||||||
{
|
|
||||||
// in strict mode, input must be completely read
|
|
||||||
if (get_token() != token_type::end_of_input)
|
|
||||||
{
|
|
||||||
sdp.parse_error(m_lexer.get_position(),
|
|
||||||
m_lexer.get_token_string(),
|
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// see above
|
|
||||||
m_lexer.release_lookahead();
|
|
||||||
}
|
|
||||||
|
|
||||||
// in case of an error, return a discarded value
|
// in case of an error, return a discarded value
|
||||||
if (sdp.is_errored())
|
if (!parse_dom(sdp, strict))
|
||||||
{
|
{
|
||||||
result = value_t::discarded;
|
result = value_t::discarded;
|
||||||
return;
|
return;
|
||||||
@@ -173,26 +138,59 @@ class parser
|
|||||||
bool accept(const bool strict = true)
|
bool accept(const bool strict = true)
|
||||||
{
|
{
|
||||||
json_sax_acceptor<BasicJsonType> sax_acceptor;
|
json_sax_acceptor<BasicJsonType> sax_acceptor;
|
||||||
return sax_parse(&sax_acceptor, strict);
|
return sax_parse_impl<false>(&sax_acceptor, strict);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief public SAX interface
|
||||||
|
|
||||||
|
If the SAX parser's parse_error() returns true, the parser recovers from
|
||||||
|
the error: it repairs the input and continues (see #3989).
|
||||||
|
|
||||||
|
@param[in] sax the SAX parser
|
||||||
|
@param[in] strict whether to expect the last token to be EOF
|
||||||
|
@return whether the input was parsed without errors and no SAX event
|
||||||
|
returned false
|
||||||
|
*/
|
||||||
template<typename SAX>
|
template<typename SAX>
|
||||||
JSON_HEDLEY_NON_NULL(2)
|
JSON_HEDLEY_NON_NULL(2)
|
||||||
bool sax_parse(SAX* sax, const bool strict = true)
|
bool sax_parse(SAX* sax, const bool strict = true)
|
||||||
|
{
|
||||||
|
return sax_parse_impl<true>(sax, strict);
|
||||||
|
}
|
||||||
|
|
||||||
|
private:
|
||||||
|
/// what sax_parse_internal() does after an object key was expected
|
||||||
|
enum class next_step : std::uint8_t
|
||||||
|
{
|
||||||
|
/// stop parsing
|
||||||
|
stop,
|
||||||
|
/// parse a value that begins with last_token
|
||||||
|
parse_value,
|
||||||
|
/// evaluate the state of the innermost container, which reads
|
||||||
|
/// last_token again
|
||||||
|
evaluate_state
|
||||||
|
};
|
||||||
|
|
||||||
|
template<bool AllowRecovery, typename SAX>
|
||||||
|
JSON_HEDLEY_NON_NULL(2)
|
||||||
|
bool sax_parse_impl(SAX* sax, const bool strict)
|
||||||
{
|
{
|
||||||
(void)detail::is_sax_static_asserts<SAX, BasicJsonType> {};
|
(void)detail::is_sax_static_asserts<SAX, BasicJsonType> {};
|
||||||
const bool result = sax_parse_internal(sax);
|
const bool result = sax_parse_internal<AllowRecovery>(sax);
|
||||||
|
|
||||||
if (result)
|
if (result)
|
||||||
{
|
{
|
||||||
if (strict)
|
if (strict)
|
||||||
{
|
{
|
||||||
// strict mode: next byte must be EOF
|
// strict mode: next byte must be EOF; after recovering from an
|
||||||
if (get_token() != token_type::end_of_input)
|
// error, the end of the input may already have been read
|
||||||
|
if (last_token != token_type::end_of_input && get_token() != token_type::end_of_input)
|
||||||
{
|
{
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
// the value is complete, so there is nothing to recover
|
||||||
m_lexer.get_token_string(),
|
static_cast<void>(report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr),
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
|
std::integral_constant<bool, AllowRecovery> {}));
|
||||||
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
@@ -203,14 +201,63 @@ class parser
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return result;
|
return result && !error_reported;
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
/*!
|
||||||
template<typename SAX>
|
@brief run a DOM SAX parser to completion and position the lexer
|
||||||
|
|
||||||
|
Shared by both branches of @ref parse(): builds no SAX parser itself,
|
||||||
|
but drives an already-constructed @a json_sax_dom_parser or
|
||||||
|
@ref json_sax_dom_callback_parser through @ref sax_parse_internal(),
|
||||||
|
then applies the strict-EOF check (reporting parse_error.101 through
|
||||||
|
@a sdp on failure) or, in non-strict mode, releases the lookahead so
|
||||||
|
the caller can keep reading the input right after the parsed value.
|
||||||
|
|
||||||
|
@param[in,out] sdp the DOM SAX parser to run
|
||||||
|
@param[in] strict whether to expect the last token to be EOF
|
||||||
|
@return whether @a sdp did not report an error
|
||||||
|
*/
|
||||||
|
template<typename DomSax>
|
||||||
|
bool parse_dom(DomSax& sdp, const bool strict)
|
||||||
|
{
|
||||||
|
sax_parse_internal<false>(&sdp);
|
||||||
|
|
||||||
|
if (strict)
|
||||||
|
{
|
||||||
|
// in strict mode, input must be completely read
|
||||||
|
if (get_token() != token_type::end_of_input)
|
||||||
|
{
|
||||||
|
sdp.parse_error(m_lexer.get_position(),
|
||||||
|
m_lexer.get_token_string(),
|
||||||
|
parse_error::create(101, m_lexer.get_position(),
|
||||||
|
exception_message(token_type::end_of_input, "value"), nullptr));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// the caller keeps using the input: position it right after
|
||||||
|
// the value by leaving the character that terminated it
|
||||||
|
m_lexer.release_lookahead();
|
||||||
|
}
|
||||||
|
|
||||||
|
return !sdp.is_errored();
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief parse a JSON value and pass it to a SAX parser
|
||||||
|
|
||||||
|
@tparam AllowRecovery whether to recover from an error if the SAX parser's
|
||||||
|
parse_error() returns true; false for the SAX parsers
|
||||||
|
of parse() and accept(), which never do, so that no
|
||||||
|
code for recovering is generated for them
|
||||||
|
*/
|
||||||
|
template<bool AllowRecovery, typename SAX>
|
||||||
JSON_HEDLEY_NON_NULL(2)
|
JSON_HEDLEY_NON_NULL(2)
|
||||||
bool sax_parse_internal(SAX* sax)
|
bool sax_parse_internal(SAX* sax)
|
||||||
{
|
{
|
||||||
|
const std::integral_constant<bool, AllowRecovery> allow_recovery{};
|
||||||
|
|
||||||
// stack to remember the hierarchy of structured values we are parsing
|
// stack to remember the hierarchy of structured values we are parsing
|
||||||
// true = array; false = object
|
// true = array; false = object
|
||||||
std::vector<bool> states;
|
std::vector<bool> states;
|
||||||
@@ -241,12 +288,18 @@ class parser
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
// parse key
|
// remember we are now inside an object
|
||||||
|
states.push_back(false);
|
||||||
|
|
||||||
|
// parse key (the steps of parse_key(), which are
|
||||||
|
// repeated here and below for speed)
|
||||||
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
||||||
{
|
{
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
if (!continue_after(key_error(sax, allow_recovery, false), skip_to_state_evaluation))
|
||||||
m_lexer.get_token_string(),
|
{
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr));
|
return false;
|
||||||
|
}
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
||||||
{
|
{
|
||||||
@@ -256,14 +309,13 @@ class parser
|
|||||||
// parse separator (:)
|
// parse separator (:)
|
||||||
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
||||||
{
|
{
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
if (!continue_after(key_error(sax, allow_recovery, true), skip_to_state_evaluation))
|
||||||
m_lexer.get_token_string(),
|
{
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr));
|
return false;
|
||||||
|
}
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
// remember we are now inside an object
|
|
||||||
states.push_back(false);
|
|
||||||
|
|
||||||
// parse values
|
// parse values
|
||||||
get_token();
|
get_token();
|
||||||
continue;
|
continue;
|
||||||
@@ -299,9 +351,11 @@ class parser
|
|||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!std::isfinite(res)))
|
if (JSON_HEDLEY_UNLIKELY(!std::isfinite(res)))
|
||||||
{
|
{
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
if (!overflow_error(sax, res, allow_recovery))
|
||||||
m_lexer.get_token_string(),
|
{
|
||||||
out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr));
|
return false;
|
||||||
|
}
|
||||||
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!sax->number_float(res, m_lexer.get_string())))
|
if (JSON_HEDLEY_UNLIKELY(!sax->number_float(res, m_lexer.get_string())))
|
||||||
@@ -369,23 +423,63 @@ class parser
|
|||||||
case token_type::parse_error:
|
case token_type::parse_error:
|
||||||
{
|
{
|
||||||
// using "uninitialized" to avoid an "expected" message
|
// using "uninitialized" to avoid an "expected" message
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::uninitialized, "value"), nullptr), allow_recovery))
|
||||||
m_lexer.get_token_string(),
|
{
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::uninitialized, "value"), nullptr));
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// recover: keep what can be read of the token
|
||||||
|
recover_token();
|
||||||
|
if (last_token != token_type::uninitialized)
|
||||||
|
{
|
||||||
|
// a string or a number
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (states.empty())
|
||||||
|
{
|
||||||
|
// look for the value after the garbage
|
||||||
|
if (!skip_to_value())
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
// nothing could be read
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!sax->null()))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
break;
|
||||||
}
|
}
|
||||||
case token_type::end_of_input:
|
case token_type::end_of_input:
|
||||||
{
|
{
|
||||||
if (JSON_HEDLEY_UNLIKELY(m_lexer.get_position().chars_read_total == 1))
|
if (JSON_HEDLEY_UNLIKELY(m_lexer.get_position().chars_read_total == 1))
|
||||||
{
|
{
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
// there is nothing to recover
|
||||||
m_lexer.get_token_string(),
|
static_cast<void>(report_error(sax, parse_error::create(101, m_lexer.get_position(),
|
||||||
parse_error::create(101, m_lexer.get_position(),
|
"attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr), allow_recovery));
|
||||||
"attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr));
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr), allow_recovery))
|
||||||
m_lexer.get_token_string(),
|
{
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr));
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// recover: the input ends where a value is missing
|
||||||
|
if (states.empty())
|
||||||
|
{
|
||||||
|
// there is no value
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (!recover_missing_value(sax, states))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// the state evaluation reads the token again
|
||||||
|
m_lexer.unget_token();
|
||||||
|
skip_to_state_evaluation = true;
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
case token_type::uninitialized:
|
case token_type::uninitialized:
|
||||||
case token_type::end_array:
|
case token_type::end_array:
|
||||||
@@ -395,9 +489,35 @@ class parser
|
|||||||
case token_type::literal_or_value:
|
case token_type::literal_or_value:
|
||||||
default: // the last token was unexpected
|
default: // the last token was unexpected
|
||||||
{
|
{
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr), allow_recovery))
|
||||||
m_lexer.get_token_string(),
|
{
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr));
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// recover
|
||||||
|
if (states.empty())
|
||||||
|
{
|
||||||
|
// look for the value after the garbage
|
||||||
|
if (!skip_to_value())
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (last_token == token_type::name_separator)
|
||||||
|
{
|
||||||
|
// a stray ':'; the value may follow
|
||||||
|
get_token();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (!recover_missing_value(sax, states))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// the state evaluation reads the token again
|
||||||
|
m_lexer.unget_token();
|
||||||
|
skip_to_state_evaluation = true;
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -439,17 +559,39 @@ class parser
|
|||||||
|
|
||||||
// We are done with this array. Before we can parse a
|
// We are done with this array. Before we can parse a
|
||||||
// new value, we need to evaluate the new state first.
|
// new value, we need to evaluate the new state first.
|
||||||
// By setting skip_to_state_evaluation to false, we
|
// By setting skip_to_state_evaluation to true, the next
|
||||||
// are effectively jumping to the beginning of this if.
|
// iteration skips parsing a value and evaluates the
|
||||||
|
// enclosing state directly.
|
||||||
JSON_ASSERT(!states.empty());
|
JSON_ASSERT(!states.empty());
|
||||||
states.pop_back();
|
states.pop_back();
|
||||||
skip_to_state_evaluation = true;
|
skip_to_state_evaluation = true;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_array, "array"), nullptr), allow_recovery))
|
||||||
m_lexer.get_token_string(),
|
{
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_array, "array"), nullptr));
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// recover
|
||||||
|
if (last_token == token_type::end_of_input)
|
||||||
|
{
|
||||||
|
// the input ends inside the array
|
||||||
|
return close_containers(sax, states);
|
||||||
|
}
|
||||||
|
if (last_token == token_type::end_object)
|
||||||
|
{
|
||||||
|
// a wrong closing bracket closes the innermost container
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!sax->end_array()))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
states.pop_back();
|
||||||
|
skip_to_state_evaluation = true;
|
||||||
|
}
|
||||||
|
// otherwise, a missing ',' (or a stray ':', which value
|
||||||
|
// parsing drops): the next value begins here
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
// states.back() is false -> object
|
// states.back() is false -> object
|
||||||
@@ -466,11 +608,12 @@ class parser
|
|||||||
// parse key
|
// parse key
|
||||||
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
||||||
{
|
{
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
if (!continue_after(key_error(sax, allow_recovery, false), skip_to_state_evaluation))
|
||||||
m_lexer.get_token_string(),
|
{
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr));
|
return false;
|
||||||
|
}
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
@@ -479,9 +622,11 @@ class parser
|
|||||||
// parse separator (:)
|
// parse separator (:)
|
||||||
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
||||||
{
|
{
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
if (!continue_after(key_error(sax, allow_recovery, true), skip_to_state_evaluation))
|
||||||
m_lexer.get_token_string(),
|
{
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr));
|
return false;
|
||||||
|
}
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
// parse values
|
// parse values
|
||||||
@@ -500,20 +645,488 @@ class parser
|
|||||||
|
|
||||||
// We are done with this object. Before we can parse a
|
// We are done with this object. Before we can parse a
|
||||||
// new value, we need to evaluate the new state first.
|
// new value, we need to evaluate the new state first.
|
||||||
// By setting skip_to_state_evaluation to false, we
|
// By setting skip_to_state_evaluation to true, the next
|
||||||
// are effectively jumping to the beginning of this if.
|
// iteration skips parsing a value and evaluates the
|
||||||
|
// enclosing state directly.
|
||||||
JSON_ASSERT(!states.empty());
|
JSON_ASSERT(!states.empty());
|
||||||
states.pop_back();
|
states.pop_back();
|
||||||
skip_to_state_evaluation = true;
|
skip_to_state_evaluation = true;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
return sax->parse_error(m_lexer.get_position(),
|
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_object, "object"), nullptr), allow_recovery))
|
||||||
m_lexer.get_token_string(),
|
{
|
||||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_object, "object"), nullptr));
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// recover
|
||||||
|
if (last_token == token_type::end_of_input)
|
||||||
|
{
|
||||||
|
// the input ends inside the object
|
||||||
|
return close_containers(sax, states);
|
||||||
|
}
|
||||||
|
if (last_token == token_type::end_array)
|
||||||
|
{
|
||||||
|
// a wrong closing bracket closes the innermost container
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!sax->end_object()))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
states.pop_back();
|
||||||
|
skip_to_state_evaluation = true;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (!continue_after(recover_member(sax, allow_recovery), skip_to_state_evaluation))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief continue sax_parse_internal() after a recovery
|
||||||
|
@return whether to continue parsing
|
||||||
|
*/
|
||||||
|
bool continue_after(const next_step step, bool& skip_to_state_evaluation)
|
||||||
|
{
|
||||||
|
if (step == next_step::evaluate_state)
|
||||||
|
{
|
||||||
|
// the state evaluation reads the token again
|
||||||
|
m_lexer.unget_token();
|
||||||
|
skip_to_state_evaluation = true;
|
||||||
|
}
|
||||||
|
return step != next_step::stop;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// the parser for parse() and accept() never recovers: stop parsing
|
||||||
|
static std::false_type continue_after(std::false_type /*step*/, bool& /*skip_to_state_evaluation*/) noexcept
|
||||||
|
{
|
||||||
|
return {};
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief parse an object key and the name separator (:) after it
|
||||||
|
|
||||||
|
last_token is the token where the key is expected. sax_parse_internal()
|
||||||
|
repeats these steps rather than calling this function, which is used
|
||||||
|
when recovering from an error.
|
||||||
|
|
||||||
|
@return next_step::parse_value if the value follows, with last_token its
|
||||||
|
first token; next_step::evaluate_state if the object's state is
|
||||||
|
to be evaluated after recovering from an error; next_step::stop
|
||||||
|
to stop parsing
|
||||||
|
*/
|
||||||
|
template<typename SAX>
|
||||||
|
next_step parse_key(SAX* sax)
|
||||||
|
{
|
||||||
|
const std::true_type allow_recovery{};
|
||||||
|
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
||||||
|
{
|
||||||
|
return key_error(sax, allow_recovery, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
||||||
|
{
|
||||||
|
return next_step::stop;
|
||||||
|
}
|
||||||
|
|
||||||
|
// parse separator (:)
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
||||||
|
{
|
||||||
|
return key_error(sax, allow_recovery, true);
|
||||||
|
}
|
||||||
|
|
||||||
|
// the value begins with the next token
|
||||||
|
get_token();
|
||||||
|
return next_step::parse_value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief report a number that is too large for number_float_t, and recover
|
||||||
|
from the error by passing the value on; the SAX parser gets the
|
||||||
|
number's text as well
|
||||||
|
|
||||||
|
This is a separate function, as reading other numbers is measurably
|
||||||
|
slower if the error is handled where they are read.
|
||||||
|
|
||||||
|
@param[in] sax the SAX parser
|
||||||
|
@param[in] value the value that is not finite
|
||||||
|
@return whether to continue parsing
|
||||||
|
*/
|
||||||
|
template<typename SAX, typename AllowRecovery>
|
||||||
|
bool overflow_error(SAX* sax, const number_float_t value, AllowRecovery allow_recovery)
|
||||||
|
{
|
||||||
|
if (!report_error(sax, out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr), allow_recovery))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
return sax->number_float(value, m_lexer.get_string());
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief report a missing key, or a missing name separator (:) after the
|
||||||
|
key; the parser for parse() and accept() never recovers
|
||||||
|
|
||||||
|
@param[in] key_read whether the key was read, so that the name separator
|
||||||
|
is missing
|
||||||
|
@return std::false_type, see report_error()
|
||||||
|
*/
|
||||||
|
template<typename SAX>
|
||||||
|
std::false_type key_error(SAX* sax, std::false_type allow_recovery, const bool key_read)
|
||||||
|
{
|
||||||
|
return report_error(sax, parse_error::create(101, m_lexer.get_position(), key_read
|
||||||
|
? exception_message(token_type::name_separator, "object separator")
|
||||||
|
: exception_message(token_type::value_string, "object key"), nullptr), allow_recovery);
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief report a missing key, or a missing name separator (:) after the
|
||||||
|
key, and recover from it
|
||||||
|
|
||||||
|
@param[in] key_read whether the key was read, so that the name separator
|
||||||
|
is missing
|
||||||
|
*/
|
||||||
|
template<typename SAX>
|
||||||
|
next_step key_error(SAX* sax, std::true_type allow_recovery, const bool key_read)
|
||||||
|
{
|
||||||
|
if (!key_read)
|
||||||
|
{
|
||||||
|
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr), allow_recovery))
|
||||||
|
{
|
||||||
|
return next_step::stop;
|
||||||
|
}
|
||||||
|
return recover_key(sax);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr), allow_recovery))
|
||||||
|
{
|
||||||
|
return next_step::stop;
|
||||||
|
}
|
||||||
|
return recover_name_separator(sax);
|
||||||
|
}
|
||||||
|
|
||||||
|
/////////////////////
|
||||||
|
// error recovery
|
||||||
|
/////////////////////
|
||||||
|
|
||||||
|
/*
|
||||||
|
The functions below repair an error after the SAX parser's parse_error()
|
||||||
|
returned true (see #3989). Each mistake is repaired by the smallest local
|
||||||
|
edit: a missing ',' or ':' is inserted, a stray token is removed, what can
|
||||||
|
be read of an invalid string or number is kept (see
|
||||||
|
lexer::recover_token()), a missing value becomes null, a wrong closing
|
||||||
|
bracket closes the innermost container, and the end of the input closes
|
||||||
|
all of them. The events stay balanced, and every key() is followed by
|
||||||
|
exactly one value.
|
||||||
|
|
||||||
|
A repair hands a token to the state evaluation, by returning it to the
|
||||||
|
lexer (lexer::unget_token()) so that the state evaluation reads it again,
|
||||||
|
only if it is ',', ']', '}', or the end of the input. The state evaluation
|
||||||
|
hands a token to value or key parsing only if it is none of them, so a
|
||||||
|
token is never handed back and forth. Every other step reads a token or
|
||||||
|
closes a container, so parsing always ends.
|
||||||
|
*/
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief report an error to the SAX parser; the parser for parse() and
|
||||||
|
accept() never recovers
|
||||||
|
|
||||||
|
@return std::false_type rather than false: its value is known where the
|
||||||
|
function is called even if the call is not inlined, so the code
|
||||||
|
for recovering is not generated
|
||||||
|
*/
|
||||||
|
template<typename SAX, typename Exception>
|
||||||
|
std::false_type report_error(SAX* sax, const Exception& ex, std::false_type /*allow_recovery*/)
|
||||||
|
{
|
||||||
|
error_reported = true;
|
||||||
|
static_cast<void>(sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), ex));
|
||||||
|
return {};
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief report an error to the SAX parser
|
||||||
|
@return whether to recover from the error
|
||||||
|
*/
|
||||||
|
template<typename SAX, typename Exception>
|
||||||
|
bool report_error(SAX* sax, const Exception& ex, std::true_type /*allow_recovery*/)
|
||||||
|
{
|
||||||
|
const std::size_t position = m_lexer.get_position().chars_read_total;
|
||||||
|
if (error_reported && position == last_error_position && last_token == last_error_token)
|
||||||
|
{
|
||||||
|
// a repair handed on the token of the error it repaired; the
|
||||||
|
// token was reported already, and the SAX parser asked to recover
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
error_reported = true;
|
||||||
|
last_error_position = position;
|
||||||
|
last_error_token = last_token;
|
||||||
|
|
||||||
|
if (!sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), ex))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// the token string of the next error begins here
|
||||||
|
m_lexer.restart_token_string();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief keep what can be read of the token that the lexer rejected
|
||||||
|
|
||||||
|
The error was reported for the rejected token, so it is not reported again
|
||||||
|
for the token it is repaired to (see lexer::recover_token()).
|
||||||
|
*/
|
||||||
|
token_type recover_token()
|
||||||
|
{
|
||||||
|
last_token = m_lexer.recover_token();
|
||||||
|
last_error_position = m_lexer.get_position().chars_read_total;
|
||||||
|
last_error_token = last_token;
|
||||||
|
return last_token;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// pass the end events of all open containers
|
||||||
|
template<typename SAX>
|
||||||
|
bool close_containers(SAX* sax, std::vector<bool>& states)
|
||||||
|
{
|
||||||
|
while (!states.empty())
|
||||||
|
{
|
||||||
|
const bool is_array = states.back();
|
||||||
|
states.pop_back();
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(is_array ? !sax->end_array() : !sax->end_object()))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief read tokens until one begins a value, skipping everything before
|
||||||
|
the top-level value
|
||||||
|
@return whether a value begins with last_token
|
||||||
|
*/
|
||||||
|
bool skip_to_value()
|
||||||
|
{
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
switch (get_token())
|
||||||
|
{
|
||||||
|
case token_type::begin_array:
|
||||||
|
case token_type::begin_object:
|
||||||
|
case token_type::literal_false:
|
||||||
|
case token_type::literal_null:
|
||||||
|
case token_type::literal_true:
|
||||||
|
case token_type::value_float:
|
||||||
|
case token_type::value_integer:
|
||||||
|
case token_type::value_string:
|
||||||
|
case token_type::value_unsigned:
|
||||||
|
return true;
|
||||||
|
|
||||||
|
case token_type::end_of_input:
|
||||||
|
return false;
|
||||||
|
|
||||||
|
case token_type::parse_error:
|
||||||
|
recover_token();
|
||||||
|
if (last_token != token_type::uninitialized)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
|
||||||
|
case token_type::uninitialized:
|
||||||
|
case token_type::end_array:
|
||||||
|
case token_type::end_object:
|
||||||
|
case token_type::name_separator:
|
||||||
|
case token_type::value_separator:
|
||||||
|
case token_type::literal_or_value:
|
||||||
|
default:
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief skip the rest of an object member that cannot be read
|
||||||
|
|
||||||
|
Reads tokens, beginning with last_token, until a ',', '}', or ']' that is
|
||||||
|
not inside a container that begins in the skipped tokens, or the end of
|
||||||
|
the input.
|
||||||
|
*/
|
||||||
|
void skip_member()
|
||||||
|
{
|
||||||
|
std::size_t depth = 0;
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
switch (last_token)
|
||||||
|
{
|
||||||
|
case token_type::begin_array:
|
||||||
|
case token_type::begin_object:
|
||||||
|
++depth;
|
||||||
|
break;
|
||||||
|
|
||||||
|
case token_type::end_array:
|
||||||
|
case token_type::end_object:
|
||||||
|
if (depth == 0)
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
--depth;
|
||||||
|
break;
|
||||||
|
|
||||||
|
case token_type::value_separator:
|
||||||
|
if (depth == 0)
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
|
||||||
|
case token_type::end_of_input:
|
||||||
|
return;
|
||||||
|
|
||||||
|
case token_type::parse_error:
|
||||||
|
recover_token();
|
||||||
|
break;
|
||||||
|
|
||||||
|
case token_type::uninitialized:
|
||||||
|
case token_type::literal_true:
|
||||||
|
case token_type::literal_false:
|
||||||
|
case token_type::literal_null:
|
||||||
|
case token_type::value_string:
|
||||||
|
case token_type::value_unsigned:
|
||||||
|
case token_type::value_integer:
|
||||||
|
case token_type::value_float:
|
||||||
|
case token_type::name_separator:
|
||||||
|
case token_type::literal_or_value:
|
||||||
|
default:
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
get_token();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief pass a value where it is missing
|
||||||
|
|
||||||
|
last_token is ',', ']', '}', or the end of the input, where a value was
|
||||||
|
expected. In an object, the key gets null; in an array, a ',' where a
|
||||||
|
value is missing stands for null (as in JavaScript), while an array that
|
||||||
|
ends there just ends.
|
||||||
|
*/
|
||||||
|
template<typename SAX>
|
||||||
|
bool recover_missing_value(SAX* sax, const std::vector<bool>& states)
|
||||||
|
{
|
||||||
|
JSON_ASSERT(!states.empty());
|
||||||
|
if (!states.back() || last_token == token_type::value_separator)
|
||||||
|
{
|
||||||
|
return sax->null();
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// recover from a missing key; last_token is where it was expected
|
||||||
|
template<typename SAX>
|
||||||
|
next_step recover_key(SAX* sax)
|
||||||
|
{
|
||||||
|
switch (last_token)
|
||||||
|
{
|
||||||
|
case token_type::value_separator:
|
||||||
|
case token_type::end_object:
|
||||||
|
case token_type::end_array:
|
||||||
|
case token_type::end_of_input:
|
||||||
|
// no member: the object's state handles the token
|
||||||
|
return next_step::evaluate_state;
|
||||||
|
|
||||||
|
case token_type::parse_error:
|
||||||
|
recover_token();
|
||||||
|
if (last_token == token_type::value_string)
|
||||||
|
{
|
||||||
|
// a key that could be repaired
|
||||||
|
return parse_key(sax);
|
||||||
|
}
|
||||||
|
skip_member();
|
||||||
|
return next_step::evaluate_state;
|
||||||
|
|
||||||
|
case token_type::uninitialized:
|
||||||
|
case token_type::literal_true:
|
||||||
|
case token_type::literal_false:
|
||||||
|
case token_type::literal_null:
|
||||||
|
case token_type::value_string:
|
||||||
|
case token_type::value_unsigned:
|
||||||
|
case token_type::value_integer:
|
||||||
|
case token_type::value_float:
|
||||||
|
case token_type::begin_array:
|
||||||
|
case token_type::begin_object:
|
||||||
|
case token_type::name_separator:
|
||||||
|
case token_type::literal_or_value:
|
||||||
|
default:
|
||||||
|
// a member without a key
|
||||||
|
skip_member();
|
||||||
|
return next_step::evaluate_state;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// recover from a missing name separator (:) after the key; last_token
|
||||||
|
/// is where it was expected
|
||||||
|
template<typename SAX>
|
||||||
|
next_step recover_name_separator(SAX* sax)
|
||||||
|
{
|
||||||
|
switch (last_token)
|
||||||
|
{
|
||||||
|
case token_type::value_separator:
|
||||||
|
case token_type::end_object:
|
||||||
|
case token_type::end_array:
|
||||||
|
case token_type::end_of_input:
|
||||||
|
// the value is missing as well
|
||||||
|
return sax->null() ? next_step::evaluate_state : next_step::stop;
|
||||||
|
|
||||||
|
case token_type::uninitialized:
|
||||||
|
case token_type::literal_true:
|
||||||
|
case token_type::literal_false:
|
||||||
|
case token_type::literal_null:
|
||||||
|
case token_type::value_string:
|
||||||
|
case token_type::value_unsigned:
|
||||||
|
case token_type::value_integer:
|
||||||
|
case token_type::value_float:
|
||||||
|
case token_type::begin_array:
|
||||||
|
case token_type::begin_object:
|
||||||
|
case token_type::name_separator:
|
||||||
|
case token_type::parse_error:
|
||||||
|
case token_type::literal_or_value:
|
||||||
|
default:
|
||||||
|
// a missing ':'; the value begins here
|
||||||
|
return next_step::parse_value;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// recover from a token after an object member that is neither ',' nor
|
||||||
|
/// '}' (nor ']' or the end of the input, which the caller handles)
|
||||||
|
template<typename SAX>
|
||||||
|
next_step recover_member(SAX* sax, std::true_type /*allow_recovery*/)
|
||||||
|
{
|
||||||
|
if (last_token == token_type::parse_error)
|
||||||
|
{
|
||||||
|
recover_token();
|
||||||
|
}
|
||||||
|
if (last_token == token_type::value_string)
|
||||||
|
{
|
||||||
|
// a missing ','; the next key begins here
|
||||||
|
return parse_key(sax);
|
||||||
|
}
|
||||||
|
skip_member();
|
||||||
|
return next_step::evaluate_state;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// the parser for parse() and accept() never recovers (and does not come
|
||||||
|
/// here, as report_error() returned false)
|
||||||
|
template<typename SAX>
|
||||||
|
std::false_type recover_member(SAX* /*sax*/, std::false_type /*allow_recovery*/) const noexcept
|
||||||
|
{
|
||||||
|
return {};
|
||||||
|
}
|
||||||
|
|
||||||
/// get next token from lexer
|
/// get next token from lexer
|
||||||
token_type get_token()
|
token_type get_token()
|
||||||
{
|
{
|
||||||
@@ -560,6 +1173,12 @@ class parser
|
|||||||
const bool allow_exceptions = true;
|
const bool allow_exceptions = true;
|
||||||
/// whether trailing commas in objects and arrays should be ignored (true) or signaled as errors (false)
|
/// whether trailing commas in objects and arrays should be ignored (true) or signaled as errors (false)
|
||||||
const bool ignore_trailing_commas = false;
|
const bool ignore_trailing_commas = false;
|
||||||
|
/// whether an error was reported to the SAX parser
|
||||||
|
bool error_reported = false;
|
||||||
|
/// the position of the last reported error
|
||||||
|
std::size_t last_error_position = 0;
|
||||||
|
/// the token of the last reported error
|
||||||
|
token_type last_error_token = token_type::uninitialized;
|
||||||
};
|
};
|
||||||
|
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
|
|||||||
@@ -28,6 +28,7 @@
|
|||||||
#include <nlohmann/detail/macro_scope.hpp>
|
#include <nlohmann/detail/macro_scope.hpp>
|
||||||
#include <nlohmann/detail/output/output_adapters.hpp>
|
#include <nlohmann/detail/output/output_adapters.hpp>
|
||||||
#include <nlohmann/detail/string_concat.hpp>
|
#include <nlohmann/detail/string_concat.hpp>
|
||||||
|
#include <nlohmann/detail/string_utils.hpp>
|
||||||
|
|
||||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||||
namespace detail
|
namespace detail
|
||||||
@@ -2128,20 +2129,10 @@ class binary_writer
|
|||||||
const std::size_t valid = valid_utf8_prefix(data, s.size());
|
const std::size_t valid = valid_utf8_prefix(data, s.size());
|
||||||
if (JSON_HEDLEY_UNLIKELY(valid != s.size()))
|
if (JSON_HEDLEY_UNLIKELY(valid != s.size()))
|
||||||
{
|
{
|
||||||
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context));
|
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", detail::hex_byte(data[valid])), &context));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @return a byte as two uppercase hexadecimal digits
|
|
||||||
static std::string hex_byte(const std::uint8_t byte)
|
|
||||||
{
|
|
||||||
std::string result = "00";
|
|
||||||
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
|
|
||||||
result[0] = nibble_to_hex[byte / 16];
|
|
||||||
result[1] = nibble_to_hex[byte % 16];
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief write an integer in the shortest encoding
|
@brief write an integer in the shortest encoding
|
||||||
|
|
||||||
|
|||||||
@@ -8,9 +8,7 @@
|
|||||||
|
|
||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
#include <algorithm> // copy
|
|
||||||
#include <cstddef> // size_t
|
#include <cstddef> // size_t
|
||||||
#include <iterator> // back_inserter
|
|
||||||
#include <memory> // shared_ptr, make_shared
|
#include <memory> // shared_ptr, make_shared
|
||||||
#include <string> // basic_string
|
#include <string> // basic_string
|
||||||
#include <utility> // move
|
#include <utility> // move
|
||||||
@@ -31,6 +29,10 @@ namespace detail
|
|||||||
template<typename CharType> struct output_adapter_protocol
|
template<typename CharType> struct output_adapter_protocol
|
||||||
{
|
{
|
||||||
virtual void write_character(CharType c) = 0;
|
virtual void write_character(CharType c) = 0;
|
||||||
|
/// @param[in] s pointer to the characters to write; binary_writer legitimately
|
||||||
|
/// passes a null pointer together with length 0 for an empty
|
||||||
|
/// string or binary value, so implementations must tolerate that
|
||||||
|
/// @param[in] length number of characters at @a s
|
||||||
virtual void write_characters(const CharType* s, std::size_t length) = 0;
|
virtual void write_characters(const CharType* s, std::size_t length) = 0;
|
||||||
virtual ~output_adapter_protocol() = default;
|
virtual ~output_adapter_protocol() = default;
|
||||||
|
|
||||||
@@ -97,7 +99,6 @@ class output_vector_adapter : public output_adapter_protocol<CharType>
|
|||||||
sink.write_character(c);
|
sink.write_character(c);
|
||||||
}
|
}
|
||||||
|
|
||||||
JSON_HEDLEY_NON_NULL(2)
|
|
||||||
void write_characters(const CharType* s, std::size_t length) override
|
void write_characters(const CharType* s, std::size_t length) override
|
||||||
{
|
{
|
||||||
sink.write_characters(s, length);
|
sink.write_characters(s, length);
|
||||||
@@ -122,7 +123,6 @@ class output_stream_adapter : public output_adapter_protocol<CharType>
|
|||||||
stream.put(c);
|
stream.put(c);
|
||||||
}
|
}
|
||||||
|
|
||||||
JSON_HEDLEY_NON_NULL(2)
|
|
||||||
void write_characters(const CharType* s, std::size_t length) override
|
void write_characters(const CharType* s, std::size_t length) override
|
||||||
{
|
{
|
||||||
stream.write(s, static_cast<std::streamsize>(length));
|
stream.write(s, static_cast<std::streamsize>(length));
|
||||||
@@ -147,7 +147,6 @@ class output_string_adapter : public output_adapter_protocol<CharType>
|
|||||||
str.push_back(c);
|
str.push_back(c);
|
||||||
}
|
}
|
||||||
|
|
||||||
JSON_HEDLEY_NON_NULL(2)
|
|
||||||
void write_characters(const CharType* s, std::size_t length) override
|
void write_characters(const CharType* s, std::size_t length) override
|
||||||
{
|
{
|
||||||
str.append(s, length);
|
str.append(s, length);
|
||||||
|
|||||||
@@ -3,24 +3,23 @@
|
|||||||
// | | |__ | | | | | | version 3.12.0
|
// | | |__ | | | | | | version 3.12.0
|
||||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||||
//
|
//
|
||||||
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
|
|
||||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||||
// SPDX-License-Identifier: MIT
|
// SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
#include <algorithm> // reverse, remove, fill, find, none_of, min
|
#include <algorithm> // remove, fill, find, none_of, min
|
||||||
#include <array> // array
|
#include <array> // array
|
||||||
#include <clocale> // localeconv, lconv
|
#include <clocale> // localeconv, lconv
|
||||||
#include <cmath> // labs, isfinite, isnan, signbit
|
#include <cmath> // isfinite
|
||||||
#include <cstddef> // size_t, ptrdiff_t
|
#include <cstddef> // size_t, ptrdiff_t
|
||||||
#include <cstdint> // uint8_t
|
#include <cstdint> // uint8_t
|
||||||
#include <cstdio> // snprintf
|
#include <cstdio> // snprintf
|
||||||
#include <cstring> // memcpy, memset
|
#include <cstring> // memcpy, memset
|
||||||
|
#include <iterator> // next
|
||||||
#include <limits> // numeric_limits
|
#include <limits> // numeric_limits
|
||||||
#include <string> // string, char_traits
|
#include <string> // string, char_traits
|
||||||
#include <type_traits> // is_same
|
#include <type_traits> // is_same
|
||||||
#include <utility> // move
|
|
||||||
#include <vector> // vector
|
#include <vector> // vector
|
||||||
|
|
||||||
#include <nlohmann/detail/conversions/to_chars.hpp>
|
#include <nlohmann/detail/conversions/to_chars.hpp>
|
||||||
@@ -28,7 +27,6 @@
|
|||||||
#include <nlohmann/detail/input/string_scan.hpp>
|
#include <nlohmann/detail/input/string_scan.hpp>
|
||||||
#include <nlohmann/detail/macro_scope.hpp>
|
#include <nlohmann/detail/macro_scope.hpp>
|
||||||
#include <nlohmann/detail/meta/cpp_future.hpp>
|
#include <nlohmann/detail/meta/cpp_future.hpp>
|
||||||
#include <nlohmann/detail/output/binary_writer.hpp>
|
|
||||||
#include <nlohmann/detail/output/output_adapters.hpp>
|
#include <nlohmann/detail/output/output_adapters.hpp>
|
||||||
#include <nlohmann/detail/recursion_depth_limit.hpp>
|
#include <nlohmann/detail/recursion_depth_limit.hpp>
|
||||||
#include <nlohmann/detail/string_concat.hpp>
|
#include <nlohmann/detail/string_concat.hpp>
|
||||||
@@ -83,7 +81,6 @@ class serializer
|
|||||||
const std::size_t indent_step_ = 0,
|
const std::size_t indent_step_ = 0,
|
||||||
error_handler_t error_handler_ = error_handler_t::strict)
|
error_handler_t error_handler_ = error_handler_t::strict)
|
||||||
: o(&s)
|
: o(&s)
|
||||||
, locale(std::localeconv())
|
|
||||||
, indent_char(ichar)
|
, indent_char(ichar)
|
||||||
, pretty_print(pretty_print_)
|
, pretty_print(pretty_print_)
|
||||||
, ensure_ascii(ensure_ascii_)
|
, ensure_ascii(ensure_ascii_)
|
||||||
@@ -106,9 +103,10 @@ class serializer
|
|||||||
additional parameter. Arrays and objects are serialized without recursion,
|
additional parameter. Arrays and objects are serialized without recursion,
|
||||||
however deeply they are nested.
|
however deeply they are nested.
|
||||||
|
|
||||||
- strings and object keys are escaped using `escape_string()`
|
- strings and object keys are escaped using @ref dump_escaped
|
||||||
- integer numbers are converted implicitly via `operator<<`
|
- integer numbers are converted using a digit-pair lookup table (@ref dump_integer)
|
||||||
- floating-point numbers are converted to a string using `"%g"` format
|
- floating-point numbers are converted to a string using @ref dump_float, which
|
||||||
|
uses `to_chars` for IEEE-754 types and `snprintf` otherwise
|
||||||
- binary values are serialized as objects containing the subtype and the
|
- binary values are serialized as objects containing the subtype and the
|
||||||
byte array
|
byte array
|
||||||
|
|
||||||
@@ -283,127 +281,16 @@ class serializer
|
|||||||
}
|
}
|
||||||
|
|
||||||
case value_t::string:
|
case value_t::string:
|
||||||
{
|
|
||||||
put_char('"');
|
|
||||||
dump_escaped(*val.m_data.m_value.string);
|
|
||||||
put_char('"');
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::binary:
|
case value_t::binary:
|
||||||
{
|
|
||||||
if (pretty_print)
|
|
||||||
{
|
|
||||||
put_literal("{\n");
|
|
||||||
|
|
||||||
// variable to hold indentation for recursive calls
|
|
||||||
const auto new_indent = next_indent(current_indent, indent_step);
|
|
||||||
|
|
||||||
put_indent(new_indent);
|
|
||||||
|
|
||||||
put_literal("\"bytes\": [");
|
|
||||||
|
|
||||||
if (!val.m_data.m_value.binary->empty())
|
|
||||||
{
|
|
||||||
for (auto i = val.m_data.m_value.binary->cbegin();
|
|
||||||
i != val.m_data.m_value.binary->cend() - 1; ++i)
|
|
||||||
{
|
|
||||||
dump_byte(*i);
|
|
||||||
put_literal(", ");
|
|
||||||
}
|
|
||||||
dump_byte(val.m_data.m_value.binary->back());
|
|
||||||
}
|
|
||||||
|
|
||||||
put_literal("],\n");
|
|
||||||
put_indent(new_indent);
|
|
||||||
|
|
||||||
put_literal("\"subtype\": ");
|
|
||||||
if (val.m_data.m_value.binary->has_subtype())
|
|
||||||
{
|
|
||||||
dump_integer(val.m_data.m_value.binary->subtype());
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
put_literal("null");
|
|
||||||
}
|
|
||||||
put_char('\n');
|
|
||||||
put_indent(current_indent);
|
|
||||||
put_char('}');
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
put_literal("{\"bytes\":[");
|
|
||||||
|
|
||||||
if (!val.m_data.m_value.binary->empty())
|
|
||||||
{
|
|
||||||
for (auto i = val.m_data.m_value.binary->cbegin();
|
|
||||||
i != val.m_data.m_value.binary->cend() - 1; ++i)
|
|
||||||
{
|
|
||||||
dump_byte(*i);
|
|
||||||
put_char(',');
|
|
||||||
}
|
|
||||||
dump_byte(val.m_data.m_value.binary->back());
|
|
||||||
}
|
|
||||||
|
|
||||||
put_literal("],\"subtype\":");
|
|
||||||
if (val.m_data.m_value.binary->has_subtype())
|
|
||||||
{
|
|
||||||
dump_integer(val.m_data.m_value.binary->subtype());
|
|
||||||
put_char('}');
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
put_literal("null}");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::boolean:
|
case value_t::boolean:
|
||||||
{
|
|
||||||
if (val.m_data.m_value.boolean)
|
|
||||||
{
|
|
||||||
put_literal("true");
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
put_literal("false");
|
|
||||||
}
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::number_integer:
|
case value_t::number_integer:
|
||||||
{
|
|
||||||
dump_integer(val.m_data.m_value.number_integer);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::number_unsigned:
|
case value_t::number_unsigned:
|
||||||
{
|
|
||||||
dump_integer(val.m_data.m_value.number_unsigned);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::number_float:
|
case value_t::number_float:
|
||||||
{
|
|
||||||
dump_float(val.m_data.m_value.number_float);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::discarded:
|
case value_t::discarded:
|
||||||
{
|
|
||||||
put_literal("<discarded>");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
case value_t::null:
|
case value_t::null:
|
||||||
{
|
default:
|
||||||
put_literal("null");
|
dump_scalar(val, current_indent);
|
||||||
return;
|
return;
|
||||||
}
|
|
||||||
|
|
||||||
default: // LCOV_EXCL_LINE
|
|
||||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -560,9 +447,9 @@ class serializer
|
|||||||
@brief serialize the value @a val, but not the elements of a container
|
@brief serialize the value @a val, but not the elements of a container
|
||||||
|
|
||||||
An object or array with elements is opened and pushed onto @a stack for
|
An object or array with elements is opened and pushed onto @a stack for
|
||||||
@ref dump_internal to walk; everything else - including a binary value,
|
@ref dump_iteratively to walk; everything else - including a binary value,
|
||||||
which looks like an object but has no elements to descend into - is written
|
which looks like an object but has no elements to descend into - is written
|
||||||
out here in full.
|
out in full by @ref dump_scalar.
|
||||||
*/
|
*/
|
||||||
void dump_value(const BasicJsonType& val,
|
void dump_value(const BasicJsonType& val,
|
||||||
const std::size_t current_indent,
|
const std::size_t current_indent,
|
||||||
@@ -620,6 +507,35 @@ class serializer
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
case value_t::string:
|
||||||
|
case value_t::binary:
|
||||||
|
case value_t::boolean:
|
||||||
|
case value_t::number_integer:
|
||||||
|
case value_t::number_unsigned:
|
||||||
|
case value_t::number_float:
|
||||||
|
case value_t::discarded:
|
||||||
|
case value_t::null:
|
||||||
|
default:
|
||||||
|
dump_scalar(val, current_indent);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief serialize the value @a val, which is neither an object nor an array
|
||||||
|
|
||||||
|
Shared by @ref dump_internal and @ref dump_value, so that a value is written
|
||||||
|
the same way however deeply it is nested. A binary value is written out here
|
||||||
|
in full: it looks like an object, but has no elements to descend into.
|
||||||
|
|
||||||
|
@param[in] val value to serialize; not an object or array
|
||||||
|
@param[in] current_indent the indentation of @a val, used for a
|
||||||
|
pretty-printed binary value
|
||||||
|
*/
|
||||||
|
void dump_scalar(const BasicJsonType& val, const std::size_t current_indent)
|
||||||
|
{
|
||||||
|
switch (val.m_data.m_type)
|
||||||
|
{
|
||||||
case value_t::string:
|
case value_t::string:
|
||||||
{
|
{
|
||||||
put_char('"');
|
put_char('"');
|
||||||
@@ -740,6 +656,8 @@ class serializer
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
case value_t::object: // LCOV_EXCL_LINE
|
||||||
|
case value_t::array: // LCOV_EXCL_LINE
|
||||||
default: // LCOV_EXCL_LINE
|
default: // LCOV_EXCL_LINE
|
||||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
||||||
}
|
}
|
||||||
@@ -768,7 +686,7 @@ class serializer
|
|||||||
Escape a string by replacing certain special characters by a sequence of an
|
Escape a string by replacing certain special characters by a sequence of an
|
||||||
escape character (backslash) and another character and other control
|
escape character (backslash) and another character and other control
|
||||||
characters by a sequence of "\u" followed by a four-digit hex
|
characters by a sequence of "\u" followed by a four-digit hex
|
||||||
representation. The escaped string is written to output stream @a o.
|
representation. The escaped string is appended to @ref write_buffer.
|
||||||
|
|
||||||
@param[in] s the string to escape
|
@param[in] s the string to escape
|
||||||
|
|
||||||
@@ -962,7 +880,7 @@ class serializer
|
|||||||
{
|
{
|
||||||
case error_handler_t::strict:
|
case error_handler_t::strict:
|
||||||
{
|
{
|
||||||
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", hex_bytes(byte | 0)), nullptr));
|
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", detail::hex_byte(byte)), nullptr));
|
||||||
}
|
}
|
||||||
|
|
||||||
case error_handler_t::ignore:
|
case error_handler_t::ignore:
|
||||||
@@ -995,9 +913,9 @@ class serializer
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xEF');
|
string_buffer[bytes++] = '\xEF';
|
||||||
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBF');
|
string_buffer[bytes++] = '\xBF';
|
||||||
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBD');
|
string_buffer[bytes++] = '\xBD';
|
||||||
}
|
}
|
||||||
|
|
||||||
// write buffer and reset index; there must be 13 bytes
|
// write buffer and reset index; there must be 13 bytes
|
||||||
@@ -1054,7 +972,7 @@ class serializer
|
|||||||
{
|
{
|
||||||
case error_handler_t::strict:
|
case error_handler_t::strict:
|
||||||
{
|
{
|
||||||
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast<std::uint8_t>(s[s.size() - 1] | 0))), nullptr));
|
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", detail::hex_byte(static_cast<std::uint8_t>(s[s.size() - 1]))), nullptr));
|
||||||
}
|
}
|
||||||
|
|
||||||
case error_handler_t::ignore:
|
case error_handler_t::ignore:
|
||||||
@@ -1275,20 +1193,6 @@ class serializer
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
|
||||||
* @brief convert a byte to a uppercase hex representation
|
|
||||||
* @param[in] byte byte to represent
|
|
||||||
* @return representation ("00".."FF")
|
|
||||||
*/
|
|
||||||
static std::string hex_bytes(std::uint8_t byte)
|
|
||||||
{
|
|
||||||
std::string result = "FF";
|
|
||||||
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
|
|
||||||
result[0] = nibble_to_hex[byte / 16];
|
|
||||||
result[1] = nibble_to_hex[byte % 16];
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
* @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer
|
* @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer
|
||||||
*
|
*
|
||||||
@@ -1402,7 +1306,7 @@ class serializer
|
|||||||
/*!
|
/*!
|
||||||
@brief dump an integer
|
@brief dump an integer
|
||||||
|
|
||||||
Dump a given integer to output stream @a o. Works internally with
|
Dump a given integer, appending it to @ref write_buffer. Works internally with
|
||||||
@a number_buffer.
|
@a number_buffer.
|
||||||
|
|
||||||
@param[in] x integer number (signed or unsigned) to dump
|
@param[in] x integer number (signed or unsigned) to dump
|
||||||
@@ -1439,7 +1343,7 @@ class serializer
|
|||||||
}
|
}
|
||||||
|
|
||||||
// use a pointer to fill the buffer
|
// use a pointer to fill the buffer
|
||||||
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto)
|
||||||
|
|
||||||
number_unsigned_t abs_value;
|
number_unsigned_t abs_value;
|
||||||
|
|
||||||
@@ -1493,7 +1397,7 @@ class serializer
|
|||||||
/*!
|
/*!
|
||||||
@brief dump a floating-point number
|
@brief dump a floating-point number
|
||||||
|
|
||||||
Dump a given floating-point number to output stream @a o. Works internally
|
Dump a given floating-point number, appending it to @ref write_buffer. Works internally
|
||||||
with @a number_buffer.
|
with @a number_buffer.
|
||||||
|
|
||||||
@param[in] x floating-point number to dump
|
@param[in] x floating-point number to dump
|
||||||
@@ -1554,21 +1458,28 @@ class serializer
|
|||||||
// check if the buffer was large enough
|
// check if the buffer was large enough
|
||||||
JSON_ASSERT(static_cast<std::size_t>(len) < number_buffer.size());
|
JSON_ASSERT(static_cast<std::size_t>(len) < number_buffer.size());
|
||||||
|
|
||||||
|
// look up the locale's thousands separator and decimal point now,
|
||||||
|
// matching what snprintf_float() just used (see lexer::get_decimal_point())
|
||||||
|
const auto* loc = std::localeconv();
|
||||||
|
JSON_ASSERT(loc != nullptr);
|
||||||
|
const char thousands_sep = (loc->thousands_sep == nullptr) ? '\0' : *loc->thousands_sep;
|
||||||
|
const char decimal_point = (loc->decimal_point == nullptr) ? '\0' : *loc->decimal_point;
|
||||||
|
|
||||||
// erase thousands separators
|
// erase thousands separators
|
||||||
if (locale.thousands_sep != '\0')
|
if (thousands_sep != '\0')
|
||||||
{
|
{
|
||||||
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081
|
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081
|
||||||
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep);
|
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep);
|
||||||
std::fill(end, number_buffer.end(), '\0');
|
std::fill(end, number_buffer.end(), '\0');
|
||||||
JSON_ASSERT((end - number_buffer.begin()) <= len);
|
JSON_ASSERT((end - number_buffer.begin()) <= len);
|
||||||
len = (end - number_buffer.begin());
|
len = (end - number_buffer.begin());
|
||||||
}
|
}
|
||||||
|
|
||||||
// convert decimal point to '.'
|
// convert decimal point to '.'
|
||||||
if (locale.decimal_point != '\0' && locale.decimal_point != '.')
|
if (decimal_point != '\0' && decimal_point != '.')
|
||||||
{
|
{
|
||||||
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081
|
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081
|
||||||
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point);
|
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point);
|
||||||
if (dec_pos != number_buffer.end())
|
if (dec_pos != number_buffer.end())
|
||||||
{
|
{
|
||||||
*dec_pos = '.';
|
*dec_pos = '.';
|
||||||
@@ -1613,34 +1524,17 @@ class serializer
|
|||||||
*/
|
*/
|
||||||
number_unsigned_t remove_sign(number_integer_t x) noexcept
|
number_unsigned_t remove_sign(number_integer_t x) noexcept
|
||||||
{
|
{
|
||||||
JSON_ASSERT(x < 0 && x < (std::numeric_limits<number_integer_t>::max)()); // NOLINT(misc-redundant-expression)
|
JSON_ASSERT(x < 0);
|
||||||
return static_cast<number_unsigned_t>(-(x + 1)) + 1;
|
return static_cast<number_unsigned_t>(-(x + 1)) + 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
/// the locale's thousand separator and decimal point characters
|
|
||||||
struct locale_chars
|
|
||||||
{
|
|
||||||
explicit locale_chars(const std::lconv* loc) noexcept
|
|
||||||
: thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->thousands_sep)))
|
|
||||||
, decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->decimal_point)))
|
|
||||||
{}
|
|
||||||
|
|
||||||
const char thousands_sep;
|
|
||||||
const char decimal_point;
|
|
||||||
};
|
|
||||||
|
|
||||||
/// the output of the serializer (non-owning; the adapter lives at the call site)
|
/// the output of the serializer (non-owning; the adapter lives at the call site)
|
||||||
output_adapter_protocol<char>* o = nullptr;
|
output_adapter_protocol<char>* o = nullptr;
|
||||||
|
|
||||||
/// a (hopefully) large enough character buffer
|
/// a (hopefully) large enough character buffer
|
||||||
std::array<char, 64> number_buffer{{}};
|
std::array<char, 64> number_buffer{{}};
|
||||||
|
|
||||||
/// computed once from std::localeconv() at construction; @ref
|
|
||||||
/// locale_chars keeps std::localeconv()'s pointer from having to be held
|
|
||||||
/// past the constructor, while still letting these stay const
|
|
||||||
const locale_chars locale;
|
|
||||||
|
|
||||||
/// string buffer
|
/// string buffer
|
||||||
std::array<char, 512> string_buffer{{}};
|
std::array<char, 512> string_buffer{{}};
|
||||||
|
|
||||||
|
|||||||
@@ -3,6 +3,7 @@
|
|||||||
// | | |__ | | | | | | version 3.12.0
|
// | | |__ | | | | | | version 3.12.0
|
||||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||||
//
|
//
|
||||||
|
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
|
||||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||||
// SPDX-License-Identifier: MIT
|
// SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
@@ -12,6 +13,7 @@
|
|||||||
#include <cstddef> // size_t
|
#include <cstddef> // size_t
|
||||||
#include <cstdint> // uint8_t, uint32_t
|
#include <cstdint> // uint8_t, uint32_t
|
||||||
#include <string> // string, to_string
|
#include <string> // string, to_string
|
||||||
|
#include <utility> // move
|
||||||
|
|
||||||
#include <nlohmann/detail/abi_macros.hpp>
|
#include <nlohmann/detail/abi_macros.hpp>
|
||||||
#include <nlohmann/detail/macro_scope.hpp>
|
#include <nlohmann/detail/macro_scope.hpp>
|
||||||
@@ -36,6 +38,71 @@ StringType to_string(std::size_t value)
|
|||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// @return a byte as two uppercase hexadecimal digits
|
||||||
|
inline std::string hex_byte(const std::uint8_t byte)
|
||||||
|
{
|
||||||
|
std::string result = "00";
|
||||||
|
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
|
||||||
|
result[0] = nibble_to_hex[byte / 16];
|
||||||
|
result[1] = nibble_to_hex[byte % 16];
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
///////////////////
|
||||||
|
// UTF-8 encoding //
|
||||||
|
///////////////////
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief encode a Unicode code point as UTF-8
|
||||||
|
|
||||||
|
Used to turn a decoded code point back into bytes: by the wide-string input
|
||||||
|
adapters in input_adapters.hpp (one code point per UTF-32 unit, per UTF-16
|
||||||
|
unit outside the surrogate range, and per valid UTF-16 surrogate pair), and
|
||||||
|
by the lexer's `\uXXXX`/`\uXXXX\uYYYY` handling in lexer.hpp. Passing a
|
||||||
|
code point above U+10FFFF, or one in the surrogate range U+D800..U+DFFF, is
|
||||||
|
undefined behavior; callers are expected to have rejected those already
|
||||||
|
(the wide-string adapters pass malformed units through unencoded instead of
|
||||||
|
calling this function, and the lexer rejects unpaired surrogates before
|
||||||
|
reaching it).
|
||||||
|
|
||||||
|
@tparam Out a callable invoked with one byte (as std::uint32_t, 0x00..0xFF)
|
||||||
|
at a time, most significant byte first
|
||||||
|
@param[in] cp the code point to encode (at most U+10FFFF)
|
||||||
|
@param[in] out called once for each byte of the UTF-8 encoding of @a cp
|
||||||
|
*/
|
||||||
|
template<typename Out>
|
||||||
|
void encode_utf8(std::uint32_t cp, Out&& out)
|
||||||
|
{
|
||||||
|
JSON_ASSERT(cp <= 0x10FFFF);
|
||||||
|
|
||||||
|
if (cp < 0x80)
|
||||||
|
{
|
||||||
|
// 1-byte characters: 0xxxxxxx (ASCII)
|
||||||
|
out(cp);
|
||||||
|
}
|
||||||
|
else if (cp <= 0x7FF)
|
||||||
|
{
|
||||||
|
// 2-byte characters: 110xxxxx 10xxxxxx
|
||||||
|
out(0xC0u | (cp >> 6u));
|
||||||
|
out(0x80u | (cp & 0x3Fu));
|
||||||
|
}
|
||||||
|
else if (cp <= 0xFFFF)
|
||||||
|
{
|
||||||
|
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
|
||||||
|
out(0xE0u | (cp >> 12u));
|
||||||
|
out(0x80u | ((cp >> 6u) & 0x3Fu));
|
||||||
|
out(0x80u | (cp & 0x3Fu));
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
|
||||||
|
out(0xF0u | (cp >> 18u));
|
||||||
|
out(0x80u | ((cp >> 12u) & 0x3Fu));
|
||||||
|
out(0x80u | ((cp >> 6u) & 0x3Fu));
|
||||||
|
out(0x80u | (cp & 0x3Fu));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
///////////////////
|
///////////////////
|
||||||
// UTF-8 decoding //
|
// UTF-8 decoding //
|
||||||
///////////////////
|
///////////////////
|
||||||
@@ -51,11 +118,23 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
|
|||||||
written by Björn Hoehrmann. See
|
written by Björn Hoehrmann. See
|
||||||
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
|
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
|
||||||
|
|
||||||
This decoder is the single source of truth for UTF-8 validation in this
|
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four
|
||||||
library: it is used both by the serializer (to escape and, in strict mode,
|
places, which differ in speed, diagnostics, and how they read the input:
|
||||||
reject ill-formed UTF-8 when dumping a string) and by the binary readers
|
|
||||||
(to reject ill-formed UTF-8 in CBOR/MessagePack/BSON/UBJSON text strings at
|
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in
|
||||||
decode time; see @ref is_valid_utf8 below).
|
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR,
|
||||||
|
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
|
||||||
|
text strings at decode time).
|
||||||
|
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
|
||||||
|
diagnostic for each kind of error.
|
||||||
|
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
|
||||||
|
bulk string scan, the bulk path of the BON8 reader, and the BON8 writer.
|
||||||
|
They must accept exactly what the lexer's switch accepts.
|
||||||
|
- the byte path of binary_reader::get_bon8_string(): BON8 input without bulk
|
||||||
|
access, and the bytes the bulk path leaves to it.
|
||||||
|
|
||||||
|
All four must accept the same set of sequences, so a change to one needs a
|
||||||
|
matching change to the others.
|
||||||
|
|
||||||
@param[in,out] state the current decoder state
|
@param[in,out] state the current decoder state
|
||||||
@param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT)
|
@param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT)
|
||||||
@@ -133,5 +212,78 @@ inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noex
|
|||||||
return state == UTF8_ACCEPT;
|
return state == UTF8_ACCEPT;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief append U+FFFD REPLACEMENT CHARACTER, encoded in UTF-8
|
||||||
|
@param[in,out] s the string to append to
|
||||||
|
*/
|
||||||
|
template<typename StringType>
|
||||||
|
inline void append_replacement_character(StringType& s)
|
||||||
|
{
|
||||||
|
s.push_back(static_cast<typename StringType::value_type>(0xEFu));
|
||||||
|
s.push_back(static_cast<typename StringType::value_type>(0xBFu));
|
||||||
|
s.push_back(static_cast<typename StringType::value_type>(0xBDu));
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief replace ill-formed UTF-8 with U+FFFD REPLACEMENT CHARACTER
|
||||||
|
|
||||||
|
Each maximal subpart of an ill-formed sequence becomes one U+FFFD, as the
|
||||||
|
Unicode Standard recommends (Section 3.9, "U+FFFD Substitution of Maximal
|
||||||
|
Subparts"), and as the parser for JSON text does when it recovers from errors.
|
||||||
|
|
||||||
|
@param[in,out] s the string to repair
|
||||||
|
@param[in] first index of the first byte to repair; the bytes before it are
|
||||||
|
assumed to be valid UTF-8 that ends on a code point boundary
|
||||||
|
*/
|
||||||
|
template<typename StringType>
|
||||||
|
inline void replace_invalid_utf8(StringType& s, const std::size_t first = 0)
|
||||||
|
{
|
||||||
|
StringType result = s;
|
||||||
|
result.resize(first);
|
||||||
|
|
||||||
|
std::uint8_t state = UTF8_ACCEPT;
|
||||||
|
std::uint32_t codepoint = 0;
|
||||||
|
// the first byte of the sequence being decoded
|
||||||
|
std::size_t sequence_start = first;
|
||||||
|
|
||||||
|
std::size_t i = first;
|
||||||
|
while (i < s.size())
|
||||||
|
{
|
||||||
|
switch (decode(state, codepoint, static_cast<std::uint8_t>(s[i])))
|
||||||
|
{
|
||||||
|
case UTF8_ACCEPT:
|
||||||
|
for (++i; sequence_start < i; ++sequence_start)
|
||||||
|
{
|
||||||
|
result.push_back(s[sequence_start]);
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
|
||||||
|
case UTF8_REJECT:
|
||||||
|
append_replacement_character(result);
|
||||||
|
// the byte that made the sequence ill-formed begins the next
|
||||||
|
// one, unless it began this one
|
||||||
|
if (i == sequence_start)
|
||||||
|
{
|
||||||
|
++i;
|
||||||
|
}
|
||||||
|
state = UTF8_ACCEPT;
|
||||||
|
sequence_start = i;
|
||||||
|
break;
|
||||||
|
|
||||||
|
default: // in the middle of a sequence
|
||||||
|
++i;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// a sequence that the string ends in the middle of
|
||||||
|
if (state != UTF8_ACCEPT)
|
||||||
|
{
|
||||||
|
append_replacement_character(result);
|
||||||
|
}
|
||||||
|
|
||||||
|
s = std::move(result);
|
||||||
|
}
|
||||||
|
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
NLOHMANN_JSON_NAMESPACE_END
|
NLOHMANN_JSON_NAMESPACE_END
|
||||||
|
|||||||
@@ -158,12 +158,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
friend class ::nlohmann::detail::iter_impl;
|
friend class ::nlohmann::detail::iter_impl;
|
||||||
template<typename BasicJsonType, typename CharType, typename OutputSinkType>
|
template<typename BasicJsonType, typename CharType, typename OutputSinkType>
|
||||||
friend class ::nlohmann::detail::binary_writer;
|
friend class ::nlohmann::detail::binary_writer;
|
||||||
template<typename BasicJsonType, typename InputType, typename SAX>
|
template<typename BasicJsonType, typename InputType, typename SAX, bool AllowRecovery>
|
||||||
friend class ::nlohmann::detail::binary_reader;
|
friend class ::nlohmann::detail::binary_reader;
|
||||||
template<typename BasicJsonType, typename InputAdapterType>
|
template<typename BasicJsonType, typename InputAdapterType>
|
||||||
friend class ::nlohmann::detail::json_sax_dom_parser;
|
friend class ::nlohmann::detail::json_sax_dom_parser;
|
||||||
template<typename BasicJsonType, typename InputAdapterType>
|
template<typename BasicJsonType, typename InputAdapterType>
|
||||||
friend class ::nlohmann::detail::json_sax_dom_callback_parser;
|
friend class ::nlohmann::detail::json_sax_dom_callback_parser;
|
||||||
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
|
friend struct ::nlohmann::detail::diagnostic_positions;
|
||||||
|
#endif
|
||||||
friend class ::nlohmann::detail::exception;
|
friend class ::nlohmann::detail::exception;
|
||||||
|
|
||||||
/// workaround type for MSVC
|
/// workaround type for MSVC
|
||||||
@@ -5264,7 +5267,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
auto ia = detail::input_adapter(std::forward<InputType>(i));
|
auto ia = detail::input_adapter(std::forward<InputType>(i));
|
||||||
return format == input_format_t::json
|
return format == input_format_t::json
|
||||||
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
||||||
: detail::binary_reader<basic_json, decltype(ia), SAX>(std::move(ia), format).sax_parse(sax, strict);
|
: detail::binary_reader<basic_json, decltype(ia), SAX, true>(std::move(ia), format).sax_parse(sax, strict);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
/// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||||
@@ -5281,7 +5284,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
auto ia = detail::input_adapter(std::move(first), std::move(last));
|
auto ia = detail::input_adapter(std::move(first), std::move(last));
|
||||||
return format == input_format_t::json
|
return format == input_format_t::json
|
||||||
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
||||||
: detail::binary_reader<basic_json, decltype(ia), SAX>(std::move(ia), format).sax_parse(sax, strict);
|
: detail::binary_reader<basic_json, decltype(ia), SAX, true>(std::move(ia), format).sax_parse(sax, strict);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief generate SAX events
|
/// @brief generate SAX events
|
||||||
@@ -5303,7 +5306,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
// NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg)
|
// NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg)
|
||||||
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
||||||
// NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg)
|
// NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg)
|
||||||
: detail::binary_reader<basic_json, decltype(ia), SAX>(std::move(ia), format).sax_parse(sax, strict);
|
: detail::binary_reader<basic_json, decltype(ia), SAX, true>(std::move(ia), format).sax_parse(sax, strict);
|
||||||
}
|
}
|
||||||
#ifndef JSON_NO_IO
|
#ifndef JSON_NO_IO
|
||||||
/// @brief deserialize from stream
|
/// @brief deserialize from stream
|
||||||
@@ -6336,21 +6339,256 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
{
|
{
|
||||||
// the patch
|
// the patch
|
||||||
basic_json result(value_t::array);
|
basic_json result(value_t::array);
|
||||||
|
diff_recursively(result, source, target, path, 0);
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
// if the values are the same, return an empty patch
|
private:
|
||||||
|
/// @brief two arrays or two objects @ref diff_iteratively is diffing
|
||||||
|
struct diff_frame
|
||||||
|
{
|
||||||
|
diff_frame(const basic_json* source_, const basic_json* target_, const std::size_t path_length_) noexcept
|
||||||
|
: source(source_), target(target_), path_length(path_length_)
|
||||||
|
{}
|
||||||
|
|
||||||
|
// declared for GCC's -Weffc++, which asks for them in a class with
|
||||||
|
// pointer members and a non-trivial destructor; the exception
|
||||||
|
// specifications are left implicit, as GCC 4.8 rejects explicit ones
|
||||||
|
// that differ from them
|
||||||
|
diff_frame(const diff_frame&) = default;
|
||||||
|
diff_frame(diff_frame&&) = default;
|
||||||
|
diff_frame& operator=(const diff_frame&) = default;
|
||||||
|
diff_frame& operator=(diff_frame&&) = default;
|
||||||
|
~diff_frame() = default;
|
||||||
|
|
||||||
|
/// the values being diffed, both arrays or both objects
|
||||||
|
const basic_json* source;
|
||||||
|
const basic_json* target;
|
||||||
|
/// the length of their path in `current_path`
|
||||||
|
std::size_t path_length;
|
||||||
|
/// arrays: the next index to diff
|
||||||
|
std::size_t index = 0;
|
||||||
|
/// objects: the next member of source to look at
|
||||||
|
const_iterator member{}; // NOLINT(readability-redundant-member-init)
|
||||||
|
/// objects: the keys common to both, in source's order
|
||||||
|
std::vector<typename object_t::key_type> common_keys{}; // NOLINT(readability-redundant-member-init)
|
||||||
|
/// objects: the next entry of common_keys
|
||||||
|
std::size_t next_common = 0;
|
||||||
|
/// objects: the "add" operations for keys only target has
|
||||||
|
basic_json added_ops{}; // NOLINT(readability-redundant-member-init)
|
||||||
|
};
|
||||||
|
|
||||||
|
// The operations of a diff are built by the functions below rather than
|
||||||
|
// where they are needed: building one takes several temporaries, and
|
||||||
|
// unoptimized builds give each temporary a stack slot of its own in the
|
||||||
|
// function it appears in. In diff_recursively, which is on the call stack
|
||||||
|
// once per nesting level, that made every level cost kilobytes of stack.
|
||||||
|
|
||||||
|
/// @brief append a "replace" operation for @a path with @a value to @a result
|
||||||
|
static void diff_replace(basic_json& result, const string_t& path, const basic_json& value)
|
||||||
|
{
|
||||||
|
result.push_back(
|
||||||
|
{
|
||||||
|
{"op", "replace"}, {"path", path}, {"value", value}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// @brief append a "remove" operation for @a path to @a result
|
||||||
|
static void diff_remove(basic_json& result, const string_t& path)
|
||||||
|
{
|
||||||
|
result.push_back(object(
|
||||||
|
{
|
||||||
|
{"op", "remove"}, {"path", path}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// @brief append an "add" operation for @a path with @a value to @a result
|
||||||
|
static void diff_add(basic_json& result, const string_t& path, const basic_json& value)
|
||||||
|
{
|
||||||
|
result.push_back(
|
||||||
|
{
|
||||||
|
{"op", "add"}, {"path", path}, {"value", value}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// @brief append the "remove" operations for the elements of array
|
||||||
|
/// @a source from @a index on, and the "add" operations for the
|
||||||
|
/// elements of array @a target from source's size on, to @a result
|
||||||
|
static void diff_array_tails(basic_json& result, const basic_json& source, const basic_json& target,
|
||||||
|
const string_t& path, const std::size_t index)
|
||||||
|
{
|
||||||
|
// remove my remaining elements, highest index first; appending
|
||||||
|
// in that order avoids the quadratic reinsertion done before
|
||||||
|
for (std::size_t j = source.size(); j > index; --j)
|
||||||
|
{
|
||||||
|
diff_remove(result, detail::concat<string_t>(path, '/', detail::to_string<string_t>(j - 1)));
|
||||||
|
}
|
||||||
|
|
||||||
|
// add other remaining elements
|
||||||
|
for (std::size_t i = source.size(); i < target.size(); ++i)
|
||||||
|
{
|
||||||
|
diff_add(result, detail::concat<string_t>(path, "/-"), target[i]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief compare the keys of objects @a source and @a target
|
||||||
|
|
||||||
|
If object_t does not keep its members in insertion order, or if the keys
|
||||||
|
both objects have are in the same order in both, and the keys only
|
||||||
|
@a target has come after them, stores the keys common to both in
|
||||||
|
source's order in @a common_keys, stores the "add" operations for the keys
|
||||||
|
only @a target has in @a added_ops, and returns true: the caller then diffs
|
||||||
|
the objects member by member. Otherwise, appends operations that remove
|
||||||
|
every member of @a source and add every member of @a target to @a result,
|
||||||
|
and returns false.
|
||||||
|
*/
|
||||||
|
static bool diff_object_keys(basic_json& result, const basic_json& source, const basic_json& target,
|
||||||
|
const string_t& path, std::vector<typename object_t::key_type>& common_keys,
|
||||||
|
basic_json& added_ops)
|
||||||
|
{
|
||||||
|
// first pass: record, for every source key, whether it is
|
||||||
|
// common to both objects (in source's iteration order) or
|
||||||
|
// was deleted (i.e., in source but not in target) -- this is
|
||||||
|
// a by-product of the target.find() call already needed to
|
||||||
|
// tell the two cases apart, so it adds no extra lookups. The
|
||||||
|
// "remove" ops themselves are emitted later, interleaved
|
||||||
|
// with the per-key diffs in the caller's fast path, to match
|
||||||
|
// source's original iteration order (as the original,
|
||||||
|
// pre-reordering-aware implementation did) instead of
|
||||||
|
// grouping all removes before all per-key diffs.
|
||||||
|
std::vector<typename object_t::key_type> common_keys_source_order;
|
||||||
|
for (auto it = source.cbegin(); it != source.cend(); ++it)
|
||||||
|
{
|
||||||
|
if (target.find(it.key()) != target.end())
|
||||||
|
{
|
||||||
|
common_keys_source_order.push_back(it.key());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// second pass: find keys that were added (i.e., in target but
|
||||||
|
// not in source), and record the keys common to both, in
|
||||||
|
// target's iteration order -- again a by-product of the
|
||||||
|
// source.find() call already needed to detect added keys. At
|
||||||
|
// the same time, determine whether every added key comes
|
||||||
|
// after every common key in target's order (a precondition
|
||||||
|
// for the fast path, which only ever appends new keys
|
||||||
|
// at the very end). Both are only needed for an object_t that
|
||||||
|
// keeps its members in insertion order, such as the one
|
||||||
|
// backing `ordered_json`; for any other object_t, the fast
|
||||||
|
// path is always taken and they are not computed.
|
||||||
|
// The patch ops for keys that were added (i.e., in target but not
|
||||||
|
// in source) are built here so the fast path can reuse
|
||||||
|
// them without a second source.find() per target key. Only
|
||||||
|
// used by the fast path -- the slow (reordering) path
|
||||||
|
// rebuilds "add" ops for every key itself.
|
||||||
|
std::vector<typename object_t::key_type> common_keys_target_order;
|
||||||
|
bool new_keys_form_suffix = true;
|
||||||
|
bool seen_new_key = false;
|
||||||
|
for (auto it = target.cbegin(); it != target.cend(); ++it)
|
||||||
|
{
|
||||||
|
if (source.find(it.key()) == source.end())
|
||||||
|
{
|
||||||
|
seen_new_key = true;
|
||||||
|
diff_add(added_ops, detail::concat<string_t>(path, '/', detail::escape(it.key())), it.value());
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
#ifdef JSON_HEDLEY_MSVC_VERSION
|
||||||
|
#pragma warning(push )
|
||||||
|
#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr
|
||||||
|
#endif
|
||||||
|
if (detail::is_ordered_map<object_t>::value)
|
||||||
|
{
|
||||||
|
common_keys_target_order.push_back(it.key());
|
||||||
|
if (seen_new_key)
|
||||||
|
{
|
||||||
|
new_keys_form_suffix = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#ifdef JSON_HEDLEY_MSVC_VERSION
|
||||||
|
#pragma warning( pop )
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Only an object type that keeps its members in insertion
|
||||||
|
// order, such as nlohmann::ordered_map, can need reordering:
|
||||||
|
// patch() appends a new member at the end of such an object.
|
||||||
|
// Any other object type places its members itself - std::map
|
||||||
|
// in key order, a hash map in an order its operator== ignores -
|
||||||
|
// so a member-by-member diff always reproduces target there.
|
||||||
|
if (!detail::is_ordered_map<object_t>::value
|
||||||
|
|| (common_keys_source_order == common_keys_target_order && new_keys_form_suffix))
|
||||||
|
{
|
||||||
|
// fast path: order of common keys already matches (or the
|
||||||
|
// object_t's iteration order does not depend on
|
||||||
|
// insertion history), so a plain per-key diff is correct
|
||||||
|
// and minimal, as before
|
||||||
|
common_keys = std::move(common_keys_source_order);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// slow path: the common keys are in a different relative
|
||||||
|
// order in source and target (only possible for a
|
||||||
|
// reorderable object_t like ordered_map). Building a
|
||||||
|
// minimal reordering patch is a nontrivial (LCS-like)
|
||||||
|
// problem; instead, remove every source key -- both
|
||||||
|
// deleted keys (which must be removed regardless) and
|
||||||
|
// common keys (removed so they can be re-added in
|
||||||
|
// target's order) -- and re-add every key that should
|
||||||
|
// remain, with its final target value, in target's
|
||||||
|
// order. basic_json::patch()'s "add" operation on an
|
||||||
|
// object uses operator[], which appends at the end for a
|
||||||
|
// vector-backed insertion-ordered map when the key does
|
||||||
|
// not already exist -- so removing a key and then adding
|
||||||
|
// it moves it to the end, fixing its position.
|
||||||
|
for (auto it = source.cbegin(); it != source.cend(); ++it)
|
||||||
|
{
|
||||||
|
diff_remove(result, detail::concat<string_t>(path, '/', detail::escape(it.key())));
|
||||||
|
}
|
||||||
|
|
||||||
|
// add every key that is either common (just removed
|
||||||
|
// above) or brand new, in target's iteration order, so
|
||||||
|
// that the final order after applying the patch matches
|
||||||
|
// target exactly
|
||||||
|
for (auto it = target.cbegin(); it != target.cend(); ++it)
|
||||||
|
{
|
||||||
|
diff_add(result, detail::concat<string_t>(path, '/', detail::escape(it.key())), it.value());
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief @ref diff, for values at nesting level @a depth, appending the
|
||||||
|
operations to @a result
|
||||||
|
|
||||||
|
Diffing two arrays or objects calls this function again, once per nesting
|
||||||
|
level, so values nested deeply enough used to exhaust the call stack and
|
||||||
|
terminate the process. The descent is bounded here: once @ref
|
||||||
|
detail::recursion_depth_limit levels have been entered, @ref
|
||||||
|
diff_iteratively diffs what is left without the call stack.
|
||||||
|
*/
|
||||||
|
static void diff_recursively(basic_json& result, const basic_json& source, const basic_json& target,
|
||||||
|
const string_t& path, const std::size_t depth)
|
||||||
|
{
|
||||||
|
// if the values are the same, there is nothing to do
|
||||||
if (source == target)
|
if (source == target)
|
||||||
{
|
{
|
||||||
return result;
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit()))
|
||||||
|
{
|
||||||
|
diff_iteratively(result, source, target, path);
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (source.type() != target.type())
|
if (source.type() != target.type())
|
||||||
{
|
{
|
||||||
// different types: replace value
|
// different types: replace value
|
||||||
result.push_back(
|
diff_replace(result, path, target);
|
||||||
{
|
return;
|
||||||
{"op", "replace"}, {"path", path}, {"value", target}
|
|
||||||
});
|
|
||||||
return result;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
switch (source.type())
|
switch (source.type())
|
||||||
@@ -6362,200 +6600,50 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
while (i < source.size() && i < target.size())
|
while (i < source.size() && i < target.size())
|
||||||
{
|
{
|
||||||
// recursive call to compare array values at index i
|
// recursive call to compare array values at index i
|
||||||
auto temp_diff = diff(source[i], target[i], detail::concat<string_t>(path, '/', detail::to_string<string_t>(i)));
|
diff_recursively(result, source[i], target[i], detail::concat<string_t>(path, '/', detail::to_string<string_t>(i)), depth + 1);
|
||||||
result.insert(result.end(), temp_diff.begin(), temp_diff.end());
|
|
||||||
++i;
|
++i;
|
||||||
}
|
}
|
||||||
|
|
||||||
// We now reached the end of at least one array
|
// We now reached the end of at least one array
|
||||||
// in a second pass, traverse the remaining elements
|
// in a second pass, traverse the remaining elements
|
||||||
|
diff_array_tails(result, source, target, path, i);
|
||||||
// remove my remaining elements, highest index first; appending
|
|
||||||
// in that order avoids the quadratic reinsertion done before
|
|
||||||
for (std::size_t j = source.size(); j > i; --j)
|
|
||||||
{
|
|
||||||
result.push_back(object(
|
|
||||||
{
|
|
||||||
{"op", "remove"},
|
|
||||||
{"path", detail::concat<string_t>(path, '/', detail::to_string<string_t>(j - 1))}
|
|
||||||
}));
|
|
||||||
}
|
|
||||||
i = source.size();
|
|
||||||
|
|
||||||
// add other remaining elements
|
|
||||||
while (i < target.size())
|
|
||||||
{
|
|
||||||
result.push_back(
|
|
||||||
{
|
|
||||||
{"op", "add"},
|
|
||||||
{"path", detail::concat<string_t>(path, "/-")},
|
|
||||||
{"value", target[i]}
|
|
||||||
});
|
|
||||||
++i;
|
|
||||||
}
|
|
||||||
|
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
case value_t::object:
|
case value_t::object:
|
||||||
{
|
{
|
||||||
// first pass: record, for every source key, whether it is
|
std::vector<typename object_t::key_type> common_keys;
|
||||||
// common to both objects (in source's iteration order) or
|
|
||||||
// was deleted (i.e., in source but not in target) -- this is
|
|
||||||
// a by-product of the target.find() call already needed to
|
|
||||||
// tell the two cases apart, so it adds no extra lookups. The
|
|
||||||
// "remove" ops themselves are emitted later, interleaved
|
|
||||||
// with the recursive per-key diffs in the fast path below,
|
|
||||||
// to match source's original iteration order (as the
|
|
||||||
// original, pre-reordering-aware implementation did) instead
|
|
||||||
// of grouping all removes before all recursive diffs.
|
|
||||||
std::vector<typename object_t::key_type> common_keys_source_order;
|
|
||||||
for (auto it = source.cbegin(); it != source.cend(); ++it)
|
|
||||||
{
|
|
||||||
if (target.find(it.key()) != target.end())
|
|
||||||
{
|
|
||||||
common_keys_source_order.push_back(it.key());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// second pass: find keys that were added (i.e., in target but
|
|
||||||
// not in source), and record the keys common to both, in
|
|
||||||
// target's iteration order -- again a by-product of the
|
|
||||||
// source.find() call already needed to detect added keys. At
|
|
||||||
// the same time, determine whether every added key comes
|
|
||||||
// after every common key in target's order (a precondition
|
|
||||||
// for the fast path below, which only ever appends new keys
|
|
||||||
// at the very end). Both are only needed for an object_t that
|
|
||||||
// keeps its members in insertion order, such as the one
|
|
||||||
// backing `ordered_json`; for any other object_t, the fast
|
|
||||||
// path is always taken and they are not computed.
|
|
||||||
// patch ops for keys that were added (i.e., in target but not
|
|
||||||
// in source); built here so the fast path below can reuse
|
|
||||||
// them without a second source.find() per target key. Only
|
|
||||||
// used by the fast path -- the slow (reordering) path
|
|
||||||
// rebuilds "add" ops for every key itself.
|
|
||||||
std::vector<typename object_t::key_type> common_keys_target_order;
|
|
||||||
basic_json added_ops(value_t::array);
|
basic_json added_ops(value_t::array);
|
||||||
bool new_keys_form_suffix = true;
|
if (diff_object_keys(result, source, target, path, common_keys, added_ops))
|
||||||
bool seen_new_key = false;
|
|
||||||
for (auto it = target.cbegin(); it != target.cend(); ++it)
|
|
||||||
{
|
{
|
||||||
if (source.find(it.key()) == source.end())
|
// fast path: common_keys is, by construction, the
|
||||||
{
|
// subsequence of source's keys that are common to both
|
||||||
seen_new_key = true;
|
// objects, in source's iteration order -- so it can be
|
||||||
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
|
// walked in lockstep with `source` using a cheap key
|
||||||
added_ops.push_back(
|
// comparison instead of another lookup. Deleted keys
|
||||||
{
|
// (those source keys not in common_keys) are interleaved
|
||||||
{"op", "add"}, {"path", path_key},
|
// here too, in source's original order, to match the
|
||||||
{"value", it.value()}
|
// historical (pre-reordering-aware) output order.
|
||||||
});
|
auto common_it = common_keys.cbegin();
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
#ifdef JSON_HEDLEY_MSVC_VERSION
|
|
||||||
#pragma warning(push )
|
|
||||||
#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr
|
|
||||||
#endif
|
|
||||||
if (detail::is_ordered_map<object_t>::value)
|
|
||||||
{
|
|
||||||
common_keys_target_order.push_back(it.key());
|
|
||||||
if (seen_new_key)
|
|
||||||
{
|
|
||||||
new_keys_form_suffix = false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
#ifdef JSON_HEDLEY_MSVC_VERSION
|
|
||||||
#pragma warning( pop )
|
|
||||||
#endif
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Only an object type that keeps its members in insertion
|
|
||||||
// order, such as nlohmann::ordered_map, can need reordering:
|
|
||||||
// patch() appends a new member at the end of such an object.
|
|
||||||
// Any other object type places its members itself - std::map
|
|
||||||
// in key order, a hash map in an order its operator== ignores -
|
|
||||||
// so a member-by-member diff always reproduces target there.
|
|
||||||
if (!detail::is_ordered_map<object_t>::value
|
|
||||||
|| (common_keys_source_order == common_keys_target_order && new_keys_form_suffix))
|
|
||||||
{
|
|
||||||
// fast path: order of common keys already matches (or the
|
|
||||||
// object_t's iteration order does not depend on
|
|
||||||
// insertion history), so a plain per-key recursive diff
|
|
||||||
// is correct and minimal, as before. common_keys_source_order
|
|
||||||
// is, by construction, the subsequence of source's keys
|
|
||||||
// that are common to both objects, in source's iteration
|
|
||||||
// order -- so it can be walked in lockstep with `source`
|
|
||||||
// using a cheap key comparison instead of another lookup.
|
|
||||||
// Deleted keys (those source keys not in common_keys_source_order)
|
|
||||||
// are interleaved here too, in source's original order, to
|
|
||||||
// match the historical (pre-reordering-aware) output order.
|
|
||||||
auto common_it = common_keys_source_order.cbegin();
|
|
||||||
for (auto it = source.cbegin(); it != source.cend(); ++it)
|
for (auto it = source.cbegin(); it != source.cend(); ++it)
|
||||||
{
|
{
|
||||||
if (common_it != common_keys_source_order.cend() && it.key() == *common_it)
|
if (common_it != common_keys.cend() && it.key() == *common_it)
|
||||||
{
|
{
|
||||||
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
|
diff_recursively(result, it.value(), target[it.key()], detail::concat<string_t>(path, '/', detail::escape(it.key())), depth + 1);
|
||||||
auto temp_diff = diff(it.value(), target[it.key()], path_key);
|
|
||||||
result.insert(result.end(), temp_diff.begin(), temp_diff.end());
|
|
||||||
++common_it;
|
++common_it;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
// found a key that is not in target -> remove it
|
// found a key that is not in target -> remove it
|
||||||
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
|
diff_remove(result, detail::concat<string_t>(path, '/', detail::escape(it.key())));
|
||||||
result.push_back(object(
|
|
||||||
{
|
|
||||||
{"op", "remove"}, {"path", path_key}
|
|
||||||
}));
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// append the "add" ops for brand-new keys collected above
|
// append the "add" ops for brand-new keys collected by
|
||||||
// during the pass over target -- no second source.find()
|
// diff_object_keys -- no second source.find() per target
|
||||||
// per target key needed
|
// key needed
|
||||||
result.insert(result.end(), added_ops.begin(), added_ops.end());
|
result.insert(result.end(), added_ops.begin(), added_ops.end());
|
||||||
}
|
}
|
||||||
else
|
|
||||||
{
|
|
||||||
// slow path: the common keys are in a different relative
|
|
||||||
// order in source and target (only possible for a
|
|
||||||
// reorderable object_t like ordered_map). Building a
|
|
||||||
// minimal reordering patch is a nontrivial (LCS-like)
|
|
||||||
// problem; instead, remove every source key -- both
|
|
||||||
// deleted keys (which must be removed regardless) and
|
|
||||||
// common keys (removed so they can be re-added in
|
|
||||||
// target's order) -- and re-add every key that should
|
|
||||||
// remain, with its final target value, in target's
|
|
||||||
// order. basic_json::patch()'s "add" operation on an
|
|
||||||
// object uses operator[], which appends at the end for a
|
|
||||||
// vector-backed insertion-ordered map when the key does
|
|
||||||
// not already exist -- so removing a key and then adding
|
|
||||||
// it moves it to the end, fixing its position.
|
|
||||||
for (auto it = source.cbegin(); it != source.cend(); ++it)
|
|
||||||
{
|
|
||||||
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
|
|
||||||
result.push_back(object(
|
|
||||||
{
|
|
||||||
{"op", "remove"}, {"path", path_key}
|
|
||||||
}));
|
|
||||||
}
|
|
||||||
|
|
||||||
// add every key that is either common (just removed
|
|
||||||
// above) or brand new, in target's iteration order, so
|
|
||||||
// that the final order after applying the patch matches
|
|
||||||
// target exactly
|
|
||||||
for (auto it = target.cbegin(); it != target.cend(); ++it)
|
|
||||||
{
|
|
||||||
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
|
|
||||||
result.push_back(
|
|
||||||
{
|
|
||||||
{"op", "add"}, {"path", path_key},
|
|
||||||
{"value", it.value()}
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -6570,16 +6658,170 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
default:
|
default:
|
||||||
{
|
{
|
||||||
// both primitive types: replace value
|
// both primitive types: replace value
|
||||||
result.push_back(
|
diff_replace(result, path, target);
|
||||||
{
|
|
||||||
{"op", "replace"}, {"path", path}, {"value", target}
|
|
||||||
});
|
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return result;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief @ref diff without the call stack, appending the operations to
|
||||||
|
@a result
|
||||||
|
|
||||||
|
Produces the same operations as @ref diff_recursively. Only reached for
|
||||||
|
values nested more deeply than @ref detail::recursion_depth_limit.
|
||||||
|
*/
|
||||||
|
static void diff_iteratively(basic_json& result, const basic_json& source, const basic_json& target,
|
||||||
|
const string_t& path)
|
||||||
|
{
|
||||||
|
// The arrays and objects being diffed are kept on an explicit stack,
|
||||||
|
// and every pair of elements is still diffed completely before the
|
||||||
|
// next one, so the operations come out in the same order as in
|
||||||
|
// diff_recursively. The path of the values being diffed is kept in
|
||||||
|
// one buffer that grows and shrinks with the stack, rather than in a
|
||||||
|
// new string per level.
|
||||||
|
std::vector<diff_frame> stack;
|
||||||
|
string_t current_path = path;
|
||||||
|
|
||||||
|
// diff `s` against `t`, whose path is current_path: primitives,
|
||||||
|
// values of different types, and objects whose members were reordered
|
||||||
|
// are handled right away; arrays and other objects get a frame
|
||||||
|
const auto enter = [&result, &stack, ¤t_path](const basic_json & s, const basic_json & t)
|
||||||
|
{
|
||||||
|
// if the values are the same, there is nothing to do. Arrays and
|
||||||
|
// objects are not compared up front: comparing them visits
|
||||||
|
// everything below them, so doing that at every level would take
|
||||||
|
// quadratic time in the nesting depth - equal ones yield no
|
||||||
|
// operations anyway.
|
||||||
|
if ((!s.is_structured() || !t.is_structured()) && s == t)
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (s.type() != t.type())
|
||||||
|
{
|
||||||
|
// different types: replace value
|
||||||
|
diff_replace(result, current_path, t);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
switch (s.type())
|
||||||
|
{
|
||||||
|
case value_t::array:
|
||||||
|
{
|
||||||
|
stack.emplace_back(&s, &t, current_path.size());
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
case value_t::object:
|
||||||
|
{
|
||||||
|
std::vector<typename object_t::key_type> common_keys;
|
||||||
|
basic_json added_ops(value_t::array);
|
||||||
|
if (diff_object_keys(result, s, t, current_path, common_keys, added_ops))
|
||||||
|
{
|
||||||
|
// fast path: the frame walks source in lockstep with
|
||||||
|
// common_keys, as diff_recursively does, and appends
|
||||||
|
// added_ops once all members are done
|
||||||
|
stack.emplace_back(&s, &t, current_path.size());
|
||||||
|
stack.back().member = s.cbegin();
|
||||||
|
stack.back().common_keys = std::move(common_keys);
|
||||||
|
stack.back().added_ops = std::move(added_ops);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
case value_t::null:
|
||||||
|
case value_t::string:
|
||||||
|
case value_t::boolean:
|
||||||
|
case value_t::number_integer:
|
||||||
|
case value_t::number_unsigned:
|
||||||
|
case value_t::number_float:
|
||||||
|
case value_t::binary:
|
||||||
|
case value_t::discarded:
|
||||||
|
default:
|
||||||
|
{
|
||||||
|
// both primitive types: replace value
|
||||||
|
diff_replace(result, current_path, t);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
enter(source, target);
|
||||||
|
while (!stack.empty())
|
||||||
|
{
|
||||||
|
// the frame is copied out member by member and changed through
|
||||||
|
// stack.back(): enter() may push a frame and the end of the loop
|
||||||
|
// pops it, either of which would invalidate a reference to it
|
||||||
|
const basic_json* const s = stack.back().source;
|
||||||
|
const basic_json* const t = stack.back().target;
|
||||||
|
const std::size_t path_length = stack.back().path_length;
|
||||||
|
const std::size_t depth = stack.size();
|
||||||
|
|
||||||
|
if (s->is_array())
|
||||||
|
{
|
||||||
|
const auto& source_array = *s->m_data.m_value.array;
|
||||||
|
const auto& target_array = *t->m_data.m_value.array;
|
||||||
|
|
||||||
|
// first pass: traverse common elements
|
||||||
|
const std::size_t i = stack.back().index;
|
||||||
|
if (i < source_array.size() && i < target_array.size())
|
||||||
|
{
|
||||||
|
++stack.back().index;
|
||||||
|
detail::concat_into(current_path, '/', detail::to_string<string_t>(i));
|
||||||
|
enter(source_array[i], target_array[i]);
|
||||||
|
if (stack.size() == depth)
|
||||||
|
{
|
||||||
|
current_path.resize(path_length);
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// We now reached the end of at least one array
|
||||||
|
// in a second pass, traverse the remaining elements
|
||||||
|
diff_array_tails(result, *s, *t, current_path, i);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
const const_iterator it = stack.back().member;
|
||||||
|
if (it != s->cend())
|
||||||
|
{
|
||||||
|
++stack.back().member;
|
||||||
|
const std::size_t next_common = stack.back().next_common;
|
||||||
|
if (next_common < stack.back().common_keys.size() && it.key() == stack.back().common_keys[next_common])
|
||||||
|
{
|
||||||
|
++stack.back().next_common;
|
||||||
|
const basic_json& target_value = (*t)[it.key()];
|
||||||
|
detail::concat_into(current_path, '/', detail::escape(it.key()));
|
||||||
|
enter(it.value(), target_value);
|
||||||
|
if (stack.size() == depth)
|
||||||
|
{
|
||||||
|
current_path.resize(path_length);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// found a key that is not in target -> remove it
|
||||||
|
diff_remove(result, detail::concat<string_t>(current_path, '/', detail::escape(it.key())));
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// append the "add" ops for brand-new keys collected when the
|
||||||
|
// object was entered
|
||||||
|
result.insert(result.end(), stack.back().added_ops.begin(), stack.back().added_ops.end());
|
||||||
|
}
|
||||||
|
|
||||||
|
// this array or object is done: continue with the one it is in
|
||||||
|
stack.pop_back();
|
||||||
|
if (!stack.empty())
|
||||||
|
{
|
||||||
|
current_path.resize(stack.back().path_length);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
public:
|
||||||
/// @}
|
/// @}
|
||||||
|
|
||||||
////////////////////////////////
|
////////////////////////////////
|
||||||
|
|||||||
@@ -47,6 +47,7 @@ inline namespace json_literals
|
|||||||
namespace detail
|
namespace detail
|
||||||
{
|
{
|
||||||
using NLOHMANN_JSON_NAMESPACE::detail::json_sax_dom_callback_parser;
|
using NLOHMANN_JSON_NAMESPACE::detail::json_sax_dom_callback_parser;
|
||||||
|
using NLOHMANN_JSON_NAMESPACE::detail::json_sax_dom_parser;
|
||||||
using NLOHMANN_JSON_NAMESPACE::detail::unknown_size;
|
using NLOHMANN_JSON_NAMESPACE::detail::unknown_size;
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
|
|
||||||
|
|||||||
@@ -10,7 +10,6 @@ set(JSON_32bitTest AUTO CACHE STRING "Enable the 32bit unit test (ON/OFF/AUT
|
|||||||
set(JSON_TestStandards "" CACHE STRING "The list of standards to test explicitly.")
|
set(JSON_TestStandards "" CACHE STRING "The list of standards to test explicitly.")
|
||||||
|
|
||||||
# using an env var, since this will also affect targets executing cmake (such as "ci_test_compiler_default")
|
# using an env var, since this will also affect targets executing cmake (such as "ci_test_compiler_default")
|
||||||
set(JSON_FORCED_GLOBAL_COMPILE_OPTIONS $ENV{JSON_FORCED_GLOBAL_COMPILE_OPTIONS})
|
|
||||||
if (NOT "" STREQUAL "$ENV{JSON_FORCED_GLOBAL_COMPILE_OPTIONS}")
|
if (NOT "" STREQUAL "$ENV{JSON_FORCED_GLOBAL_COMPILE_OPTIONS}")
|
||||||
add_compile_options($ENV{JSON_FORCED_GLOBAL_COMPILE_OPTIONS})
|
add_compile_options($ENV{JSON_FORCED_GLOBAL_COMPILE_OPTIONS})
|
||||||
endif()
|
endif()
|
||||||
@@ -95,10 +94,15 @@ target_compile_options(test_main PUBLIC
|
|||||||
$<$<CXX_COMPILER_ID:Intel>:-diag-disable=1786>)
|
$<$<CXX_COMPILER_ID:Intel>:-diag-disable=1786>)
|
||||||
target_include_directories(test_main PUBLIC
|
target_include_directories(test_main PUBLIC
|
||||||
thirdparty/doctest
|
thirdparty/doctest
|
||||||
thirdparty/fifo_map
|
|
||||||
${PROJECT_BINARY_DIR}/include)
|
${PROJECT_BINARY_DIR}/include)
|
||||||
target_link_libraries(test_main PUBLIC ${NLOHMANN_JSON_TARGET_NAME})
|
target_link_libraries(test_main PUBLIC ${NLOHMANN_JSON_TARGET_NAME})
|
||||||
|
|
||||||
|
# thirdparty/fifo_map is only used by the #972 regression test, so only
|
||||||
|
# test-regression1 needs it on its include path (see json_test_set_test_options
|
||||||
|
# below), rather than every test-* target via test_main.
|
||||||
|
add_library(fifo_map_include INTERFACE)
|
||||||
|
target_include_directories(fifo_map_include INTERFACE thirdparty/fifo_map)
|
||||||
|
|
||||||
#############################################################################
|
#############################################################################
|
||||||
# define test- and standard-specific build settings
|
# define test- and standard-specific build settings
|
||||||
#############################################################################
|
#############################################################################
|
||||||
@@ -132,6 +136,9 @@ json_test_set_test_options(test-disabled_exceptions
|
|||||||
# raise timeout of expensive Unicode test
|
# raise timeout of expensive Unicode test
|
||||||
json_test_set_test_options(test-unicode4 TEST_PROPERTIES TIMEOUT 3000)
|
json_test_set_test_options(test-unicode4 TEST_PROPERTIES TIMEOUT 3000)
|
||||||
|
|
||||||
|
# only the #972 regression test needs thirdparty/fifo_map on its include path
|
||||||
|
json_test_set_test_options(test-regression1 LINK_LIBRARIES fifo_map_include)
|
||||||
|
|
||||||
#############################################################################
|
#############################################################################
|
||||||
# add unit tests
|
# add unit tests
|
||||||
#############################################################################
|
#############################################################################
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
add_test(NAME cmake_add_subdirectory_configure
|
add_test(NAME cmake_add_subdirectory_configure
|
||||||
COMMAND ${CMAKE_COMMAND}
|
COMMAND ${CMAKE_COMMAND}
|
||||||
-G "${CMAKE_GENERATOR}"
|
-G "${CMAKE_GENERATOR}"
|
||||||
|
-A "${CMAKE_GENERATOR_PLATFORM}"
|
||||||
|
-T "${CMAKE_GENERATOR_TOOLSET}"
|
||||||
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
||||||
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
||||||
-Dnlohmann_json_source=${PROJECT_SOURCE_DIR}
|
-Dnlohmann_json_source=${PROJECT_SOURCE_DIR}
|
||||||
|
|||||||
@@ -1,21 +1,20 @@
|
|||||||
if (${CMAKE_VERSION} VERSION_GREATER "3.11.0")
|
add_test(NAME cmake_fetch_content_configure
|
||||||
add_test(NAME cmake_fetch_content_configure
|
COMMAND ${CMAKE_COMMAND}
|
||||||
COMMAND ${CMAKE_COMMAND}
|
|
||||||
-G "${CMAKE_GENERATOR}"
|
-G "${CMAKE_GENERATOR}"
|
||||||
|
-A "${CMAKE_GENERATOR_PLATFORM}"
|
||||||
|
-T "${CMAKE_GENERATOR_TOOLSET}"
|
||||||
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
||||||
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
||||||
-Dnlohmann_json_source=${PROJECT_SOURCE_DIR}
|
|
||||||
${CMAKE_CURRENT_SOURCE_DIR}/project
|
${CMAKE_CURRENT_SOURCE_DIR}/project
|
||||||
)
|
)
|
||||||
add_test(NAME cmake_fetch_content_build
|
add_test(NAME cmake_fetch_content_build
|
||||||
COMMAND ${CMAKE_COMMAND} --build .
|
COMMAND ${CMAKE_COMMAND} --build .
|
||||||
)
|
)
|
||||||
set_tests_properties(cmake_fetch_content_configure PROPERTIES
|
set_tests_properties(cmake_fetch_content_configure PROPERTIES
|
||||||
FIXTURES_SETUP cmake_fetch_content
|
FIXTURES_SETUP cmake_fetch_content
|
||||||
LABELS "git_required;not_reproducible"
|
LABELS "git_required;not_reproducible"
|
||||||
)
|
)
|
||||||
set_tests_properties(cmake_fetch_content_build PROPERTIES
|
set_tests_properties(cmake_fetch_content_build PROPERTIES
|
||||||
FIXTURES_REQUIRED cmake_fetch_content
|
FIXTURES_REQUIRED cmake_fetch_content
|
||||||
LABELS "git_required;not_reproducible"
|
LABELS "git_required;not_reproducible"
|
||||||
)
|
)
|
||||||
endif()
|
|
||||||
|
|||||||
@@ -4,6 +4,14 @@ project(DummyImport CXX)
|
|||||||
|
|
||||||
include(FetchContent)
|
include(FetchContent)
|
||||||
|
|
||||||
|
# This test deliberately covers the pre-3.14 FetchContent_Populate pattern
|
||||||
|
# (see FetchContent_MakeAvailable in ../../cmake_fetch_content2/project for
|
||||||
|
# the 3.14+ alternative), which CMake 3.30 deprecated as CMP0169. Silence the
|
||||||
|
# deprecation warning on purpose instead of switching to the new pattern.
|
||||||
|
if(POLICY CMP0169)
|
||||||
|
cmake_policy(SET CMP0169 OLD)
|
||||||
|
endif()
|
||||||
|
|
||||||
get_filename_component(GIT_REPOSITORY_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/../../.. ABSOLUTE)
|
get_filename_component(GIT_REPOSITORY_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/../../.. ABSOLUTE)
|
||||||
FetchContent_Declare(json GIT_REPOSITORY ${GIT_REPOSITORY_DIRECTORY} GIT_TAG HEAD)
|
FetchContent_Declare(json GIT_REPOSITORY ${GIT_REPOSITORY_DIRECTORY} GIT_TAG HEAD)
|
||||||
|
|
||||||
|
|||||||
@@ -1,10 +1,11 @@
|
|||||||
if (${CMAKE_VERSION} VERSION_GREATER "3.14.0")
|
if (${CMAKE_VERSION} VERSION_GREATER_EQUAL "3.14")
|
||||||
add_test(NAME cmake_fetch_content2_configure
|
add_test(NAME cmake_fetch_content2_configure
|
||||||
COMMAND ${CMAKE_COMMAND}
|
COMMAND ${CMAKE_COMMAND}
|
||||||
-G "${CMAKE_GENERATOR}"
|
-G "${CMAKE_GENERATOR}"
|
||||||
|
-A "${CMAKE_GENERATOR_PLATFORM}"
|
||||||
|
-T "${CMAKE_GENERATOR_TOOLSET}"
|
||||||
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
||||||
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
||||||
-Dnlohmann_json_source=${PROJECT_SOURCE_DIR}
|
|
||||||
${CMAKE_CURRENT_SOURCE_DIR}/project
|
${CMAKE_CURRENT_SOURCE_DIR}/project
|
||||||
)
|
)
|
||||||
add_test(NAME cmake_fetch_content2_build
|
add_test(NAME cmake_fetch_content2_build
|
||||||
|
|||||||
@@ -2,6 +2,7 @@ add_test(NAME cmake_import_configure
|
|||||||
COMMAND ${CMAKE_COMMAND}
|
COMMAND ${CMAKE_COMMAND}
|
||||||
-G "${CMAKE_GENERATOR}"
|
-G "${CMAKE_GENERATOR}"
|
||||||
-A "${CMAKE_GENERATOR_PLATFORM}"
|
-A "${CMAKE_GENERATOR_PLATFORM}"
|
||||||
|
-T "${CMAKE_GENERATOR_TOOLSET}"
|
||||||
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
||||||
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
||||||
-Dnlohmann_json_DIR=${PROJECT_BINARY_DIR}
|
-Dnlohmann_json_DIR=${PROJECT_BINARY_DIR}
|
||||||
|
|||||||
@@ -2,6 +2,7 @@ add_test(NAME cmake_import_minver_configure
|
|||||||
COMMAND ${CMAKE_COMMAND}
|
COMMAND ${CMAKE_COMMAND}
|
||||||
-G "${CMAKE_GENERATOR}"
|
-G "${CMAKE_GENERATOR}"
|
||||||
-A "${CMAKE_GENERATOR_PLATFORM}"
|
-A "${CMAKE_GENERATOR_PLATFORM}"
|
||||||
|
-T "${CMAKE_GENERATOR_TOOLSET}"
|
||||||
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
||||||
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
||||||
-Dnlohmann_json_DIR=${PROJECT_BINARY_DIR}
|
-Dnlohmann_json_DIR=${PROJECT_BINARY_DIR}
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
add_test(NAME cmake_target_include_directories_configure
|
add_test(NAME cmake_target_include_directories_configure
|
||||||
COMMAND ${CMAKE_COMMAND}
|
COMMAND ${CMAKE_COMMAND}
|
||||||
-G "${CMAKE_GENERATOR}"
|
-G "${CMAKE_GENERATOR}"
|
||||||
|
-A "${CMAKE_GENERATOR_PLATFORM}"
|
||||||
|
-T "${CMAKE_GENERATOR_TOOLSET}"
|
||||||
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
-DCMAKE_CXX_COMPILER=${CMAKE_CXX_COMPILER}
|
||||||
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}
|
||||||
-Dnlohmann_json_source=${PROJECT_SOURCE_DIR}
|
-Dnlohmann_json_source=${PROJECT_SOURCE_DIR}
|
||||||
|
|||||||
@@ -93,9 +93,10 @@ conventions:
|
|||||||
add it as a regression test to the unit test of the affected format (e.g., `tests/src/unit-bjdata.cpp`), with a
|
add it as a regression test to the unit test of the affected format (e.g., `tests/src/unit-bjdata.cpp`), with a
|
||||||
comment naming the OSS-Fuzz issue. This way the input is checked by every CI run rather than only by OSS-Fuzz, and
|
comment naming the OSS-Fuzz issue. This way the input is checked by every CI run rather than only by OSS-Fuzz, and
|
||||||
it stays covered even if OSS-Fuzz later closes the report as not reproducible.
|
it stays covered even if OSS-Fuzz later closes the report as not reproducible.
|
||||||
- **Keep the fuzzer drivers and the unit tests in sync.** The round-trip checks of the UBJSON and BJData drivers are
|
- **Keep the fuzzer drivers and the unit tests in sync.** The round-trip checks of the BJData, BON8, BSON, CBOR,
|
||||||
also run on a fixed corpus in the unit tests (see `tests/src/round_trip_corpus.hpp` and the "round-trip invariants"
|
MessagePack and UBJSON drivers are also run on a fixed corpus in the unit tests (see
|
||||||
test cases), so a regression shows up in CI first. When a driver's checks change, change the unit tests with them.
|
`tests/src/round_trip_corpus.hpp` and the "round-trip invariants" test cases), so a regression shows up in CI
|
||||||
|
first. When a driver's checks change, change the unit tests with them.
|
||||||
- **Record in the report whether the bug shipped.** OSS-Fuzz asks whether a crash was a short-lived regression or
|
- **Record in the report whether the bug shipped.** OSS-Fuzz asks whether a crash was a short-lived regression or
|
||||||
affects a released version; answer it when the fix is merged, as it decides whether the fix needs a release note or
|
affects a released version; answer it when the fix is merged, as it decides whether the fix needs a release note or
|
||||||
a security advisory (see the [security policy](../.github/SECURITY.md)).
|
a security advisory (see the [security policy](../.github/SECURITY.md)).
|
||||||
|
|||||||
|
Before Width: | Height: | Size: 28 KiB |
|
Before Width: | Height: | Size: 26 KiB |
@@ -1,10 +0,0 @@
|
|||||||
<table style="font-family: 'Trebuchet MS', 'Tahoma', 'Arial', 'Helvetica'">
|
|
||||||
<tr><td style="width: 18ex"><b>Banner:</b></td><td>fuzz</td></tr>
|
|
||||||
<tr><td><b>Directory:</b></td><td>fuzz-testing/out</td></tr>
|
|
||||||
<tr><td><b>Generated on:</b></td><td>Mo 29 Aug 2016 22:14:22 CEST</td></tr>
|
|
||||||
</table>
|
|
||||||
<p>
|
|
||||||
<img src="high_freq.png" width=1000 height=300><p>
|
|
||||||
<img src="low_freq.png" width=1000 height=200><p>
|
|
||||||
<img src="exec_speed.png" width=1000 height=200>
|
|
||||||
|
|
||||||
|
Before Width: | Height: | Size: 12 KiB |
@@ -1,31 +0,0 @@
|
|||||||
Results of the latest benchmark from <https://github.com/miloyip/nativejson-benchmark>.
|
|
||||||
|
|
||||||
See <https://github.com/nlohmann/json/issues/307> for discussion.
|
|
||||||
|
|
||||||
Original post at 2016-09-09 to <json@yahoogroups.com>:
|
|
||||||
|
|
||||||
> Hi,
|
|
||||||
>
|
|
||||||
> This benchmark evaluated conformance, parse/stringify speed/memory, and
|
|
||||||
> code size. It can also be viewed as a long list of open source C/C++ JSON
|
|
||||||
> libraries.
|
|
||||||
>
|
|
||||||
> You can run the benchmark on your own machine by checkout this project.
|
|
||||||
>
|
|
||||||
> https://github.com/miloyip/nativejson-benchmark
|
|
||||||
>
|
|
||||||
> You can also view some sample results here:
|
|
||||||
>
|
|
||||||
> https://rawgit.com/miloyip/nativejson-benchmark/master/sample/conformance.html
|
|
||||||
> https://rawgit.com/miloyip/nativejson-benchmark/master/sample/performance_Corei7-4980HQ@2.80GHz_mac64_clang7.0.html
|
|
||||||
>
|
|
||||||
> If you make a new library, you may use this for testing conformance and
|
|
||||||
> performance. Afterwards, please submit a pull request.
|
|
||||||
>
|
|
||||||
> Enjoy!
|
|
||||||
>
|
|
||||||
> --
|
|
||||||
> Milo Yip
|
|
||||||
>
|
|
||||||
> https://github.com/miloyip/
|
|
||||||
> http://twitter.com/miloyip/
|
|
||||||
@@ -1,670 +0,0 @@
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<!DOCTYPE html>
|
|
||||||
<html lang="en" class="">
|
|
||||||
<head prefix="og: http://ogp.me/ns# fb: http://ogp.me/ns/fb# object: http://ogp.me/ns/object# article: http://ogp.me/ns/article# profile: http://ogp.me/ns/profile#">
|
|
||||||
<meta charset='utf-8'>
|
|
||||||
|
|
||||||
|
|
||||||
<link crossorigin="anonymous" href="https://assets-cdn.github.com/assets/frameworks-3a71f36dec04358c4f2f42280fb2cf5c38856f935a3f609eab0a1ae31b1d635a.css" media="all" rel="stylesheet" />
|
|
||||||
<link crossorigin="anonymous" href="https://assets-cdn.github.com/assets/github-5c885e880e980dc7f119393287b84906d930ccaff83f6bff8ae2c086b87ca4d8.css" media="all" rel="stylesheet" />
|
|
||||||
|
|
||||||
|
|
||||||
<link crossorigin="anonymous" href="https://assets-cdn.github.com/assets/site-4ef7bbe907458c89cb1f7f5c5e6c4cd87e03acf66b1817325e644920d2a83330.css" media="all" rel="stylesheet" />
|
|
||||||
|
|
||||||
|
|
||||||
<link as="script" href="https://assets-cdn.github.com/assets/frameworks-88471af1fec40ff9418efbe2ddd15b6896af8d772f8179004c254dffc25ea490.js" rel="preload" />
|
|
||||||
|
|
||||||
<link as="script" href="https://assets-cdn.github.com/assets/github-e18e11a943ff2eb9394c72d4ec8b76592c454915b5839ae177d422777a046e29.js" rel="preload" />
|
|
||||||
|
|
||||||
<meta http-equiv="X-UA-Compatible" content="IE=edge">
|
|
||||||
<meta http-equiv="Content-Language" content="en">
|
|
||||||
<meta name="viewport" content="width=device-width">
|
|
||||||
|
|
||||||
<title>nativejson-benchmark/conformance_Nlohmann (C++11).md at master · miloyip/nativejson-benchmark · GitHub</title>
|
|
||||||
<link rel="search" type="application/opensearchdescription+xml" href="/opensearch.xml" title="GitHub">
|
|
||||||
<link rel="fluid-icon" href="https://github.com/fluidicon.png" title="GitHub">
|
|
||||||
<link rel="apple-touch-icon" href="/apple-touch-icon.png">
|
|
||||||
<link rel="apple-touch-icon" sizes="57x57" href="/apple-touch-icon-57x57.png">
|
|
||||||
<link rel="apple-touch-icon" sizes="60x60" href="/apple-touch-icon-60x60.png">
|
|
||||||
<link rel="apple-touch-icon" sizes="72x72" href="/apple-touch-icon-72x72.png">
|
|
||||||
<link rel="apple-touch-icon" sizes="76x76" href="/apple-touch-icon-76x76.png">
|
|
||||||
<link rel="apple-touch-icon" sizes="114x114" href="/apple-touch-icon-114x114.png">
|
|
||||||
<link rel="apple-touch-icon" sizes="120x120" href="/apple-touch-icon-120x120.png">
|
|
||||||
<link rel="apple-touch-icon" sizes="144x144" href="/apple-touch-icon-144x144.png">
|
|
||||||
<link rel="apple-touch-icon" sizes="152x152" href="/apple-touch-icon-152x152.png">
|
|
||||||
<link rel="apple-touch-icon" sizes="180x180" href="/apple-touch-icon-180x180.png">
|
|
||||||
<meta property="fb:app_id" content="1401488693436528">
|
|
||||||
|
|
||||||
<meta content="https://avatars2.githubusercontent.com/u/1195774?v=3&s=400" name="twitter:image:src" /><meta content="@github" name="twitter:site" /><meta content="summary" name="twitter:card" /><meta content="miloyip/nativejson-benchmark" name="twitter:title" /><meta content="nativejson-benchmark - C/C++ JSON parser/generator benchmark" name="twitter:description" />
|
|
||||||
<meta content="https://avatars2.githubusercontent.com/u/1195774?v=3&s=400" property="og:image" /><meta content="GitHub" property="og:site_name" /><meta content="object" property="og:type" /><meta content="miloyip/nativejson-benchmark" property="og:title" /><meta content="https://github.com/miloyip/nativejson-benchmark" property="og:url" /><meta content="nativejson-benchmark - C/C++ JSON parser/generator benchmark" property="og:description" />
|
|
||||||
<meta name="browser-stats-url" content="https://api.github.com/_private/browser/stats">
|
|
||||||
<meta name="browser-errors-url" content="https://api.github.com/_private/browser/errors">
|
|
||||||
<link rel="assets" href="https://assets-cdn.github.com/">
|
|
||||||
|
|
||||||
<meta name="pjax-timeout" content="1000">
|
|
||||||
|
|
||||||
<meta name="request-id" content="D4563DA7:75ED:525C921:57D6FEBB" data-pjax-transient>
|
|
||||||
|
|
||||||
<meta name="msapplication-TileImage" content="/windows-tile.png">
|
|
||||||
<meta name="msapplication-TileColor" content="#ffffff">
|
|
||||||
<meta name="selected-link" value="repo_source" data-pjax-transient>
|
|
||||||
|
|
||||||
<meta name="google-site-verification" content="KT5gs8h0wvaagLKAVWq8bbeNwnZZK1r1XQysX3xurLU">
|
|
||||||
<meta name="google-site-verification" content="ZzhVyEFwb7w3e0-uOTltm8Jsck2F5StVihD0exw2fsA">
|
|
||||||
<meta name="google-analytics" content="UA-3769691-2">
|
|
||||||
|
|
||||||
<meta content="collector.githubapp.com" name="octolytics-host" /><meta content="github" name="octolytics-app-id" /><meta content="D4563DA7:75ED:525C921:57D6FEBB" name="octolytics-dimension-request_id" />
|
|
||||||
<meta content="/<user-name>/<repo-name>/blob/show" data-pjax-transient="true" name="analytics-location" />
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<meta class="js-ga-set" name="dimension1" content="Logged Out">
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<meta name="hostname" content="github.com">
|
|
||||||
<meta name="user-login" content="">
|
|
||||||
|
|
||||||
<meta name="expected-hostname" content="github.com">
|
|
||||||
<meta name="js-proxy-site-detection-payload" content="N2RkMmY1ZjE1MTA4MzRhYTA5NDNkMjliNDU3OTA3ZTdlMGNmNDZjN2QyODBiZjM3MmYzMjFjNzY1ZjIwNjY4NHx7InJlbW90ZV9hZGRyZXNzIjoiMjEyLjg2LjYxLjE2NyIsInJlcXVlc3RfaWQiOiJENDU2M0RBNzo3NUVEOjUyNUM5MjE6NTdENkZFQkIiLCJ0aW1lc3RhbXAiOjE0NzM3MDc3MDh9">
|
|
||||||
|
|
||||||
|
|
||||||
<link rel="mask-icon" href="https://assets-cdn.github.com/pinned-octocat.svg" color="#4078c0">
|
|
||||||
<link rel="icon" type="image/x-icon" href="https://assets-cdn.github.com/favicon.ico">
|
|
||||||
|
|
||||||
<meta name="html-safe-nonce" content="deea0f406df2fb95709865c5ee80a255d8c47fa9">
|
|
||||||
<meta content="736eea9d74cf34fe850d2180e8a5f0a1cc7bc0be" name="form-nonce" />
|
|
||||||
|
|
||||||
<meta http-equiv="x-pjax-version" content="f302977937abe3fe1c9a4d4bed913565">
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<meta name="description" content="nativejson-benchmark - C/C++ JSON parser/generator benchmark">
|
|
||||||
<meta name="go-import" content="github.com/miloyip/nativejson-benchmark git https://github.com/miloyip/nativejson-benchmark.git">
|
|
||||||
|
|
||||||
<meta content="1195774" name="octolytics-dimension-user_id" /><meta content="miloyip" name="octolytics-dimension-user_login" /><meta content="22976798" name="octolytics-dimension-repository_id" /><meta content="miloyip/nativejson-benchmark" name="octolytics-dimension-repository_nwo" /><meta content="true" name="octolytics-dimension-repository_public" /><meta content="false" name="octolytics-dimension-repository_is_fork" /><meta content="22976798" name="octolytics-dimension-repository_network_root_id" /><meta content="miloyip/nativejson-benchmark" name="octolytics-dimension-repository_network_root_nwo" />
|
|
||||||
<link href="https://github.com/miloyip/nativejson-benchmark/commits/master.atom" rel="alternate" title="Recent Commits to nativejson-benchmark:master" type="application/atom+xml">
|
|
||||||
|
|
||||||
|
|
||||||
<link rel="canonical" href="https://github.com/miloyip/nativejson-benchmark/blob/master/sample/conformance_Nlohmann%20(C%2B%2B11).md" data-pjax-transient>
|
|
||||||
</head>
|
|
||||||
|
|
||||||
|
|
||||||
<body class="logged-out env-production vis-public page-blob">
|
|
||||||
<div id="js-pjax-loader-bar" class="pjax-loader-bar"><div class="progress"></div></div>
|
|
||||||
<a href="#start-of-content" tabindex="1" class="accessibility-aid js-skip-to-content">Skip to content</a>
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<header class="site-header js-details-container" role="banner">
|
|
||||||
<div class="container-responsive">
|
|
||||||
<a class="header-logo-invertocat" href="https://github.com/" aria-label="Homepage" data-ga-click="(Logged out) Header, go to homepage, icon:logo-wordmark">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-mark-github" height="32" version="1.1" viewBox="0 0 16 16" width="32"><path d="M8 0C3.58 0 0 3.58 0 8c0 3.54 2.29 6.53 5.47 7.59.4.07.55-.17.55-.38 0-.19-.01-.82-.01-1.49-2.01.37-2.53-.49-2.69-.94-.09-.23-.48-.94-.82-1.13-.28-.15-.68-.52-.01-.53.63-.01 1.08.58 1.23.82.72 1.21 1.87.87 2.33.66.07-.52.28-.87.51-1.07-1.78-.2-3.64-.89-3.64-3.95 0-.87.31-1.59.82-2.15-.08-.2-.36-1.02.08-2.12 0 0 .67-.21 2.2.82.64-.18 1.32-.27 2-.27.68 0 1.36.09 2 .27 1.53-1.04 2.2-.82 2.2-.82.44 1.1.16 1.92.08 2.12.51.56.82 1.27.82 2.15 0 3.07-1.87 3.75-3.65 3.95.29.25.54.73.54 1.48 0 1.07-.01 1.93-.01 2.2 0 .21.15.46.55.38A8.013 8.013 0 0 0 16 8c0-4.42-3.58-8-8-8z"></path></svg>
|
|
||||||
</a>
|
|
||||||
|
|
||||||
<button class="btn-link float-right site-header-toggle js-details-target" type="button" aria-label="Toggle navigation">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-three-bars" height="24" version="1.1" viewBox="0 0 12 16" width="18"><path d="M11.41 9H.59C0 9 0 8.59 0 8c0-.59 0-1 .59-1H11.4c.59 0 .59.41.59 1 0 .59 0 1-.59 1h.01zm0-4H.59C0 5 0 4.59 0 4c0-.59 0-1 .59-1H11.4c.59 0 .59.41.59 1 0 .59 0 1-.59 1h.01zM.59 11H11.4c.59 0 .59.41.59 1 0 .59 0 1-.59 1H.59C0 13 0 12.59 0 12c0-.59 0-1 .59-1z"></path></svg>
|
|
||||||
</button>
|
|
||||||
|
|
||||||
<div class="site-header-menu">
|
|
||||||
<nav class="site-header-nav site-header-nav-main">
|
|
||||||
<a href="/personal" class="js-selected-navigation-item nav-item nav-item-personal" data-ga-click="Header, click, Nav menu - item:personal" data-selected-links="/personal /personal">
|
|
||||||
Personal
|
|
||||||
</a> <a href="/open-source" class="js-selected-navigation-item nav-item nav-item-opensource" data-ga-click="Header, click, Nav menu - item:opensource" data-selected-links="/open-source /open-source">
|
|
||||||
Open source
|
|
||||||
</a> <a href="/business" class="js-selected-navigation-item nav-item nav-item-business" data-ga-click="Header, click, Nav menu - item:business" data-selected-links="/business /business/partners /business/features /business/customers /business">
|
|
||||||
Business
|
|
||||||
</a> <a href="/explore" class="js-selected-navigation-item nav-item nav-item-explore" data-ga-click="Header, click, Nav menu - item:explore" data-selected-links="/explore /trending /trending/developers /integrations /integrations/feature/code /integrations/feature/collaborate /integrations/feature/ship /explore">
|
|
||||||
Explore
|
|
||||||
</a> </nav>
|
|
||||||
|
|
||||||
<div class="site-header-actions">
|
|
||||||
<a class="btn btn-primary site-header-actions-btn" href="/join?source=header-repo" data-ga-click="(Logged out) Header, clicked Sign up, text:sign-up">Sign up</a>
|
|
||||||
<a class="btn site-header-actions-btn mr-2" href="/login?return_to=%2Fmiloyip%2Fnativejson-benchmark%2Fblob%2Fmaster%2Fsample%2Fconformance_Nlohmann%2520%28C%252B%252B11%29.md" data-ga-click="(Logged out) Header, clicked Sign in, text:sign-in">Sign in</a>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<nav class="site-header-nav site-header-nav-secondary">
|
|
||||||
<a class="nav-item" href="/pricing">Pricing</a>
|
|
||||||
<a class="nav-item" href="/blog">Blog</a>
|
|
||||||
<a class="nav-item" href="https://help.github.com">Support</a>
|
|
||||||
<a class="nav-item header-search-link" href="https://github.com/search">Search GitHub</a>
|
|
||||||
<div class="header-search scoped-search site-scoped-search js-site-search" role="search">
|
|
||||||
<!-- </textarea> --><!-- '"` --><form accept-charset="UTF-8" action="/miloyip/nativejson-benchmark/search" class="js-site-search-form" data-scoped-search-url="/miloyip/nativejson-benchmark/search" data-unscoped-search-url="/search" method="get"><div style="margin:0;padding:0;display:inline"><input name="utf8" type="hidden" value="✓" /></div>
|
|
||||||
<label class="form-control header-search-wrapper js-chromeless-input-container">
|
|
||||||
<div class="header-search-scope">This repository</div>
|
|
||||||
<input type="text"
|
|
||||||
class="form-control header-search-input js-site-search-focus js-site-search-field is-clearable"
|
|
||||||
data-hotkey="s"
|
|
||||||
name="q"
|
|
||||||
placeholder="Search"
|
|
||||||
aria-label="Search this repository"
|
|
||||||
data-unscoped-placeholder="Search GitHub"
|
|
||||||
data-scoped-placeholder="Search"
|
|
||||||
autocapitalize="off">
|
|
||||||
</label>
|
|
||||||
</form></div>
|
|
||||||
|
|
||||||
</nav>
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
</header>
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<div id="start-of-content" class="accessibility-aid"></div>
|
|
||||||
|
|
||||||
<div id="js-flash-container">
|
|
||||||
</div>
|
|
||||||
|
|
||||||
|
|
||||||
<div role="main">
|
|
||||||
<div itemscope itemtype="http://schema.org/SoftwareSourceCode">
|
|
||||||
<div id="js-repo-pjax-container" data-pjax-container>
|
|
||||||
|
|
||||||
<div class="pagehead repohead instapaper_ignore readability-menu experiment-repo-nav">
|
|
||||||
<div class="container repohead-details-container">
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<ul class="pagehead-actions">
|
|
||||||
|
|
||||||
<li>
|
|
||||||
<a href="/login?return_to=%2Fmiloyip%2Fnativejson-benchmark"
|
|
||||||
class="btn btn-sm btn-with-count tooltipped tooltipped-n"
|
|
||||||
aria-label="You must be signed in to watch a repository" rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-eye" height="16" version="1.1" viewBox="0 0 16 16" width="16"><path d="M8.06 2C3 2 0 8 0 8s3 6 8.06 6C13 14 16 8 16 8s-3-6-7.94-6zM8 12c-2.2 0-4-1.78-4-4 0-2.2 1.8-4 4-4 2.22 0 4 1.8 4 4 0 2.22-1.78 4-4 4zm2-4c0 1.11-.89 2-2 2-1.11 0-2-.89-2-2 0-1.11.89-2 2-2 1.11 0 2 .89 2 2z"></path></svg>
|
|
||||||
Watch
|
|
||||||
</a>
|
|
||||||
<a class="social-count" href="/miloyip/nativejson-benchmark/watchers"
|
|
||||||
aria-label="40 users are watching this repository">
|
|
||||||
40
|
|
||||||
</a>
|
|
||||||
|
|
||||||
</li>
|
|
||||||
|
|
||||||
<li>
|
|
||||||
<a href="/login?return_to=%2Fmiloyip%2Fnativejson-benchmark"
|
|
||||||
class="btn btn-sm btn-with-count tooltipped tooltipped-n"
|
|
||||||
aria-label="You must be signed in to star a repository" rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-star" height="16" version="1.1" viewBox="0 0 14 16" width="14"><path d="M14 6l-4.9-.64L7 1 4.9 5.36 0 6l3.6 3.26L2.67 14 7 11.67 11.33 14l-.93-4.74z"></path></svg>
|
|
||||||
Star
|
|
||||||
</a>
|
|
||||||
|
|
||||||
<a class="social-count js-social-count" href="/miloyip/nativejson-benchmark/stargazers"
|
|
||||||
aria-label="352 users starred this repository">
|
|
||||||
352
|
|
||||||
</a>
|
|
||||||
|
|
||||||
</li>
|
|
||||||
|
|
||||||
<li>
|
|
||||||
<a href="/login?return_to=%2Fmiloyip%2Fnativejson-benchmark"
|
|
||||||
class="btn btn-sm btn-with-count tooltipped tooltipped-n"
|
|
||||||
aria-label="You must be signed in to fork a repository" rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-repo-forked" height="16" version="1.1" viewBox="0 0 10 16" width="10"><path d="M8 1a1.993 1.993 0 0 0-1 3.72V6L5 8 3 6V4.72A1.993 1.993 0 0 0 2 1a1.993 1.993 0 0 0-1 3.72V6.5l3 3v1.78A1.993 1.993 0 0 0 5 15a1.993 1.993 0 0 0 1-3.72V9.5l3-3V4.72A1.993 1.993 0 0 0 8 1zM2 4.2C1.34 4.2.8 3.65.8 3c0-.65.55-1.2 1.2-1.2.65 0 1.2.55 1.2 1.2 0 .65-.55 1.2-1.2 1.2zm3 10c-.66 0-1.2-.55-1.2-1.2 0-.65.55-1.2 1.2-1.2.65 0 1.2.55 1.2 1.2 0 .65-.55 1.2-1.2 1.2zm3-10c-.66 0-1.2-.55-1.2-1.2 0-.65.55-1.2 1.2-1.2.65 0 1.2.55 1.2 1.2 0 .65-.55 1.2-1.2 1.2z"></path></svg>
|
|
||||||
Fork
|
|
||||||
</a>
|
|
||||||
|
|
||||||
<a href="/miloyip/nativejson-benchmark/network" class="social-count"
|
|
||||||
aria-label="59 users are forked this repository">
|
|
||||||
59
|
|
||||||
</a>
|
|
||||||
</li>
|
|
||||||
</ul>
|
|
||||||
|
|
||||||
<h1 class="public ">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-repo" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M4 9H3V8h1v1zm0-3H3v1h1V6zm0-2H3v1h1V4zm0-2H3v1h1V2zm8-1v12c0 .55-.45 1-1 1H6v2l-1.5-1.5L3 16v-2H1c-.55 0-1-.45-1-1V1c0-.55.45-1 1-1h10c.55 0 1 .45 1 1zm-1 10H1v2h2v-1h3v1h5v-2zm0-10H2v9h9V1z"></path></svg>
|
|
||||||
<span class="author" itemprop="author"><a href="/miloyip" class="url fn" rel="author">miloyip</a></span><!--
|
|
||||||
--><span class="path-divider">/</span><!--
|
|
||||||
--><strong itemprop="name"><a href="/miloyip/nativejson-benchmark" data-pjax="#js-repo-pjax-container">nativejson-benchmark</a></strong>
|
|
||||||
|
|
||||||
</h1>
|
|
||||||
|
|
||||||
</div>
|
|
||||||
<div class="container">
|
|
||||||
|
|
||||||
<nav class="reponav js-repo-nav js-sidenav-container-pjax"
|
|
||||||
itemscope
|
|
||||||
itemtype="http://schema.org/BreadcrumbList"
|
|
||||||
role="navigation"
|
|
||||||
data-pjax="#js-repo-pjax-container">
|
|
||||||
|
|
||||||
<span itemscope itemtype="http://schema.org/ListItem" itemprop="itemListElement">
|
|
||||||
<a href="/miloyip/nativejson-benchmark" aria-selected="true" class="js-selected-navigation-item selected reponav-item" data-hotkey="g c" data-selected-links="repo_source repo_downloads repo_commits repo_releases repo_tags repo_branches /miloyip/nativejson-benchmark" itemprop="url">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-code" height="16" version="1.1" viewBox="0 0 14 16" width="14"><path d="M9.5 3L8 4.5 11.5 8 8 11.5 9.5 13 14 8 9.5 3zm-5 0L0 8l4.5 5L6 11.5 2.5 8 6 4.5 4.5 3z"></path></svg>
|
|
||||||
<span itemprop="name">Code</span>
|
|
||||||
<meta itemprop="position" content="1">
|
|
||||||
</a> </span>
|
|
||||||
|
|
||||||
<span itemscope itemtype="http://schema.org/ListItem" itemprop="itemListElement">
|
|
||||||
<a href="/miloyip/nativejson-benchmark/issues" class="js-selected-navigation-item reponav-item" data-hotkey="g i" data-selected-links="repo_issues repo_labels repo_milestones /miloyip/nativejson-benchmark/issues" itemprop="url">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-issue-opened" height="16" version="1.1" viewBox="0 0 14 16" width="14"><path d="M7 2.3c3.14 0 5.7 2.56 5.7 5.7s-2.56 5.7-5.7 5.7A5.71 5.71 0 0 1 1.3 8c0-3.14 2.56-5.7 5.7-5.7zM7 1C3.14 1 0 4.14 0 8s3.14 7 7 7 7-3.14 7-7-3.14-7-7-7zm1 3H6v5h2V4zm0 6H6v2h2v-2z"></path></svg>
|
|
||||||
<span itemprop="name">Issues</span>
|
|
||||||
<span class="counter">8</span>
|
|
||||||
<meta itemprop="position" content="2">
|
|
||||||
</a> </span>
|
|
||||||
|
|
||||||
<span itemscope itemtype="http://schema.org/ListItem" itemprop="itemListElement">
|
|
||||||
<a href="/miloyip/nativejson-benchmark/pulls" class="js-selected-navigation-item reponav-item" data-hotkey="g p" data-selected-links="repo_pulls /miloyip/nativejson-benchmark/pulls" itemprop="url">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-git-pull-request" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M11 11.28V5c-.03-.78-.34-1.47-.94-2.06C9.46 2.35 8.78 2.03 8 2H7V0L4 3l3 3V4h1c.27.02.48.11.69.31.21.2.3.42.31.69v6.28A1.993 1.993 0 0 0 10 15a1.993 1.993 0 0 0 1-3.72zm-1 2.92c-.66 0-1.2-.55-1.2-1.2 0-.65.55-1.2 1.2-1.2.65 0 1.2.55 1.2 1.2 0 .65-.55 1.2-1.2 1.2zM4 3c0-1.11-.89-2-2-2a1.993 1.993 0 0 0-1 3.72v6.56A1.993 1.993 0 0 0 2 15a1.993 1.993 0 0 0 1-3.72V4.72c.59-.34 1-.98 1-1.72zm-.8 10c0 .66-.55 1.2-1.2 1.2-.65 0-1.2-.55-1.2-1.2 0-.65.55-1.2 1.2-1.2.65 0 1.2.55 1.2 1.2zM2 4.2C1.34 4.2.8 3.65.8 3c0-.65.55-1.2 1.2-1.2.65 0 1.2.55 1.2 1.2 0 .65-.55 1.2-1.2 1.2z"></path></svg>
|
|
||||||
<span itemprop="name">Pull requests</span>
|
|
||||||
<span class="counter">1</span>
|
|
||||||
<meta itemprop="position" content="3">
|
|
||||||
</a> </span>
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<a href="/miloyip/nativejson-benchmark/pulse" class="js-selected-navigation-item reponav-item" data-selected-links="pulse /miloyip/nativejson-benchmark/pulse">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-pulse" height="16" version="1.1" viewBox="0 0 14 16" width="14"><path d="M11.5 8L8.8 5.4 6.6 8.5 5.5 1.6 2.38 8H0v2h3.6l.9-1.8.9 5.4L9 8.5l1.6 1.5H14V8z"></path></svg>
|
|
||||||
Pulse
|
|
||||||
</a>
|
|
||||||
<a href="/miloyip/nativejson-benchmark/graphs" class="js-selected-navigation-item reponav-item" data-selected-links="repo_graphs repo_contributors /miloyip/nativejson-benchmark/graphs">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-graph" height="16" version="1.1" viewBox="0 0 16 16" width="16"><path d="M16 14v1H0V0h1v14h15zM5 13H3V8h2v5zm4 0H7V3h2v10zm4 0h-2V6h2v7z"></path></svg>
|
|
||||||
Graphs
|
|
||||||
</a>
|
|
||||||
|
|
||||||
</nav>
|
|
||||||
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="container new-discussion-timeline experiment-repo-nav">
|
|
||||||
<div class="repository-content">
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<a href="/miloyip/nativejson-benchmark/blob/95f27ebcf9a96c7ca4cee26467ed5420140090fb/sample/conformance_Nlohmann%20(C%2B%2B11).md" class="d-none js-permalink-shortcut" data-hotkey="y">Permalink</a>
|
|
||||||
|
|
||||||
<!-- blob contrib key: blob_contributors:v21:0bf9e3593dedd91db6c9dc69e13b7f95 -->
|
|
||||||
|
|
||||||
<div class="file-navigation js-zeroclipboard-container">
|
|
||||||
|
|
||||||
<div class="select-menu branch-select-menu js-menu-container js-select-menu float-left">
|
|
||||||
<button class="btn btn-sm select-menu-button js-menu-target css-truncate" data-hotkey="w"
|
|
||||||
|
|
||||||
type="button" aria-label="Switch branches or tags" tabindex="0" aria-haspopup="true">
|
|
||||||
<i>Branch:</i>
|
|
||||||
<span class="js-select-button css-truncate-target">master</span>
|
|
||||||
</button>
|
|
||||||
|
|
||||||
<div class="select-menu-modal-holder js-menu-content js-navigation-container" data-pjax aria-hidden="true">
|
|
||||||
|
|
||||||
<div class="select-menu-modal">
|
|
||||||
<div class="select-menu-header">
|
|
||||||
<svg aria-label="Close" class="octicon octicon-x js-menu-close" height="16" role="img" version="1.1" viewBox="0 0 12 16" width="12"><path d="M7.48 8l3.75 3.75-1.48 1.48L6 9.48l-3.75 3.75-1.48-1.48L4.52 8 .77 4.25l1.48-1.48L6 6.52l3.75-3.75 1.48 1.48z"></path></svg>
|
|
||||||
<span class="select-menu-title">Switch branches/tags</span>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="select-menu-filters">
|
|
||||||
<div class="select-menu-text-filter">
|
|
||||||
<input type="text" aria-label="Filter branches/tags" id="context-commitish-filter-field" class="form-control js-filterable-field js-navigation-enable" placeholder="Filter branches/tags">
|
|
||||||
</div>
|
|
||||||
<div class="select-menu-tabs">
|
|
||||||
<ul>
|
|
||||||
<li class="select-menu-tab">
|
|
||||||
<a href="#" data-tab-filter="branches" data-filter-placeholder="Filter branches/tags" class="js-select-menu-tab" role="tab">Branches</a>
|
|
||||||
</li>
|
|
||||||
<li class="select-menu-tab">
|
|
||||||
<a href="#" data-tab-filter="tags" data-filter-placeholder="Find a tag…" class="js-select-menu-tab" role="tab">Tags</a>
|
|
||||||
</li>
|
|
||||||
</ul>
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="select-menu-list select-menu-tab-bucket js-select-menu-tab-bucket" data-tab-filter="branches" role="menu">
|
|
||||||
|
|
||||||
<div data-filterable-for="context-commitish-filter-field" data-filterable-type="substring">
|
|
||||||
|
|
||||||
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open "
|
|
||||||
href="/miloyip/nativejson-benchmark/blob/Stixjson/sample/conformance_Nlohmann%20(C++11).md"
|
|
||||||
data-name="Stixjson"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target js-select-menu-filter-text">
|
|
||||||
Stixjson
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open "
|
|
||||||
href="/miloyip/nativejson-benchmark/blob/ccan/sample/conformance_Nlohmann%20(C++11).md"
|
|
||||||
data-name="ccan"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target js-select-menu-filter-text">
|
|
||||||
ccan
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open "
|
|
||||||
href="/miloyip/nativejson-benchmark/blob/jbson/sample/conformance_Nlohmann%20(C++11).md"
|
|
||||||
data-name="jbson"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target js-select-menu-filter-text">
|
|
||||||
jbson
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open "
|
|
||||||
href="/miloyip/nativejson-benchmark/blob/jute/sample/conformance_Nlohmann%20(C++11).md"
|
|
||||||
data-name="jute"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target js-select-menu-filter-text">
|
|
||||||
jute
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open "
|
|
||||||
href="/miloyip/nativejson-benchmark/blob/lastjson/sample/conformance_Nlohmann%20(C++11).md"
|
|
||||||
data-name="lastjson"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target js-select-menu-filter-text">
|
|
||||||
lastjson
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open "
|
|
||||||
href="/miloyip/nativejson-benchmark/blob/libjson/sample/conformance_Nlohmann%20(C++11).md"
|
|
||||||
data-name="libjson"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target js-select-menu-filter-text">
|
|
||||||
libjson
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open selected"
|
|
||||||
href="/miloyip/nativejson-benchmark/blob/master/sample/conformance_Nlohmann%20(C++11).md"
|
|
||||||
data-name="master"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target js-select-menu-filter-text">
|
|
||||||
master
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open "
|
|
||||||
href="/miloyip/nativejson-benchmark/blob/qt/sample/conformance_Nlohmann%20(C++11).md"
|
|
||||||
data-name="qt"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target js-select-menu-filter-text">
|
|
||||||
qt
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open "
|
|
||||||
href="/miloyip/nativejson-benchmark/blob/tunnuz/sample/conformance_Nlohmann%20(C++11).md"
|
|
||||||
data-name="tunnuz"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target js-select-menu-filter-text">
|
|
||||||
tunnuz
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open "
|
|
||||||
href="/miloyip/nativejson-benchmark/blob/ujson/sample/conformance_Nlohmann%20(C++11).md"
|
|
||||||
data-name="ujson"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target js-select-menu-filter-text">
|
|
||||||
ujson
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="select-menu-no-results">Nothing to show</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="select-menu-list select-menu-tab-bucket js-select-menu-tab-bucket" data-tab-filter="tags">
|
|
||||||
<div data-filterable-for="context-commitish-filter-field" data-filterable-type="substring">
|
|
||||||
|
|
||||||
|
|
||||||
<a class="select-menu-item js-navigation-item js-navigation-open "
|
|
||||||
href="/miloyip/nativejson-benchmark/tree/v1.0.0/sample/conformance_Nlohmann%20(C%2B%2B11).md"
|
|
||||||
data-name="v1.0.0"
|
|
||||||
data-skip-pjax="true"
|
|
||||||
rel="nofollow">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-check select-menu-item-icon" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M12 5l-8 8-4-4 1.5-1.5L4 10l6.5-6.5z"></path></svg>
|
|
||||||
<span class="select-menu-item-text css-truncate-target" title="v1.0.0">
|
|
||||||
v1.0.0
|
|
||||||
</span>
|
|
||||||
</a>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="select-menu-no-results">Nothing to show</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="btn-group float-right">
|
|
||||||
<a href="/miloyip/nativejson-benchmark/find/master"
|
|
||||||
class="js-pjax-capture-input btn btn-sm"
|
|
||||||
data-pjax
|
|
||||||
data-hotkey="t">
|
|
||||||
Find file
|
|
||||||
</a>
|
|
||||||
<button aria-label="Copy file path to clipboard" class="js-zeroclipboard btn btn-sm zeroclipboard-button tooltipped tooltipped-s" data-copied-hint="Copied!" type="button">Copy path</button>
|
|
||||||
</div>
|
|
||||||
<div class="breadcrumb js-zeroclipboard-target">
|
|
||||||
<span class="repo-root js-repo-root"><span class="js-path-segment"><a href="/miloyip/nativejson-benchmark"><span>nativejson-benchmark</span></a></span></span><span class="separator">/</span><span class="js-path-segment"><a href="/miloyip/nativejson-benchmark/tree/master/sample"><span>sample</span></a></span><span class="separator">/</span><strong class="final-path">conformance_Nlohmann (C++11).md</strong>
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
|
|
||||||
<div class="commit-tease">
|
|
||||||
<span class="float-right">
|
|
||||||
<a class="commit-tease-sha" href="/miloyip/nativejson-benchmark/commit/a4a9f10f41c515d6abb0f019ab9f5d021ed4bb9e" data-pjax>
|
|
||||||
a4a9f10
|
|
||||||
</a>
|
|
||||||
<relative-time datetime="2016-09-09T03:15:21Z">Sep 9, 2016</relative-time>
|
|
||||||
</span>
|
|
||||||
<div>
|
|
||||||
<img alt="@miloyip" class="avatar" height="20" src="https://avatars3.githubusercontent.com/u/1195774?v=3&s=40" width="20" />
|
|
||||||
<a href="/miloyip" class="user-mention" rel="author">miloyip</a>
|
|
||||||
<a href="/miloyip/nativejson-benchmark/commit/a4a9f10f41c515d6abb0f019ab9f5d021ed4bb9e" class="message" data-pjax="true" title="Update sample result for 41 libraries
|
|
||||||
|
|
||||||
Fixed #43">Update sample result for 41 libraries</a>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="commit-tease-contributors">
|
|
||||||
<button type="button" class="btn-link muted-link contributors-toggle" data-facebox="#blob_contributors_box">
|
|
||||||
<strong>1</strong>
|
|
||||||
contributor
|
|
||||||
</button>
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div id="blob_contributors_box" style="display:none">
|
|
||||||
<h2 class="facebox-header" data-facebox-id="facebox-header">Users who have contributed to this file</h2>
|
|
||||||
<ul class="facebox-user-list" data-facebox-id="facebox-description">
|
|
||||||
<li class="facebox-user-list-item">
|
|
||||||
<img alt="@miloyip" height="24" src="https://avatars1.githubusercontent.com/u/1195774?v=3&s=48" width="24" />
|
|
||||||
<a href="/miloyip">miloyip</a>
|
|
||||||
</li>
|
|
||||||
</ul>
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="file">
|
|
||||||
<div class="file-header">
|
|
||||||
<div class="file-actions">
|
|
||||||
|
|
||||||
<div class="btn-group">
|
|
||||||
<a href="/miloyip/nativejson-benchmark/raw/master/sample/conformance_Nlohmann%20(C%2B%2B11).md" class="btn btn-sm " id="raw-url">Raw</a>
|
|
||||||
<a href="/miloyip/nativejson-benchmark/blame/master/sample/conformance_Nlohmann%20(C%2B%2B11).md" class="btn btn-sm js-update-url-with-hash">Blame</a>
|
|
||||||
<a href="/miloyip/nativejson-benchmark/commits/master/sample/conformance_Nlohmann%20(C%2B%2B11).md" class="btn btn-sm " rel="nofollow">History</a>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
|
|
||||||
<button type="button" class="btn-octicon disabled tooltipped tooltipped-nw"
|
|
||||||
aria-label="You must be signed in to make or propose changes">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-pencil" height="16" version="1.1" viewBox="0 0 14 16" width="14"><path d="M0 12v3h3l8-8-3-3-8 8zm3 2H1v-2h1v1h1v1zm10.3-9.3L12 6 9 3l1.3-1.3a.996.996 0 0 1 1.41 0l1.59 1.59c.39.39.39 1.02 0 1.41z"></path></svg>
|
|
||||||
</button>
|
|
||||||
<button type="button" class="btn-octicon btn-octicon-danger disabled tooltipped tooltipped-nw"
|
|
||||||
aria-label="You must be signed in to make or propose changes">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-trashcan" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M11 2H9c0-.55-.45-1-1-1H5c-.55 0-1 .45-1 1H2c-.55 0-1 .45-1 1v1c0 .55.45 1 1 1v9c0 .55.45 1 1 1h7c.55 0 1-.45 1-1V5c.55 0 1-.45 1-1V3c0-.55-.45-1-1-1zm-1 12H3V5h1v8h1V5h1v8h1V5h1v8h1V5h1v9zm1-10H2V3h9v1z"></path></svg>
|
|
||||||
</button>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="file-info">
|
|
||||||
59 lines (37 sloc)
|
|
||||||
<span class="file-info-divider"></span>
|
|
||||||
545 Bytes
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
|
|
||||||
<div id="readme" class="readme blob instapaper_body">
|
|
||||||
<article class="markdown-body entry-content" itemprop="text"><h1><a id="user-content-conformance-of-nlohmann-c11" class="anchor" href="#conformance-of-nlohmann-c11" aria-hidden="true"><svg aria-hidden="true" class="octicon octicon-link" height="16" version="1.1" viewBox="0 0 16 16" width="16"><path d="M4 9h1v1H4c-1.5 0-3-1.69-3-3.5S2.55 3 4 3h4c1.45 0 3 1.69 3 3.5 0 1.41-.91 2.72-2 3.25V8.59c.58-.45 1-1.27 1-2.09C10 5.22 8.98 4 8 4H4c-.98 0-2 1.22-2 2.5S3 9 4 9zm9-3h-1v1h1c1 0 2 1.22 2 2.5S13.98 12 13 12H9c-.98 0-2-1.22-2-2.5 0-.83.42-1.64 1-2.09V6.25c-1.09.53-2 1.84-2 3.25C6 11.31 7.55 13 9 13h4c1.45 0 3-1.69 3-3.5S14.5 6 13 6z"></path></svg></a>Conformance of Nlohmann (C++11)</h1>
|
|
||||||
|
|
||||||
<h2><a id="user-content-1-parse-validation" class="anchor" href="#1-parse-validation" aria-hidden="true"><svg aria-hidden="true" class="octicon octicon-link" height="16" version="1.1" viewBox="0 0 16 16" width="16"><path d="M4 9h1v1H4c-1.5 0-3-1.69-3-3.5S2.55 3 4 3h4c1.45 0 3 1.69 3 3.5 0 1.41-.91 2.72-2 3.25V8.59c.58-.45 1-1.27 1-2.09C10 5.22 8.98 4 8 4H4c-.98 0-2 1.22-2 2.5S3 9 4 9zm9-3h-1v1h1c1 0 2 1.22 2 2.5S13.98 12 13 12H9c-.98 0-2-1.22-2-2.5 0-.83.42-1.64 1-2.09V6.25c-1.09.53-2 1.84-2 3.25C6 11.31 7.55 13 9 13h4c1.45 0 3-1.69 3-3.5S14.5 6 13 6z"></path></svg></a>1. Parse Validation</h2>
|
|
||||||
|
|
||||||
<p>Summary: 34 of 34 are correct.</p>
|
|
||||||
|
|
||||||
<h2><a id="user-content-2-parse-double" class="anchor" href="#2-parse-double" aria-hidden="true"><svg aria-hidden="true" class="octicon octicon-link" height="16" version="1.1" viewBox="0 0 16 16" width="16"><path d="M4 9h1v1H4c-1.5 0-3-1.69-3-3.5S2.55 3 4 3h4c1.45 0 3 1.69 3 3.5 0 1.41-.91 2.72-2 3.25V8.59c.58-.45 1-1.27 1-2.09C10 5.22 8.98 4 8 4H4c-.98 0-2 1.22-2 2.5S3 9 4 9zm9-3h-1v1h1c1 0 2 1.22 2 2.5S13.98 12 13 12H9c-.98 0-2-1.22-2-2.5 0-.83.42-1.64 1-2.09V6.25c-1.09.53-2 1.84-2 3.25C6 11.31 7.55 13 9 13h4c1.45 0 3-1.69 3-3.5S14.5 6 13 6z"></path></svg></a>2. Parse Double</h2>
|
|
||||||
|
|
||||||
<p>Summary: 66 of 66 are correct.</p>
|
|
||||||
|
|
||||||
<h2><a id="user-content-3-parse-string" class="anchor" href="#3-parse-string" aria-hidden="true"><svg aria-hidden="true" class="octicon octicon-link" height="16" version="1.1" viewBox="0 0 16 16" width="16"><path d="M4 9h1v1H4c-1.5 0-3-1.69-3-3.5S2.55 3 4 3h4c1.45 0 3 1.69 3 3.5 0 1.41-.91 2.72-2 3.25V8.59c.58-.45 1-1.27 1-2.09C10 5.22 8.98 4 8 4H4c-.98 0-2 1.22-2 2.5S3 9 4 9zm9-3h-1v1h1c1 0 2 1.22 2 2.5S13.98 12 13 12H9c-.98 0-2-1.22-2-2.5 0-.83.42-1.64 1-2.09V6.25c-1.09.53-2 1.84-2 3.25C6 11.31 7.55 13 9 13h4c1.45 0 3-1.69 3-3.5S14.5 6 13 6z"></path></svg></a>3. Parse String</h2>
|
|
||||||
|
|
||||||
<p>Summary: 9 of 9 are correct.</p>
|
|
||||||
|
|
||||||
<h2><a id="user-content-4-roundtrip" class="anchor" href="#4-roundtrip" aria-hidden="true"><svg aria-hidden="true" class="octicon octicon-link" height="16" version="1.1" viewBox="0 0 16 16" width="16"><path d="M4 9h1v1H4c-1.5 0-3-1.69-3-3.5S2.55 3 4 3h4c1.45 0 3 1.69 3 3.5 0 1.41-.91 2.72-2 3.25V8.59c.58-.45 1-1.27 1-2.09C10 5.22 8.98 4 8 4H4c-.98 0-2 1.22-2 2.5S3 9 4 9zm9-3h-1v1h1c1 0 2 1.22 2 2.5S13.98 12 13 12H9c-.98 0-2-1.22-2-2.5 0-.83.42-1.64 1-2.09V6.25c-1.09.53-2 1.84-2 3.25C6 11.31 7.55 13 9 13h4c1.45 0 3-1.69 3-3.5S14.5 6 13 6z"></path></svg></a>4. Roundtrip</h2>
|
|
||||||
|
|
||||||
<ul>
|
|
||||||
<li>Fail:</li>
|
|
||||||
</ul>
|
|
||||||
|
|
||||||
<div class="highlight highlight-source-js"><pre>[<span class="pl-c1">5e-324</span>]</pre></div>
|
|
||||||
|
|
||||||
<div class="highlight highlight-source-js"><pre>[<span class="pl-c1">4.94065645841247e-324</span>]</pre></div>
|
|
||||||
|
|
||||||
<ul>
|
|
||||||
<li>Fail:</li>
|
|
||||||
</ul>
|
|
||||||
|
|
||||||
<div class="highlight highlight-source-js"><pre>[<span class="pl-c1">2.225073858507201e-308</span>]</pre></div>
|
|
||||||
|
|
||||||
<div class="highlight highlight-source-js"><pre>[<span class="pl-c1">2.2250738585072e-308</span>]</pre></div>
|
|
||||||
|
|
||||||
<ul>
|
|
||||||
<li>Fail:</li>
|
|
||||||
</ul>
|
|
||||||
|
|
||||||
<div class="highlight highlight-source-js"><pre>[<span class="pl-c1">2.2250738585072014e-308</span>]</pre></div>
|
|
||||||
|
|
||||||
<div class="highlight highlight-source-js"><pre>[<span class="pl-c1">2.2250738585072e-308</span>]</pre></div>
|
|
||||||
|
|
||||||
<ul>
|
|
||||||
<li>Fail:</li>
|
|
||||||
</ul>
|
|
||||||
|
|
||||||
<div class="highlight highlight-source-js"><pre>[<span class="pl-c1">1.7976931348623157e308</span>]</pre></div>
|
|
||||||
|
|
||||||
<div class="highlight highlight-source-js"><pre>[<span class="pl-c1">1.79769313486232e+308</span>]</pre></div>
|
|
||||||
|
|
||||||
<p>Summary: 23 of 27 are correct.</p>
|
|
||||||
</article>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<button type="button" data-facebox="#jump-to-line" data-facebox-class="linejump" data-hotkey="l" class="d-none">Jump to Line</button>
|
|
||||||
<div id="jump-to-line" style="display:none">
|
|
||||||
<!-- </textarea> --><!-- '"` --><form accept-charset="UTF-8" action="" class="js-jump-to-line-form" method="get"><div style="margin:0;padding:0;display:inline"><input name="utf8" type="hidden" value="✓" /></div>
|
|
||||||
<input class="form-control linejump-input js-jump-to-line-field" type="text" placeholder="Jump to line…" aria-label="Jump to line" autofocus>
|
|
||||||
<button type="submit" class="btn">Go</button>
|
|
||||||
</form></div>
|
|
||||||
|
|
||||||
</div>
|
|
||||||
<div class="modal-backdrop js-touch-events"></div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
<div class="container site-footer-container">
|
|
||||||
<div class="site-footer" role="contentinfo">
|
|
||||||
<ul class="site-footer-links float-right">
|
|
||||||
<li><a href="https://github.com/contact" data-ga-click="Footer, go to contact, text:contact">Contact GitHub</a></li>
|
|
||||||
<li><a href="https://developer.github.com" data-ga-click="Footer, go to api, text:api">API</a></li>
|
|
||||||
<li><a href="https://training.github.com" data-ga-click="Footer, go to training, text:training">Training</a></li>
|
|
||||||
<li><a href="https://shop.github.com" data-ga-click="Footer, go to shop, text:shop">Shop</a></li>
|
|
||||||
<li><a href="https://github.com/blog" data-ga-click="Footer, go to blog, text:blog">Blog</a></li>
|
|
||||||
<li><a href="https://github.com/about" data-ga-click="Footer, go to about, text:about">About</a></li>
|
|
||||||
|
|
||||||
</ul>
|
|
||||||
|
|
||||||
<a href="https://github.com" aria-label="Homepage" class="site-footer-mark" title="GitHub">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-mark-github" height="24" version="1.1" viewBox="0 0 16 16" width="24"><path d="M8 0C3.58 0 0 3.58 0 8c0 3.54 2.29 6.53 5.47 7.59.4.07.55-.17.55-.38 0-.19-.01-.82-.01-1.49-2.01.37-2.53-.49-2.69-.94-.09-.23-.48-.94-.82-1.13-.28-.15-.68-.52-.01-.53.63-.01 1.08.58 1.23.82.72 1.21 1.87.87 2.33.66.07-.52.28-.87.51-1.07-1.78-.2-3.64-.89-3.64-3.95 0-.87.31-1.59.82-2.15-.08-.2-.36-1.02.08-2.12 0 0 .67-.21 2.2.82.64-.18 1.32-.27 2-.27.68 0 1.36.09 2 .27 1.53-1.04 2.2-.82 2.2-.82.44 1.1.16 1.92.08 2.12.51.56.82 1.27.82 2.15 0 3.07-1.87 3.75-3.65 3.95.29.25.54.73.54 1.48 0 1.07-.01 1.93-.01 2.2 0 .21.15.46.55.38A8.013 8.013 0 0 0 16 8c0-4.42-3.58-8-8-8z"></path></svg>
|
|
||||||
</a>
|
|
||||||
<ul class="site-footer-links">
|
|
||||||
<li>© 2016 <span title="0.18611s from github-fe151-cp1-prd.iad.github.net">GitHub</span>, Inc.</li>
|
|
||||||
<li><a href="https://github.com/site/terms" data-ga-click="Footer, go to terms, text:terms">Terms</a></li>
|
|
||||||
<li><a href="https://github.com/site/privacy" data-ga-click="Footer, go to privacy, text:privacy">Privacy</a></li>
|
|
||||||
<li><a href="https://github.com/security" data-ga-click="Footer, go to security, text:security">Security</a></li>
|
|
||||||
<li><a href="https://status.github.com/" data-ga-click="Footer, go to status, text:status">Status</a></li>
|
|
||||||
<li><a href="https://help.github.com" data-ga-click="Footer, go to help, text:help">Help</a></li>
|
|
||||||
</ul>
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<div id="ajax-error-message" class="ajax-error-message flash flash-error">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-alert" height="16" version="1.1" viewBox="0 0 16 16" width="16"><path d="M8.865 1.52c-.18-.31-.51-.5-.87-.5s-.69.19-.87.5L.275 13.5c-.18.31-.18.69 0 1 .19.31.52.5.87.5h13.7c.36 0 .69-.19.86-.5.17-.31.18-.69.01-1L8.865 1.52zM8.995 13h-2v-2h2v2zm0-3h-2V6h2v4z"></path></svg>
|
|
||||||
<button type="button" class="flash-close js-flash-close js-ajax-error-dismiss" aria-label="Dismiss error">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-x" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M7.48 8l3.75 3.75-1.48 1.48L6 9.48l-3.75 3.75-1.48-1.48L4.52 8 .77 4.25l1.48-1.48L6 6.52l3.75-3.75 1.48 1.48z"></path></svg>
|
|
||||||
</button>
|
|
||||||
You can't perform that action at this time.
|
|
||||||
</div>
|
|
||||||
|
|
||||||
|
|
||||||
<script crossorigin="anonymous" src="https://assets-cdn.github.com/assets/compat-40e365359d1c4db1e36a55be458e60f2b7c24d58b5a00ae13398480e7ba768e0.js"></script>
|
|
||||||
<script crossorigin="anonymous" src="https://assets-cdn.github.com/assets/frameworks-88471af1fec40ff9418efbe2ddd15b6896af8d772f8179004c254dffc25ea490.js"></script>
|
|
||||||
<script async="async" crossorigin="anonymous" src="https://assets-cdn.github.com/assets/github-e18e11a943ff2eb9394c72d4ec8b76592c454915b5839ae177d422777a046e29.js"></script>
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<div class="js-stale-session-flash stale-session-flash flash flash-warn flash-banner d-none">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-alert" height="16" version="1.1" viewBox="0 0 16 16" width="16"><path d="M8.865 1.52c-.18-.31-.51-.5-.87-.5s-.69.19-.87.5L.275 13.5c-.18.31-.18.69 0 1 .19.31.52.5.87.5h13.7c.36 0 .69-.19.86-.5.17-.31.18-.69.01-1L8.865 1.52zM8.995 13h-2v-2h2v2zm0-3h-2V6h2v4z"></path></svg>
|
|
||||||
<span class="signed-in-tab-flash">You signed in with another tab or window. <a href="">Reload</a> to refresh your session.</span>
|
|
||||||
<span class="signed-out-tab-flash">You signed out in another tab or window. <a href="">Reload</a> to refresh your session.</span>
|
|
||||||
</div>
|
|
||||||
<div class="facebox" id="facebox" style="display:none;">
|
|
||||||
<div class="facebox-popup">
|
|
||||||
<div class="facebox-content" role="dialog" aria-labelledby="facebox-header" aria-describedby="facebox-description">
|
|
||||||
</div>
|
|
||||||
<button type="button" class="facebox-close js-facebox-close" aria-label="Close modal">
|
|
||||||
<svg aria-hidden="true" class="octicon octicon-x" height="16" version="1.1" viewBox="0 0 12 16" width="12"><path d="M7.48 8l3.75 3.75-1.48 1.48L6 9.48l-3.75 3.75-1.48-1.48L4.52 8 .77 4.25l1.48-1.48L6 6.52l3.75-3.75 1.48 1.48z"></path></svg>
|
|
||||||
</button>
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
|
|
||||||
</body>
|
|
||||||
</html>
|
|
||||||
|
|
||||||
|
Before Width: | Height: | Size: 166 KiB |
|
Before Width: | Height: | Size: 192 KiB |
|
Before Width: | Height: | Size: 146 KiB |
|
Before Width: | Height: | Size: 136 KiB |
|
Before Width: | Height: | Size: 98 KiB |
|
Before Width: | Height: | Size: 182 KiB |
|
Before Width: | Height: | Size: 31 KiB |
|
Before Width: | Height: | Size: 22 KiB |
@@ -1,10 +0,0 @@
|
|||||||
<table style="font-family: 'Trebuchet MS', 'Tahoma', 'Arial', 'Helvetica'">
|
|
||||||
<tr><td style="width: 18ex"><b>Banner:</b></td><td>fuzz</td></tr>
|
|
||||||
<tr><td><b>Directory:</b></td><td>fuzz-testing/out</td></tr>
|
|
||||||
<tr><td><b>Generated on:</b></td><td>Sun Oct 2 08:51:02 CEST 2016</td></tr>
|
|
||||||
</table>
|
|
||||||
<p>
|
|
||||||
<img src="high_freq.png" width=1000 height=300><p>
|
|
||||||
<img src="low_freq.png" width=1000 height=200><p>
|
|
||||||
<img src="exec_speed.png" width=1000 height=200>
|
|
||||||
|
|
||||||
|
Before Width: | Height: | Size: 14 KiB |
@@ -11,15 +11,15 @@ This file implements a parser test suitable for fuzz testing. Given a byte
|
|||||||
array data, it performs the following steps:
|
array data, it performs the following steps:
|
||||||
|
|
||||||
- j1 = from_bjdata(data)
|
- j1 = from_bjdata(data)
|
||||||
- vec = to_bjdata(j1)
|
- vec2 = to_bjdata(j1, use_size = false, use_type = false)
|
||||||
- j2 = from_bjdata(vec)
|
- vec3 = to_bjdata(j1, use_size = true, use_type = false)
|
||||||
- assert(j1 == j2)
|
- vec4 = to_bjdata(j1, use_size = true, use_type = true)
|
||||||
- vec2 = to_bjdata(j1, use_size = true, use_type = false)
|
- j2 = from_bjdata(vec2)
|
||||||
- j3 = from_bjdata(vec2)
|
- j3 = from_bjdata(vec3)
|
||||||
- assert(j1 == j3)
|
- j4 = from_bjdata(vec4)
|
||||||
- vec3 = to_bjdata(j1, use_size = true, use_type = true)
|
- assert(from_bjdata(to_bjdata(j2, use_size = false, use_type = false)) is value-stable with j2)
|
||||||
- j4 = from_bjdata(vec3)
|
- assert(from_bjdata(to_bjdata(j3, use_size = true, use_type = false)) is value-stable with j3)
|
||||||
- assert(j1 == j4)
|
- assert(from_bjdata(to_bjdata(j4, use_size = true, use_type = true)) is value-stable with j4)
|
||||||
|
|
||||||
Re-serializing j2/j3/j4 with the same use_size/use_type settings is checked
|
Re-serializing j2/j3/j4 with the same use_size/use_type settings is checked
|
||||||
for value-stability rather than byte-exact stability: from_bjdata(to_bjdata(j2))
|
for value-stability rather than byte-exact stability: from_bjdata(to_bjdata(j2))
|
||||||
@@ -45,13 +45,15 @@ dumps is stable under exactly the same values that break operator==.
|
|||||||
The unit tests run the same checks on a fixed corpus (see the "BJData round-trip
|
The unit tests run the same checks on a fixed corpus (see the "BJData round-trip
|
||||||
invariants" test case), so keep both in sync.
|
invariants" test case), so keep both in sync.
|
||||||
|
|
||||||
|
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||||
|
and checks that the events are balanced, that reading ends, and that it
|
||||||
|
reports an error exactly when from_bjdata() fails (see #3989).
|
||||||
|
|
||||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||||
drivers.
|
drivers.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#include <cassert>
|
#include <cassert>
|
||||||
#include <iostream>
|
|
||||||
#include <sstream>
|
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
// the round-trip checks below are assertions; NDEBUG would compile them away
|
// the round-trip checks below are assertions; NDEBUG would compile them away
|
||||||
@@ -59,6 +61,8 @@ drivers.
|
|||||||
#error "the fuzzer drivers must be built without NDEBUG"
|
#error "the fuzzer drivers must be built without NDEBUG"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#include "fuzzer-recovering_checker.hpp"
|
||||||
|
|
||||||
using json = nlohmann::json;
|
using json = nlohmann::json;
|
||||||
|
|
||||||
// value-stable comparison for the round-trip checks below; see the note
|
// value-stable comparison for the round-trip checks below; see the note
|
||||||
@@ -71,11 +75,15 @@ static bool is_value_stable(const json& lhs, const json& rhs)
|
|||||||
// see http://llvm.org/docs/LibFuzzer.html
|
// see http://llvm.org/docs/LibFuzzer.html
|
||||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||||
{
|
{
|
||||||
|
// step 0: recover from all errors, reading from memory and from a stream
|
||||||
|
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bjdata).errors == 0;
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
// step 1: parse input
|
// step 1: parse input
|
||||||
std::vector<uint8_t> const vec1(data, data + size);
|
std::vector<uint8_t> const vec1(data, data + size);
|
||||||
json const j1 = json::from_bjdata(vec1);
|
json const j1 = json::from_bjdata(vec1);
|
||||||
|
assert(recovered_without_errors);
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
@@ -109,6 +117,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::parse_error&)
|
catch (const json::parse_error&)
|
||||||
{
|
{
|
||||||
// parse errors are ok, because input may be random bytes
|
// parse errors are ok, because input may be random bytes
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
catch (const json::type_error&)
|
catch (const json::type_error&)
|
||||||
{
|
{
|
||||||
@@ -117,6 +126,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::out_of_range&)
|
catch (const json::out_of_range&)
|
||||||
{
|
{
|
||||||
// out of range errors may happen if provided sizes are excessive
|
// out of range errors may happen if provided sizes are excessive
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
|
|
||||||
// return 0 - non-zero return values are reserved for future use
|
// return 0 - non-zero return values are reserved for future use
|
||||||
|
|||||||
@@ -13,18 +13,21 @@ array data, it performs the following steps:
|
|||||||
- j1 = from_bon8(data)
|
- j1 = from_bon8(data)
|
||||||
- vec = to_bon8(j1)
|
- vec = to_bon8(j1)
|
||||||
- j2 = from_bon8(vec)
|
- j2 = from_bon8(vec)
|
||||||
- assert(j1 == j2)
|
- assert(to_bon8(j2) == vec)
|
||||||
|
|
||||||
It also checks that reading the data from a stream, which reads strings byte by
|
It also checks that reading the data from a stream, which reads strings byte by
|
||||||
byte, gives the same value or error as reading it from contiguous memory, which
|
byte, gives the same value or error as reading it from contiguous memory, which
|
||||||
copies strings in bulk.
|
copies strings in bulk.
|
||||||
|
|
||||||
|
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||||
|
and checks that the events are balanced, that reading ends, and that it
|
||||||
|
reports an error exactly when from_bon8() fails (see #3989).
|
||||||
|
|
||||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||||
drivers.
|
drivers.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#include <cassert>
|
#include <cassert>
|
||||||
#include <iostream>
|
|
||||||
#include <sstream>
|
#include <sstream>
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
@@ -33,6 +36,8 @@ drivers.
|
|||||||
#error "the fuzzer drivers must be built without NDEBUG"
|
#error "the fuzzer drivers must be built without NDEBUG"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#include "fuzzer-recovering_checker.hpp"
|
||||||
|
|
||||||
using json = nlohmann::json;
|
using json = nlohmann::json;
|
||||||
|
|
||||||
namespace
|
namespace
|
||||||
@@ -56,6 +61,9 @@ std::string read_bon8(InputType&& input)
|
|||||||
// see http://llvm.org/docs/LibFuzzer.html
|
// see http://llvm.org/docs/LibFuzzer.html
|
||||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||||
{
|
{
|
||||||
|
// step 0: recover from all errors, reading from memory and from a stream
|
||||||
|
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bon8).errors == 0;
|
||||||
|
|
||||||
// contiguous and stream input must be read alike
|
// contiguous and stream input must be read alike
|
||||||
{
|
{
|
||||||
std::istringstream stream(std::string(reinterpret_cast<const char*>(data), size));
|
std::istringstream stream(std::string(reinterpret_cast<const char*>(data), size));
|
||||||
@@ -67,6 +75,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
// step 1: parse input
|
// step 1: parse input
|
||||||
std::vector<uint8_t> const vec1(data, data + size);
|
std::vector<uint8_t> const vec1(data, data + size);
|
||||||
json const j1 = json::from_bon8(vec1);
|
json const j1 = json::from_bon8(vec1);
|
||||||
|
assert(recovered_without_errors);
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
@@ -88,6 +97,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::parse_error&)
|
catch (const json::parse_error&)
|
||||||
{
|
{
|
||||||
// parse errors are ok, because input may be random bytes
|
// parse errors are ok, because input may be random bytes
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
catch (const json::type_error&)
|
catch (const json::type_error&)
|
||||||
{
|
{
|
||||||
@@ -96,6 +106,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::out_of_range&)
|
catch (const json::out_of_range&)
|
||||||
{
|
{
|
||||||
// out of range errors may happen if provided sizes are excessive
|
// out of range errors may happen if provided sizes are excessive
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
|
|
||||||
// return 0 - non-zero return values are reserved for future use
|
// return 0 - non-zero return values are reserved for future use
|
||||||
|
|||||||
@@ -13,15 +13,17 @@ array data, it performs the following steps:
|
|||||||
- j1 = from_bson(data)
|
- j1 = from_bson(data)
|
||||||
- vec = to_bson(j1)
|
- vec = to_bson(j1)
|
||||||
- j2 = from_bson(vec)
|
- j2 = from_bson(vec)
|
||||||
- assert(j1 == j2)
|
- assert(to_bson(j2) == vec)
|
||||||
|
|
||||||
|
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||||
|
and checks that the events are balanced, that reading ends, and that it
|
||||||
|
reports an error exactly when from_bson() fails (see #3989).
|
||||||
|
|
||||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||||
drivers.
|
drivers.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#include <cassert>
|
#include <cassert>
|
||||||
#include <iostream>
|
|
||||||
#include <sstream>
|
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
// the round-trip checks below are assertions; NDEBUG would compile them away
|
// the round-trip checks below are assertions; NDEBUG would compile them away
|
||||||
@@ -29,21 +31,22 @@ drivers.
|
|||||||
#error "the fuzzer drivers must be built without NDEBUG"
|
#error "the fuzzer drivers must be built without NDEBUG"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#include "fuzzer-recovering_checker.hpp"
|
||||||
|
|
||||||
using json = nlohmann::json;
|
using json = nlohmann::json;
|
||||||
|
|
||||||
// see http://llvm.org/docs/LibFuzzer.html
|
// see http://llvm.org/docs/LibFuzzer.html
|
||||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||||
{
|
{
|
||||||
|
// step 0: recover from all errors, reading from memory and from a stream
|
||||||
|
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bson).errors == 0;
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
// step 1: parse input
|
// step 1: parse input
|
||||||
std::vector<uint8_t> const vec1(data, data + size);
|
std::vector<uint8_t> const vec1(data, data + size);
|
||||||
json const j1 = json::from_bson(vec1);
|
json const j1 = json::from_bson(vec1);
|
||||||
|
assert(recovered_without_errors);
|
||||||
if (j1.is_discarded())
|
|
||||||
{
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
@@ -65,6 +68,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::parse_error&)
|
catch (const json::parse_error&)
|
||||||
{
|
{
|
||||||
// parse errors are ok, because input may be random bytes
|
// parse errors are ok, because input may be random bytes
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
catch (const json::type_error&)
|
catch (const json::type_error&)
|
||||||
{
|
{
|
||||||
@@ -73,6 +77,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::out_of_range&)
|
catch (const json::out_of_range&)
|
||||||
{
|
{
|
||||||
// out of range errors can occur during parsing, too
|
// out of range errors can occur during parsing, too
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
|
|
||||||
// return 0 - non-zero return values are reserved for future use
|
// return 0 - non-zero return values are reserved for future use
|
||||||
|
|||||||
@@ -13,15 +13,17 @@ array data, it performs the following steps:
|
|||||||
- j1 = from_cbor(data)
|
- j1 = from_cbor(data)
|
||||||
- vec = to_cbor(j1)
|
- vec = to_cbor(j1)
|
||||||
- j2 = from_cbor(vec)
|
- j2 = from_cbor(vec)
|
||||||
- assert(j1 == j2)
|
- assert(to_cbor(j2) == vec)
|
||||||
|
|
||||||
|
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||||
|
and checks that the events are balanced, that reading ends, and that it
|
||||||
|
reports an error exactly when from_cbor() fails (see #3989).
|
||||||
|
|
||||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||||
drivers.
|
drivers.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#include <cassert>
|
#include <cassert>
|
||||||
#include <iostream>
|
|
||||||
#include <sstream>
|
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
// the round-trip checks below are assertions; NDEBUG would compile them away
|
// the round-trip checks below are assertions; NDEBUG would compile them away
|
||||||
@@ -29,16 +31,22 @@ drivers.
|
|||||||
#error "the fuzzer drivers must be built without NDEBUG"
|
#error "the fuzzer drivers must be built without NDEBUG"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#include "fuzzer-recovering_checker.hpp"
|
||||||
|
|
||||||
using json = nlohmann::json;
|
using json = nlohmann::json;
|
||||||
|
|
||||||
// see http://llvm.org/docs/LibFuzzer.html
|
// see http://llvm.org/docs/LibFuzzer.html
|
||||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||||
{
|
{
|
||||||
|
// step 0: recover from all errors, reading from memory and from a stream
|
||||||
|
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::cbor).errors == 0;
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
// step 1: parse input
|
// step 1: parse input
|
||||||
std::vector<uint8_t> const vec1(data, data + size);
|
std::vector<uint8_t> const vec1(data, data + size);
|
||||||
json const j1 = json::from_cbor(vec1);
|
json const j1 = json::from_cbor(vec1);
|
||||||
|
assert(recovered_without_errors);
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
@@ -60,6 +68,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::parse_error&)
|
catch (const json::parse_error&)
|
||||||
{
|
{
|
||||||
// parse errors are ok, because input may be random bytes
|
// parse errors are ok, because input may be random bytes
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
catch (const json::type_error&)
|
catch (const json::type_error&)
|
||||||
{
|
{
|
||||||
@@ -68,6 +77,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::out_of_range&)
|
catch (const json::out_of_range&)
|
||||||
{
|
{
|
||||||
// out of range errors can occur during parsing, too
|
// out of range errors can occur during parsing, too
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
|
|
||||||
// return 0 - non-zero return values are reserved for future use
|
// return 0 - non-zero return values are reserved for future use
|
||||||
|
|||||||
@@ -16,13 +16,15 @@ array data, it performs the following steps:
|
|||||||
- s2 = serialize(j2)
|
- s2 = serialize(j2)
|
||||||
- assert(s1 == s2)
|
- assert(s1 == s2)
|
||||||
|
|
||||||
|
Furthermore, it parses data with a SAX parser that recovers from every error
|
||||||
|
and checks that the events are balanced, that parsing ends, and that valid
|
||||||
|
input is parsed without errors (see #3989).
|
||||||
|
|
||||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||||
drivers.
|
drivers.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#include <cassert>
|
#include <cassert>
|
||||||
#include <iostream>
|
|
||||||
#include <sstream>
|
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
// the round-trip checks below are assertions; NDEBUG would compile them away
|
// the round-trip checks below are assertions; NDEBUG would compile them away
|
||||||
@@ -30,11 +32,20 @@ drivers.
|
|||||||
#error "the fuzzer drivers must be built without NDEBUG"
|
#error "the fuzzer drivers must be built without NDEBUG"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#include "fuzzer-recovering_checker.hpp"
|
||||||
|
|
||||||
using json = nlohmann::json;
|
using json = nlohmann::json;
|
||||||
|
|
||||||
// see http://llvm.org/docs/LibFuzzer.html
|
// see http://llvm.org/docs/LibFuzzer.html
|
||||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||||
{
|
{
|
||||||
|
// step 0: recover from all errors, reading from memory and from a stream
|
||||||
|
{
|
||||||
|
const auto checker = check_recovering_parse(data, size, json::input_format_t::json);
|
||||||
|
assert(checker.events <= (4 * size) + 4);
|
||||||
|
assert((checker.errors == 0) == json::accept(data, data + size));
|
||||||
|
}
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
// step 1: parse input
|
// step 1: parse input
|
||||||
|
|||||||
@@ -13,15 +13,17 @@ array data, it performs the following steps:
|
|||||||
- j1 = from_msgpack(data)
|
- j1 = from_msgpack(data)
|
||||||
- vec = to_msgpack(j1)
|
- vec = to_msgpack(j1)
|
||||||
- j2 = from_msgpack(vec)
|
- j2 = from_msgpack(vec)
|
||||||
- assert(j1 == j2)
|
- assert(to_msgpack(j2) == vec)
|
||||||
|
|
||||||
|
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||||
|
and checks that the events are balanced, that reading ends, and that it
|
||||||
|
reports an error exactly when from_msgpack() fails (see #3989).
|
||||||
|
|
||||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||||
drivers.
|
drivers.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#include <cassert>
|
#include <cassert>
|
||||||
#include <iostream>
|
|
||||||
#include <sstream>
|
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
// the round-trip checks below are assertions; NDEBUG would compile them away
|
// the round-trip checks below are assertions; NDEBUG would compile them away
|
||||||
@@ -29,16 +31,22 @@ drivers.
|
|||||||
#error "the fuzzer drivers must be built without NDEBUG"
|
#error "the fuzzer drivers must be built without NDEBUG"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#include "fuzzer-recovering_checker.hpp"
|
||||||
|
|
||||||
using json = nlohmann::json;
|
using json = nlohmann::json;
|
||||||
|
|
||||||
// see http://llvm.org/docs/LibFuzzer.html
|
// see http://llvm.org/docs/LibFuzzer.html
|
||||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||||
{
|
{
|
||||||
|
// step 0: recover from all errors, reading from memory and from a stream
|
||||||
|
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::msgpack).errors == 0;
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
// step 1: parse input
|
// step 1: parse input
|
||||||
std::vector<uint8_t> const vec1(data, data + size);
|
std::vector<uint8_t> const vec1(data, data + size);
|
||||||
json const j1 = json::from_msgpack(vec1);
|
json const j1 = json::from_msgpack(vec1);
|
||||||
|
assert(recovered_without_errors);
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
@@ -60,6 +68,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::parse_error&)
|
catch (const json::parse_error&)
|
||||||
{
|
{
|
||||||
// parse errors are ok, because input may be random bytes
|
// parse errors are ok, because input may be random bytes
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
catch (const json::type_error&)
|
catch (const json::type_error&)
|
||||||
{
|
{
|
||||||
@@ -68,6 +77,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::out_of_range&)
|
catch (const json::out_of_range&)
|
||||||
{
|
{
|
||||||
// out of range errors may happen if provided sizes are excessive
|
// out of range errors may happen if provided sizes are excessive
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
|
|
||||||
// return 0 - non-zero return values are reserved for future use
|
// return 0 - non-zero return values are reserved for future use
|
||||||
|
|||||||
@@ -11,26 +11,28 @@ This file implements a parser test suitable for fuzz testing. Given a byte
|
|||||||
array data, it performs the following steps:
|
array data, it performs the following steps:
|
||||||
|
|
||||||
- j1 = from_ubjson(data)
|
- j1 = from_ubjson(data)
|
||||||
- vec = to_ubjson(j1)
|
- vec2 = to_ubjson(j1, use_size = false, use_type = false)
|
||||||
- j2 = from_ubjson(vec)
|
- vec3 = to_ubjson(j1, use_size = true, use_type = false)
|
||||||
- assert(j1 == j2)
|
- vec4 = to_ubjson(j1, use_size = true, use_type = true)
|
||||||
- vec2 = to_ubjson(j1, use_size = true, use_type = false)
|
- j2 = from_ubjson(vec2)
|
||||||
- j3 = from_ubjson(vec2)
|
- j3 = from_ubjson(vec3)
|
||||||
- assert(j1 == j3)
|
- j4 = from_ubjson(vec4)
|
||||||
- vec3 = to_ubjson(j1, use_size = true, use_type = true)
|
- assert(to_ubjson(j2, use_size = false, use_type = false) == vec2)
|
||||||
- j4 = from_ubjson(vec3)
|
- assert(to_ubjson(j3, use_size = true, use_type = false) == vec3)
|
||||||
- assert(j1 == j4)
|
- assert(to_ubjson(j4, use_size = true, use_type = true) == vec4)
|
||||||
|
|
||||||
The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip
|
The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip
|
||||||
invariants" test case), so keep both in sync.
|
invariants" test case), so keep both in sync.
|
||||||
|
|
||||||
|
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||||
|
and checks that the events are balanced, that reading ends, and that it
|
||||||
|
reports an error exactly when from_ubjson() fails (see #3989).
|
||||||
|
|
||||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||||
drivers.
|
drivers.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#include <cassert>
|
#include <cassert>
|
||||||
#include <iostream>
|
|
||||||
#include <sstream>
|
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
// the round-trip checks below are assertions; NDEBUG would compile them away
|
// the round-trip checks below are assertions; NDEBUG would compile them away
|
||||||
@@ -38,16 +40,22 @@ drivers.
|
|||||||
#error "the fuzzer drivers must be built without NDEBUG"
|
#error "the fuzzer drivers must be built without NDEBUG"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#include "fuzzer-recovering_checker.hpp"
|
||||||
|
|
||||||
using json = nlohmann::json;
|
using json = nlohmann::json;
|
||||||
|
|
||||||
// see http://llvm.org/docs/LibFuzzer.html
|
// see http://llvm.org/docs/LibFuzzer.html
|
||||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||||
{
|
{
|
||||||
|
// step 0: recover from all errors, reading from memory and from a stream
|
||||||
|
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::ubjson).errors == 0;
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
// step 1: parse input
|
// step 1: parse input
|
||||||
std::vector<uint8_t> const vec1(data, data + size);
|
std::vector<uint8_t> const vec1(data, data + size);
|
||||||
json const j1 = json::from_ubjson(vec1);
|
json const j1 = json::from_ubjson(vec1);
|
||||||
|
assert(recovered_without_errors);
|
||||||
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
@@ -79,6 +87,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::parse_error&)
|
catch (const json::parse_error&)
|
||||||
{
|
{
|
||||||
// parse errors are ok, because input may be random bytes
|
// parse errors are ok, because input may be random bytes
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
catch (const json::type_error&)
|
catch (const json::type_error&)
|
||||||
{
|
{
|
||||||
@@ -87,6 +96,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
|||||||
catch (const json::out_of_range&)
|
catch (const json::out_of_range&)
|
||||||
{
|
{
|
||||||
// out of range errors may happen if provided sizes are excessive
|
// out of range errors may happen if provided sizes are excessive
|
||||||
|
assert(!recovered_without_errors);
|
||||||
}
|
}
|
||||||
|
|
||||||
// return 0 - non-zero return values are reserved for future use
|
// return 0 - non-zero return values are reserved for future use
|
||||||
|
|||||||
@@ -0,0 +1,154 @@
|
|||||||
|
// __ _____ _____ _____
|
||||||
|
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||||
|
// | | |__ | | | | | | version 3.12.0
|
||||||
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||||
|
//
|
||||||
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||||
|
// SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <cassert>
|
||||||
|
#include <cstddef>
|
||||||
|
#include <cstdint>
|
||||||
|
#include <sstream>
|
||||||
|
#include <string>
|
||||||
|
#include <vector>
|
||||||
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
|
namespace
|
||||||
|
{
|
||||||
|
// a SAX parser that recovers from every error and checks that the events are
|
||||||
|
// balanced and that every key is followed by exactly one value
|
||||||
|
class recovering_checker : public nlohmann::json_sax<nlohmann::json>
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
bool null() override
|
||||||
|
{
|
||||||
|
return value();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool boolean(bool /*val*/) override
|
||||||
|
{
|
||||||
|
return value();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_integer(number_integer_t /*val*/) override
|
||||||
|
{
|
||||||
|
return value();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_unsigned(number_unsigned_t /*val*/) override
|
||||||
|
{
|
||||||
|
return value();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_float(number_float_t /*val*/, const string_t& /*s*/) override
|
||||||
|
{
|
||||||
|
return value();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool string(string_t& /*val*/) override
|
||||||
|
{
|
||||||
|
return value();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool binary(binary_t& /*val*/) override
|
||||||
|
{
|
||||||
|
return value();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool start_object(std::size_t /*elements*/) override
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
stack.push_back('o');
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool key(string_t& /*val*/) override
|
||||||
|
{
|
||||||
|
++events;
|
||||||
|
assert(!stack.empty() && stack.back() == 'o');
|
||||||
|
stack.back() = 'v';
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool end_object() override
|
||||||
|
{
|
||||||
|
++events;
|
||||||
|
assert(!stack.empty() && stack.back() == 'o');
|
||||||
|
stack.pop_back();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool start_array(std::size_t /*elements*/) override
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
stack.push_back('a');
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool end_array() override
|
||||||
|
{
|
||||||
|
++events;
|
||||||
|
assert(!stack.empty() && stack.back() == 'a');
|
||||||
|
stack.pop_back();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool parse_error(std::size_t /*position*/, const std::string& /*last_token*/, const nlohmann::detail::exception& /*ex*/) override
|
||||||
|
{
|
||||||
|
++errors;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool complete() const
|
||||||
|
{
|
||||||
|
return stack.empty();
|
||||||
|
}
|
||||||
|
|
||||||
|
std::size_t events = 0;
|
||||||
|
std::size_t errors = 0;
|
||||||
|
|
||||||
|
private:
|
||||||
|
bool value()
|
||||||
|
{
|
||||||
|
++events;
|
||||||
|
if (!stack.empty())
|
||||||
|
{
|
||||||
|
// an array element, or the value of a key
|
||||||
|
assert(stack.back() != 'o');
|
||||||
|
if (stack.back() == 'v')
|
||||||
|
{
|
||||||
|
stack.back() = 'o';
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// 'a' for an array, 'o' for an object that expects a key, 'v' for an
|
||||||
|
// object that expects the value of a key
|
||||||
|
std::vector<char> stack {}; // NOLINT(readability-redundant-member-init)
|
||||||
|
};
|
||||||
|
|
||||||
|
/// parses @a data with a recovering_checker from memory and from a stream,
|
||||||
|
/// checks that both see the same, that the events are balanced, and that the
|
||||||
|
/// number of errors is bounded, and returns the checker (see #3989)
|
||||||
|
inline recovering_checker check_recovering_parse(const std::uint8_t* data, const std::size_t size, const nlohmann::json::input_format_t format)
|
||||||
|
{
|
||||||
|
recovering_checker checker;
|
||||||
|
const bool ok = nlohmann::json::sax_parse(data, data + size, &checker, format);
|
||||||
|
assert(checker.complete());
|
||||||
|
assert(checker.errors <= size + 1);
|
||||||
|
assert(ok == (checker.errors == 0));
|
||||||
|
|
||||||
|
std::istringstream stream(std::string(reinterpret_cast<const char*>(data), size));
|
||||||
|
recovering_checker stream_checker;
|
||||||
|
assert(nlohmann::json::sax_parse(stream, &stream_checker, format) == ok);
|
||||||
|
assert(stream_checker.complete());
|
||||||
|
assert(stream_checker.events == checker.events);
|
||||||
|
assert(stream_checker.errors == checker.errors);
|
||||||
|
|
||||||
|
return checker;
|
||||||
|
}
|
||||||
|
} // namespace
|
||||||
@@ -19,13 +19,14 @@
|
|||||||
|
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
// Values for the round-trip property tests of the UBJSON and BJData writers.
|
// Values for the round-trip property tests of the binary format writers
|
||||||
|
// (BJData, BON8, BSON, CBOR, MessagePack and UBJSON).
|
||||||
//
|
//
|
||||||
// The fuzzer drivers (tests/src/fuzzer-parse_ubjson.cpp and
|
// The fuzzer drivers (tests/src/fuzzer-parse_*.cpp) check that anything the
|
||||||
// fuzzer-parse_bjdata.cpp) check that anything the library parses can be
|
// library parses can be serialized, parsed back, and serialized again
|
||||||
// serialized, parsed back, and serialized again without loss. Those checks
|
// without loss. Those checks only run at OSS-Fuzz, so a regression used to
|
||||||
// only run at OSS-Fuzz, so a regression used to surface days later as an
|
// surface days later as an external report. The unit tests run the same
|
||||||
// external report. The unit tests run the same checks on this corpus in CI.
|
// checks on this corpus in CI.
|
||||||
//
|
//
|
||||||
// The corpus is deterministic: std::mt19937's output sequence is fixed by
|
// The corpus is deterministic: std::mt19937's output sequence is fixed by
|
||||||
// the standard, and it is used directly rather than through a distribution
|
// the standard, and it is used directly rather than through a distribution
|
||||||
|
|||||||
@@ -0,0 +1,99 @@
|
|||||||
|
// __ _____ _____ _____
|
||||||
|
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||||
|
// | | |__ | | | | | | version 3.12.0
|
||||||
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||||
|
//
|
||||||
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||||
|
// SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <cstddef>
|
||||||
|
#include <cstdint>
|
||||||
|
#include <string>
|
||||||
|
#include <vector>
|
||||||
|
|
||||||
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
|
namespace utils
|
||||||
|
{
|
||||||
|
/// a SAX event consumer that stops accepting events after a fixed count,
|
||||||
|
/// used by the binary-format tests to check behavior when the SAX consumer
|
||||||
|
/// rejects an event partway through parsing
|
||||||
|
class SaxCountdown
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
using json = nlohmann::json;
|
||||||
|
|
||||||
|
explicit SaxCountdown(const int count) : events_left(count)
|
||||||
|
{}
|
||||||
|
|
||||||
|
bool null()
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool boolean(bool /*unused*/)
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_integer(json::number_integer_t /*unused*/)
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_unsigned(json::number_unsigned_t /*unused*/)
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_float(json::number_float_t /*unused*/, const std::string& /*unused*/)
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool string(std::string& /*unused*/)
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool binary(std::vector<std::uint8_t>& /*unused*/)
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool start_object(std::size_t /*unused*/)
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool key(std::string& /*unused*/)
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool end_object()
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool start_array(std::size_t /*unused*/)
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool end_array()
|
||||||
|
{
|
||||||
|
return events_left-- > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& /*unused*/) // NOLINT(readability-convert-member-functions-to-static)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
private:
|
||||||
|
int events_left = 0;
|
||||||
|
};
|
||||||
|
} // namespace utils
|
||||||
@@ -12,80 +12,11 @@
|
|||||||
using nlohmann::json;
|
using nlohmann::json;
|
||||||
|
|
||||||
#include <climits> // SIZE_MAX
|
#include <climits> // SIZE_MAX
|
||||||
#include <limits> // numeric_limits
|
|
||||||
|
|
||||||
template <typename OfType, typename T, bool MinInRange, bool MaxInRange>
|
// JSON_32bitTest=ONLY builds only this file, so it must keep its own
|
||||||
struct trait_test_arg
|
// include of the shared trait/TEST_CASE_TEMPLATE_DEFINE rather than relying
|
||||||
{
|
// on unit-bjdata.cpp to provide it
|
||||||
using of_type = OfType;
|
#include "value_in_range_of_test.hpp"
|
||||||
using type = T;
|
|
||||||
static constexpr bool min_in_range = MinInRange;
|
|
||||||
static constexpr bool max_in_range = MaxInRange;
|
|
||||||
};
|
|
||||||
|
|
||||||
TEST_CASE_TEMPLATE_DEFINE("value_in_range_of trait", T, value_in_range_of_test) // NOLINT(readability-math-missing-parentheses)
|
|
||||||
{
|
|
||||||
using nlohmann::detail::value_in_range_of;
|
|
||||||
|
|
||||||
using of_type = typename T::of_type;
|
|
||||||
using type = typename T::type;
|
|
||||||
constexpr bool min_in_range = T::min_in_range;
|
|
||||||
constexpr bool max_in_range = T::max_in_range;
|
|
||||||
|
|
||||||
type const val_min = std::numeric_limits<type>::min();
|
|
||||||
type const val_min2 = val_min + 1;
|
|
||||||
type const val_max = std::numeric_limits<type>::max();
|
|
||||||
type const val_max2 = val_max - 1;
|
|
||||||
|
|
||||||
REQUIRE(CHAR_BIT == 8);
|
|
||||||
|
|
||||||
std::string of_type_str;
|
|
||||||
if (std::is_unsigned<of_type>::value)
|
|
||||||
{
|
|
||||||
of_type_str += "u";
|
|
||||||
}
|
|
||||||
of_type_str += "int";
|
|
||||||
of_type_str += std::to_string(sizeof(of_type) * 8);
|
|
||||||
|
|
||||||
INFO("of_type := ", of_type_str);
|
|
||||||
|
|
||||||
std::string type_str;
|
|
||||||
if (std::is_unsigned<type>::value)
|
|
||||||
{
|
|
||||||
type_str += "u";
|
|
||||||
}
|
|
||||||
type_str += "int";
|
|
||||||
type_str += std::to_string(sizeof(type) * 8);
|
|
||||||
|
|
||||||
INFO("type := ", type_str);
|
|
||||||
|
|
||||||
CAPTURE(val_min);
|
|
||||||
CAPTURE(min_in_range);
|
|
||||||
CAPTURE(val_max);
|
|
||||||
CAPTURE(max_in_range);
|
|
||||||
|
|
||||||
if (min_in_range)
|
|
||||||
{
|
|
||||||
CHECK(value_in_range_of<of_type>(val_min));
|
|
||||||
CHECK(value_in_range_of<of_type>(val_min2));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
CHECK_FALSE(value_in_range_of<of_type>(val_min));
|
|
||||||
CHECK_FALSE(value_in_range_of<of_type>(val_min2));
|
|
||||||
}
|
|
||||||
|
|
||||||
if (max_in_range)
|
|
||||||
{
|
|
||||||
CHECK(value_in_range_of<of_type>(val_max));
|
|
||||||
CHECK(value_in_range_of<of_type>(val_max2));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
CHECK_FALSE(value_in_range_of<of_type>(val_max));
|
|
||||||
CHECK_FALSE(value_in_range_of<of_type>(val_max2));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST_CASE("32bit")
|
TEST_CASE("32bit")
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -419,6 +419,46 @@ TEST_CASE("alternative string type")
|
|||||||
CHECK(j2.dump() == R"({"/foo/0":"bar","/foo/1":"baz"})");
|
CHECK(j2.dump() == R"({"/foo/0":"bar","/foo/1":"baz"})");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("error recovery")
|
||||||
|
{
|
||||||
|
// a SAX parser that recovers from every error (see #3989)
|
||||||
|
struct recovering_parser : nlohmann::detail::json_sax_dom_parser<alt_json>
|
||||||
|
{
|
||||||
|
explicit recovering_parser(alt_json& j)
|
||||||
|
: nlohmann::detail::json_sax_dom_parser<alt_json>(j, false)
|
||||||
|
{}
|
||||||
|
|
||||||
|
// sax_parse() calls the SAX parser's own parse_error(), so hiding
|
||||||
|
// the one of the base class is what recovering takes
|
||||||
|
// NOLINTNEXTLINE(bugprone-derived-method-shadowing-base-method)
|
||||||
|
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const nlohmann::detail::exception& /*unused*/)
|
||||||
|
{
|
||||||
|
++errors;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
std::size_t errors = 0;
|
||||||
|
};
|
||||||
|
|
||||||
|
alt_json j;
|
||||||
|
recovering_parser sax(j);
|
||||||
|
// not inside CHECK(): MSVC reads the escape in a stringized raw string
|
||||||
|
const std::string input = R"([1., "a\qb", tru, {"k" 2}])";
|
||||||
|
CHECK(!alt_json::sax_parse(input, &sax));
|
||||||
|
CHECK(sax.errors == 4);
|
||||||
|
CHECK(j.dump() == R"([1,"aqb",null,{"k":2}])");
|
||||||
|
|
||||||
|
// a UBJSON high-precision number, a CBOR key that is not a string
|
||||||
|
alt_json u;
|
||||||
|
recovering_parser ubjson_sax(u);
|
||||||
|
CHECK(!alt_json::sax_parse(std::vector<std::uint8_t> {'[', 'H', 'i', 2, '1', '.', ']'}, &ubjson_sax, alt_json::input_format_t::ubjson));
|
||||||
|
CHECK(u.dump() == "[1]");
|
||||||
|
alt_json c;
|
||||||
|
recovering_parser cbor_sax(c);
|
||||||
|
CHECK(!alt_json::sax_parse(std::vector<std::uint8_t> {0xA2, 0x01, 0x02, 0x61, 'a', 0x03}, &cbor_sax, alt_json::input_format_t::cbor));
|
||||||
|
CHECK(c.dump() == R"({"a":3})");
|
||||||
|
}
|
||||||
|
|
||||||
SECTION("strict enum")
|
SECTION("strict enum")
|
||||||
{
|
{
|
||||||
// regression test for #5667: NLOHMANN_JSON_SERIALIZE_ENUM_STRICT's from_json
|
// regression test for #5667: NLOHMANN_JSON_SERIALIZE_ENUM_STRICT's from_json
|
||||||
|
|||||||
@@ -21,158 +21,13 @@ using nlohmann::json;
|
|||||||
#include "make_test_data_available.hpp"
|
#include "make_test_data_available.hpp"
|
||||||
#include "round_trip_corpus.hpp"
|
#include "round_trip_corpus.hpp"
|
||||||
#include "test_utils.hpp"
|
#include "test_utils.hpp"
|
||||||
|
#include "sax_countdown.hpp"
|
||||||
|
using utils::SaxCountdown;
|
||||||
|
|
||||||
namespace
|
|
||||||
{
|
|
||||||
class SaxCountdown
|
|
||||||
{
|
|
||||||
public:
|
|
||||||
explicit SaxCountdown(const int count) : events_left(count)
|
|
||||||
{}
|
|
||||||
|
|
||||||
bool null()
|
// trait_test_arg and the "value_in_range_of trait" TEST_CASE_TEMPLATE_DEFINE
|
||||||
{
|
// are shared with unit-32bit.cpp
|
||||||
return events_left-- > 0;
|
#include "value_in_range_of_test.hpp"
|
||||||
}
|
|
||||||
|
|
||||||
bool boolean(bool /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_integer(json::number_integer_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_unsigned(json::number_unsigned_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_float(json::number_float_t /*unused*/, const std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool string(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool binary(std::vector<std::uint8_t>& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_object(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool key(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_object()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_array(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_array()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& /*unused*/) // NOLINT(readability-convert-member-functions-to-static)
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
int events_left = 0;
|
|
||||||
};
|
|
||||||
} // namespace
|
|
||||||
|
|
||||||
// at some point in the future, a unit test dedicated to type traits might be a good idea
|
|
||||||
template <typename OfType, typename T, bool MinInRange, bool MaxInRange>
|
|
||||||
struct trait_test_arg
|
|
||||||
{
|
|
||||||
using of_type = OfType;
|
|
||||||
using type = T;
|
|
||||||
static constexpr bool min_in_range = MinInRange;
|
|
||||||
static constexpr bool max_in_range = MaxInRange;
|
|
||||||
};
|
|
||||||
|
|
||||||
TEST_CASE_TEMPLATE_DEFINE("value_in_range_of trait", T, value_in_range_of_test) // NOLINT(readability-math-missing-parentheses)
|
|
||||||
{
|
|
||||||
using nlohmann::detail::value_in_range_of;
|
|
||||||
|
|
||||||
using of_type = typename T::of_type;
|
|
||||||
using type = typename T::type;
|
|
||||||
constexpr bool min_in_range = T::min_in_range;
|
|
||||||
constexpr bool max_in_range = T::max_in_range;
|
|
||||||
|
|
||||||
type const val_min = std::numeric_limits<type>::min();
|
|
||||||
type const val_min2 = val_min + 1;
|
|
||||||
type const val_max = std::numeric_limits<type>::max();
|
|
||||||
type const val_max2 = val_max - 1;
|
|
||||||
|
|
||||||
REQUIRE(CHAR_BIT == 8);
|
|
||||||
|
|
||||||
std::string of_type_str;
|
|
||||||
if (std::is_unsigned<of_type>::value)
|
|
||||||
{
|
|
||||||
of_type_str += "u";
|
|
||||||
}
|
|
||||||
of_type_str += "int";
|
|
||||||
of_type_str += std::to_string(sizeof(of_type) * 8);
|
|
||||||
|
|
||||||
INFO("of_type := ", of_type_str);
|
|
||||||
|
|
||||||
std::string type_str;
|
|
||||||
if (std::is_unsigned<type>::value)
|
|
||||||
{
|
|
||||||
type_str += "u";
|
|
||||||
}
|
|
||||||
type_str += "int";
|
|
||||||
type_str += std::to_string(sizeof(type) * 8);
|
|
||||||
|
|
||||||
INFO("type := ", type_str);
|
|
||||||
|
|
||||||
CAPTURE(val_min);
|
|
||||||
CAPTURE(min_in_range);
|
|
||||||
CAPTURE(val_max);
|
|
||||||
CAPTURE(max_in_range);
|
|
||||||
|
|
||||||
if (min_in_range)
|
|
||||||
{
|
|
||||||
CHECK(value_in_range_of<of_type>(val_min));
|
|
||||||
CHECK(value_in_range_of<of_type>(val_min2));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
CHECK_FALSE(value_in_range_of<of_type>(val_min));
|
|
||||||
CHECK_FALSE(value_in_range_of<of_type>(val_min2));
|
|
||||||
}
|
|
||||||
|
|
||||||
if (max_in_range)
|
|
||||||
{
|
|
||||||
CHECK(value_in_range_of<of_type>(val_max));
|
|
||||||
CHECK(value_in_range_of<of_type>(val_max2));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
CHECK_FALSE(value_in_range_of<of_type>(val_max));
|
|
||||||
CHECK_FALSE(value_in_range_of<of_type>(val_max2));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// NOLINTNEXTLINE(bugprone-throwing-static-initialization)
|
// NOLINTNEXTLINE(bugprone-throwing-static-initialization)
|
||||||
TEST_CASE_TEMPLATE_INVOKE(value_in_range_of_test, \
|
TEST_CASE_TEMPLATE_INVOKE(value_in_range_of_test, \
|
||||||
|
|||||||
@@ -21,85 +21,13 @@ using nlohmann::json;
|
|||||||
#include <string>
|
#include <string>
|
||||||
#include <vector>
|
#include <vector>
|
||||||
#include "make_test_data_available.hpp"
|
#include "make_test_data_available.hpp"
|
||||||
|
#include "round_trip_corpus.hpp"
|
||||||
#include "test_utils.hpp"
|
#include "test_utils.hpp"
|
||||||
|
#include "sax_countdown.hpp"
|
||||||
|
using utils::SaxCountdown;
|
||||||
|
|
||||||
namespace
|
namespace
|
||||||
{
|
{
|
||||||
class SaxCountdown
|
|
||||||
{
|
|
||||||
public:
|
|
||||||
explicit SaxCountdown(const int count) : events_left(count)
|
|
||||||
{}
|
|
||||||
|
|
||||||
bool null()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool boolean(bool /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_integer(json::number_integer_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_unsigned(json::number_unsigned_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_float(json::number_float_t /*unused*/, const std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool string(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool binary(std::vector<std::uint8_t>& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_object(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool key(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_object()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_array(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_array()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& /*unused*/) // NOLINT(readability-convert-member-functions-to-static)
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
int events_left = 0;
|
|
||||||
};
|
|
||||||
|
|
||||||
using bytes = std::vector<std::uint8_t>;
|
using bytes = std::vector<std::uint8_t>;
|
||||||
|
|
||||||
/// @return the string with the given bytes
|
/// @return the string with the given bytes
|
||||||
@@ -817,6 +745,39 @@ TEST_CASE("Parse BON8 directly from a file using iterator and sentinel")
|
|||||||
CHECK((parsed.is_object() || parsed.is_array()));
|
CHECK((parsed.is_object() || parsed.is_array()));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("BON8 round-trip invariants")
|
||||||
|
{
|
||||||
|
// This checks what the parse_bon8_fuzzer driver checks (see
|
||||||
|
// tests/src/fuzzer-parse_bon8.cpp), so that a regression shows up in CI
|
||||||
|
// rather than as an OSS-Fuzz report: anything from_bon8() returns (j1)
|
||||||
|
// can be serialized, parsed back (j2), and serialized again to reproduce
|
||||||
|
// the exact bytes. The stream-versus-contiguous input check the driver
|
||||||
|
// also performs is not covered here (see #5601).
|
||||||
|
for (const auto& j0 : utils::round_trip_corpus::values())
|
||||||
|
{
|
||||||
|
json j1;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
// turn the corpus value into a value as from_bon8() returns it
|
||||||
|
j1 = json::from_bon8(json::to_bon8(j0));
|
||||||
|
}
|
||||||
|
catch (const json::exception&)
|
||||||
|
{
|
||||||
|
// BON8 cannot represent an unsigned integer above INT64_MAX, and
|
||||||
|
// the fuzzer driver only ever sees values from_bon8() actually
|
||||||
|
// produced, so skip such corpus values here, too
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
INFO("j1 = " << j1.dump());
|
||||||
|
const std::vector<std::uint8_t> vec = json::to_bon8(j1);
|
||||||
|
json j2;
|
||||||
|
// anything the library writes must be parsable by the library
|
||||||
|
REQUIRE_NOTHROW(j2 = json::from_bon8(vec));
|
||||||
|
CHECK(json::to_bon8(j2) == vec);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("BON8 roundtrips" * doctest::skip())
|
TEST_CASE("BON8 roundtrips" * doctest::skip())
|
||||||
{
|
{
|
||||||
SECTION("input from HikoGUI")
|
SECTION("input from HikoGUI")
|
||||||
|
|||||||
@@ -17,7 +17,10 @@ using nlohmann::json;
|
|||||||
#include <sstream>
|
#include <sstream>
|
||||||
#include <vector>
|
#include <vector>
|
||||||
#include "make_test_data_available.hpp"
|
#include "make_test_data_available.hpp"
|
||||||
|
#include "round_trip_corpus.hpp"
|
||||||
#include "test_utils.hpp"
|
#include "test_utils.hpp"
|
||||||
|
#include "sax_countdown.hpp"
|
||||||
|
using utils::SaxCountdown;
|
||||||
|
|
||||||
namespace
|
namespace
|
||||||
{
|
{
|
||||||
@@ -861,83 +864,6 @@ TEST_CASE("BSON input/output_adapters")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
namespace
|
|
||||||
{
|
|
||||||
class SaxCountdown
|
|
||||||
{
|
|
||||||
public:
|
|
||||||
explicit SaxCountdown(const int count) : events_left(count)
|
|
||||||
{}
|
|
||||||
|
|
||||||
bool null()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool boolean(bool /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_integer(json::number_integer_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_unsigned(json::number_unsigned_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_float(json::number_float_t /*unused*/, const std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool string(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool binary(std::vector<std::uint8_t>& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_object(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool key(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_object()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_array(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_array()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& /*unused*/) // NOLINT(readability-convert-member-functions-to-static)
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
int events_left = 0;
|
|
||||||
};
|
|
||||||
} // namespace
|
|
||||||
|
|
||||||
TEST_CASE("Incomplete BSON Input")
|
TEST_CASE("Incomplete BSON Input")
|
||||||
{
|
{
|
||||||
@@ -1692,6 +1618,44 @@ TEST_CASE("Parse BSON directly from a file using iterator and sentinel")
|
|||||||
CHECK(parsed == expected);
|
CHECK(parsed == expected);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("BSON round-trip invariants")
|
||||||
|
{
|
||||||
|
// This checks what the parse_bson_fuzzer driver checks (see
|
||||||
|
// tests/src/fuzzer-parse_bson.cpp), so that a regression shows up in CI
|
||||||
|
// rather than as an OSS-Fuzz report: anything from_bson() returns (j1)
|
||||||
|
// can be serialized, parsed back (j2), and serialized again to reproduce
|
||||||
|
// the exact bytes. BSON only serializes objects, so non-object corpus
|
||||||
|
// values are skipped.
|
||||||
|
for (const auto& j0 : utils::round_trip_corpus::values())
|
||||||
|
{
|
||||||
|
if (!j0.is_object())
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
json j1;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
// turn the corpus value into a value as from_bson() returns it
|
||||||
|
j1 = json::from_bson(json::to_bson(j0));
|
||||||
|
}
|
||||||
|
catch (const json::exception&)
|
||||||
|
{
|
||||||
|
// the fuzzer driver only ever sees values from_bson() actually
|
||||||
|
// produced, so skip corpus values that do not survive the
|
||||||
|
// round trip here, too
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
INFO("j1 = " << j1.dump());
|
||||||
|
const std::vector<std::uint8_t> vec = json::to_bson(j1);
|
||||||
|
json j2;
|
||||||
|
// anything the library writes must be parsable by the library
|
||||||
|
REQUIRE_NOTHROW(j2 = json::from_bson(vec));
|
||||||
|
CHECK(json::to_bson(j2) == vec);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("BSON roundtrips" * doctest::skip())
|
TEST_CASE("BSON roundtrips" * doctest::skip())
|
||||||
{
|
{
|
||||||
SECTION("reference files")
|
SECTION("reference files")
|
||||||
|
|||||||
@@ -18,85 +18,11 @@ using nlohmann::json;
|
|||||||
#include <list>
|
#include <list>
|
||||||
#include <set>
|
#include <set>
|
||||||
#include "make_test_data_available.hpp"
|
#include "make_test_data_available.hpp"
|
||||||
|
#include "round_trip_corpus.hpp"
|
||||||
#include "test_utils.hpp"
|
#include "test_utils.hpp"
|
||||||
|
#include "sax_countdown.hpp"
|
||||||
|
using utils::SaxCountdown;
|
||||||
|
|
||||||
namespace
|
|
||||||
{
|
|
||||||
class SaxCountdown
|
|
||||||
{
|
|
||||||
public:
|
|
||||||
explicit SaxCountdown(const int count) : events_left(count)
|
|
||||||
{}
|
|
||||||
|
|
||||||
bool null()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool boolean(bool /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_integer(json::number_integer_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_unsigned(json::number_unsigned_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_float(json::number_float_t /*unused*/, const std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool string(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool binary(std::vector<std::uint8_t>& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_object(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool key(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_object()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_array(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_array()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& /*unused*/) // NOLINT(readability-convert-member-functions-to-static)
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
int events_left = 0;
|
|
||||||
};
|
|
||||||
} // namespace
|
|
||||||
|
|
||||||
TEST_CASE("CBOR")
|
TEST_CASE("CBOR")
|
||||||
{
|
{
|
||||||
@@ -2415,6 +2341,39 @@ TEST_CASE("issue #5405 - array reserve for definite-length CBOR arrays")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("CBOR round-trip invariants")
|
||||||
|
{
|
||||||
|
// This checks what the parse_cbor_fuzzer driver checks (see
|
||||||
|
// tests/src/fuzzer-parse_cbor.cpp), so that a regression shows up in CI
|
||||||
|
// rather than as an OSS-Fuzz report: anything from_cbor() returns (j1)
|
||||||
|
// can be serialized, parsed back (j2), and serialized again to reproduce
|
||||||
|
// the exact bytes.
|
||||||
|
for (const auto& j0 : utils::round_trip_corpus::values())
|
||||||
|
{
|
||||||
|
json j1;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
// turn the corpus value into a value as from_cbor() returns it
|
||||||
|
j1 = json::from_cbor(json::to_cbor(j0));
|
||||||
|
}
|
||||||
|
catch (const json::exception&)
|
||||||
|
{
|
||||||
|
// not every corpus value survives a CBOR round trip (e.g., a
|
||||||
|
// binary subtype is written with a tag the default tag handler
|
||||||
|
// then rejects); the fuzzer driver only ever sees values
|
||||||
|
// from_cbor() actually produced, so skip those here, too
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
INFO("j1 = " << j1.dump());
|
||||||
|
const std::vector<std::uint8_t> vec = json::to_cbor(j1);
|
||||||
|
json j2;
|
||||||
|
// anything the library writes must be parsable by the library
|
||||||
|
REQUIRE_NOTHROW(j2 = json::from_cbor(vec));
|
||||||
|
CHECK(json::to_cbor(j2) == vec);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("CBOR roundtrips" * doctest::skip())
|
TEST_CASE("CBOR roundtrips" * doctest::skip())
|
||||||
{
|
{
|
||||||
SECTION("input from flynn")
|
SECTION("input from flynn")
|
||||||
@@ -2774,33 +2733,21 @@ TEST_CASE("examples from RFC 8949 Appendix A")
|
|||||||
CHECK(json::to_cbor(json::parse("1.1")) == std::vector<uint8_t>({0xfb, 0x3f, 0xf1, 0x99, 0x99, 0x99, 0x99, 0x99, 0x9a}));
|
CHECK(json::to_cbor(json::parse("1.1")) == std::vector<uint8_t>({0xfb, 0x3f, 0xf1, 0x99, 0x99, 0x99, 0x99, 0x99, 0x9a}));
|
||||||
CHECK(json::parse("1.1") == json::from_cbor(std::vector<uint8_t>({0xfb, 0x3f, 0xf1, 0x99, 0x99, 0x99, 0x99, 0x99, 0x9a})));
|
CHECK(json::parse("1.1") == json::from_cbor(std::vector<uint8_t>({0xfb, 0x3f, 0xf1, 0x99, 0x99, 0x99, 0x99, 0x99, 0x9a})));
|
||||||
|
|
||||||
// half-precision float
|
// the writer never emits half-precision floats, so these can only be decoded, not encoded
|
||||||
//CHECK(json::to_cbor(json::parse("1.5")) == std::vector<uint8_t>({0xf9, 0x3e, 0x00}));
|
|
||||||
CHECK(json::parse("1.5") == json::from_cbor(std::vector<uint8_t>({0xf9, 0x3e, 0x00})));
|
CHECK(json::parse("1.5") == json::from_cbor(std::vector<uint8_t>({0xf9, 0x3e, 0x00})));
|
||||||
|
|
||||||
// half-precision float
|
|
||||||
//CHECK(json::to_cbor(json::parse("65504.0")) == std::vector<uint8_t>({0xf9, 0x7b, 0xff}));
|
|
||||||
CHECK(json::parse("65504.0") == json::from_cbor(std::vector<uint8_t>({0xf9, 0x7b, 0xff})));
|
CHECK(json::parse("65504.0") == json::from_cbor(std::vector<uint8_t>({0xf9, 0x7b, 0xff})));
|
||||||
|
|
||||||
//CHECK(json::to_cbor(json::parse("100000.0")) == std::vector<uint8_t>({0xfa, 0x47, 0xc3, 0x50, 0x00}));
|
CHECK(json::to_cbor(json::parse("100000.0")) == std::vector<uint8_t>({0xfa, 0x47, 0xc3, 0x50, 0x00}));
|
||||||
CHECK(json::parse("100000.0") == json::from_cbor(std::vector<uint8_t>({0xfa, 0x47, 0xc3, 0x50, 0x00})));
|
CHECK(json::parse("100000.0") == json::from_cbor(std::vector<uint8_t>({0xfa, 0x47, 0xc3, 0x50, 0x00})));
|
||||||
|
|
||||||
//CHECK(json::to_cbor(json::parse("3.4028234663852886e+38")) == std::vector<uint8_t>({0xfa, 0x7f, 0x7f, 0xff, 0xff}));
|
CHECK(json::to_cbor(json::parse("3.4028234663852886e+38")) == std::vector<uint8_t>({0xfa, 0x7f, 0x7f, 0xff, 0xff}));
|
||||||
CHECK(json::parse("3.4028234663852886e+38") == json::from_cbor(std::vector<uint8_t>({0xfa, 0x7f, 0x7f, 0xff, 0xff})));
|
CHECK(json::parse("3.4028234663852886e+38") == json::from_cbor(std::vector<uint8_t>({0xfa, 0x7f, 0x7f, 0xff, 0xff})));
|
||||||
|
|
||||||
CHECK(json::to_cbor(json::parse("1.0e+300")) == std::vector<uint8_t>({0xfb, 0x7e, 0x37, 0xe4, 0x3c, 0x88, 0x00, 0x75, 0x9c}));
|
CHECK(json::to_cbor(json::parse("1.0e+300")) == std::vector<uint8_t>({0xfb, 0x7e, 0x37, 0xe4, 0x3c, 0x88, 0x00, 0x75, 0x9c}));
|
||||||
CHECK(json::parse("1.0e+300") == json::from_cbor(std::vector<uint8_t>({0xfb, 0x7e, 0x37, 0xe4, 0x3c, 0x88, 0x00, 0x75, 0x9c})));
|
CHECK(json::parse("1.0e+300") == json::from_cbor(std::vector<uint8_t>({0xfb, 0x7e, 0x37, 0xe4, 0x3c, 0x88, 0x00, 0x75, 0x9c})));
|
||||||
|
|
||||||
// half-precision float
|
CHECK(json::parse("5.960464477539063e-8") == json::from_cbor(std::vector<uint8_t>({0xf9, 0x00, 0x01})));
|
||||||
//CHECK(json::to_cbor(json::parse("5.960464477539063e-8")) == std::vector<uint8_t>({0xf9, 0x00, 0x01}));
|
CHECK(json::parse("0.00006103515625") == json::from_cbor(std::vector<uint8_t>({0xf9, 0x04, 0x00})));
|
||||||
CHECK(json::parse("-4.0") == json::from_cbor(std::vector<uint8_t>({0xf9, 0xc4, 0x00})));
|
|
||||||
|
|
||||||
// half-precision float
|
|
||||||
//CHECK(json::to_cbor(json::parse("0.00006103515625")) == std::vector<uint8_t>({0xf9, 0x04, 0x00}));
|
|
||||||
CHECK(json::parse("-4.0") == json::from_cbor(std::vector<uint8_t>({0xf9, 0xc4, 0x00})));
|
|
||||||
|
|
||||||
// half-precision float
|
|
||||||
//CHECK(json::to_cbor(json::parse("-4.0")) == std::vector<uint8_t>({0xf9, 0xc4, 0x00}));
|
|
||||||
CHECK(json::parse("-4.0") == json::from_cbor(std::vector<uint8_t>({0xf9, 0xc4, 0x00})));
|
CHECK(json::parse("-4.0") == json::from_cbor(std::vector<uint8_t>({0xf9, 0xc4, 0x00})));
|
||||||
|
|
||||||
CHECK(json::to_cbor(json::parse("-4.1")) == std::vector<uint8_t>({0xfb, 0xc0, 0x10, 0x66, 0x66, 0x66, 0x66, 0x66, 0x66}));
|
CHECK(json::to_cbor(json::parse("-4.1")) == std::vector<uint8_t>({0xfb, 0xc0, 0x10, 0x66, 0x66, 0x66, 0x66, 0x66, 0x66}));
|
||||||
|
|||||||
@@ -143,11 +143,13 @@ class SaxEventLogger
|
|||||||
{
|
{
|
||||||
errored = true;
|
errored = true;
|
||||||
events.push_back("parse_error(" + std::to_string(position) + ")");
|
events.push_back("parse_error(" + std::to_string(position) + ")");
|
||||||
return false;
|
return recover;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::vector<std::string> events {}; // NOLINT(readability-redundant-member-init)
|
std::vector<std::string> events {}; // NOLINT(readability-redundant-member-init)
|
||||||
bool errored = false;
|
bool errored = false;
|
||||||
|
/// whether parse_error() asks the parser to recover from the error (see #3989)
|
||||||
|
bool recover = false;
|
||||||
};
|
};
|
||||||
|
|
||||||
class SaxCountdown : public nlohmann::json::json_sax_t
|
class SaxCountdown : public nlohmann::json::json_sax_t
|
||||||
@@ -2937,3 +2939,583 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
namespace
|
||||||
|
{
|
||||||
|
/// builds a value like json::parse(), but asks the parser to recover from
|
||||||
|
/// errors (see #3989), and checks that the events it receives are balanced
|
||||||
|
class RecoveringDomParser
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
explicit RecoveringDomParser(json& j, std::size_t max_errors_ = static_cast<std::size_t>(-1))
|
||||||
|
: dom(j, false)
|
||||||
|
, max_errors(max_errors_)
|
||||||
|
{}
|
||||||
|
|
||||||
|
bool null()
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.null();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool boolean(bool val)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.boolean(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_integer(json::number_integer_t val)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.number_integer(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_unsigned(json::number_unsigned_t val)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.number_unsigned(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_float(json::number_float_t val, const std::string& s)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.number_float(val, s);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool string(std::string& val)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.string(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool binary(json::binary_t& val)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.binary(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool start_object(std::size_t elements)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
stack.push_back('o');
|
||||||
|
return dom.start_object(elements);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool key(std::string& val)
|
||||||
|
{
|
||||||
|
++events;
|
||||||
|
if (stack.empty() || stack.back() != 'o')
|
||||||
|
{
|
||||||
|
well_formed = false;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
stack.back() = 'v';
|
||||||
|
return dom.key(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool end_object()
|
||||||
|
{
|
||||||
|
++events;
|
||||||
|
if (stack.empty() || stack.back() != 'o')
|
||||||
|
{
|
||||||
|
well_formed = false;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
stack.pop_back();
|
||||||
|
return dom.end_object();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool start_array(std::size_t elements)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
stack.push_back('a');
|
||||||
|
return dom.start_array(elements);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool end_array()
|
||||||
|
{
|
||||||
|
++events;
|
||||||
|
if (stack.empty() || stack.back() != 'a')
|
||||||
|
{
|
||||||
|
well_formed = false;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
stack.pop_back();
|
||||||
|
return dom.end_array();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& ex)
|
||||||
|
{
|
||||||
|
errors.emplace_back(ex.what());
|
||||||
|
return errors.size() < max_errors;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// whether the events were balanced and every key was followed by a value
|
||||||
|
bool balanced() const
|
||||||
|
{
|
||||||
|
return well_formed && stack.empty();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// builds the value
|
||||||
|
nlohmann::detail::json_sax_dom_parser<json> dom;
|
||||||
|
std::vector<std::string> errors {}; // NOLINT(readability-redundant-member-init)
|
||||||
|
std::size_t events = 0;
|
||||||
|
/// the open containers: 'a' for an array, 'o' for an object that expects
|
||||||
|
/// a key, 'v' for an object that expects the value of a key
|
||||||
|
std::vector<char> stack {}; // NOLINT(readability-redundant-member-init)
|
||||||
|
bool well_formed = true;
|
||||||
|
std::size_t max_errors;
|
||||||
|
|
||||||
|
private:
|
||||||
|
/// a value is passed: it is an array element, or the value of a key
|
||||||
|
void value()
|
||||||
|
{
|
||||||
|
++events;
|
||||||
|
if (!stack.empty())
|
||||||
|
{
|
||||||
|
if (stack.back() == 'v')
|
||||||
|
{
|
||||||
|
stack.back() = 'o';
|
||||||
|
}
|
||||||
|
else if (stack.back() == 'o')
|
||||||
|
{
|
||||||
|
// a value without a key
|
||||||
|
well_formed = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
};
|
||||||
|
|
||||||
|
struct RecoveryResult
|
||||||
|
{
|
||||||
|
json value;
|
||||||
|
std::vector<std::string> errors;
|
||||||
|
std::size_t events;
|
||||||
|
bool ok;
|
||||||
|
bool balanced;
|
||||||
|
};
|
||||||
|
|
||||||
|
template<typename InputType>
|
||||||
|
RecoveryResult parse_recovering(InputType&& input, const bool strict = true,
|
||||||
|
const bool ignore_comments = false, const bool ignore_trailing_commas = false)
|
||||||
|
{
|
||||||
|
json j;
|
||||||
|
RecoveringDomParser sax(j);
|
||||||
|
const bool ok = json::sax_parse(std::forward<InputType>(input), &sax, json::input_format_t::json,
|
||||||
|
strict, ignore_comments, ignore_trailing_commas);
|
||||||
|
return {j, sax.errors, sax.events, ok, sax.balanced()};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// stops after a number of events, but recovers from errors
|
||||||
|
class RecoveringCountdown : public SaxCountdown
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
using SaxCountdown::SaxCountdown;
|
||||||
|
|
||||||
|
bool parse_error(std::size_t /*position*/, const std::string& /*last_token*/, const json::exception& /*ex*/) override
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// a repaired input: the value it is repaired to, and the number of errors
|
||||||
|
struct Repair
|
||||||
|
{
|
||||||
|
const char* input;
|
||||||
|
const char* expected;
|
||||||
|
std::size_t errors;
|
||||||
|
};
|
||||||
|
} // namespace
|
||||||
|
|
||||||
|
TEST_CASE("parser error recovery (#3989)")
|
||||||
|
{
|
||||||
|
SECTION("repairs")
|
||||||
|
{
|
||||||
|
const std::vector<Repair> repairs =
|
||||||
|
{
|
||||||
|
// a missing separator is inserted
|
||||||
|
{"[1 2]", "[1,2]", 1},
|
||||||
|
{R"({"a":1 "b":2})", R"({"a":1,"b":2})", 1},
|
||||||
|
{R"({"a" 1})", R"({"a":1})", 1},
|
||||||
|
{"[1 tru 2]", "[1,null,2]", 2},
|
||||||
|
{R"({"a" "b": 1})", R"({"a":"b"})", 2},
|
||||||
|
|
||||||
|
// a missing value is null in an object; in an array, a ',' stands
|
||||||
|
// for null, while an array that ends there just ends
|
||||||
|
{R"({"a":})", R"({"a":null})", 1},
|
||||||
|
{R"({"a"})", R"({"a":null})", 1},
|
||||||
|
{R"({"a","b":1})", R"({"a":null,"b":1})", 1},
|
||||||
|
{"[1,,2]", "[1,null,2]", 1},
|
||||||
|
{"[,1]", "[null,1]", 1},
|
||||||
|
{"[1,]", "[1]", 1},
|
||||||
|
{"[1,2,3,]", "[1,2,3]", 1},
|
||||||
|
{R"({"a":1,})", R"({"a":1})", 1},
|
||||||
|
|
||||||
|
// a broken string keeps what can be read
|
||||||
|
{R"(["a\qb"])", R"(["aqb"])", 1},
|
||||||
|
{R"({"na\me":1})", R"({"name":1})", 1},
|
||||||
|
{"[\"\xFF\"]", R"(["\uFFFD"])", 1},
|
||||||
|
{"[\"a\xC3(\"]", R"(["a\uFFFD("])", 1},
|
||||||
|
{"[\"\xE2\x82\"]", R"(["\uFFFD"])", 1},
|
||||||
|
{"[\"\xC3\\\\\", 1]", R"(["\uFFFD\\",1])", 1},
|
||||||
|
{R"(["\u12"])", R"(["\uFFFD"])", 1},
|
||||||
|
{R"(["\u12G4"])", R"(["\uFFFDG4"])", 1},
|
||||||
|
{R"(["\uDC00x"])", R"(["\uFFFDx"])", 1},
|
||||||
|
{R"(["\uD800x"])", R"(["\uFFFDx"])", 1},
|
||||||
|
{R"(["\uD800\u0041"])", R"(["\uFFFDA"])", 1},
|
||||||
|
{R"(["\uD800\uD800\uDC00"])", R"(["\uFFFD\uD800\uDC00"])", 1},
|
||||||
|
{R"(["\uD800\uD800\uD800x"])", R"(["\uFFFD\uFFFD\uFFFDx"])", 1},
|
||||||
|
{
|
||||||
|
R"(["\uD800\"x", 1])", R"(["\uFFFD\"x",1])", 1
|
||||||
|
},
|
||||||
|
{R"(["\uD800\q"])", R"(["\uFFFDq"])", 1},
|
||||||
|
{"[\"a\tb\"]", R"(["a\tb"])", 1},
|
||||||
|
{R"(["a\qb\u0041\x"])", R"(["aqbAx"])", 1},
|
||||||
|
|
||||||
|
// a broken number keeps its longest valid prefix
|
||||||
|
{"[1.]", "[1]", 1},
|
||||||
|
{"[-2.]", "[-2]", 1},
|
||||||
|
{"[1.5e]", "[1.5]", 1},
|
||||||
|
{"[1e+]", "[1]", 1},
|
||||||
|
{"[1.x2, 3]", "[1,3]", 1},
|
||||||
|
|
||||||
|
// what cannot be read at all is null
|
||||||
|
{"[1,NaN,3]", "[1,null,3]", 1},
|
||||||
|
{"[tru]", "[null]", 1},
|
||||||
|
{"[-]", "[null]", 1},
|
||||||
|
{R"({"a":Infinity})", R"({"a":null})", 1},
|
||||||
|
|
||||||
|
// a stray token is dropped
|
||||||
|
{"[:1]", "[1]", 1},
|
||||||
|
{R"(["a":1])", R"(["a",1])", 1},
|
||||||
|
{R"({"a"::1})", R"({"a":1})", 1},
|
||||||
|
|
||||||
|
// a member that cannot be read is skipped
|
||||||
|
{R"({1:2,"b":3})", R"({"b":3})", 1},
|
||||||
|
{R"({"a":1 2})", R"({"a":1})", 1},
|
||||||
|
{R"({,"a":1})", R"({"a":1})", 1},
|
||||||
|
{R"({"a":1,,"b":2})", R"({"a":1,"b":2})", 1},
|
||||||
|
{"{a:1}", "{}", 1},
|
||||||
|
{R"({"a":1 [1,{"b":2}], "c":3})", R"({"a":1,"c":3})", 1},
|
||||||
|
{R"([{1}, "a"])", R"([{},"a"])", 1},
|
||||||
|
|
||||||
|
// a wrong closing bracket closes the innermost container
|
||||||
|
{R"({"a":[1,2}, "b":3})", R"({"a":[1,2],"b":3})", 1},
|
||||||
|
{R"([{"a":1], 2])", R"([{"a":1},2])", 1},
|
||||||
|
{"{]", "{}", 1},
|
||||||
|
{"[}", "[]", 1},
|
||||||
|
|
||||||
|
// the end of the input closes all containers
|
||||||
|
{R"({"a":[1,2)", R"({"a":[1,2]})", 1},
|
||||||
|
{"[", "[]", 1},
|
||||||
|
{"{", "{}", 1},
|
||||||
|
{R"({"a")", R"({"a":null})", 1},
|
||||||
|
{R"({"a":)", R"({"a":null})", 1},
|
||||||
|
{"[1,", "[1]", 1},
|
||||||
|
{"[[[1", "[[[1]]]", 1},
|
||||||
|
{
|
||||||
|
R"(["abc)", R"(["abc"])", 2
|
||||||
|
},
|
||||||
|
{"[1,tr", "[1,null]", 2},
|
||||||
|
{"\"abc", "\"abc\"", 1},
|
||||||
|
{"[\"ab\ncd\"]", R"(["ab",null,"]"])", 4},
|
||||||
|
|
||||||
|
// what comes before the top-level value is skipped
|
||||||
|
{")]}'\n{\"a\":1}", R"({"a":1})", 1},
|
||||||
|
{R"(data: {"a":1})", R"({"a":1})", 1},
|
||||||
|
{"\xEF\xBB[1]", "[1]", 1},
|
||||||
|
|
||||||
|
// what comes after it is an error that ends parsing
|
||||||
|
{R"({"a":1}})", R"({"a":1})", 1},
|
||||||
|
{"[1}]", "[1]", 2},
|
||||||
|
{"[1] [2]", "[1]", 1},
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const auto& repair : repairs)
|
||||||
|
{
|
||||||
|
CAPTURE(repair.input);
|
||||||
|
const auto result = parse_recovering(std::string(repair.input));
|
||||||
|
CHECK(!result.ok);
|
||||||
|
CHECK(result.balanced);
|
||||||
|
CHECK(result.value == json::parse(repair.expected));
|
||||||
|
CHECK(result.errors.size() == repair.errors);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("number overflow")
|
||||||
|
{
|
||||||
|
const auto result = parse_recovering(std::string("[1e999,-1e999]"));
|
||||||
|
CHECK(!result.ok);
|
||||||
|
CHECK(result.balanced);
|
||||||
|
CHECK(result.errors.size() == 2);
|
||||||
|
CHECK(result.errors[0] == "[json.exception.out_of_range.406] number overflow parsing '1e999'");
|
||||||
|
REQUIRE(result.value.size() == 2);
|
||||||
|
CHECK(result.value[0].is_number_float());
|
||||||
|
CHECK(result.value[0].get<double>() == std::numeric_limits<double>::infinity());
|
||||||
|
CHECK(result.value[1].get<double>() == -std::numeric_limits<double>::infinity());
|
||||||
|
|
||||||
|
// the SAX parser gets the number's text
|
||||||
|
SaxEventLogger logger;
|
||||||
|
logger.recover = true;
|
||||||
|
CHECK(!json::sax_parse("1e999", &logger));
|
||||||
|
CHECK(logger.events == std::vector<std::string>({"parse_error(5)", "number_float(1e999)"}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("nothing to recover")
|
||||||
|
{
|
||||||
|
for (const std::string s :
|
||||||
|
{
|
||||||
|
"", " ", "]", "tru", "NaN", ",:", "/* comment"
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(s);
|
||||||
|
const auto result = parse_recovering(s, true, true);
|
||||||
|
CHECK(!result.ok);
|
||||||
|
CHECK(result.balanced);
|
||||||
|
CHECK(result.events == 0);
|
||||||
|
CHECK(result.value == nullptr);
|
||||||
|
CHECK(result.errors.size() == 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("error messages")
|
||||||
|
{
|
||||||
|
// the first error is reported as without recovery
|
||||||
|
for (const std::string s :
|
||||||
|
{
|
||||||
|
"[1 2]", R"({"a":1 "b":2})", R"({"a" 1})", R"({"a":})", "[1,]", "[1.]",
|
||||||
|
R"(["a\qb"])", "[1e999]", "{1:2}", R"({"a":[1,2}})", "[1,", "[1] [2]", "{a:1}"
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(s);
|
||||||
|
const auto result = parse_recovering(s);
|
||||||
|
REQUIRE(!result.errors.empty());
|
||||||
|
json _;
|
||||||
|
CHECK_THROWS_WITH_STD_STR(_ = json::parse(s), result.errors.front());
|
||||||
|
}
|
||||||
|
|
||||||
|
// the token of an error begins where the previous error was
|
||||||
|
const auto result = parse_recovering(std::string("[tru, fals, nul]"));
|
||||||
|
CHECK(result.errors == std::vector<std::string>(
|
||||||
|
{
|
||||||
|
"[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: '[tru,'",
|
||||||
|
"[json.exception.parse_error.101] parse error at line 1, column 11: syntax error while parsing value - invalid literal; last read: ', fals,'",
|
||||||
|
"[json.exception.parse_error.101] parse error at line 1, column 16: syntax error while parsing value - invalid literal; last read: ', nul]'"
|
||||||
|
}));
|
||||||
|
CHECK(result.value == json::parse("[null,null,null]"));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("events")
|
||||||
|
{
|
||||||
|
// see #4522
|
||||||
|
SaxEventLogger logger;
|
||||||
|
logger.recover = true;
|
||||||
|
CHECK(!json::sax_parse(R"([{1}, "a"])", &logger));
|
||||||
|
CHECK(logger.events == std::vector<std::string>(
|
||||||
|
{
|
||||||
|
"start_array()", "start_object()", "parse_error(3)", "end_object()", "string(a)", "end_array()"
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("options")
|
||||||
|
{
|
||||||
|
SECTION("strict")
|
||||||
|
{
|
||||||
|
const auto result = parse_recovering(std::string("[1 2] [3]"), false);
|
||||||
|
CHECK(!result.ok);
|
||||||
|
CHECK(result.value == json::parse("[1,2]"));
|
||||||
|
CHECK(result.errors.size() == 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("ignore_trailing_commas")
|
||||||
|
{
|
||||||
|
for (const std::string s :
|
||||||
|
{
|
||||||
|
"[1,]", R"({"a":1,})", "[[1,],]"
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(s);
|
||||||
|
const auto result = parse_recovering(s, true, false, true);
|
||||||
|
CHECK(result.ok);
|
||||||
|
CHECK(result.errors.empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
auto result = parse_recovering(std::string("[1,,]"), true, false, true);
|
||||||
|
CHECK(result.value == json::parse("[1,null]"));
|
||||||
|
CHECK(result.errors.size() == 1);
|
||||||
|
|
||||||
|
result = parse_recovering(std::string(R"({"a":1,,})"), true, false, true);
|
||||||
|
CHECK(result.value == json::parse(R"({"a":1})"));
|
||||||
|
CHECK(result.errors.size() == 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("ignore_comments")
|
||||||
|
{
|
||||||
|
auto result = parse_recovering(std::string("[1 /* one */ 2]"), true, true);
|
||||||
|
CHECK(result.value == json::parse("[1,2]"));
|
||||||
|
CHECK(result.errors.size() == 1);
|
||||||
|
|
||||||
|
// a comment that is not closed runs to the end of the input, which
|
||||||
|
// is not reported again
|
||||||
|
result = parse_recovering(std::string("[1, 2 /* unterminated"), true, true);
|
||||||
|
CHECK(result.balanced);
|
||||||
|
CHECK(result.value == json::parse("[1,2]"));
|
||||||
|
CHECK(result.errors.size() == 1);
|
||||||
|
|
||||||
|
// a '/' that does not begin a comment is garbage
|
||||||
|
result = parse_recovering(std::string("[1, /x, 2]"), true, true);
|
||||||
|
CHECK(result.balanced);
|
||||||
|
CHECK(result.value == json::parse("[1,null,2]"));
|
||||||
|
CHECK(result.errors.size() == 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("null bytes")
|
||||||
|
{
|
||||||
|
// a null byte ends the input, unless JSON_STRICT_NUL_HANDLING is set
|
||||||
|
const auto result = parse_recovering(std::string("[1,\0x", 5));
|
||||||
|
CHECK(result.balanced);
|
||||||
|
CHECK(!result.ok);
|
||||||
|
#ifdef JSON_TEST_STRICT_NUL_HANDLING_ENABLED
|
||||||
|
CHECK(result.value == json::parse("[1,null]"));
|
||||||
|
#else
|
||||||
|
CHECK(result.value == json::parse("[1]"));
|
||||||
|
CHECK(result.errors.size() == 1);
|
||||||
|
#endif
|
||||||
|
|
||||||
|
const auto in_string = parse_recovering(std::string("[\"a\0b\"]", 7));
|
||||||
|
CHECK(in_string.balanced);
|
||||||
|
#ifdef JSON_TEST_STRICT_NUL_HANDLING_ENABLED
|
||||||
|
CHECK(in_string.value == json::array({std::string("a\0b", 3)}));
|
||||||
|
#else
|
||||||
|
CHECK(in_string.value == json::parse(R"(["a"])"));
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("the SAX parser stops recovering")
|
||||||
|
{
|
||||||
|
json j;
|
||||||
|
RecoveringDomParser sax(j, 2);
|
||||||
|
CHECK(!json::sax_parse("[1 2 3 4 5]", &sax));
|
||||||
|
CHECK(sax.errors.size() == 2);
|
||||||
|
|
||||||
|
// an error at a delimiter that an invalid token consumed is reported
|
||||||
|
// to the SAX parser, too
|
||||||
|
json j2;
|
||||||
|
RecoveringDomParser sax2(j2, 2);
|
||||||
|
CHECK(!json::sax_parse("[tru}, 1]", &sax2));
|
||||||
|
CHECK(sax2.errors.size() == 2);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("an event stops parsing during a repair")
|
||||||
|
{
|
||||||
|
// start_object() and key() are passed, then null() for the missing
|
||||||
|
// value returns false
|
||||||
|
RecoveringCountdown countdown(2);
|
||||||
|
CHECK(!json::sax_parse(R"({"a":})", &countdown));
|
||||||
|
|
||||||
|
// the end of the input: end_array() for the second array returns false
|
||||||
|
RecoveringCountdown countdown2(4);
|
||||||
|
CHECK(!json::sax_parse("[[1", &countdown2));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("input adapters")
|
||||||
|
{
|
||||||
|
// the lexer reads contiguous and streaming input differently, and it
|
||||||
|
// puts back a character that ended an invalid token
|
||||||
|
for (const std::string s :
|
||||||
|
{
|
||||||
|
"[1 2]", "[tru}, 1]", R"({"a" "b\q", "c":[1.x, 2}})", "[\"\xFF\xC3(\", -, 1e+]", "{a:1,\"b\":2", ")]}' [1]"
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(s);
|
||||||
|
const auto reference = parse_recovering(s);
|
||||||
|
CHECK(reference.balanced);
|
||||||
|
|
||||||
|
const auto from_c_string = parse_recovering(s.c_str());
|
||||||
|
CHECK(from_c_string.value == reference.value);
|
||||||
|
CHECK(from_c_string.errors == reference.errors);
|
||||||
|
|
||||||
|
const std::list<char> l(s.begin(), s.end());
|
||||||
|
json j;
|
||||||
|
RecoveringDomParser sax(j);
|
||||||
|
CHECK(!json::sax_parse(l.begin(), l.end(), &sax));
|
||||||
|
CHECK(j == reference.value);
|
||||||
|
CHECK(sax.errors == reference.errors);
|
||||||
|
|
||||||
|
std::istringstream ss(s);
|
||||||
|
const auto from_stream = parse_recovering(ss);
|
||||||
|
CHECK(from_stream.value == reference.value);
|
||||||
|
CHECK(from_stream.errors == reference.errors);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("long runs of errors")
|
||||||
|
{
|
||||||
|
// no error may copy all the input read before it
|
||||||
|
const auto closing = parse_recovering("[" + std::string(100000, '}'));
|
||||||
|
CHECK(closing.balanced);
|
||||||
|
CHECK(closing.value == json::array());
|
||||||
|
|
||||||
|
const auto garbage = parse_recovering("[" + std::string(100000, 'x') + "]");
|
||||||
|
CHECK(garbage.balanced);
|
||||||
|
CHECK(garbage.errors.size() == 1);
|
||||||
|
|
||||||
|
const auto commas = parse_recovering("{" + std::string(100000, ',') + "}");
|
||||||
|
CHECK(commas.balanced);
|
||||||
|
CHECK(commas.value == json::object());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("mutations of valid input")
|
||||||
|
{
|
||||||
|
// whatever the input, the events are balanced, every error is reported
|
||||||
|
// at most once, and valid input is parsed as usual
|
||||||
|
const std::vector<std::string> documents =
|
||||||
|
{
|
||||||
|
R"({"name": "value", "list": [1, -2.5, true, null, {"x": [[]]}], "e": "\u00e9"})",
|
||||||
|
R"([{"a": [1, 2, {"b": "c"}]}, [], {}, "\ud83d\ude00", 1e10])",
|
||||||
|
"{\"\xC3\xA9\": \"\xF0\x9F\x98\x80\"}",
|
||||||
|
R"( {"k" : [ "v" , 0 ] } )",
|
||||||
|
};
|
||||||
|
// each character that can be inserted, including a null byte
|
||||||
|
const std::string insertions("[]{},:\"x\\\0\xFF", 11);
|
||||||
|
|
||||||
|
std::vector<std::string> inputs;
|
||||||
|
for (const auto& doc : documents)
|
||||||
|
{
|
||||||
|
for (std::size_t i = 0; i <= doc.size(); ++i)
|
||||||
|
{
|
||||||
|
inputs.push_back(doc.substr(0, i));
|
||||||
|
if (i < doc.size())
|
||||||
|
{
|
||||||
|
inputs.push_back(doc.substr(0, i) + doc.substr(i + 1));
|
||||||
|
}
|
||||||
|
for (const char c : insertions)
|
||||||
|
{
|
||||||
|
inputs.push_back(doc.substr(0, i) + c + doc.substr(i));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for (const auto& s : inputs)
|
||||||
|
{
|
||||||
|
CAPTURE(s);
|
||||||
|
const auto result = parse_recovering(s);
|
||||||
|
CHECK(result.balanced);
|
||||||
|
CHECK(result.errors.size() <= s.size() + 1);
|
||||||
|
CHECK(result.events <= (4 * s.size()) + 4);
|
||||||
|
if (json::accept(s))
|
||||||
|
{
|
||||||
|
CHECK(result.ok);
|
||||||
|
CHECK(result.errors.empty());
|
||||||
|
CHECK(result.value == json::parse(s));
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
CHECK(!result.ok);
|
||||||
|
CHECK(!result.errors.empty());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -1380,21 +1380,22 @@ TEST_CASE("value conversion")
|
|||||||
|
|
||||||
SECTION("std::map")
|
SECTION("std::map")
|
||||||
{
|
{
|
||||||
j1.get<std::map<std::string, int>>();
|
CHECK(j1.get<std::map<std::string, int>>() == (std::map<std::string, int> {{"one", 1}, {"two", 2}, {"three", 3}}));
|
||||||
j2.get<std::map<std::string, unsigned int>>();
|
CHECK(j2.get<std::map<std::string, unsigned int>>() == (std::map<std::string, unsigned int> {{"one", 1u}, {"two", 2u}, {"three", 3u}}));
|
||||||
j3.get<std::map<std::string, double>>();
|
CHECK(j3.get<std::map<std::string, double>>() == (std::map<std::string, double> {{"one", 1.1}, {"two", 2.2}, {"three", 3.3}}));
|
||||||
j4.get<std::map<std::string, bool>>();
|
CHECK(j4.get<std::map<std::string, bool>>() == (std::map<std::string, bool> {{"one", true}, {"two", false}, {"three", true}}));
|
||||||
j5.get<std::map<std::string, std::string>>();
|
CHECK(j5.get<std::map<std::string, std::string>>() == (std::map<std::string, std::string> {{"one", "eins"}, {"two", "zwei"}, {"three", "drei"}}));
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::unordered_map")
|
SECTION("std::unordered_map")
|
||||||
{
|
{
|
||||||
j1.get<std::unordered_map<std::string, int>>();
|
CHECK(j1.get<std::unordered_map<std::string, int>>() == (std::unordered_map<std::string, int> {{"one", 1}, {"two", 2}, {"three", 3}}));
|
||||||
j2.get<std::unordered_map<std::string, unsigned int>>();
|
CHECK(j2.get<std::unordered_map<std::string, unsigned int>>() == (std::unordered_map<std::string, unsigned int> {{"one", 1u}, {"two", 2u}, {"three", 3u}}));
|
||||||
j3.get<std::unordered_map<std::string, double>>();
|
CHECK(j3.get<std::unordered_map<std::string, double>>() == (std::unordered_map<std::string, double> {{"one", 1.1}, {"two", 2.2}, {"three", 3.3}}));
|
||||||
j4.get<std::unordered_map<std::string, bool>>();
|
CHECK(j4.get<std::unordered_map<std::string, bool>>() == (std::unordered_map<std::string, bool> {{"one", true}, {"two", false}, {"three", true}}));
|
||||||
j5.get<std::unordered_map<std::string, std::string>>();
|
const auto m5 = j5.get<std::unordered_map<std::string, std::string>>();
|
||||||
// CHECK(m5["one"] == "eins");
|
CHECK(m5 == (std::unordered_map<std::string, std::string> {{"one", "eins"}, {"two", "zwei"}, {"three", "drei"}}));
|
||||||
|
CHECK(m5.at("one") == "eins");
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("reserve is called on containers that support it (#5406)")
|
SECTION("reserve is called on containers that support it (#5406)")
|
||||||
@@ -1430,22 +1431,24 @@ TEST_CASE("value conversion")
|
|||||||
|
|
||||||
SECTION("std::multimap")
|
SECTION("std::multimap")
|
||||||
{
|
{
|
||||||
j1.get<std::multimap<std::string, int>>();
|
CHECK(j1.get<std::multimap<std::string, int>>() == (std::multimap<std::string, int> {{"one", 1}, {"two", 2}, {"three", 3}}));
|
||||||
j2.get<std::multimap<std::string, unsigned int>>();
|
CHECK(j2.get<std::multimap<std::string, unsigned int>>() == (std::multimap<std::string, unsigned int> {{"one", 1u}, {"two", 2u}, {"three", 3u}}));
|
||||||
j3.get<std::multimap<std::string, double>>();
|
CHECK(j3.get<std::multimap<std::string, double>>() == (std::multimap<std::string, double> {{"one", 1.1}, {"two", 2.2}, {"three", 3.3}}));
|
||||||
j4.get<std::multimap<std::string, bool>>();
|
CHECK(j4.get<std::multimap<std::string, bool>>() == (std::multimap<std::string, bool> {{"one", true}, {"two", false}, {"three", true}}));
|
||||||
j5.get<std::multimap<std::string, std::string>>();
|
const auto m5 = j5.get<std::multimap<std::string, std::string>>();
|
||||||
// CHECK(m5["one"] == "eins");
|
CHECK(m5 == (std::multimap<std::string, std::string> {{"one", "eins"}, {"two", "zwei"}, {"three", "drei"}}));
|
||||||
|
CHECK(m5.find("one")->second == "eins");
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::unordered_multimap")
|
SECTION("std::unordered_multimap")
|
||||||
{
|
{
|
||||||
j1.get<std::unordered_multimap<std::string, int>>();
|
CHECK(j1.get<std::unordered_multimap<std::string, int>>() == (std::unordered_multimap<std::string, int> {{"one", 1}, {"two", 2}, {"three", 3}}));
|
||||||
j2.get<std::unordered_multimap<std::string, unsigned int>>();
|
CHECK(j2.get<std::unordered_multimap<std::string, unsigned int>>() == (std::unordered_multimap<std::string, unsigned int> {{"one", 1u}, {"two", 2u}, {"three", 3u}}));
|
||||||
j3.get<std::unordered_multimap<std::string, double>>();
|
CHECK(j3.get<std::unordered_multimap<std::string, double>>() == (std::unordered_multimap<std::string, double> {{"one", 1.1}, {"two", 2.2}, {"three", 3.3}}));
|
||||||
j4.get<std::unordered_multimap<std::string, bool>>();
|
CHECK(j4.get<std::unordered_multimap<std::string, bool>>() == (std::unordered_multimap<std::string, bool> {{"one", true}, {"two", false}, {"three", true}}));
|
||||||
j5.get<std::unordered_multimap<std::string, std::string>>();
|
const auto m5 = j5.get<std::unordered_multimap<std::string, std::string>>();
|
||||||
// CHECK(m5["one"] == "eins");
|
CHECK(m5 == (std::unordered_multimap<std::string, std::string> {{"one", "eins"}, {"two", "zwei"}, {"three", "drei"}}));
|
||||||
|
CHECK(m5.find("one")->second == "eins");
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("exception in case of a non-object type")
|
SECTION("exception in case of a non-object type")
|
||||||
@@ -1466,29 +1469,30 @@ TEST_CASE("value conversion")
|
|||||||
|
|
||||||
SECTION("std::list")
|
SECTION("std::list")
|
||||||
{
|
{
|
||||||
j1.get<std::list<int>>();
|
CHECK(j1.get<std::list<int>>() == (std::list<int> {1, 2, 3, 4}));
|
||||||
j2.get<std::list<unsigned int>>();
|
CHECK(j2.get<std::list<unsigned int>>() == (std::list<unsigned int> {1u, 2u, 3u, 4u}));
|
||||||
j3.get<std::list<double>>();
|
CHECK(j3.get<std::list<double>>() == (std::list<double> {1.2, 2.3, 3.4, 4.5}));
|
||||||
j4.get<std::list<bool>>();
|
CHECK(j4.get<std::list<bool>>() == (std::list<bool> {true, false, true}));
|
||||||
j5.get<std::list<std::string>>();
|
CHECK(j5.get<std::list<std::string>>() == (std::list<std::string> {"one", "two", "three"}));
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::forward_list")
|
SECTION("std::forward_list")
|
||||||
{
|
{
|
||||||
j1.get<std::forward_list<int>>();
|
CHECK(j1.get<std::forward_list<int>>() == (std::forward_list<int> {1, 2, 3, 4}));
|
||||||
j2.get<std::forward_list<unsigned int>>();
|
CHECK(j2.get<std::forward_list<unsigned int>>() == (std::forward_list<unsigned int> {1u, 2u, 3u, 4u}));
|
||||||
j3.get<std::forward_list<double>>();
|
CHECK(j3.get<std::forward_list<double>>() == (std::forward_list<double> {1.2, 2.3, 3.4, 4.5}));
|
||||||
j4.get<std::forward_list<bool>>();
|
CHECK(j4.get<std::forward_list<bool>>() == (std::forward_list<bool> {true, false, true}));
|
||||||
j5.get<std::forward_list<std::string>>();
|
CHECK(j5.get<std::forward_list<std::string>>() == (std::forward_list<std::string> {"one", "two", "three"}));
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::array")
|
SECTION("std::array")
|
||||||
{
|
{
|
||||||
j1.get<std::array<int, 4>>();
|
CHECK(j1.get<std::array<int, 4>>() == (std::array<int, 4> {{1, 2, 3, 4}}));
|
||||||
j2.get<std::array<unsigned int, 3>>();
|
// only the first 3 elements of j2 are converted, since the target array is smaller
|
||||||
j3.get<std::array<double, 4>>();
|
CHECK(j2.get<std::array<unsigned int, 3>>() == (std::array<unsigned int, 3> {{1u, 2u, 3u}}));
|
||||||
j4.get<std::array<bool, 3>>();
|
CHECK(j3.get<std::array<double, 4>>() == (std::array<double, 4> {{1.2, 2.3, 3.4, 4.5}}));
|
||||||
j5.get<std::array<std::string, 3>>();
|
CHECK(j4.get<std::array<bool, 3>>() == (std::array<bool, 3> {{true, false, true}}));
|
||||||
|
CHECK(j5.get<std::array<std::string, 3>>() == (std::array<std::string, 3> {{"one", "two", "three"}}));
|
||||||
|
|
||||||
SECTION("std::array is larger than JSON")
|
SECTION("std::array is larger than JSON")
|
||||||
{
|
{
|
||||||
@@ -1508,47 +1512,53 @@ TEST_CASE("value conversion")
|
|||||||
|
|
||||||
SECTION("std::valarray")
|
SECTION("std::valarray")
|
||||||
{
|
{
|
||||||
j1.get<std::valarray<int>>();
|
// valarray has no operator== that returns bool, so compare via a vector copy
|
||||||
j2.get<std::valarray<unsigned int>>();
|
const auto v1 = j1.get<std::valarray<int>>();
|
||||||
j3.get<std::valarray<double>>();
|
CHECK((std::vector<int>(std::begin(v1), std::end(v1)) == std::vector<int> {1, 2, 3, 4}));
|
||||||
j4.get<std::valarray<bool>>();
|
const auto v2 = j2.get<std::valarray<unsigned int>>();
|
||||||
j5.get<std::valarray<std::string>>();
|
CHECK((std::vector<unsigned int>(std::begin(v2), std::end(v2)) == std::vector<unsigned int> {1u, 2u, 3u, 4u}));
|
||||||
|
const auto v3 = j3.get<std::valarray<double>>();
|
||||||
|
CHECK((std::vector<double>(std::begin(v3), std::end(v3)) == std::vector<double> {1.2, 2.3, 3.4, 4.5}));
|
||||||
|
const auto v4 = j4.get<std::valarray<bool>>();
|
||||||
|
CHECK((std::vector<bool>(std::begin(v4), std::end(v4)) == std::vector<bool> {true, false, true}));
|
||||||
|
const auto v5 = j5.get<std::valarray<std::string>>();
|
||||||
|
CHECK((std::vector<std::string>(std::begin(v5), std::end(v5)) == std::vector<std::string> {"one", "two", "three"}));
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::vector")
|
SECTION("std::vector")
|
||||||
{
|
{
|
||||||
j1.get<std::vector<int>>();
|
CHECK(j1.get<std::vector<int>>() == (std::vector<int> {1, 2, 3, 4}));
|
||||||
j2.get<std::vector<unsigned int>>();
|
CHECK(j2.get<std::vector<unsigned int>>() == (std::vector<unsigned int> {1u, 2u, 3u, 4u}));
|
||||||
j3.get<std::vector<double>>();
|
CHECK(j3.get<std::vector<double>>() == (std::vector<double> {1.2, 2.3, 3.4, 4.5}));
|
||||||
j4.get<std::vector<bool>>();
|
CHECK(j4.get<std::vector<bool>>() == (std::vector<bool> {true, false, true}));
|
||||||
j5.get<std::vector<std::string>>();
|
CHECK(j5.get<std::vector<std::string>>() == (std::vector<std::string> {"one", "two", "three"}));
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::deque")
|
SECTION("std::deque")
|
||||||
{
|
{
|
||||||
j1.get<std::deque<int>>();
|
CHECK(j1.get<std::deque<int>>() == (std::deque<int> {1, 2, 3, 4}));
|
||||||
j2.get<std::deque<unsigned int>>();
|
CHECK(j2.get<std::deque<unsigned int>>() == (std::deque<unsigned int> {1u, 2u, 3u, 4u}));
|
||||||
j2.get<std::deque<double>>();
|
CHECK(j3.get<std::deque<double>>() == (std::deque<double> {1.2, 2.3, 3.4, 4.5}));
|
||||||
j4.get<std::deque<bool>>();
|
CHECK(j4.get<std::deque<bool>>() == (std::deque<bool> {true, false, true}));
|
||||||
j5.get<std::deque<std::string>>();
|
CHECK(j5.get<std::deque<std::string>>() == (std::deque<std::string> {"one", "two", "three"}));
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::set")
|
SECTION("std::set")
|
||||||
{
|
{
|
||||||
j1.get<std::set<int>>();
|
CHECK(j1.get<std::set<int>>() == (std::set<int> {1, 2, 3, 4}));
|
||||||
j2.get<std::set<unsigned int>>();
|
CHECK(j2.get<std::set<unsigned int>>() == (std::set<unsigned int> {1u, 2u, 3u, 4u}));
|
||||||
j3.get<std::set<double>>();
|
CHECK(j3.get<std::set<double>>() == (std::set<double> {1.2, 2.3, 3.4, 4.5}));
|
||||||
j4.get<std::set<bool>>();
|
CHECK(j4.get<std::set<bool>>() == (std::set<bool> {true, false, true}));
|
||||||
j5.get<std::set<std::string>>();
|
CHECK(j5.get<std::set<std::string>>() == (std::set<std::string> {"one", "two", "three"}));
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::unordered_set")
|
SECTION("std::unordered_set")
|
||||||
{
|
{
|
||||||
j1.get<std::unordered_set<int>>();
|
CHECK(j1.get<std::unordered_set<int>>() == (std::unordered_set<int> {1, 2, 3, 4}));
|
||||||
j2.get<std::unordered_set<unsigned int>>();
|
CHECK(j2.get<std::unordered_set<unsigned int>>() == (std::unordered_set<unsigned int> {1u, 2u, 3u, 4u}));
|
||||||
j3.get<std::unordered_set<double>>();
|
CHECK(j3.get<std::unordered_set<double>>() == (std::unordered_set<double> {1.2, 2.3, 3.4, 4.5}));
|
||||||
j4.get<std::unordered_set<bool>>();
|
CHECK(j4.get<std::unordered_set<bool>>() == (std::unordered_set<bool> {true, false, true}));
|
||||||
j5.get<std::unordered_set<std::string>>();
|
CHECK(j5.get<std::unordered_set<std::string>>() == (std::unordered_set<std::string> {"one", "two", "three"}));
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::map (array of pairs)")
|
SECTION("std::map (array of pairs)")
|
||||||
|
|||||||
@@ -28,11 +28,6 @@ using nlohmann::json;
|
|||||||
#include <string>
|
#include <string>
|
||||||
#include <valarray>
|
#include <valarray>
|
||||||
|
|
||||||
#if defined(_WIN32)
|
|
||||||
#define NOMINMAX
|
|
||||||
#include <windows.h> // for GetACP()
|
|
||||||
#endif
|
|
||||||
|
|
||||||
namespace
|
namespace
|
||||||
{
|
{
|
||||||
struct SaxEventLogger : public nlohmann::json_sax<json>
|
struct SaxEventLogger : public nlohmann::json_sax<json>
|
||||||
@@ -228,24 +223,6 @@ class proxy_iterator
|
|||||||
iterator* m_it = nullptr;
|
iterator* m_it = nullptr;
|
||||||
};
|
};
|
||||||
|
|
||||||
// JSON_HAS_CPP_20
|
|
||||||
#if defined(__cpp_char8_t)
|
|
||||||
bool check_utf8()
|
|
||||||
{
|
|
||||||
#if defined(_WIN32)
|
|
||||||
// Runtime check of the active ANSI code page
|
|
||||||
// 65001 == UTF-8
|
|
||||||
return GetACP() == 65001;
|
|
||||||
#elif defined(__ICC) || defined(__INTEL_COMPILER)
|
|
||||||
// classic Intel ICC does not encode narrow string literals containing
|
|
||||||
// non-ASCII source characters as UTF-8, so comparing a decoded u8 literal
|
|
||||||
// against a narrow string literal containing the same characters fails
|
|
||||||
return false;
|
|
||||||
#else
|
|
||||||
return true;
|
|
||||||
#endif
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
TEST_CASE("deserialization")
|
TEST_CASE("deserialization")
|
||||||
@@ -1328,14 +1305,15 @@ TEST_CASE("deserialization")
|
|||||||
CHECK(j1["key"] == "value");
|
CHECK(j1["key"] == "value");
|
||||||
CHECK(j1["num"] == 42);
|
CHECK(j1["num"] == 42);
|
||||||
|
|
||||||
// UTF-8 prefixed literal (C++20 and later);
|
// UTF-8 prefixed literal (C++20 and later); the emoji is written as a
|
||||||
// MSVC may not set /utf-8, so we need to check
|
// \U escape rather than a raw multibyte character so this does not
|
||||||
if (check_utf8())
|
// depend on the compiler's source-file encoding (e.g., MSVC without
|
||||||
{
|
// /utf-8, or classic ICC, which does not encode non-ASCII narrow
|
||||||
const auto j2 = u8R"({"emoji": "😀", "msg": "hello"})"_json;
|
// string literals as UTF-8 - compare against a \x-escaped expectation
|
||||||
CHECK(j2["emoji"] == "😀");
|
// for the same reason)
|
||||||
CHECK(j2["msg"] == "hello");
|
const auto j2 = u8"{\"emoji\": \"\U0001F600\", \"msg\": \"hello\"}"_json;
|
||||||
}
|
CHECK(j2["emoji"] == "\xF0\x9F\x98\x80");
|
||||||
|
CHECK(j2["msg"] == "hello");
|
||||||
|
|
||||||
const auto j3 = u8R"({"key": "value", "num": 42})"_json;
|
const auto j3 = u8R"({"key": "value", "num": 42})"_json;
|
||||||
CHECK(j3["key"] == "value");
|
CHECK(j3["key"] == "value");
|
||||||
|
|||||||
@@ -893,8 +893,6 @@ TEST_CASE("iterators 2")
|
|||||||
CHECK(std::ranges::input_range<items_type>);
|
CHECK(std::ranges::input_range<items_type>);
|
||||||
}
|
}
|
||||||
|
|
||||||
// libstdc++ algorithms don't work with Clang 15 (04/2022)
|
|
||||||
#if !DOCTEST_CLANG || (DOCTEST_CLANG && defined(__GLIBCXX__))
|
|
||||||
SECTION("algorithms")
|
SECTION("algorithms")
|
||||||
{
|
{
|
||||||
SECTION("copy")
|
SECTION("copy")
|
||||||
@@ -929,11 +927,7 @@ TEST_CASE("iterators 2")
|
|||||||
CHECK(*it == 2);
|
CHECK(*it == 2);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
#endif
|
|
||||||
|
|
||||||
// libstdc++ views don't work with Clang 15 (04/2022)
|
|
||||||
// libc++ hides limited ranges implementation behind guard macro
|
|
||||||
#if !(DOCTEST_CLANG && (defined(__GLIBCXX__) || defined(_LIBCPP_HAS_NO_INCOMPLETE_RANGES)))
|
|
||||||
SECTION("views")
|
SECTION("views")
|
||||||
{
|
{
|
||||||
SECTION("reverse")
|
SECTION("reverse")
|
||||||
@@ -966,7 +960,6 @@ TEST_CASE("iterators 2")
|
|||||||
CHECK(j_transformed == j_expected);
|
CHECK(j_transformed == j_expected);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
#endif
|
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -15,8 +15,65 @@ using nlohmann::json;
|
|||||||
#endif
|
#endif
|
||||||
|
|
||||||
#include <fstream>
|
#include <fstream>
|
||||||
|
#include <string>
|
||||||
|
#include <vector>
|
||||||
#include "make_test_data_available.hpp"
|
#include "make_test_data_available.hpp"
|
||||||
|
|
||||||
|
namespace
|
||||||
|
{
|
||||||
|
// alternating objects and arrays nested `depth` levels deep, with members that
|
||||||
|
// depend on `variant` at some levels, so diffing two variants yields
|
||||||
|
// operations on many levels: replacing the innermost value, adding, removing,
|
||||||
|
// and (for ordered_json) reordering members, and changing array lengths
|
||||||
|
template<typename BasicJsonType>
|
||||||
|
BasicJsonType nested(const std::size_t depth, const int variant)
|
||||||
|
{
|
||||||
|
BasicJsonType value = variant;
|
||||||
|
for (std::size_t i = 0; i < depth; ++i)
|
||||||
|
{
|
||||||
|
if (i % 2 == 0)
|
||||||
|
{
|
||||||
|
BasicJsonType object = BasicJsonType::object();
|
||||||
|
if ((i + static_cast<std::size_t>(variant)) % 7 == 0)
|
||||||
|
{
|
||||||
|
object["x"] = i;
|
||||||
|
}
|
||||||
|
if (variant == 2 && i % 11 == 0)
|
||||||
|
{
|
||||||
|
object["z"] = "z";
|
||||||
|
}
|
||||||
|
object["a"] = std::move(value);
|
||||||
|
if (variant == 1 && i % 5 == 0)
|
||||||
|
{
|
||||||
|
object["y"] = 1;
|
||||||
|
}
|
||||||
|
value = std::move(object);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
BasicJsonType array = BasicJsonType::array({std::move(value)});
|
||||||
|
if ((i + static_cast<std::size_t>(variant)) % 3 == 0)
|
||||||
|
{
|
||||||
|
array.push_back(i);
|
||||||
|
}
|
||||||
|
value = std::move(array);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
// a path of `depth` reference tokens, as nested() nests its values
|
||||||
|
std::string nested_path(const std::size_t depth)
|
||||||
|
{
|
||||||
|
std::string path;
|
||||||
|
for (std::size_t i = depth; i > 0; --i)
|
||||||
|
{
|
||||||
|
path += (i - 1) % 2 == 0 ? "/a" : "/0";
|
||||||
|
}
|
||||||
|
return path;
|
||||||
|
}
|
||||||
|
} // namespace
|
||||||
|
|
||||||
TEST_CASE("JSON patch")
|
TEST_CASE("JSON patch")
|
||||||
{
|
{
|
||||||
SECTION("examples from RFC 6902")
|
SECTION("examples from RFC 6902")
|
||||||
@@ -1752,6 +1809,102 @@ TEST_CASE("JSON patch - diff emits array removals in descending index order")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("JSON patch: diff of deeply nested values")
|
||||||
|
{
|
||||||
|
SECTION("the diff reproduces the target at every depth")
|
||||||
|
{
|
||||||
|
// depths on either side of the nesting depth up to which diff()
|
||||||
|
// recurses (detail::recursion_depth_limit(), 128); not every depth up
|
||||||
|
// to 300, as the test would then time out under Valgrind
|
||||||
|
std::vector<std::size_t> depths;
|
||||||
|
for (std::size_t depth = 0; depth <= 16; ++depth)
|
||||||
|
{
|
||||||
|
depths.push_back(depth);
|
||||||
|
}
|
||||||
|
for (std::size_t depth = 120; depth <= 136; ++depth)
|
||||||
|
{
|
||||||
|
depths.push_back(depth);
|
||||||
|
}
|
||||||
|
depths.push_back(300);
|
||||||
|
|
||||||
|
for (const auto depth : depths)
|
||||||
|
{
|
||||||
|
CAPTURE(depth);
|
||||||
|
for (int from = 0; from < 3; ++from)
|
||||||
|
{
|
||||||
|
for (int to = 0; to < 3; ++to)
|
||||||
|
{
|
||||||
|
CAPTURE(from);
|
||||||
|
CAPTURE(to);
|
||||||
|
const auto source = nested<json>(depth, from);
|
||||||
|
const auto target = nested<json>(depth, to);
|
||||||
|
const auto patch = json::diff(source, target);
|
||||||
|
CHECK(source.patch(patch) == target);
|
||||||
|
CHECK(patch.empty() == (from == to));
|
||||||
|
|
||||||
|
const auto ordered_source = nested<nlohmann::ordered_json>(depth, from);
|
||||||
|
const auto ordered_target = nested<nlohmann::ordered_json>(depth, to);
|
||||||
|
CHECK(ordered_source.patch(nlohmann::ordered_json::diff(ordered_source, ordered_target)) == ordered_target);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("a difference only in the innermost value is one replace operation")
|
||||||
|
{
|
||||||
|
for (std::size_t depth = 0; depth <= 300; ++depth)
|
||||||
|
{
|
||||||
|
CAPTURE(depth);
|
||||||
|
json source = 1;
|
||||||
|
json target = 2;
|
||||||
|
for (std::size_t i = 0; i < depth; ++i)
|
||||||
|
{
|
||||||
|
source = i % 2 == 0 ? json::object({{"a", std::move(source)}}) : json::array({std::move(source)});
|
||||||
|
target = i % 2 == 0 ? json::object({{"a", std::move(target)}}) : json::array({std::move(target)});
|
||||||
|
}
|
||||||
|
CHECK(json::diff(source, target, "/root") == json::array({{{"op", "replace"}, {"path", "/root" + nested_path(depth)}, {"value", 2}}}));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("values nested too deeply for the call stack (#5393)")
|
||||||
|
{
|
||||||
|
// diff() used to recurse once per nesting level, and compared the
|
||||||
|
// values with operator== on every level. The values are only
|
||||||
|
// parsed and diffed, never copied or compared, since those recurse
|
||||||
|
// too.
|
||||||
|
const std::size_t depth = 100000;
|
||||||
|
for (const bool objects :
|
||||||
|
{
|
||||||
|
false, true
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(objects);
|
||||||
|
std::string source_text;
|
||||||
|
std::string target_text;
|
||||||
|
std::string equal_text;
|
||||||
|
std::string path;
|
||||||
|
for (std::size_t i = 0; i < depth; ++i)
|
||||||
|
{
|
||||||
|
source_text += objects ? "{\"a\":" : "[";
|
||||||
|
path += objects ? "/a" : "/0";
|
||||||
|
}
|
||||||
|
target_text = source_text + "2";
|
||||||
|
equal_text = source_text + "1";
|
||||||
|
source_text += "1";
|
||||||
|
const std::string closing(depth, objects ? '}' : ']');
|
||||||
|
const auto source = json::parse(source_text + closing);
|
||||||
|
|
||||||
|
const auto patch = json::diff(source, json::parse(target_text + closing));
|
||||||
|
REQUIRE(patch.size() == 1);
|
||||||
|
CHECK(patch[0]["op"] == "replace");
|
||||||
|
CHECK(patch[0]["path"] == path);
|
||||||
|
CHECK(patch[0]["value"] == 2);
|
||||||
|
|
||||||
|
CHECK(json::diff(source, json::parse(equal_text + closing)).empty());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("JSON patch - diff() takes the fast path for non-reorderable object types (regression #5639)")
|
TEST_CASE("JSON patch - diff() takes the fast path for non-reorderable object types (regression #5639)")
|
||||||
{
|
{
|
||||||
// #5465 added an order check to diff()'s object handling so a
|
// #5465 added an order check to diff()'s object handling so a
|
||||||
|
|||||||
@@ -14,7 +14,10 @@ using nlohmann::json;
|
|||||||
|
|
||||||
#include <array>
|
#include <array>
|
||||||
#include <clocale>
|
#include <clocale>
|
||||||
|
#include <limits>
|
||||||
#include <map>
|
#include <map>
|
||||||
|
#include <ostream>
|
||||||
|
#include <streambuf>
|
||||||
#include <string>
|
#include <string>
|
||||||
#include <utility>
|
#include <utility>
|
||||||
#include <vector>
|
#include <vector>
|
||||||
@@ -385,3 +388,92 @@ TEST_CASE("locale with a multi-byte decimal point")
|
|||||||
|
|
||||||
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
|
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
namespace
|
||||||
|
{
|
||||||
|
// a streambuf that switches LC_NUMERIC the first time anything is written to
|
||||||
|
// it, so a dump() in progress can be made to change locale mid-flight: after
|
||||||
|
// the serializer was constructed (and, before #5709 item 3, after it had
|
||||||
|
// cached std::localeconv() for the whole call) but before a later float is
|
||||||
|
// converted
|
||||||
|
struct LocaleSwitchingStreambuf final : std::streambuf
|
||||||
|
{
|
||||||
|
explicit LocaleSwitchingStreambuf(const char* switch_to)
|
||||||
|
: locale_after_first_write(switch_to)
|
||||||
|
{}
|
||||||
|
|
||||||
|
std::string data {}; // NOLINT(readability-redundant-member-init)
|
||||||
|
std::string locale_after_first_write;
|
||||||
|
bool switched = false;
|
||||||
|
|
||||||
|
std::streamsize xsputn(const char* s, std::streamsize n) override
|
||||||
|
{
|
||||||
|
if (!switched)
|
||||||
|
{
|
||||||
|
switched = std::setlocale(LC_NUMERIC, locale_after_first_write.c_str()) != nullptr;
|
||||||
|
}
|
||||||
|
data.append(s, static_cast<std::size_t>(n));
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
} // namespace
|
||||||
|
|
||||||
|
TEST_CASE("locale changes during a single dump() (#5709 item 3)")
|
||||||
|
{
|
||||||
|
// dump_float() only reads the locale on the snprintf path, taken for a
|
||||||
|
// number_float_t that is not an IEEE-754 single or double, i.e. not
|
||||||
|
// (is_iec559 && digits == 24 && max_exponent == 128) and not (is_iec559
|
||||||
|
// && digits == 53 && max_exponent == 1024) - see dump_float(). Checking
|
||||||
|
// is_iec559 alone is not enough: on x86_64, long double is a 64-bit
|
||||||
|
// (80-bit extended) format for which is_iec559 is also true, so it still
|
||||||
|
// takes the snprintf path this test means to exercise. Only a
|
||||||
|
// number_float_t whose digits/max_exponent match float or double (e.g.
|
||||||
|
// long double on 64-bit Arm, where it is IEEE-754 double) takes the
|
||||||
|
// locale-independent to_chars() path instead, and this test is a no-op
|
||||||
|
// there.
|
||||||
|
using long_double_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, long double>;
|
||||||
|
using ld_limits = std::numeric_limits<long_double_json::number_float_t>;
|
||||||
|
const bool is_ieee_single_or_double =
|
||||||
|
(ld_limits::is_iec559 && ld_limits::digits == 24 && ld_limits::max_exponent == 128) ||
|
||||||
|
(ld_limits::is_iec559 && ld_limits::digits == 53 && ld_limits::max_exponent == 1024);
|
||||||
|
if (is_ieee_single_or_double)
|
||||||
|
{
|
||||||
|
MESSAGE("long double is IEEE-754 single or double on this platform; dump_float()'s snprintf/locale path is not exercised here");
|
||||||
|
}
|
||||||
|
|
||||||
|
const char* de_DE_name = "de_DE.UTF-8";
|
||||||
|
if (std::setlocale(LC_NUMERIC, de_DE_name) == nullptr)
|
||||||
|
{
|
||||||
|
de_DE_name = "de_DE";
|
||||||
|
if (std::setlocale(LC_NUMERIC, de_DE_name) == nullptr)
|
||||||
|
{
|
||||||
|
MESSAGE("locale de_DE is not usable");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const std::string decimal_point = std::localeconv()->decimal_point;
|
||||||
|
REQUIRE(std::setlocale(LC_NUMERIC, "C") != nullptr);
|
||||||
|
if (decimal_point != ",")
|
||||||
|
{
|
||||||
|
MESSAGE("de_DE's decimal point is not ',' on this platform, skipping");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// a string long enough to overflow the serializer's internal write
|
||||||
|
// buffer, so that it is flushed to the output adapter - and the locale
|
||||||
|
// switched - before the number after it is converted
|
||||||
|
const std::string padding(5000, 'a');
|
||||||
|
const long_double_json j = { padding, 1234.5L };
|
||||||
|
|
||||||
|
LocaleSwitchingStreambuf buf(de_DE_name);
|
||||||
|
std::ostream os(&buf);
|
||||||
|
os << j;
|
||||||
|
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
|
||||||
|
|
||||||
|
REQUIRE(buf.switched);
|
||||||
|
// whatever locale was in effect when the float was actually converted,
|
||||||
|
// the output is normalized to use '.' as the decimal point: it must be
|
||||||
|
// looked up at conversion time, not once for the whole dump() - the same
|
||||||
|
// fix #5597 made on the parser side
|
||||||
|
CHECK(buf.data == "[\"" + padding + "\",1234.5]");
|
||||||
|
}
|
||||||
|
|||||||
@@ -21,85 +21,11 @@ using nlohmann::json;
|
|||||||
#include <limits>
|
#include <limits>
|
||||||
#include <set>
|
#include <set>
|
||||||
#include "make_test_data_available.hpp"
|
#include "make_test_data_available.hpp"
|
||||||
|
#include "round_trip_corpus.hpp"
|
||||||
#include "test_utils.hpp"
|
#include "test_utils.hpp"
|
||||||
|
#include "sax_countdown.hpp"
|
||||||
|
using utils::SaxCountdown;
|
||||||
|
|
||||||
namespace
|
|
||||||
{
|
|
||||||
class SaxCountdown
|
|
||||||
{
|
|
||||||
public:
|
|
||||||
explicit SaxCountdown(const int count) : events_left(count)
|
|
||||||
{}
|
|
||||||
|
|
||||||
bool null()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool boolean(bool /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_integer(json::number_integer_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_unsigned(json::number_unsigned_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_float(json::number_float_t /*unused*/, const std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool string(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool binary(std::vector<std::uint8_t>& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_object(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool key(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_object()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_array(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_array()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& /*unused*/) // NOLINT(readability-convert-member-functions-to-static)
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
int events_left = 0;
|
|
||||||
};
|
|
||||||
} // namespace
|
|
||||||
|
|
||||||
TEST_CASE("MessagePack")
|
TEST_CASE("MessagePack")
|
||||||
{
|
{
|
||||||
@@ -1930,6 +1856,38 @@ TEST_CASE("Parse MessagePack directly from a file using iterator and sentinel")
|
|||||||
CHECK((parsed.is_object() || parsed.is_array()));
|
CHECK((parsed.is_object() || parsed.is_array()));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("MessagePack round-trip invariants")
|
||||||
|
{
|
||||||
|
// This checks what the parse_msgpack_fuzzer driver checks (see
|
||||||
|
// tests/src/fuzzer-parse_msgpack.cpp), so that a regression shows up in
|
||||||
|
// CI rather than as an OSS-Fuzz report: anything from_msgpack() returns
|
||||||
|
// (j1) can be serialized, parsed back (j2), and serialized again to
|
||||||
|
// reproduce the exact bytes.
|
||||||
|
for (const auto& j0 : utils::round_trip_corpus::values())
|
||||||
|
{
|
||||||
|
json j1;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
// turn the corpus value into a value as from_msgpack() returns it
|
||||||
|
j1 = json::from_msgpack(json::to_msgpack(j0));
|
||||||
|
}
|
||||||
|
catch (const json::exception&)
|
||||||
|
{
|
||||||
|
// the fuzzer driver only ever sees values from_msgpack() actually
|
||||||
|
// produced, so skip corpus values that do not survive the
|
||||||
|
// round trip here, too
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
INFO("j1 = " << j1.dump());
|
||||||
|
const std::vector<std::uint8_t> vec = json::to_msgpack(j1);
|
||||||
|
json j2;
|
||||||
|
// anything the library writes must be parsable by the library
|
||||||
|
REQUIRE_NOTHROW(j2 = json::from_msgpack(vec));
|
||||||
|
CHECK(json::to_msgpack(j2) == vec);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("MessagePack roundtrips" * doctest::skip())
|
TEST_CASE("MessagePack roundtrips" * doctest::skip())
|
||||||
{
|
{
|
||||||
SECTION("input from msgpack-python")
|
SECTION("input from msgpack-python")
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ using nlohmann::json;
|
|||||||
DOCTEST_MSVC_SUPPRESS_WARNING_PUSH
|
DOCTEST_MSVC_SUPPRESS_WARNING_PUSH
|
||||||
DOCTEST_MSVC_SUPPRESS_WARNING(4189)
|
DOCTEST_MSVC_SUPPRESS_WARNING(4189)
|
||||||
|
|
||||||
TEST_CASE("README" * doctest::skip())
|
TEST_CASE("README")
|
||||||
{
|
{
|
||||||
{
|
{
|
||||||
// redirect std::cout for the README file
|
// redirect std::cout for the README file
|
||||||
|
|||||||
@@ -1482,7 +1482,23 @@ TEST_CASE("regression tests 1")
|
|||||||
|
|
||||||
SECTION("issue #972 - Segmentation fault on G++ when trying to assign json string literal to custom json type")
|
SECTION("issue #972 - Segmentation fault on G++ when trying to assign json string literal to custom json type")
|
||||||
{
|
{
|
||||||
|
// this assignment used to crash outright
|
||||||
my_json const foo = R"([1, 2, 3])"_json;
|
my_json const foo = R"([1, 2, 3])"_json;
|
||||||
|
|
||||||
|
// fifo_map is the adapter the docs recommend for keeping object keys
|
||||||
|
// in insertion order (see docs/mkdocs/docs/features/object_order.md
|
||||||
|
// and docs/mkdocs/docs/features/types/template_parameters.md); check
|
||||||
|
// that recommendation actually holds, including through erase() and
|
||||||
|
// inserting a new key. The comparator is stateful, so this avoids
|
||||||
|
// deep copies of "order" (see #1763, #5649).
|
||||||
|
my_json order = my_json::parse(R"({"z":1,"a":2,"m":{"y":1,"b":2}})");
|
||||||
|
CHECK(order.dump() == R"({"z":1,"a":2,"m":{"y":1,"b":2}})");
|
||||||
|
|
||||||
|
order.erase("z");
|
||||||
|
CHECK(order.dump() == R"({"a":2,"m":{"y":1,"b":2}})");
|
||||||
|
|
||||||
|
order["new_key"] = 3;
|
||||||
|
CHECK(order.dump() == R"({"a":2,"m":{"y":1,"b":2},"new_key":3})");
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("issue #977 - Assigning between different json types")
|
SECTION("issue #977 - Assigning between different json types")
|
||||||
|
|||||||
@@ -213,28 +213,6 @@ struct adl_serializer<NonDefaultConstructible>
|
|||||||
};
|
};
|
||||||
} // namespace nlohmann
|
} // namespace nlohmann
|
||||||
|
|
||||||
/////////////////////////////////////////////////////////////////////
|
|
||||||
// for #2824
|
|
||||||
/////////////////////////////////////////////////////////////////////
|
|
||||||
|
|
||||||
class sax_no_exception : public nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type>
|
|
||||||
{
|
|
||||||
public:
|
|
||||||
explicit sax_no_exception(json& j)
|
|
||||||
: nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type>(j, false)
|
|
||||||
{}
|
|
||||||
|
|
||||||
static bool parse_error(std::size_t /*position*/, const std::string& /*last_token*/, const json::exception& ex)
|
|
||||||
{
|
|
||||||
error_string = new std::string(ex.what()); // NOLINT(cppcoreguidelines-owning-memory)
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
static std::string* error_string;
|
|
||||||
};
|
|
||||||
|
|
||||||
std::string* sax_no_exception::error_string = nullptr;
|
|
||||||
|
|
||||||
/////////////////////////////////////////////////////////////////////
|
/////////////////////////////////////////////////////////////////////
|
||||||
// for #2982
|
// for #2982
|
||||||
/////////////////////////////////////////////////////////////////////
|
/////////////////////////////////////////////////////////////////////
|
||||||
@@ -751,16 +729,6 @@ TEST_CASE("regression tests 2")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("issue #2824 - encoding of json::exception::what()")
|
|
||||||
{
|
|
||||||
json j;
|
|
||||||
sax_no_exception sax(j);
|
|
||||||
|
|
||||||
CHECK(!json::sax_parse("xyz", &sax));
|
|
||||||
CHECK(*sax_no_exception::error_string == "[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - invalid literal; last read: 'x'");
|
|
||||||
delete sax_no_exception::error_string; // NOLINT(cppcoreguidelines-owning-memory)
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("issue #2825 - Properly constrain the basic_json conversion operator")
|
SECTION("issue #2825 - Properly constrain the basic_json conversion operator")
|
||||||
{
|
{
|
||||||
static_assert(std::is_copy_assignable<nlohmann::ordered_json>::value, "ordered_json must be copy assignable");
|
static_assert(std::is_copy_assignable<nlohmann::ordered_json>::value, "ordered_json must be copy assignable");
|
||||||
@@ -898,4 +866,585 @@ TEST_CASE("regression test - excessive binary container size honors allow_except
|
|||||||
CHECK(json::from_cbor(std::vector<std::uint8_t> {0x9b, 0, 0, 0, 0, 0, 0, 0, 0x02}, true, false).is_discarded());
|
CHECK(json::from_cbor(std::vector<std::uint8_t> {0x9b, 0, 0, 0, 0, 0, 0, 0, 0x02}, true, false).is_discarded());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
namespace
|
||||||
|
{
|
||||||
|
/// builds a value from SAX events, asks the parser to recover from its first
|
||||||
|
/// 100 errors, and checks that the events are balanced (see #3989)
|
||||||
|
class RecoveringParser
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
explicit RecoveringParser(json& j)
|
||||||
|
: dom(j, false)
|
||||||
|
{}
|
||||||
|
|
||||||
|
bool null()
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.null();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool boolean(bool val)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.boolean(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_integer(json::number_integer_t val)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.number_integer(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_unsigned(json::number_unsigned_t val)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.number_unsigned(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool number_float(json::number_float_t val, const std::string& s)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.number_float(val, s);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool string(std::string& val)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.string(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool binary(json::binary_t& val)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
return dom.binary(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool start_object(std::size_t elements)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
stack.push_back('o');
|
||||||
|
return dom.start_object(elements);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool key(std::string& val)
|
||||||
|
{
|
||||||
|
if (stack.empty() || stack.back() != 'o')
|
||||||
|
{
|
||||||
|
well_formed = false;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
stack.back() = 'v';
|
||||||
|
return dom.key(val);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool end_object()
|
||||||
|
{
|
||||||
|
if (stack.empty() || stack.back() != 'o')
|
||||||
|
{
|
||||||
|
well_formed = false;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
stack.pop_back();
|
||||||
|
return dom.end_object();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool start_array(std::size_t elements)
|
||||||
|
{
|
||||||
|
value();
|
||||||
|
stack.push_back('a');
|
||||||
|
return dom.start_array(elements);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool end_array()
|
||||||
|
{
|
||||||
|
if (stack.empty() || stack.back() != 'a')
|
||||||
|
{
|
||||||
|
well_formed = false;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
stack.pop_back();
|
||||||
|
return dom.end_array();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& ex)
|
||||||
|
{
|
||||||
|
messages.emplace_back(ex.what());
|
||||||
|
// a limit, so that a reader that does not stop fails the test
|
||||||
|
// instead of making it hang
|
||||||
|
return ++errors < 100;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// whether the events were balanced and every key was followed by a value
|
||||||
|
bool balanced() const
|
||||||
|
{
|
||||||
|
return well_formed && stack.empty();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// builds the value
|
||||||
|
nlohmann::detail::json_sax_dom_parser<json> dom;
|
||||||
|
std::size_t errors = 0;
|
||||||
|
std::vector<std::string> messages {}; // NOLINT(readability-redundant-member-init)
|
||||||
|
std::vector<char> stack {}; // NOLINT(readability-redundant-member-init)
|
||||||
|
bool well_formed = true;
|
||||||
|
|
||||||
|
private:
|
||||||
|
void value()
|
||||||
|
{
|
||||||
|
if (!stack.empty())
|
||||||
|
{
|
||||||
|
if (stack.back() == 'v')
|
||||||
|
{
|
||||||
|
stack.back() = 'o';
|
||||||
|
}
|
||||||
|
else if (stack.back() == 'o')
|
||||||
|
{
|
||||||
|
well_formed = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
};
|
||||||
|
|
||||||
|
struct BinaryParseResult
|
||||||
|
{
|
||||||
|
json value;
|
||||||
|
std::size_t errors;
|
||||||
|
std::vector<std::string> messages;
|
||||||
|
bool ok;
|
||||||
|
bool balanced;
|
||||||
|
};
|
||||||
|
|
||||||
|
BinaryParseResult parse_binary_recovering(const std::vector<std::uint8_t>& input, const json::input_format_t format)
|
||||||
|
{
|
||||||
|
json j;
|
||||||
|
RecoveringParser sax(j);
|
||||||
|
const bool ok = json::sax_parse(input, &sax, format);
|
||||||
|
return {j, sax.errors, sax.messages, ok, sax.balanced()};
|
||||||
|
}
|
||||||
|
|
||||||
|
#if !defined(JSON_NOEXCEPTION)
|
||||||
|
/// the message of the exception that reading @a input into a JSON value
|
||||||
|
/// throws, or an empty string if reading succeeds
|
||||||
|
std::string binary_error_message(const std::vector<std::uint8_t>& input, const json::input_format_t format)
|
||||||
|
{
|
||||||
|
try
|
||||||
|
{
|
||||||
|
json _;
|
||||||
|
switch (format)
|
||||||
|
{
|
||||||
|
case json::input_format_t::cbor:
|
||||||
|
_ = json::from_cbor(input);
|
||||||
|
break;
|
||||||
|
case json::input_format_t::msgpack:
|
||||||
|
_ = json::from_msgpack(input);
|
||||||
|
break;
|
||||||
|
case json::input_format_t::ubjson:
|
||||||
|
_ = json::from_ubjson(input);
|
||||||
|
break;
|
||||||
|
case json::input_format_t::bjdata:
|
||||||
|
_ = json::from_bjdata(input);
|
||||||
|
break;
|
||||||
|
case json::input_format_t::bson:
|
||||||
|
_ = json::from_bson(input);
|
||||||
|
break;
|
||||||
|
case json::input_format_t::bon8:
|
||||||
|
_ = json::from_bon8(input);
|
||||||
|
break;
|
||||||
|
case json::input_format_t::json:
|
||||||
|
default:
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
catch (const json::exception& e)
|
||||||
|
{
|
||||||
|
return e.what();
|
||||||
|
}
|
||||||
|
return "";
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
/// a BSON element: its type, its name, and its value
|
||||||
|
std::vector<std::uint8_t> bson_element(const std::uint8_t type, const std::string& name, const std::vector<std::uint8_t>& value)
|
||||||
|
{
|
||||||
|
std::vector<std::uint8_t> result = {type};
|
||||||
|
result.insert(result.end(), name.begin(), name.end());
|
||||||
|
result.push_back(0x00);
|
||||||
|
result.insert(result.end(), value.begin(), value.end());
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// a BSON document of the given elements; @a size_offset is added to the
|
||||||
|
/// size it declares
|
||||||
|
std::vector<std::uint8_t> bson_document(const std::vector<std::vector<std::uint8_t>>& elements, const int size_offset = 0)
|
||||||
|
{
|
||||||
|
std::vector<std::uint8_t> body;
|
||||||
|
for (const auto& element : elements)
|
||||||
|
{
|
||||||
|
body.insert(body.end(), element.begin(), element.end());
|
||||||
|
}
|
||||||
|
const auto size = static_cast<std::uint32_t>(static_cast<int>(body.size()) + 5 + size_offset);
|
||||||
|
std::vector<std::uint8_t> result = {static_cast<std::uint8_t>(size & 0xFFu), static_cast<std::uint8_t>((size >> 8u) & 0xFFu),
|
||||||
|
static_cast<std::uint8_t>((size >> 16u) & 0xFFu), static_cast<std::uint8_t>((size >> 24u) & 0xFFu)
|
||||||
|
};
|
||||||
|
result.insert(result.end(), body.begin(), body.end());
|
||||||
|
result.push_back(0x00);
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// a BSON int32 value
|
||||||
|
std::vector<std::uint8_t> bson_int32(const std::int32_t value)
|
||||||
|
{
|
||||||
|
const auto u = static_cast<std::uint32_t>(value);
|
||||||
|
return {static_cast<std::uint8_t>(u & 0xFFu), static_cast<std::uint8_t>((u >> 8u) & 0xFFu),
|
||||||
|
static_cast<std::uint8_t>((u >> 16u) & 0xFFu), static_cast<std::uint8_t>((u >> 24u) & 0xFFu)};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// a BSON string value, whose length is @a length_offset off
|
||||||
|
std::vector<std::uint8_t> bson_string(const std::string& value, const std::int32_t length_offset = 0)
|
||||||
|
{
|
||||||
|
auto result = bson_int32(static_cast<std::int32_t>(value.size() + 1) + length_offset);
|
||||||
|
result.insert(result.end(), value.begin(), value.end());
|
||||||
|
result.push_back(0x00);
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// @a count bytes of value 0xAB
|
||||||
|
std::vector<std::uint8_t> bytes(const std::size_t count)
|
||||||
|
{
|
||||||
|
return std::vector<std::uint8_t>(count, 0xAB);
|
||||||
|
}
|
||||||
|
|
||||||
|
template<typename... Parts>
|
||||||
|
std::vector<std::uint8_t> concatenated(const std::vector<std::uint8_t>& first, const Parts& ... rest)
|
||||||
|
{
|
||||||
|
std::vector<std::uint8_t> result = first;
|
||||||
|
for (const auto& part : std::initializer_list<std::vector<std::uint8_t>> {rest...})
|
||||||
|
{
|
||||||
|
result.insert(result.end(), part.begin(), part.end());
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// U+FFFD REPLACEMENT CHARACTER
|
||||||
|
std::string replacement_character()
|
||||||
|
{
|
||||||
|
return "\xEF\xBF\xBD";
|
||||||
|
}
|
||||||
|
} // namespace
|
||||||
|
|
||||||
|
TEST_CASE("regression test - #3989 SAX parse_error() returning true")
|
||||||
|
{
|
||||||
|
SECTION("binary formats complete what was read before the input ends")
|
||||||
|
{
|
||||||
|
const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}};
|
||||||
|
|
||||||
|
const std::vector<std::pair<json::input_format_t, std::vector<std::uint8_t>>> encodings =
|
||||||
|
{
|
||||||
|
{json::input_format_t::cbor, json::to_cbor(j)},
|
||||||
|
{json::input_format_t::msgpack, json::to_msgpack(j)},
|
||||||
|
{json::input_format_t::ubjson, json::to_ubjson(j)},
|
||||||
|
{json::input_format_t::ubjson, json::to_ubjson(j, true, true)},
|
||||||
|
{json::input_format_t::bjdata, json::to_bjdata(j)},
|
||||||
|
{json::input_format_t::bjdata, json::to_bjdata(j, true, true)},
|
||||||
|
{json::input_format_t::bson, json::to_bson(j)},
|
||||||
|
{json::input_format_t::bon8, json::to_bon8(j)},
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const auto& encoding : encodings)
|
||||||
|
{
|
||||||
|
const auto format = encoding.first;
|
||||||
|
const auto& bytes = encoding.second;
|
||||||
|
CAPTURE(format);
|
||||||
|
|
||||||
|
// every prefix is truncated input
|
||||||
|
for (std::size_t length = 0; length < bytes.size(); ++length)
|
||||||
|
{
|
||||||
|
CAPTURE(length);
|
||||||
|
const auto result = parse_binary_recovering(std::vector<std::uint8_t>(bytes.begin(), bytes.begin() + static_cast<std::ptrdiff_t>(length)), format);
|
||||||
|
CHECK(!result.ok);
|
||||||
|
CHECK(result.errors == 1);
|
||||||
|
CHECK(result.balanced);
|
||||||
|
}
|
||||||
|
|
||||||
|
// the complete input is read as usual (binary values do not
|
||||||
|
// round-trip through every format, so compare with a plain parse)
|
||||||
|
json expected;
|
||||||
|
nlohmann::detail::json_sax_dom_parser<json> dom(expected);
|
||||||
|
CHECK(json::sax_parse(bytes, &dom, format));
|
||||||
|
const auto complete = parse_binary_recovering(bytes, format);
|
||||||
|
CHECK(complete.ok);
|
||||||
|
CHECK(complete.errors == 0);
|
||||||
|
CHECK(complete.value == expected);
|
||||||
|
|
||||||
|
// a byte after the value
|
||||||
|
auto trailing_bytes = bytes;
|
||||||
|
trailing_bytes.push_back(0x01);
|
||||||
|
const auto trailing = parse_binary_recovering(trailing_bytes, format);
|
||||||
|
CHECK(!trailing.ok);
|
||||||
|
CHECK(trailing.errors == 1);
|
||||||
|
CHECK(trailing.value == expected);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("containers without an end")
|
||||||
|
{
|
||||||
|
// these made the readers loop, or read on, after the error
|
||||||
|
const auto cbor_array = parse_binary_recovering({0x9F}, json::input_format_t::cbor);
|
||||||
|
CHECK(cbor_array.errors == 1);
|
||||||
|
CHECK(cbor_array.value == json::array());
|
||||||
|
|
||||||
|
const auto cbor_map = parse_binary_recovering({0xBF, 0x61, 'a'}, json::input_format_t::cbor);
|
||||||
|
CHECK(cbor_map.errors == 1);
|
||||||
|
CHECK(cbor_map.value == json({{"a", nullptr}}));
|
||||||
|
|
||||||
|
const auto msgpack_array = parse_binary_recovering({0xDD, 0xFF, 0xFF, 0xFF, 0xFF}, json::input_format_t::msgpack);
|
||||||
|
CHECK(msgpack_array.errors == 1);
|
||||||
|
CHECK(msgpack_array.value == json::array());
|
||||||
|
|
||||||
|
const auto msgpack_map = parse_binary_recovering({0x81, 0xA1, 'a', 0x92, 0x01}, json::input_format_t::msgpack);
|
||||||
|
CHECK(msgpack_map.errors == 1);
|
||||||
|
CHECK(msgpack_map.value == json({{"a", {1}}}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("BJData ndarray")
|
||||||
|
{
|
||||||
|
// a 2x3 int8 array with two of its six elements; the annotated array
|
||||||
|
// format opens an object and two arrays of its own
|
||||||
|
const auto result = parse_binary_recovering({'[', '$', 'i', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2}, json::input_format_t::bjdata);
|
||||||
|
CHECK(result.errors == 1);
|
||||||
|
CHECK(result.balanced);
|
||||||
|
CHECK(result.value == json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2}}}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("binary formats repair items whose end is known")
|
||||||
|
{
|
||||||
|
struct Repair
|
||||||
|
{
|
||||||
|
json::input_format_t format;
|
||||||
|
std::vector<std::uint8_t> input;
|
||||||
|
json expected;
|
||||||
|
std::size_t errors;
|
||||||
|
};
|
||||||
|
|
||||||
|
const std::vector<Repair> repairs =
|
||||||
|
{
|
||||||
|
// CBOR: tags are ignored (here tag 1 and the self-describe tag 55799)
|
||||||
|
{json::input_format_t::cbor, {0x82, 0xC1, 0x05, 0xD9, 0xD9, 0xF7, 0x06}, {5, 6}, 2},
|
||||||
|
// CBOR: undefined and other simple values become null
|
||||||
|
{json::input_format_t::cbor, {0x84, 0xF7, 0xE0, 0xF8, 0x20, 0x01}, {nullptr, nullptr, nullptr, 1}, 3},
|
||||||
|
// CBOR: ill-formed UTF-8 becomes U+FFFD, also in keys
|
||||||
|
{json::input_format_t::cbor, {0xA1, 0x61, 0xFF, 0x62, 0xC3, 0x28}, {{replacement_character(), replacement_character() + "("}}, 2},
|
||||||
|
// CBOR: members whose key is not a string are skipped, whatever their key and value
|
||||||
|
{json::input_format_t::cbor, {0xA4, 0x01, 0x02, 0x82, 0x01, 0x02, 0xA1, 0x61, 'x', 0x9F, 0xFF, 0xC1, 0x01, 0x5F, 0x41, 0x00, 0xFF, 0x61, 'a', 0x03}, {{"a", 3}}, 3},
|
||||||
|
{json::input_format_t::cbor, {0xBF, 0xF5, 0xBF, 0x61, 'x', 0x7F, 0x61, 'y', 0xFF, 0xFF, 0x61, 'a', 0x03, 0xFF}, {{"a", 3}}, 1},
|
||||||
|
// MessagePack: members whose key is not a string are skipped
|
||||||
|
{json::input_format_t::msgpack, {0x84, 0x01, 0x02, 0x81, 0xA1, 'x', 0x01, 0x92, 0x01, 0x02, 0xD4, 0x01, 0x02, 0xC0, 0xA1, 'a', 0x04}, {{"a", 4}}, 3},
|
||||||
|
// MessagePack: ill-formed UTF-8 becomes U+FFFD
|
||||||
|
{json::input_format_t::msgpack, {0x92, 0xA2, 0xC3, 0x28, 0xA3, 0xE2, 0x82, 'x'}, {replacement_character() + "(", replacement_character() + "x"}, 2},
|
||||||
|
// UBJSON: a char that is not ASCII becomes U+FFFD
|
||||||
|
{json::input_format_t::ubjson, {'[', 'C', 0x80, 'C', 'A', ']'}, {replacement_character(), "A"}, 1},
|
||||||
|
// UBJSON: the longest beginning of a high-precision number is kept
|
||||||
|
{json::input_format_t::ubjson, {'[', 'H', 'i', 5, '1', '2', 'a', 'b', 'c', 'H', 'i', 2, '1', '.', 'H', 'i', 3, 'a', 'b', 'c', 'H', 'i', 3, '4', '.', '5', ']'}, {12, 1, nullptr, 4.5}, 3},
|
||||||
|
// BJData, too
|
||||||
|
{json::input_format_t::bjdata, {'[', 'C', 0xFF, 'H', 'i', 2, '-', '1', 'H', 'i', 2, '-', 'x', ']'}, {replacement_character(), -1, nullptr}, 2},
|
||||||
|
// BON8: members whose key is not a string are skipped
|
||||||
|
{json::input_format_t::bon8, {0x89, 0x91, 0x92, 0xC9, 0x40, 0x82, 0x91, 0x92, 0x61, 0x93}, {{"a", 3}}, 2},
|
||||||
|
{json::input_format_t::bon8, {0x8B, 0x91, 0x85, 0x91, 0xFE, 0xFA, 0x8B, 'x', 0x91, 0xFE, 0x61, 0x93, 0xFE}, {{"a", 3}}, 2},
|
||||||
|
// BSON: elements of types the library does not read become null
|
||||||
|
{
|
||||||
|
json::input_format_t::bson, bson_document(
|
||||||
|
{
|
||||||
|
bson_element(0x07, "_id", bytes(12)), // ObjectId
|
||||||
|
bson_element(0x09, "date", bytes(8)), // UTC datetime
|
||||||
|
bson_element(0x13, "decimal", bytes(16)), // 128-bit decimal
|
||||||
|
bson_element(0x0B, "regex", {'a', '+', 0, 'i', 0}), // regular expression
|
||||||
|
bson_element(0x0D, "code", bson_string("f()")), // JavaScript code
|
||||||
|
bson_element(0x0E, "symbol", bson_string("s")), // symbol
|
||||||
|
bson_element(0x0C, "pointer", concatenated(bson_string("c"), bytes(12))), // DBPointer
|
||||||
|
bson_element(0x0F, "scope", concatenated(bson_int32(15), bson_string("g"), bson_document({}))), // code with scope
|
||||||
|
bson_element(0x06, "undefined", {}), // undefined
|
||||||
|
bson_element(0xFF, "min", {}), // min key
|
||||||
|
bson_element(0x7F, "max", {}), // max key
|
||||||
|
bson_element(0x10, "z", bson_int32(7)),
|
||||||
|
}),
|
||||||
|
{{"_id", nullptr}, {"date", nullptr}, {"decimal", nullptr}, {"regex", nullptr}, {"code", nullptr}, {"symbol", nullptr}, {"pointer", nullptr}, {"scope", nullptr}, {"undefined", nullptr}, {"min", nullptr}, {"max", nullptr}, {"z", 7}},
|
||||||
|
11
|
||||||
|
},
|
||||||
|
// BSON: an element of an unknown type becomes null, and the rest of its document is skipped
|
||||||
|
{
|
||||||
|
json::input_format_t::bson, bson_document(
|
||||||
|
{
|
||||||
|
bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3)), bson_element(0x10, "b", bson_int32(2))})),
|
||||||
|
bson_element(0x04, "array", bson_document({bson_element(0x10, "0", bson_int32(1)), bson_element(0x42, "1", bytes(3))})),
|
||||||
|
bson_element(0x10, "after", bson_int32(3)),
|
||||||
|
}),
|
||||||
|
{{"inner", {{"a", 1}, {"x", nullptr}}}, {"array", {1, nullptr}}, {"after", 3}},
|
||||||
|
2
|
||||||
|
},
|
||||||
|
// BSON: so does a string or byte array whose length cannot be right
|
||||||
|
{
|
||||||
|
json::input_format_t::bson, bson_document(
|
||||||
|
{
|
||||||
|
bson_element(0x03, "inner", bson_document({bson_element(0x02, "s", bson_string("abc", -10)), bson_element(0x10, "b", bson_int32(2))})),
|
||||||
|
bson_element(0x03, "bin", bson_document({bson_element(0x05, "b", concatenated(bson_int32(-1), bytes(1))), bson_element(0x10, "b", bson_int32(2))})),
|
||||||
|
bson_element(0x10, "after", bson_int32(3)),
|
||||||
|
}),
|
||||||
|
{{"inner", {{"s", nullptr}}}, {"bin", {{"b", nullptr}}}, {"after", 3}},
|
||||||
|
2
|
||||||
|
},
|
||||||
|
// BSON: a string without its terminator, and a document whose size does not match, are kept
|
||||||
|
{
|
||||||
|
json::input_format_t::bson, bson_document(
|
||||||
|
{
|
||||||
|
bson_element(0x02, "s", {2, 0, 0, 0, 'a', 'X'}),
|
||||||
|
bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1))}, 1)),
|
||||||
|
}),
|
||||||
|
{{"s", "a"}, {"inner", {{"a", 1}}}},
|
||||||
|
2
|
||||||
|
},
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const auto& repair : repairs)
|
||||||
|
{
|
||||||
|
CAPTURE(repair.format);
|
||||||
|
CAPTURE(repair.input);
|
||||||
|
const auto result = parse_binary_recovering(repair.input, repair.format);
|
||||||
|
CHECK(!result.ok);
|
||||||
|
CHECK(result.balanced);
|
||||||
|
CHECK(result.errors == repair.errors);
|
||||||
|
CHECK(result.value == repair.expected);
|
||||||
|
REQUIRE(!result.messages.empty());
|
||||||
|
#if !defined(JSON_NOEXCEPTION)
|
||||||
|
// the first error is the one reported without recovering; under
|
||||||
|
// JSON_NOEXCEPTION, reading without recovering aborts instead of
|
||||||
|
// throwing, so there is no message to compare with
|
||||||
|
CHECK(result.messages.front() == binary_error_message(repair.input, repair.format));
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("binary formats repair numbers that are out of range")
|
||||||
|
{
|
||||||
|
// CBOR: a negative integer below the range of number_integer_t
|
||||||
|
const auto cbor = parse_binary_recovering({0x3B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}, json::input_format_t::cbor);
|
||||||
|
CHECK(cbor.errors == 1);
|
||||||
|
CHECK(cbor.value.is_number_float());
|
||||||
|
CHECK(cbor.value.get<double>() == -18446744073709551616.0);
|
||||||
|
|
||||||
|
// UBJSON: a high-precision number too large for number_float_t
|
||||||
|
const auto ubjson = parse_binary_recovering({'H', 'i', 5, '1', 'e', '9', '9', '9'}, json::input_format_t::ubjson);
|
||||||
|
CHECK(ubjson.errors == 1);
|
||||||
|
CHECK(ubjson.value.is_number_float());
|
||||||
|
CHECK(std::isinf(ubjson.value.get<double>()));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("binary formats stop where the end of an item is not known")
|
||||||
|
{
|
||||||
|
// a byte that begins no item
|
||||||
|
const auto cbor = parse_binary_recovering({0x82, 0x01, 0x1C, 0x02}, json::input_format_t::cbor);
|
||||||
|
CHECK(cbor.errors == 1);
|
||||||
|
CHECK(cbor.value == json({1}));
|
||||||
|
|
||||||
|
// a key that is no item: the unused MessagePack byte, a CBOR break
|
||||||
|
// in a map of known size, and the end of a BON8 container
|
||||||
|
const auto msgpack = parse_binary_recovering({0x82, 0xA1, 'a', 0x01, 0xC1, 0x02}, json::input_format_t::msgpack);
|
||||||
|
CHECK(msgpack.errors == 1);
|
||||||
|
CHECK(msgpack.value == json({{"a", 1}}));
|
||||||
|
const auto cbor_break = parse_binary_recovering({0xA2, 0x61, 'a', 0x01, 0xFF, 0x02}, json::input_format_t::cbor);
|
||||||
|
CHECK(cbor_break.errors == 1);
|
||||||
|
CHECK(cbor_break.value == json({{"a", 1}}));
|
||||||
|
const auto bon8 = parse_binary_recovering({0x88, 0x61, 0x91, 0xFE}, json::input_format_t::bon8);
|
||||||
|
CHECK(bon8.errors == 1);
|
||||||
|
CHECK(bon8.value == json({{"a", 1}}));
|
||||||
|
|
||||||
|
// a skipped member that the input ends in
|
||||||
|
const auto truncated = parse_binary_recovering({0xA2, 0x01, 0x82, 0x01}, json::input_format_t::cbor);
|
||||||
|
CHECK(truncated.errors == 2);
|
||||||
|
CHECK(truncated.balanced);
|
||||||
|
CHECK(truncated.value == json::object());
|
||||||
|
|
||||||
|
// a BSON element of an unknown type in a document whose size cannot be right
|
||||||
|
const auto bson = parse_binary_recovering(bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3))}, -10), json::input_format_t::bson);
|
||||||
|
CHECK(bson.errors == 1);
|
||||||
|
CHECK(bson.value == json({{"a", 1}, {"x", nullptr}}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("changed bytes in binary input")
|
||||||
|
{
|
||||||
|
const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}, {"i", "\xC3\xA4"}};
|
||||||
|
|
||||||
|
const std::vector<std::pair<json::input_format_t, std::vector<std::uint8_t>>> encodings =
|
||||||
|
{
|
||||||
|
{json::input_format_t::cbor, json::to_cbor(j)},
|
||||||
|
{json::input_format_t::msgpack, json::to_msgpack(j)},
|
||||||
|
{json::input_format_t::ubjson, json::to_ubjson(j)},
|
||||||
|
{json::input_format_t::ubjson, json::to_ubjson(j, true, true)},
|
||||||
|
{json::input_format_t::bjdata, json::to_bjdata(j)},
|
||||||
|
{json::input_format_t::bjdata, json::to_bjdata(j, true, true)},
|
||||||
|
{json::input_format_t::bson, json::to_bson(j)},
|
||||||
|
{json::input_format_t::bon8, json::to_bon8(j)},
|
||||||
|
};
|
||||||
|
const std::vector<std::uint8_t> replacements = {0x00, 0x01, 0x7F, 0x80, 0xC1, 0xD9, 0xE0, 0xF7, 0xFE, 0xFF};
|
||||||
|
|
||||||
|
for (const auto& encoding : encodings)
|
||||||
|
{
|
||||||
|
const auto format = encoding.first;
|
||||||
|
const auto& original = encoding.second;
|
||||||
|
CAPTURE(format);
|
||||||
|
|
||||||
|
std::vector<std::vector<std::uint8_t>> inputs;
|
||||||
|
for (std::size_t position = 0; position < original.size(); ++position)
|
||||||
|
{
|
||||||
|
for (const auto replacement : replacements)
|
||||||
|
{
|
||||||
|
auto changed = original;
|
||||||
|
changed[position] = replacement;
|
||||||
|
inputs.push_back(changed);
|
||||||
|
}
|
||||||
|
auto removed = original;
|
||||||
|
removed.erase(removed.begin() + static_cast<std::ptrdiff_t>(position));
|
||||||
|
inputs.push_back(removed);
|
||||||
|
}
|
||||||
|
|
||||||
|
for (const auto& input : inputs)
|
||||||
|
{
|
||||||
|
CAPTURE(input);
|
||||||
|
const auto result = parse_binary_recovering(input, format);
|
||||||
|
CHECK(result.balanced);
|
||||||
|
CHECK(result.errors <= input.size() + 1);
|
||||||
|
#if !defined(JSON_NOEXCEPTION)
|
||||||
|
// an error is reported exactly if reading into a JSON value
|
||||||
|
// fails, and the first one is the same (under JSON_NOEXCEPTION,
|
||||||
|
// that reading aborts instead of throwing)
|
||||||
|
const auto message = binary_error_message(input, format);
|
||||||
|
CHECK(result.ok == message.empty());
|
||||||
|
if (!result.ok && result.errors < 100)
|
||||||
|
{
|
||||||
|
CHECK(result.messages.front() == message);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("JSON text")
|
||||||
|
{
|
||||||
|
// the parser stopped, but reported success
|
||||||
|
json j;
|
||||||
|
RecoveringParser sax(j);
|
||||||
|
CHECK(!json::sax_parse("[1,2,3,]", &sax));
|
||||||
|
CHECK(sax.errors == 1);
|
||||||
|
CHECK(j == json({1, 2, 3}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("the SAX parsers of the library stop")
|
||||||
|
{
|
||||||
|
json _;
|
||||||
|
CHECK(json::from_cbor(std::vector<std::uint8_t> {0x9F}, true, false).is_discarded());
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<std::uint8_t> {0x9F}), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&);
|
||||||
|
CHECK(json::parse("[1,2,3,]", nullptr, false).is_discarded());
|
||||||
|
CHECK(!json::accept("[1,2,3,]"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
DOCTEST_CLANG_SUPPRESS_WARNING_POP
|
DOCTEST_CLANG_SUPPRESS_WARNING_POP
|
||||||
|
|||||||
@@ -17,84 +17,9 @@ using nlohmann::json;
|
|||||||
#include "make_test_data_available.hpp"
|
#include "make_test_data_available.hpp"
|
||||||
#include "round_trip_corpus.hpp"
|
#include "round_trip_corpus.hpp"
|
||||||
#include "test_utils.hpp"
|
#include "test_utils.hpp"
|
||||||
|
#include "sax_countdown.hpp"
|
||||||
|
using utils::SaxCountdown;
|
||||||
|
|
||||||
namespace
|
|
||||||
{
|
|
||||||
class SaxCountdown
|
|
||||||
{
|
|
||||||
public:
|
|
||||||
explicit SaxCountdown(const int count) : events_left(count)
|
|
||||||
{}
|
|
||||||
|
|
||||||
bool null()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool boolean(bool /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_integer(json::number_integer_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_unsigned(json::number_unsigned_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool number_float(json::number_float_t /*unused*/, const std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool string(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool binary(std::vector<std::uint8_t>& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_object(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool key(std::string& /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_object()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool start_array(std::size_t /*unused*/)
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool end_array()
|
|
||||||
{
|
|
||||||
return events_left-- > 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& /*unused*/) // NOLINT(readability-convert-member-functions-to-static)
|
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
int events_left = 0;
|
|
||||||
};
|
|
||||||
} // namespace
|
|
||||||
|
|
||||||
TEST_CASE("UBJSON")
|
TEST_CASE("UBJSON")
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -11,135 +11,97 @@
|
|||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
using nlohmann::json;
|
using nlohmann::json;
|
||||||
|
|
||||||
// ICPC errors out on multibyte character sequences in source files
|
|
||||||
#ifndef __INTEL_COMPILER
|
|
||||||
namespace
|
|
||||||
{
|
|
||||||
bool wstring_is_utf16();
|
|
||||||
bool wstring_is_utf16()
|
|
||||||
{
|
|
||||||
return (std::wstring(L"💩") == std::wstring(L"\U0001F4A9"));
|
|
||||||
}
|
|
||||||
|
|
||||||
bool u16string_is_utf16();
|
|
||||||
bool u16string_is_utf16()
|
|
||||||
{
|
|
||||||
return (std::u16string(u"💩") == std::u16string(u"\U0001F4A9"));
|
|
||||||
}
|
|
||||||
|
|
||||||
bool u32string_is_utf32();
|
|
||||||
bool u32string_is_utf32()
|
|
||||||
{
|
|
||||||
return (std::u32string(U"💩") == std::u32string(U"\U0001F4A9"));
|
|
||||||
}
|
|
||||||
} // namespace
|
|
||||||
|
|
||||||
TEST_CASE("wide strings")
|
TEST_CASE("wide strings")
|
||||||
{
|
{
|
||||||
SECTION("std::wstring")
|
SECTION("std::wstring")
|
||||||
{
|
{
|
||||||
if (wstring_is_utf16())
|
// U+10C5 U+0061(a) U+00E4 U+00F6 U+1F4A4 U+1F9E2, written with \u/\U
|
||||||
{
|
// escapes rather than as raw multibyte characters so this file
|
||||||
std::wstring const w = L"[12.2,\"Ⴥaäö💤🧢\"]";
|
// compiles on toolchains (e.g. classic ICC) that error out on
|
||||||
json const j = json::parse(w);
|
// multibyte character sequences in source files
|
||||||
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
|
std::wstring const w = L"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]";
|
||||||
}
|
json const j = json::parse(w);
|
||||||
|
CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]");
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("invalid std::wstring")
|
SECTION("invalid std::wstring")
|
||||||
{
|
{
|
||||||
if (wstring_is_utf16())
|
std::wstring const w = L"\"\xDBFF";
|
||||||
{
|
json _;
|
||||||
std::wstring const w = L"\"\xDBFF";
|
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||||
json _;
|
|
||||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
|
||||||
|
|
||||||
// the exact message depends on the width of wchar_t: a 16-bit
|
// the exact message depends on the width of wchar_t: a 16-bit
|
||||||
// wchar_t passes the lone surrogate to the UTF-8 decoder unchanged
|
// wchar_t passes the lone surrogate to the UTF-8 decoder unchanged
|
||||||
// (rejected as a single ill-formed byte at column 2), while a
|
// (rejected as a single ill-formed byte at column 2), while a
|
||||||
// 32-bit wchar_t first encodes it as an ill-formed three-byte
|
// 32-bit wchar_t first encodes it as an ill-formed three-byte
|
||||||
// sequence (rejected one byte later, at column 3)
|
// sequence (rejected one byte later, at column 3)
|
||||||
const char* const error_low_surrogate = sizeof(wchar_t) == 2
|
const char* const error_low_surrogate = sizeof(wchar_t) == 2
|
||||||
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
||||||
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'";
|
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'";
|
||||||
const char* const error_high_surrogate = sizeof(wchar_t) == 2
|
const char* const error_high_surrogate = sizeof(wchar_t) == 2
|
||||||
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
||||||
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'";
|
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'";
|
||||||
|
|
||||||
// a lone low surrogate cannot start a pair
|
// a lone low surrogate cannot start a pair
|
||||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'"'}), error_low_surrogate, json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'"'}), error_low_surrogate, json::parse_error&);
|
||||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&);
|
||||||
// ... also when the unit is above the low surrogates
|
// ... also when the unit is above the low surrogates
|
||||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), static_cast<wchar_t>(0xE000), L'"'}), error_high_surrogate, json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), static_cast<wchar_t>(0xE000), L'"'}), error_high_surrogate, json::parse_error&);
|
||||||
// a lone low surrogate must not swallow the following unit: pairing
|
// a lone low surrogate must not swallow the following unit: pairing
|
||||||
// it with any second unit would produce valid UTF-8, so the error
|
// it with any second unit would produce valid UTF-8, so the error
|
||||||
// has to report an ill-formed byte at the surrogate's own position
|
// has to report an ill-formed byte at the surrogate's own position
|
||||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&);
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::u16string")
|
SECTION("std::u16string")
|
||||||
{
|
{
|
||||||
if (u16string_is_utf16())
|
std::u16string const w = u"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]";
|
||||||
{
|
json const j = json::parse(w);
|
||||||
std::u16string const w = u"[12.2,\"Ⴥaäö💤🧢\"]";
|
CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]");
|
||||||
json const j = json::parse(w);
|
|
||||||
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("invalid std::u16string")
|
SECTION("invalid std::u16string")
|
||||||
{
|
{
|
||||||
if (u16string_is_utf16())
|
std::u16string const w = u"\"\xDBFF";
|
||||||
{
|
json _;
|
||||||
std::u16string const w = u"\"\xDBFF";
|
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||||
json _;
|
|
||||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
|
||||||
|
|
||||||
// a lone low surrogate cannot start a pair
|
// a lone low surrogate cannot start a pair
|
||||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||||
// ... also when the unit is above the low surrogates
|
// ... also when the unit is above the low surrogates
|
||||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||||
// a lone low surrogate must not swallow the following unit: pairing
|
// a lone low surrogate must not swallow the following unit: pairing
|
||||||
// it with any second unit would produce valid UTF-8, so the error
|
// it with any second unit would produce valid UTF-8, so the error
|
||||||
// has to report an ill-formed byte at the surrogate's own position
|
// has to report an ill-formed byte at the surrogate's own position
|
||||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||||
// a valid surrogate pair is still decoded (U+1F600)
|
// a valid surrogate pair is still decoded (U+1F600)
|
||||||
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
|
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("std::u32string")
|
SECTION("std::u32string")
|
||||||
{
|
{
|
||||||
if (u32string_is_utf32())
|
std::u32string const w = U"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]";
|
||||||
{
|
json const j = json::parse(w);
|
||||||
std::u32string const w = U"[12.2,\"Ⴥaäö💤🧢\"]";
|
CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]");
|
||||||
json const j = json::parse(w);
|
|
||||||
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("invalid std::u32string")
|
SECTION("invalid std::u32string")
|
||||||
{
|
{
|
||||||
if (u32string_is_utf32())
|
std::u32string const w = U"\"\x110000";
|
||||||
{
|
json _;
|
||||||
std::u32string const w = U"\"\x110000";
|
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||||
json _;
|
|
||||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
|
||||||
|
|
||||||
// a code unit above U+10FFFF must not be narrowed onto the EOF
|
// a code unit above U+10FFFF must not be narrowed onto the EOF
|
||||||
// sentinel: 0xFFFFFFFF would otherwise end the document silently and
|
// sentinel: 0xFFFFFFFF would otherwise end the document silently and
|
||||||
// let everything following it pass the strict end-of-input check
|
// let everything following it pass the strict end-of-input check
|
||||||
std::u32string const trailing{U'[', U'1', U']', static_cast<char32_t>(0xFFFFFFFF), U'x'};
|
std::u32string const trailing{U'[', U'1', U']', static_cast<char32_t>(0xFFFFFFFF), U'x'};
|
||||||
CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
|
||||||
CHECK(!json::accept(trailing));
|
CHECK(!json::accept(trailing));
|
||||||
|
|
||||||
// the same unit inside a string is reported as an ill-formed byte
|
// the same unit inside a string is reported as an ill-formed byte
|
||||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
#endif
|
|
||||||
|
|||||||
@@ -0,0 +1,92 @@
|
|||||||
|
// __ _____ _____ _____
|
||||||
|
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||||
|
// | | |__ | | | | | | version 3.12.0
|
||||||
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||||
|
//
|
||||||
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||||
|
// SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
// shared between unit-32bit.cpp (which must keep including this header,
|
||||||
|
// because JSON_32bitTest=ONLY builds only that file) and unit-bjdata.cpp
|
||||||
|
|
||||||
|
#include <climits> // CHAR_BIT
|
||||||
|
#include <limits>
|
||||||
|
#include <string>
|
||||||
|
#include <type_traits>
|
||||||
|
|
||||||
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
|
template <typename OfType, typename T, bool MinInRange, bool MaxInRange>
|
||||||
|
struct trait_test_arg
|
||||||
|
{
|
||||||
|
using of_type = OfType;
|
||||||
|
using type = T;
|
||||||
|
static constexpr bool min_in_range = MinInRange;
|
||||||
|
static constexpr bool max_in_range = MaxInRange;
|
||||||
|
};
|
||||||
|
|
||||||
|
TEST_CASE_TEMPLATE_DEFINE("value_in_range_of trait", T, value_in_range_of_test) // NOLINT(readability-math-missing-parentheses)
|
||||||
|
{
|
||||||
|
using nlohmann::detail::value_in_range_of;
|
||||||
|
|
||||||
|
using of_type = typename T::of_type;
|
||||||
|
using type = typename T::type;
|
||||||
|
constexpr bool min_in_range = T::min_in_range;
|
||||||
|
constexpr bool max_in_range = T::max_in_range;
|
||||||
|
|
||||||
|
type const val_min = std::numeric_limits<type>::min();
|
||||||
|
type const val_min2 = val_min + 1;
|
||||||
|
type const val_max = std::numeric_limits<type>::max();
|
||||||
|
type const val_max2 = val_max - 1;
|
||||||
|
|
||||||
|
REQUIRE(CHAR_BIT == 8);
|
||||||
|
|
||||||
|
std::string of_type_str;
|
||||||
|
if (std::is_unsigned<of_type>::value)
|
||||||
|
{
|
||||||
|
of_type_str += "u";
|
||||||
|
}
|
||||||
|
of_type_str += "int";
|
||||||
|
of_type_str += std::to_string(sizeof(of_type) * 8);
|
||||||
|
|
||||||
|
INFO("of_type := ", of_type_str);
|
||||||
|
|
||||||
|
std::string type_str;
|
||||||
|
if (std::is_unsigned<type>::value)
|
||||||
|
{
|
||||||
|
type_str += "u";
|
||||||
|
}
|
||||||
|
type_str += "int";
|
||||||
|
type_str += std::to_string(sizeof(type) * 8);
|
||||||
|
|
||||||
|
INFO("type := ", type_str);
|
||||||
|
|
||||||
|
CAPTURE(val_min);
|
||||||
|
CAPTURE(min_in_range);
|
||||||
|
CAPTURE(val_max);
|
||||||
|
CAPTURE(max_in_range);
|
||||||
|
|
||||||
|
if (min_in_range)
|
||||||
|
{
|
||||||
|
CHECK(value_in_range_of<of_type>(val_min));
|
||||||
|
CHECK(value_in_range_of<of_type>(val_min2));
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
CHECK_FALSE(value_in_range_of<of_type>(val_min));
|
||||||
|
CHECK_FALSE(value_in_range_of<of_type>(val_min2));
|
||||||
|
}
|
||||||
|
|
||||||
|
if (max_in_range)
|
||||||
|
{
|
||||||
|
CHECK(value_in_range_of<of_type>(val_max));
|
||||||
|
CHECK(value_in_range_of<of_type>(val_max2));
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
CHECK_FALSE(value_in_range_of<of_type>(val_max));
|
||||||
|
CHECK_FALSE(value_in_range_of<of_type>(val_max2));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,45 +0,0 @@
|
|||||||
set(LIBFUZZER_FLAGS_BASE "${CMAKE_CXX_FLAGS}")
|
|
||||||
# Disable the coverage and sanitizer instrumentation for the fuzzer itself.
|
|
||||||
set(CMAKE_CXX_FLAGS "${LIBFUZZER_FLAGS_BASE} -fno-sanitize-coverage=trace-pc-guard,edge,trace-cmp,indirect-calls,8bit-counters -Werror")
|
|
||||||
if( LLVM_USE_SANITIZE_COVERAGE )
|
|
||||||
if(NOT "${LLVM_USE_SANITIZER}" STREQUAL "Address")
|
|
||||||
message(FATAL_ERROR
|
|
||||||
"LibFuzzer and its tests require LLVM_USE_SANITIZER=Address and "
|
|
||||||
"LLVM_USE_SANITIZE_COVERAGE=YES to be set."
|
|
||||||
)
|
|
||||||
endif()
|
|
||||||
add_library(LLVMFuzzerNoMainObjects OBJECT
|
|
||||||
FuzzerCrossOver.cpp
|
|
||||||
FuzzerDriver.cpp
|
|
||||||
FuzzerExtFunctionsDlsym.cpp
|
|
||||||
FuzzerExtFunctionsWeak.cpp
|
|
||||||
FuzzerExtFunctionsWeakAlias.cpp
|
|
||||||
FuzzerIO.cpp
|
|
||||||
FuzzerIOPosix.cpp
|
|
||||||
FuzzerIOWindows.cpp
|
|
||||||
FuzzerLoop.cpp
|
|
||||||
FuzzerMerge.cpp
|
|
||||||
FuzzerMutate.cpp
|
|
||||||
FuzzerSHA1.cpp
|
|
||||||
FuzzerTracePC.cpp
|
|
||||||
FuzzerTraceState.cpp
|
|
||||||
FuzzerUtil.cpp
|
|
||||||
FuzzerUtilDarwin.cpp
|
|
||||||
FuzzerUtilLinux.cpp
|
|
||||||
FuzzerUtilPosix.cpp
|
|
||||||
FuzzerUtilWindows.cpp
|
|
||||||
)
|
|
||||||
add_library(LLVMFuzzerNoMain STATIC
|
|
||||||
$<TARGET_OBJECTS:LLVMFuzzerNoMainObjects>
|
|
||||||
)
|
|
||||||
target_link_libraries(LLVMFuzzerNoMain ${PTHREAD_LIB})
|
|
||||||
add_library(LLVMFuzzer STATIC
|
|
||||||
FuzzerMain.cpp
|
|
||||||
$<TARGET_OBJECTS:LLVMFuzzerNoMainObjects>
|
|
||||||
)
|
|
||||||
target_link_libraries(LLVMFuzzer ${PTHREAD_LIB})
|
|
||||||
|
|
||||||
if( LLVM_INCLUDE_TESTS )
|
|
||||||
add_subdirectory(test)
|
|
||||||
endif()
|
|
||||||
endif()
|
|
||||||
@@ -1,217 +0,0 @@
|
|||||||
//===- FuzzerCorpus.h - Internal header for the Fuzzer ----------*- C++ -* ===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// fuzzer::InputCorpus
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
|
|
||||||
#ifndef LLVM_FUZZER_CORPUS
|
|
||||||
#define LLVM_FUZZER_CORPUS
|
|
||||||
|
|
||||||
#include "FuzzerDefs.h"
|
|
||||||
#include "FuzzerIO.h"
|
|
||||||
#include "FuzzerRandom.h"
|
|
||||||
#include "FuzzerSHA1.h"
|
|
||||||
#include "FuzzerTracePC.h"
|
|
||||||
#include <numeric>
|
|
||||||
#include <random>
|
|
||||||
#include <unordered_set>
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
struct InputInfo {
|
|
||||||
Unit U; // The actual input data.
|
|
||||||
uint8_t Sha1[kSHA1NumBytes]; // Checksum.
|
|
||||||
// Number of features that this input has and no smaller input has.
|
|
||||||
size_t NumFeatures = 0;
|
|
||||||
size_t Tmp = 0; // Used by ValidateFeatureSet.
|
|
||||||
// Stats.
|
|
||||||
size_t NumExecutedMutations = 0;
|
|
||||||
size_t NumSuccessfullMutations = 0;
|
|
||||||
bool MayDeleteFile = false;
|
|
||||||
};
|
|
||||||
|
|
||||||
class InputCorpus {
|
|
||||||
public:
|
|
||||||
static const size_t kFeatureSetSize = 1 << 16;
|
|
||||||
InputCorpus(const std::string &OutputCorpus) : OutputCorpus(OutputCorpus) {
|
|
||||||
memset(InputSizesPerFeature, 0, sizeof(InputSizesPerFeature));
|
|
||||||
memset(SmallestElementPerFeature, 0, sizeof(SmallestElementPerFeature));
|
|
||||||
}
|
|
||||||
~InputCorpus() {
|
|
||||||
for (auto II : Inputs)
|
|
||||||
delete II;
|
|
||||||
}
|
|
||||||
size_t size() const { return Inputs.size(); }
|
|
||||||
size_t SizeInBytes() const {
|
|
||||||
size_t Res = 0;
|
|
||||||
for (auto II : Inputs)
|
|
||||||
Res += II->U.size();
|
|
||||||
return Res;
|
|
||||||
}
|
|
||||||
size_t NumActiveUnits() const {
|
|
||||||
size_t Res = 0;
|
|
||||||
for (auto II : Inputs)
|
|
||||||
Res += !II->U.empty();
|
|
||||||
return Res;
|
|
||||||
}
|
|
||||||
bool empty() const { return Inputs.empty(); }
|
|
||||||
const Unit &operator[] (size_t Idx) const { return Inputs[Idx]->U; }
|
|
||||||
void AddToCorpus(const Unit &U, size_t NumFeatures, bool MayDeleteFile = false) {
|
|
||||||
assert(!U.empty());
|
|
||||||
uint8_t Hash[kSHA1NumBytes];
|
|
||||||
if (FeatureDebug)
|
|
||||||
Printf("ADD_TO_CORPUS %zd NF %zd\n", Inputs.size(), NumFeatures);
|
|
||||||
ComputeSHA1(U.data(), U.size(), Hash);
|
|
||||||
Hashes.insert(Sha1ToString(Hash));
|
|
||||||
Inputs.push_back(new InputInfo());
|
|
||||||
InputInfo &II = *Inputs.back();
|
|
||||||
II.U = U;
|
|
||||||
II.NumFeatures = NumFeatures;
|
|
||||||
II.MayDeleteFile = MayDeleteFile;
|
|
||||||
memcpy(II.Sha1, Hash, kSHA1NumBytes);
|
|
||||||
UpdateCorpusDistribution();
|
|
||||||
ValidateFeatureSet();
|
|
||||||
}
|
|
||||||
|
|
||||||
bool HasUnit(const Unit &U) { return Hashes.count(Hash(U)); }
|
|
||||||
bool HasUnit(const std::string &H) { return Hashes.count(H); }
|
|
||||||
InputInfo &ChooseUnitToMutate(Random &Rand) {
|
|
||||||
InputInfo &II = *Inputs[ChooseUnitIdxToMutate(Rand)];
|
|
||||||
assert(!II.U.empty());
|
|
||||||
return II;
|
|
||||||
};
|
|
||||||
|
|
||||||
// Returns an index of random unit from the corpus to mutate.
|
|
||||||
// Hypothesis: units added to the corpus last are more likely to be
|
|
||||||
// interesting. This function gives more weight to the more recent units.
|
|
||||||
size_t ChooseUnitIdxToMutate(Random &Rand) {
|
|
||||||
size_t Idx = static_cast<size_t>(CorpusDistribution(Rand.Get_mt19937()));
|
|
||||||
assert(Idx < Inputs.size());
|
|
||||||
return Idx;
|
|
||||||
}
|
|
||||||
|
|
||||||
void PrintStats() {
|
|
||||||
for (size_t i = 0; i < Inputs.size(); i++) {
|
|
||||||
const auto &II = *Inputs[i];
|
|
||||||
Printf(" [%zd %s]\tsz: %zd\truns: %zd\tsucc: %zd\n", i,
|
|
||||||
Sha1ToString(II.Sha1).c_str(), II.U.size(),
|
|
||||||
II.NumExecutedMutations, II.NumSuccessfullMutations);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void PrintFeatureSet() {
|
|
||||||
for (size_t i = 0; i < kFeatureSetSize; i++) {
|
|
||||||
if(size_t Sz = GetFeature(i))
|
|
||||||
Printf("[%zd: id %zd sz%zd] ", i, SmallestElementPerFeature[i], Sz);
|
|
||||||
}
|
|
||||||
Printf("\n\t");
|
|
||||||
for (size_t i = 0; i < Inputs.size(); i++)
|
|
||||||
if (size_t N = Inputs[i]->NumFeatures)
|
|
||||||
Printf(" %zd=>%zd ", i, N);
|
|
||||||
Printf("\n");
|
|
||||||
}
|
|
||||||
|
|
||||||
void DeleteInput(size_t Idx) {
|
|
||||||
InputInfo &II = *Inputs[Idx];
|
|
||||||
if (!OutputCorpus.empty() && II.MayDeleteFile)
|
|
||||||
RemoveFile(DirPlusFile(OutputCorpus, Sha1ToString(II.Sha1)));
|
|
||||||
Unit().swap(II.U);
|
|
||||||
if (FeatureDebug)
|
|
||||||
Printf("EVICTED %zd\n", Idx);
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AddFeature(size_t Idx, uint32_t NewSize, bool Shrink) {
|
|
||||||
assert(NewSize);
|
|
||||||
Idx = Idx % kFeatureSetSize;
|
|
||||||
uint32_t OldSize = GetFeature(Idx);
|
|
||||||
if (OldSize == 0 || (Shrink && OldSize > NewSize)) {
|
|
||||||
if (OldSize > 0) {
|
|
||||||
size_t OldIdx = SmallestElementPerFeature[Idx];
|
|
||||||
InputInfo &II = *Inputs[OldIdx];
|
|
||||||
assert(II.NumFeatures > 0);
|
|
||||||
II.NumFeatures--;
|
|
||||||
if (II.NumFeatures == 0)
|
|
||||||
DeleteInput(OldIdx);
|
|
||||||
}
|
|
||||||
if (FeatureDebug)
|
|
||||||
Printf("ADD FEATURE %zd sz %d\n", Idx, NewSize);
|
|
||||||
SmallestElementPerFeature[Idx] = Inputs.size();
|
|
||||||
InputSizesPerFeature[Idx] = NewSize;
|
|
||||||
CountingFeatures = true;
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
size_t NumFeatures() const {
|
|
||||||
size_t Res = 0;
|
|
||||||
for (size_t i = 0; i < kFeatureSetSize; i++)
|
|
||||||
Res += GetFeature(i) != 0;
|
|
||||||
return Res;
|
|
||||||
}
|
|
||||||
|
|
||||||
void ResetFeatureSet() {
|
|
||||||
assert(Inputs.empty());
|
|
||||||
memset(InputSizesPerFeature, 0, sizeof(InputSizesPerFeature));
|
|
||||||
memset(SmallestElementPerFeature, 0, sizeof(SmallestElementPerFeature));
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
|
|
||||||
static const bool FeatureDebug = false;
|
|
||||||
|
|
||||||
size_t GetFeature(size_t Idx) const { return InputSizesPerFeature[Idx]; }
|
|
||||||
|
|
||||||
void ValidateFeatureSet() {
|
|
||||||
if (!CountingFeatures) return;
|
|
||||||
if (FeatureDebug)
|
|
||||||
PrintFeatureSet();
|
|
||||||
for (size_t Idx = 0; Idx < kFeatureSetSize; Idx++)
|
|
||||||
if (GetFeature(Idx))
|
|
||||||
Inputs[SmallestElementPerFeature[Idx]]->Tmp++;
|
|
||||||
for (auto II: Inputs) {
|
|
||||||
if (II->Tmp != II->NumFeatures)
|
|
||||||
Printf("ZZZ %zd %zd\n", II->Tmp, II->NumFeatures);
|
|
||||||
assert(II->Tmp == II->NumFeatures);
|
|
||||||
II->Tmp = 0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Updates the probability distribution for the units in the corpus.
|
|
||||||
// Must be called whenever the corpus or unit weights are changed.
|
|
||||||
void UpdateCorpusDistribution() {
|
|
||||||
size_t N = Inputs.size();
|
|
||||||
Intervals.resize(N + 1);
|
|
||||||
Weights.resize(N);
|
|
||||||
std::iota(Intervals.begin(), Intervals.end(), 0);
|
|
||||||
if (CountingFeatures)
|
|
||||||
for (size_t i = 0; i < N; i++)
|
|
||||||
Weights[i] = Inputs[i]->NumFeatures * (i + 1);
|
|
||||||
else
|
|
||||||
std::iota(Weights.begin(), Weights.end(), 1);
|
|
||||||
CorpusDistribution = std::piecewise_constant_distribution<double>(
|
|
||||||
Intervals.begin(), Intervals.end(), Weights.begin());
|
|
||||||
}
|
|
||||||
std::piecewise_constant_distribution<double> CorpusDistribution;
|
|
||||||
|
|
||||||
std::vector<double> Intervals;
|
|
||||||
std::vector<double> Weights;
|
|
||||||
|
|
||||||
std::unordered_set<std::string> Hashes;
|
|
||||||
std::vector<InputInfo*> Inputs;
|
|
||||||
|
|
||||||
bool CountingFeatures = false;
|
|
||||||
uint32_t InputSizesPerFeature[kFeatureSetSize];
|
|
||||||
uint32_t SmallestElementPerFeature[kFeatureSetSize];
|
|
||||||
|
|
||||||
std::string OutputCorpus;
|
|
||||||
};
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
|
|
||||||
#endif // LLVM_FUZZER_CORPUS
|
|
||||||
@@ -1,52 +0,0 @@
|
|||||||
//===- FuzzerCrossOver.cpp - Cross over two test inputs -------------------===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// Cross over test inputs.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
|
|
||||||
#include "FuzzerDefs.h"
|
|
||||||
#include "FuzzerMutate.h"
|
|
||||||
#include "FuzzerRandom.h"
|
|
||||||
#include <cstring>
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
// Cross Data1 and Data2, store the result (up to MaxOutSize bytes) in Out.
|
|
||||||
size_t MutationDispatcher::CrossOver(const uint8_t *Data1, size_t Size1,
|
|
||||||
const uint8_t *Data2, size_t Size2,
|
|
||||||
uint8_t *Out, size_t MaxOutSize) {
|
|
||||||
assert(Size1 || Size2);
|
|
||||||
MaxOutSize = Rand(MaxOutSize) + 1;
|
|
||||||
size_t OutPos = 0;
|
|
||||||
size_t Pos1 = 0;
|
|
||||||
size_t Pos2 = 0;
|
|
||||||
size_t *InPos = &Pos1;
|
|
||||||
size_t InSize = Size1;
|
|
||||||
const uint8_t *Data = Data1;
|
|
||||||
bool CurrentlyUsingFirstData = true;
|
|
||||||
while (OutPos < MaxOutSize && (Pos1 < Size1 || Pos2 < Size2)) {
|
|
||||||
// Merge a part of Data into Out.
|
|
||||||
size_t OutSizeLeft = MaxOutSize - OutPos;
|
|
||||||
if (*InPos < InSize) {
|
|
||||||
size_t InSizeLeft = InSize - *InPos;
|
|
||||||
size_t MaxExtraSize = std::min(OutSizeLeft, InSizeLeft);
|
|
||||||
size_t ExtraSize = Rand(MaxExtraSize) + 1;
|
|
||||||
memcpy(Out + OutPos, Data + *InPos, ExtraSize);
|
|
||||||
OutPos += ExtraSize;
|
|
||||||
(*InPos) += ExtraSize;
|
|
||||||
}
|
|
||||||
// Use the other input data on the next iteration.
|
|
||||||
InPos = CurrentlyUsingFirstData ? &Pos2 : &Pos1;
|
|
||||||
InSize = CurrentlyUsingFirstData ? Size2 : Size1;
|
|
||||||
Data = CurrentlyUsingFirstData ? Data2 : Data1;
|
|
||||||
CurrentlyUsingFirstData = !CurrentlyUsingFirstData;
|
|
||||||
}
|
|
||||||
return OutPos;
|
|
||||||
}
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
@@ -1,89 +0,0 @@
|
|||||||
//===- FuzzerDefs.h - Internal header for the Fuzzer ------------*- C++ -* ===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// Basic definitions.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
|
|
||||||
#ifndef LLVM_FUZZER_DEFS_H
|
|
||||||
#define LLVM_FUZZER_DEFS_H
|
|
||||||
|
|
||||||
#include <cassert>
|
|
||||||
#include <cstddef>
|
|
||||||
#include <cstdint>
|
|
||||||
#include <cstring>
|
|
||||||
#include <string>
|
|
||||||
#include <vector>
|
|
||||||
|
|
||||||
// Platform detection.
|
|
||||||
#ifdef __linux__
|
|
||||||
#define LIBFUZZER_APPLE 0
|
|
||||||
#define LIBFUZZER_LINUX 1
|
|
||||||
#define LIBFUZZER_WINDOWS 0
|
|
||||||
#elif __APPLE__
|
|
||||||
#define LIBFUZZER_APPLE 1
|
|
||||||
#define LIBFUZZER_LINUX 0
|
|
||||||
#define LIBFUZZER_WINDOWS 0
|
|
||||||
#elif _WIN32
|
|
||||||
#define LIBFUZZER_APPLE 0
|
|
||||||
#define LIBFUZZER_LINUX 0
|
|
||||||
#define LIBFUZZER_WINDOWS 1
|
|
||||||
#else
|
|
||||||
#error "Support for your platform has not been implemented"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#define LIBFUZZER_POSIX LIBFUZZER_APPLE || LIBFUZZER_LINUX
|
|
||||||
|
|
||||||
#ifdef __x86_64
|
|
||||||
#define ATTRIBUTE_TARGET_POPCNT __attribute__((target("popcnt")))
|
|
||||||
#else
|
|
||||||
#define ATTRIBUTE_TARGET_POPCNT
|
|
||||||
#endif
|
|
||||||
|
|
||||||
|
|
||||||
#ifdef __clang__ // avoid gcc warning.
|
|
||||||
# define ATTRIBUTE_NO_SANITIZE_MEMORY __attribute__((no_sanitize("memory")))
|
|
||||||
#else
|
|
||||||
# define ATTRIBUTE_NO_SANITIZE_MEMORY
|
|
||||||
#endif
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
template <class T> T Min(T a, T b) { return a < b ? a : b; }
|
|
||||||
template <class T> T Max(T a, T b) { return a > b ? a : b; }
|
|
||||||
|
|
||||||
class Random;
|
|
||||||
class Dictionary;
|
|
||||||
class DictionaryEntry;
|
|
||||||
class MutationDispatcher;
|
|
||||||
struct FuzzingOptions;
|
|
||||||
class InputCorpus;
|
|
||||||
struct InputInfo;
|
|
||||||
struct ExternalFunctions;
|
|
||||||
|
|
||||||
// Global interface to functions that may or may not be available.
|
|
||||||
extern ExternalFunctions *EF;
|
|
||||||
|
|
||||||
typedef std::vector<uint8_t> Unit;
|
|
||||||
typedef std::vector<Unit> UnitVector;
|
|
||||||
typedef int (*UserCallback)(const uint8_t *Data, size_t Size);
|
|
||||||
|
|
||||||
int FuzzerDriver(int *argc, char ***argv, UserCallback Callback);
|
|
||||||
|
|
||||||
struct ScopedDoingMyOwnMemmem {
|
|
||||||
ScopedDoingMyOwnMemmem();
|
|
||||||
~ScopedDoingMyOwnMemmem();
|
|
||||||
};
|
|
||||||
|
|
||||||
inline uint8_t Bswap(uint8_t x) { return x; }
|
|
||||||
inline uint16_t Bswap(uint16_t x) { return __builtin_bswap16(x); }
|
|
||||||
inline uint32_t Bswap(uint32_t x) { return __builtin_bswap32(x); }
|
|
||||||
inline uint64_t Bswap(uint64_t x) { return __builtin_bswap64(x); }
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
|
|
||||||
#endif // LLVM_FUZZER_DEFS_H
|
|
||||||
@@ -1,124 +0,0 @@
|
|||||||
//===- FuzzerDictionary.h - Internal header for the Fuzzer ------*- C++ -* ===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// fuzzer::Dictionary
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
|
|
||||||
#ifndef LLVM_FUZZER_DICTIONARY_H
|
|
||||||
#define LLVM_FUZZER_DICTIONARY_H
|
|
||||||
|
|
||||||
#include "FuzzerDefs.h"
|
|
||||||
#include "FuzzerIO.h"
|
|
||||||
#include "FuzzerUtil.h"
|
|
||||||
#include <algorithm>
|
|
||||||
#include <limits>
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
// A simple POD sized array of bytes.
|
|
||||||
template <size_t kMaxSize> class FixedWord {
|
|
||||||
public:
|
|
||||||
FixedWord() {}
|
|
||||||
FixedWord(const uint8_t *B, uint8_t S) { Set(B, S); }
|
|
||||||
|
|
||||||
void Set(const uint8_t *B, uint8_t S) {
|
|
||||||
assert(S <= kMaxSize);
|
|
||||||
memcpy(Data, B, S);
|
|
||||||
Size = S;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool operator==(const FixedWord<kMaxSize> &w) const {
|
|
||||||
return Size == w.Size && 0 == memcmp(Data, w.Data, Size);
|
|
||||||
}
|
|
||||||
|
|
||||||
bool operator<(const FixedWord<kMaxSize> &w) const {
|
|
||||||
if (Size != w.Size)
|
|
||||||
return Size < w.Size;
|
|
||||||
return memcmp(Data, w.Data, Size) < 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
static size_t GetMaxSize() { return kMaxSize; }
|
|
||||||
const uint8_t *data() const { return Data; }
|
|
||||||
uint8_t size() const { return Size; }
|
|
||||||
|
|
||||||
private:
|
|
||||||
uint8_t Size = 0;
|
|
||||||
uint8_t Data[kMaxSize];
|
|
||||||
};
|
|
||||||
|
|
||||||
typedef FixedWord<27> Word; // 28 bytes.
|
|
||||||
|
|
||||||
class DictionaryEntry {
|
|
||||||
public:
|
|
||||||
DictionaryEntry() {}
|
|
||||||
DictionaryEntry(Word W) : W(W) {}
|
|
||||||
DictionaryEntry(Word W, size_t PositionHint) : W(W), PositionHint(PositionHint) {}
|
|
||||||
const Word &GetW() const { return W; }
|
|
||||||
|
|
||||||
bool HasPositionHint() const { return PositionHint != std::numeric_limits<size_t>::max(); }
|
|
||||||
size_t GetPositionHint() const {
|
|
||||||
assert(HasPositionHint());
|
|
||||||
return PositionHint;
|
|
||||||
}
|
|
||||||
void IncUseCount() { UseCount++; }
|
|
||||||
void IncSuccessCount() { SuccessCount++; }
|
|
||||||
size_t GetUseCount() const { return UseCount; }
|
|
||||||
size_t GetSuccessCount() const {return SuccessCount; }
|
|
||||||
|
|
||||||
void Print(const char *PrintAfter = "\n") {
|
|
||||||
PrintASCII(W.data(), W.size());
|
|
||||||
if (HasPositionHint())
|
|
||||||
Printf("@%zd", GetPositionHint());
|
|
||||||
Printf("%s", PrintAfter);
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
Word W;
|
|
||||||
size_t PositionHint = std::numeric_limits<size_t>::max();
|
|
||||||
size_t UseCount = 0;
|
|
||||||
size_t SuccessCount = 0;
|
|
||||||
};
|
|
||||||
|
|
||||||
class Dictionary {
|
|
||||||
public:
|
|
||||||
static const size_t kMaxDictSize = 1 << 14;
|
|
||||||
|
|
||||||
bool ContainsWord(const Word &W) const {
|
|
||||||
return std::any_of(begin(), end(), [&](const DictionaryEntry &DE) {
|
|
||||||
return DE.GetW() == W;
|
|
||||||
});
|
|
||||||
}
|
|
||||||
const DictionaryEntry *begin() const { return &DE[0]; }
|
|
||||||
const DictionaryEntry *end() const { return begin() + Size; }
|
|
||||||
DictionaryEntry & operator[] (size_t Idx) {
|
|
||||||
assert(Idx < Size);
|
|
||||||
return DE[Idx];
|
|
||||||
}
|
|
||||||
void push_back(DictionaryEntry DE) {
|
|
||||||
if (Size < kMaxDictSize)
|
|
||||||
this->DE[Size++] = DE;
|
|
||||||
}
|
|
||||||
void clear() { Size = 0; }
|
|
||||||
bool empty() const { return Size == 0; }
|
|
||||||
size_t size() const { return Size; }
|
|
||||||
|
|
||||||
private:
|
|
||||||
DictionaryEntry DE[kMaxDictSize];
|
|
||||||
size_t Size = 0;
|
|
||||||
};
|
|
||||||
|
|
||||||
// Parses one dictionary entry.
|
|
||||||
// If successfull, write the enty to Unit and returns true,
|
|
||||||
// otherwise returns false.
|
|
||||||
bool ParseOneDictionaryEntry(const std::string &Str, Unit *U);
|
|
||||||
// Parses the dictionary file, fills Units, returns true iff all lines
|
|
||||||
// were parsed succesfully.
|
|
||||||
bool ParseDictionaryFile(const std::string &Text, std::vector<Unit> *Units);
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
|
|
||||||
#endif // LLVM_FUZZER_DICTIONARY_H
|
|
||||||
@@ -1,545 +0,0 @@
|
|||||||
//===- FuzzerDriver.cpp - FuzzerDriver function and flags -----------------===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// FuzzerDriver and flag parsing.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
|
|
||||||
#include "FuzzerCorpus.h"
|
|
||||||
#include "FuzzerInterface.h"
|
|
||||||
#include "FuzzerInternal.h"
|
|
||||||
#include "FuzzerIO.h"
|
|
||||||
#include "FuzzerMutate.h"
|
|
||||||
#include "FuzzerRandom.h"
|
|
||||||
#include "FuzzerTracePC.h"
|
|
||||||
#include <algorithm>
|
|
||||||
#include <atomic>
|
|
||||||
#include <chrono>
|
|
||||||
#include <cstring>
|
|
||||||
#include <mutex>
|
|
||||||
#include <string>
|
|
||||||
#include <thread>
|
|
||||||
|
|
||||||
// This function should be present in the libFuzzer so that the client
|
|
||||||
// binary can test for its existence.
|
|
||||||
extern "C" __attribute__((used)) void __libfuzzer_is_present() {}
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
// Program arguments.
|
|
||||||
struct FlagDescription {
|
|
||||||
const char *Name;
|
|
||||||
const char *Description;
|
|
||||||
int Default;
|
|
||||||
int *IntFlag;
|
|
||||||
const char **StrFlag;
|
|
||||||
unsigned int *UIntFlag;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct {
|
|
||||||
#define FUZZER_DEPRECATED_FLAG(Name)
|
|
||||||
#define FUZZER_FLAG_INT(Name, Default, Description) int Name;
|
|
||||||
#define FUZZER_FLAG_UNSIGNED(Name, Default, Description) unsigned int Name;
|
|
||||||
#define FUZZER_FLAG_STRING(Name, Description) const char *Name;
|
|
||||||
#include "FuzzerFlags.def"
|
|
||||||
#undef FUZZER_DEPRECATED_FLAG
|
|
||||||
#undef FUZZER_FLAG_INT
|
|
||||||
#undef FUZZER_FLAG_UNSIGNED
|
|
||||||
#undef FUZZER_FLAG_STRING
|
|
||||||
} Flags;
|
|
||||||
|
|
||||||
static const FlagDescription FlagDescriptions [] {
|
|
||||||
#define FUZZER_DEPRECATED_FLAG(Name) \
|
|
||||||
{#Name, "Deprecated; don't use", 0, nullptr, nullptr, nullptr},
|
|
||||||
#define FUZZER_FLAG_INT(Name, Default, Description) \
|
|
||||||
{#Name, Description, Default, &Flags.Name, nullptr, nullptr},
|
|
||||||
#define FUZZER_FLAG_UNSIGNED(Name, Default, Description) \
|
|
||||||
{#Name, Description, static_cast<int>(Default), \
|
|
||||||
nullptr, nullptr, &Flags.Name},
|
|
||||||
#define FUZZER_FLAG_STRING(Name, Description) \
|
|
||||||
{#Name, Description, 0, nullptr, &Flags.Name, nullptr},
|
|
||||||
#include "FuzzerFlags.def"
|
|
||||||
#undef FUZZER_DEPRECATED_FLAG
|
|
||||||
#undef FUZZER_FLAG_INT
|
|
||||||
#undef FUZZER_FLAG_UNSIGNED
|
|
||||||
#undef FUZZER_FLAG_STRING
|
|
||||||
};
|
|
||||||
|
|
||||||
static const size_t kNumFlags =
|
|
||||||
sizeof(FlagDescriptions) / sizeof(FlagDescriptions[0]);
|
|
||||||
|
|
||||||
static std::vector<std::string> *Inputs;
|
|
||||||
static std::string *ProgName;
|
|
||||||
|
|
||||||
static void PrintHelp() {
|
|
||||||
Printf("Usage:\n");
|
|
||||||
auto Prog = ProgName->c_str();
|
|
||||||
Printf("\nTo run fuzzing pass 0 or more directories.\n");
|
|
||||||
Printf("%s [-flag1=val1 [-flag2=val2 ...] ] [dir1 [dir2 ...] ]\n", Prog);
|
|
||||||
|
|
||||||
Printf("\nTo run individual tests without fuzzing pass 1 or more files:\n");
|
|
||||||
Printf("%s [-flag1=val1 [-flag2=val2 ...] ] file1 [file2 ...]\n", Prog);
|
|
||||||
|
|
||||||
Printf("\nFlags: (strictly in form -flag=value)\n");
|
|
||||||
size_t MaxFlagLen = 0;
|
|
||||||
for (size_t F = 0; F < kNumFlags; F++)
|
|
||||||
MaxFlagLen = std::max(strlen(FlagDescriptions[F].Name), MaxFlagLen);
|
|
||||||
|
|
||||||
for (size_t F = 0; F < kNumFlags; F++) {
|
|
||||||
const auto &D = FlagDescriptions[F];
|
|
||||||
if (strstr(D.Description, "internal flag") == D.Description) continue;
|
|
||||||
Printf(" %s", D.Name);
|
|
||||||
for (size_t i = 0, n = MaxFlagLen - strlen(D.Name); i < n; i++)
|
|
||||||
Printf(" ");
|
|
||||||
Printf("\t");
|
|
||||||
Printf("%d\t%s\n", D.Default, D.Description);
|
|
||||||
}
|
|
||||||
Printf("\nFlags starting with '--' will be ignored and "
|
|
||||||
"will be passed verbatim to subprocesses.\n");
|
|
||||||
}
|
|
||||||
|
|
||||||
static const char *FlagValue(const char *Param, const char *Name) {
|
|
||||||
size_t Len = strlen(Name);
|
|
||||||
if (Param[0] == '-' && strstr(Param + 1, Name) == Param + 1 &&
|
|
||||||
Param[Len + 1] == '=')
|
|
||||||
return &Param[Len + 2];
|
|
||||||
return nullptr;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Avoid calling stol as it triggers a bug in clang/glibc build.
|
|
||||||
static long MyStol(const char *Str) {
|
|
||||||
long Res = 0;
|
|
||||||
long Sign = 1;
|
|
||||||
if (*Str == '-') {
|
|
||||||
Str++;
|
|
||||||
Sign = -1;
|
|
||||||
}
|
|
||||||
for (size_t i = 0; Str[i]; i++) {
|
|
||||||
char Ch = Str[i];
|
|
||||||
if (Ch < '0' || Ch > '9')
|
|
||||||
return Res;
|
|
||||||
Res = Res * 10 + (Ch - '0');
|
|
||||||
}
|
|
||||||
return Res * Sign;
|
|
||||||
}
|
|
||||||
|
|
||||||
static bool ParseOneFlag(const char *Param) {
|
|
||||||
if (Param[0] != '-') return false;
|
|
||||||
if (Param[1] == '-') {
|
|
||||||
static bool PrintedWarning = false;
|
|
||||||
if (!PrintedWarning) {
|
|
||||||
PrintedWarning = true;
|
|
||||||
Printf("INFO: libFuzzer ignores flags that start with '--'\n");
|
|
||||||
}
|
|
||||||
for (size_t F = 0; F < kNumFlags; F++)
|
|
||||||
if (FlagValue(Param + 1, FlagDescriptions[F].Name))
|
|
||||||
Printf("WARNING: did you mean '%s' (single dash)?\n", Param + 1);
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
for (size_t F = 0; F < kNumFlags; F++) {
|
|
||||||
const char *Name = FlagDescriptions[F].Name;
|
|
||||||
const char *Str = FlagValue(Param, Name);
|
|
||||||
if (Str) {
|
|
||||||
if (FlagDescriptions[F].IntFlag) {
|
|
||||||
int Val = MyStol(Str);
|
|
||||||
*FlagDescriptions[F].IntFlag = Val;
|
|
||||||
if (Flags.verbosity >= 2)
|
|
||||||
Printf("Flag: %s %d\n", Name, Val);
|
|
||||||
return true;
|
|
||||||
} else if (FlagDescriptions[F].UIntFlag) {
|
|
||||||
unsigned int Val = std::stoul(Str);
|
|
||||||
*FlagDescriptions[F].UIntFlag = Val;
|
|
||||||
if (Flags.verbosity >= 2)
|
|
||||||
Printf("Flag: %s %u\n", Name, Val);
|
|
||||||
return true;
|
|
||||||
} else if (FlagDescriptions[F].StrFlag) {
|
|
||||||
*FlagDescriptions[F].StrFlag = Str;
|
|
||||||
if (Flags.verbosity >= 2)
|
|
||||||
Printf("Flag: %s %s\n", Name, Str);
|
|
||||||
return true;
|
|
||||||
} else { // Deprecated flag.
|
|
||||||
Printf("Flag: %s: deprecated, don't use\n", Name);
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Printf("\n\nWARNING: unrecognized flag '%s'; "
|
|
||||||
"use -help=1 to list all flags\n\n", Param);
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// We don't use any library to minimize dependencies.
|
|
||||||
static void ParseFlags(const std::vector<std::string> &Args) {
|
|
||||||
for (size_t F = 0; F < kNumFlags; F++) {
|
|
||||||
if (FlagDescriptions[F].IntFlag)
|
|
||||||
*FlagDescriptions[F].IntFlag = FlagDescriptions[F].Default;
|
|
||||||
if (FlagDescriptions[F].UIntFlag)
|
|
||||||
*FlagDescriptions[F].UIntFlag =
|
|
||||||
static_cast<unsigned int>(FlagDescriptions[F].Default);
|
|
||||||
if (FlagDescriptions[F].StrFlag)
|
|
||||||
*FlagDescriptions[F].StrFlag = nullptr;
|
|
||||||
}
|
|
||||||
Inputs = new std::vector<std::string>;
|
|
||||||
for (size_t A = 1; A < Args.size(); A++) {
|
|
||||||
if (ParseOneFlag(Args[A].c_str())) continue;
|
|
||||||
Inputs->push_back(Args[A]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
static std::mutex Mu;
|
|
||||||
|
|
||||||
static void PulseThread() {
|
|
||||||
while (true) {
|
|
||||||
SleepSeconds(600);
|
|
||||||
std::lock_guard<std::mutex> Lock(Mu);
|
|
||||||
Printf("pulse...\n");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
static void WorkerThread(const std::string &Cmd, std::atomic<unsigned> *Counter,
|
|
||||||
unsigned NumJobs, std::atomic<bool> *HasErrors) {
|
|
||||||
while (true) {
|
|
||||||
unsigned C = (*Counter)++;
|
|
||||||
if (C >= NumJobs) break;
|
|
||||||
std::string Log = "fuzz-" + std::to_string(C) + ".log";
|
|
||||||
std::string ToRun = Cmd + " > " + Log + " 2>&1\n";
|
|
||||||
if (Flags.verbosity)
|
|
||||||
Printf("%s", ToRun.c_str());
|
|
||||||
int ExitCode = ExecuteCommand(ToRun);
|
|
||||||
if (ExitCode != 0)
|
|
||||||
*HasErrors = true;
|
|
||||||
std::lock_guard<std::mutex> Lock(Mu);
|
|
||||||
Printf("================== Job %u exited with exit code %d ============\n",
|
|
||||||
C, ExitCode);
|
|
||||||
fuzzer::CopyFileToErr(Log);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string CloneArgsWithoutX(const std::vector<std::string> &Args,
|
|
||||||
const char *X1, const char *X2) {
|
|
||||||
std::string Cmd;
|
|
||||||
for (auto &S : Args) {
|
|
||||||
if (FlagValue(S.c_str(), X1) || FlagValue(S.c_str(), X2))
|
|
||||||
continue;
|
|
||||||
Cmd += S + " ";
|
|
||||||
}
|
|
||||||
return Cmd;
|
|
||||||
}
|
|
||||||
|
|
||||||
static int RunInMultipleProcesses(const std::vector<std::string> &Args,
|
|
||||||
unsigned NumWorkers, unsigned NumJobs) {
|
|
||||||
std::atomic<unsigned> Counter(0);
|
|
||||||
std::atomic<bool> HasErrors(false);
|
|
||||||
std::string Cmd = CloneArgsWithoutX(Args, "jobs", "workers");
|
|
||||||
std::vector<std::thread> V;
|
|
||||||
std::thread Pulse(PulseThread);
|
|
||||||
Pulse.detach();
|
|
||||||
for (unsigned i = 0; i < NumWorkers; i++)
|
|
||||||
V.push_back(std::thread(WorkerThread, Cmd, &Counter, NumJobs, &HasErrors));
|
|
||||||
for (auto &T : V)
|
|
||||||
T.join();
|
|
||||||
return HasErrors ? 1 : 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
static void RssThread(Fuzzer *F, size_t RssLimitMb) {
|
|
||||||
while (true) {
|
|
||||||
SleepSeconds(1);
|
|
||||||
size_t Peak = GetPeakRSSMb();
|
|
||||||
if (Peak > RssLimitMb)
|
|
||||||
F->RssLimitCallback();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
static void StartRssThread(Fuzzer *F, size_t RssLimitMb) {
|
|
||||||
if (!RssLimitMb) return;
|
|
||||||
std::thread T(RssThread, F, RssLimitMb);
|
|
||||||
T.detach();
|
|
||||||
}
|
|
||||||
|
|
||||||
int RunOneTest(Fuzzer *F, const char *InputFilePath, size_t MaxLen) {
|
|
||||||
Unit U = FileToVector(InputFilePath);
|
|
||||||
if (MaxLen && MaxLen < U.size())
|
|
||||||
U.resize(MaxLen);
|
|
||||||
F->RunOne(U.data(), U.size());
|
|
||||||
F->TryDetectingAMemoryLeak(U.data(), U.size(), true);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
static bool AllInputsAreFiles() {
|
|
||||||
if (Inputs->empty()) return false;
|
|
||||||
for (auto &Path : *Inputs)
|
|
||||||
if (!IsFile(Path))
|
|
||||||
return false;
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
int MinimizeCrashInput(const std::vector<std::string> &Args) {
|
|
||||||
if (Inputs->size() != 1) {
|
|
||||||
Printf("ERROR: -minimize_crash should be given one input file\n");
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
std::string InputFilePath = Inputs->at(0);
|
|
||||||
std::string BaseCmd =
|
|
||||||
CloneArgsWithoutX(Args, "minimize_crash", "exact_artifact_path");
|
|
||||||
auto InputPos = BaseCmd.find(" " + InputFilePath + " ");
|
|
||||||
assert(InputPos != std::string::npos);
|
|
||||||
BaseCmd.erase(InputPos, InputFilePath.size() + 1);
|
|
||||||
if (Flags.runs <= 0 && Flags.max_total_time == 0) {
|
|
||||||
Printf("INFO: you need to specify -runs=N or "
|
|
||||||
"-max_total_time=N with -minimize_crash=1\n"
|
|
||||||
"INFO: defaulting to -max_total_time=600\n");
|
|
||||||
BaseCmd += " -max_total_time=600";
|
|
||||||
}
|
|
||||||
// BaseCmd += " > /dev/null 2>&1 ";
|
|
||||||
|
|
||||||
std::string CurrentFilePath = InputFilePath;
|
|
||||||
while (true) {
|
|
||||||
Unit U = FileToVector(CurrentFilePath);
|
|
||||||
if (U.size() < 2) {
|
|
||||||
Printf("CRASH_MIN: '%s' is small enough\n", CurrentFilePath.c_str());
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
Printf("CRASH_MIN: minimizing crash input: '%s' (%zd bytes)\n",
|
|
||||||
CurrentFilePath.c_str(), U.size());
|
|
||||||
|
|
||||||
auto Cmd = BaseCmd + " " + CurrentFilePath;
|
|
||||||
|
|
||||||
Printf("CRASH_MIN: executing: %s\n", Cmd.c_str());
|
|
||||||
int ExitCode = ExecuteCommand(Cmd);
|
|
||||||
if (ExitCode == 0) {
|
|
||||||
Printf("ERROR: the input %s did not crash\n", CurrentFilePath.c_str());
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
Printf("CRASH_MIN: '%s' (%zd bytes) caused a crash. Will try to minimize "
|
|
||||||
"it further\n",
|
|
||||||
CurrentFilePath.c_str(), U.size());
|
|
||||||
|
|
||||||
std::string ArtifactPath = "minimized-from-" + Hash(U);
|
|
||||||
Cmd += " -minimize_crash_internal_step=1 -exact_artifact_path=" +
|
|
||||||
ArtifactPath;
|
|
||||||
Printf("CRASH_MIN: executing: %s\n", Cmd.c_str());
|
|
||||||
ExitCode = ExecuteCommand(Cmd);
|
|
||||||
if (ExitCode == 0) {
|
|
||||||
if (Flags.exact_artifact_path) {
|
|
||||||
CurrentFilePath = Flags.exact_artifact_path;
|
|
||||||
WriteToFile(U, CurrentFilePath);
|
|
||||||
}
|
|
||||||
Printf("CRASH_MIN: failed to minimize beyond %s (%d bytes), exiting\n",
|
|
||||||
CurrentFilePath.c_str(), U.size());
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
CurrentFilePath = ArtifactPath;
|
|
||||||
Printf("\n\n\n\n\n\n*********************************\n");
|
|
||||||
}
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
int MinimizeCrashInputInternalStep(Fuzzer *F, InputCorpus *Corpus) {
|
|
||||||
assert(Inputs->size() == 1);
|
|
||||||
std::string InputFilePath = Inputs->at(0);
|
|
||||||
Unit U = FileToVector(InputFilePath);
|
|
||||||
assert(U.size() > 2);
|
|
||||||
Printf("INFO: Starting MinimizeCrashInputInternalStep: %zd\n", U.size());
|
|
||||||
Corpus->AddToCorpus(U, 0);
|
|
||||||
F->SetMaxInputLen(U.size());
|
|
||||||
F->SetMaxMutationLen(U.size() - 1);
|
|
||||||
F->MinimizeCrashLoop(U);
|
|
||||||
Printf("INFO: Done MinimizeCrashInputInternalStep, no crashes found\n");
|
|
||||||
exit(0);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
int FuzzerDriver(int *argc, char ***argv, UserCallback Callback) {
|
|
||||||
using namespace fuzzer;
|
|
||||||
assert(argc && argv && "Argument pointers cannot be nullptr");
|
|
||||||
EF = new ExternalFunctions();
|
|
||||||
if (EF->LLVMFuzzerInitialize)
|
|
||||||
EF->LLVMFuzzerInitialize(argc, argv);
|
|
||||||
const std::vector<std::string> Args(*argv, *argv + *argc);
|
|
||||||
assert(!Args.empty());
|
|
||||||
ProgName = new std::string(Args[0]);
|
|
||||||
ParseFlags(Args);
|
|
||||||
if (Flags.help) {
|
|
||||||
PrintHelp();
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (Flags.minimize_crash)
|
|
||||||
return MinimizeCrashInput(Args);
|
|
||||||
|
|
||||||
if (Flags.close_fd_mask & 2)
|
|
||||||
DupAndCloseStderr();
|
|
||||||
if (Flags.close_fd_mask & 1)
|
|
||||||
CloseStdout();
|
|
||||||
|
|
||||||
if (Flags.jobs > 0 && Flags.workers == 0) {
|
|
||||||
Flags.workers = std::min(NumberOfCpuCores() / 2, Flags.jobs);
|
|
||||||
if (Flags.workers > 1)
|
|
||||||
Printf("Running %u workers\n", Flags.workers);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (Flags.workers > 0 && Flags.jobs > 0)
|
|
||||||
return RunInMultipleProcesses(Args, Flags.workers, Flags.jobs);
|
|
||||||
|
|
||||||
const size_t kMaxSaneLen = 1 << 20;
|
|
||||||
const size_t kMinDefaultLen = 64;
|
|
||||||
FuzzingOptions Options;
|
|
||||||
Options.Verbosity = Flags.verbosity;
|
|
||||||
Options.MaxLen = Flags.max_len;
|
|
||||||
Options.UnitTimeoutSec = Flags.timeout;
|
|
||||||
Options.ErrorExitCode = Flags.error_exitcode;
|
|
||||||
Options.TimeoutExitCode = Flags.timeout_exitcode;
|
|
||||||
Options.MaxTotalTimeSec = Flags.max_total_time;
|
|
||||||
Options.DoCrossOver = Flags.cross_over;
|
|
||||||
Options.MutateDepth = Flags.mutate_depth;
|
|
||||||
Options.UseCounters = Flags.use_counters;
|
|
||||||
Options.UseIndirCalls = Flags.use_indir_calls;
|
|
||||||
Options.UseMemcmp = Flags.use_memcmp;
|
|
||||||
Options.UseMemmem = Flags.use_memmem;
|
|
||||||
Options.UseCmp = Flags.use_cmp;
|
|
||||||
Options.UseValueProfile = Flags.use_value_profile;
|
|
||||||
Options.Shrink = Flags.shrink;
|
|
||||||
Options.ShuffleAtStartUp = Flags.shuffle;
|
|
||||||
Options.PreferSmall = Flags.prefer_small;
|
|
||||||
Options.ReloadIntervalSec = Flags.reload;
|
|
||||||
Options.OnlyASCII = Flags.only_ascii;
|
|
||||||
Options.OutputCSV = Flags.output_csv;
|
|
||||||
Options.DetectLeaks = Flags.detect_leaks;
|
|
||||||
Options.TraceMalloc = Flags.trace_malloc;
|
|
||||||
Options.RssLimitMb = Flags.rss_limit_mb;
|
|
||||||
if (Flags.runs >= 0)
|
|
||||||
Options.MaxNumberOfRuns = Flags.runs;
|
|
||||||
if (!Inputs->empty() && !Flags.minimize_crash_internal_step)
|
|
||||||
Options.OutputCorpus = (*Inputs)[0];
|
|
||||||
Options.ReportSlowUnits = Flags.report_slow_units;
|
|
||||||
if (Flags.artifact_prefix)
|
|
||||||
Options.ArtifactPrefix = Flags.artifact_prefix;
|
|
||||||
if (Flags.exact_artifact_path)
|
|
||||||
Options.ExactArtifactPath = Flags.exact_artifact_path;
|
|
||||||
std::vector<Unit> Dictionary;
|
|
||||||
if (Flags.dict)
|
|
||||||
if (!ParseDictionaryFile(FileToString(Flags.dict), &Dictionary))
|
|
||||||
return 1;
|
|
||||||
if (Flags.verbosity > 0 && !Dictionary.empty())
|
|
||||||
Printf("Dictionary: %zd entries\n", Dictionary.size());
|
|
||||||
bool DoPlainRun = AllInputsAreFiles();
|
|
||||||
Options.SaveArtifacts =
|
|
||||||
!DoPlainRun || Flags.minimize_crash_internal_step;
|
|
||||||
Options.PrintNewCovPcs = Flags.print_pcs;
|
|
||||||
Options.PrintFinalStats = Flags.print_final_stats;
|
|
||||||
Options.PrintCorpusStats = Flags.print_corpus_stats;
|
|
||||||
Options.PrintCoverage = Flags.print_coverage;
|
|
||||||
Options.DumpCoverage = Flags.dump_coverage;
|
|
||||||
if (Flags.exit_on_src_pos)
|
|
||||||
Options.ExitOnSrcPos = Flags.exit_on_src_pos;
|
|
||||||
if (Flags.exit_on_item)
|
|
||||||
Options.ExitOnItem = Flags.exit_on_item;
|
|
||||||
|
|
||||||
unsigned Seed = Flags.seed;
|
|
||||||
// Initialize Seed.
|
|
||||||
if (Seed == 0)
|
|
||||||
Seed = (std::chrono::system_clock::now().time_since_epoch().count() << 10) +
|
|
||||||
GetPid();
|
|
||||||
if (Flags.verbosity)
|
|
||||||
Printf("INFO: Seed: %u\n", Seed);
|
|
||||||
|
|
||||||
Random Rand(Seed);
|
|
||||||
auto *MD = new MutationDispatcher(Rand, Options);
|
|
||||||
auto *Corpus = new InputCorpus(Options.OutputCorpus);
|
|
||||||
auto *F = new Fuzzer(Callback, *Corpus, *MD, Options);
|
|
||||||
|
|
||||||
for (auto &U: Dictionary)
|
|
||||||
if (U.size() <= Word::GetMaxSize())
|
|
||||||
MD->AddWordToManualDictionary(Word(U.data(), U.size()));
|
|
||||||
|
|
||||||
StartRssThread(F, Flags.rss_limit_mb);
|
|
||||||
|
|
||||||
Options.HandleAbrt = Flags.handle_abrt;
|
|
||||||
Options.HandleBus = Flags.handle_bus;
|
|
||||||
Options.HandleFpe = Flags.handle_fpe;
|
|
||||||
Options.HandleIll = Flags.handle_ill;
|
|
||||||
Options.HandleInt = Flags.handle_int;
|
|
||||||
Options.HandleSegv = Flags.handle_segv;
|
|
||||||
Options.HandleTerm = Flags.handle_term;
|
|
||||||
SetSignalHandler(Options);
|
|
||||||
|
|
||||||
if (Flags.minimize_crash_internal_step)
|
|
||||||
return MinimizeCrashInputInternalStep(F, Corpus);
|
|
||||||
|
|
||||||
if (DoPlainRun) {
|
|
||||||
Options.SaveArtifacts = false;
|
|
||||||
int Runs = std::max(1, Flags.runs);
|
|
||||||
Printf("%s: Running %zd inputs %d time(s) each.\n", ProgName->c_str(),
|
|
||||||
Inputs->size(), Runs);
|
|
||||||
for (auto &Path : *Inputs) {
|
|
||||||
auto StartTime = system_clock::now();
|
|
||||||
Printf("Running: %s\n", Path.c_str());
|
|
||||||
for (int Iter = 0; Iter < Runs; Iter++)
|
|
||||||
RunOneTest(F, Path.c_str(), Options.MaxLen);
|
|
||||||
auto StopTime = system_clock::now();
|
|
||||||
auto MS = duration_cast<milliseconds>(StopTime - StartTime).count();
|
|
||||||
Printf("Executed %s in %zd ms\n", Path.c_str(), (long)MS);
|
|
||||||
}
|
|
||||||
Printf("***\n"
|
|
||||||
"*** NOTE: fuzzing was not performed, you have only\n"
|
|
||||||
"*** executed the target code on a fixed set of inputs.\n"
|
|
||||||
"***\n");
|
|
||||||
F->PrintFinalStats();
|
|
||||||
exit(0);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (Flags.merge) {
|
|
||||||
if (Options.MaxLen == 0)
|
|
||||||
F->SetMaxInputLen(kMaxSaneLen);
|
|
||||||
if (TPC.UsingTracePcGuard()) {
|
|
||||||
if (Flags.merge_control_file)
|
|
||||||
F->CrashResistantMergeInternalStep(Flags.merge_control_file);
|
|
||||||
else
|
|
||||||
F->CrashResistantMerge(Args, *Inputs);
|
|
||||||
} else {
|
|
||||||
F->Merge(*Inputs);
|
|
||||||
}
|
|
||||||
exit(0);
|
|
||||||
}
|
|
||||||
|
|
||||||
size_t TemporaryMaxLen = Options.MaxLen ? Options.MaxLen : kMaxSaneLen;
|
|
||||||
|
|
||||||
UnitVector InitialCorpus;
|
|
||||||
for (auto &Inp : *Inputs) {
|
|
||||||
Printf("Loading corpus dir: %s\n", Inp.c_str());
|
|
||||||
ReadDirToVectorOfUnits(Inp.c_str(), &InitialCorpus, nullptr,
|
|
||||||
TemporaryMaxLen, /*ExitOnError=*/false);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (Options.MaxLen == 0) {
|
|
||||||
size_t MaxLen = 0;
|
|
||||||
for (auto &U : InitialCorpus)
|
|
||||||
MaxLen = std::max(U.size(), MaxLen);
|
|
||||||
F->SetMaxInputLen(std::min(std::max(kMinDefaultLen, MaxLen), kMaxSaneLen));
|
|
||||||
}
|
|
||||||
|
|
||||||
if (InitialCorpus.empty()) {
|
|
||||||
InitialCorpus.push_back(Unit({'\n'})); // Valid ASCII input.
|
|
||||||
if (Options.Verbosity)
|
|
||||||
Printf("INFO: A corpus is not provided, starting from an empty corpus\n");
|
|
||||||
}
|
|
||||||
F->ShuffleAndMinimize(&InitialCorpus);
|
|
||||||
InitialCorpus.clear(); // Don't need this memory any more.
|
|
||||||
F->Loop();
|
|
||||||
|
|
||||||
if (Flags.verbosity)
|
|
||||||
Printf("Done %d runs in %zd second(s)\n", F->getTotalNumberOfRuns(),
|
|
||||||
F->secondsSinceProcessStartUp());
|
|
||||||
F->PrintFinalStats();
|
|
||||||
|
|
||||||
exit(0); // Don't let F destroy itself.
|
|
||||||
}
|
|
||||||
|
|
||||||
// Storage for global ExternalFunctions object.
|
|
||||||
ExternalFunctions *EF = nullptr;
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
@@ -1,50 +0,0 @@
|
|||||||
//===- FuzzerExtFunctions.def - External functions --------------*- C++ -* ===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// This defines the external function pointers that
|
|
||||||
// ``fuzzer::ExternalFunctions`` should contain and try to initialize. The
|
|
||||||
// EXT_FUNC macro must be defined at the point of inclusion. The signature of
|
|
||||||
// the macro is:
|
|
||||||
//
|
|
||||||
// EXT_FUNC(<name>, <return_type>, <function_signature>, <warn_if_missing>)
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
|
|
||||||
// Optional user functions
|
|
||||||
EXT_FUNC(LLVMFuzzerInitialize, int, (int *argc, char ***argv), false);
|
|
||||||
EXT_FUNC(LLVMFuzzerCustomMutator, size_t,
|
|
||||||
(uint8_t * Data, size_t Size, size_t MaxSize, unsigned int Seed),
|
|
||||||
false);
|
|
||||||
EXT_FUNC(LLVMFuzzerCustomCrossOver, size_t,
|
|
||||||
(const uint8_t * Data1, size_t Size1,
|
|
||||||
const uint8_t * Data2, size_t Size2,
|
|
||||||
uint8_t * Out, size_t MaxOutSize, unsigned int Seed),
|
|
||||||
false);
|
|
||||||
|
|
||||||
// Sanitizer functions
|
|
||||||
EXT_FUNC(__lsan_enable, void, (), false);
|
|
||||||
EXT_FUNC(__lsan_disable, void, (), false);
|
|
||||||
EXT_FUNC(__lsan_do_recoverable_leak_check, int, (), false);
|
|
||||||
EXT_FUNC(__sanitizer_get_number_of_counters, size_t, (), false);
|
|
||||||
EXT_FUNC(__sanitizer_install_malloc_and_free_hooks, int,
|
|
||||||
(void (*malloc_hook)(const volatile void *, size_t),
|
|
||||||
void (*free_hook)(const volatile void *)),
|
|
||||||
false);
|
|
||||||
EXT_FUNC(__sanitizer_get_total_unique_caller_callee_pairs, size_t, (), false);
|
|
||||||
EXT_FUNC(__sanitizer_get_total_unique_coverage, size_t, (), true);
|
|
||||||
EXT_FUNC(__sanitizer_print_memory_profile, int, (size_t), false);
|
|
||||||
EXT_FUNC(__sanitizer_print_stack_trace, void, (), true);
|
|
||||||
EXT_FUNC(__sanitizer_symbolize_pc, void,
|
|
||||||
(void *, const char *fmt, char *out_buf, size_t out_buf_size), false);
|
|
||||||
EXT_FUNC(__sanitizer_get_module_and_offset_for_pc, int,
|
|
||||||
(void *pc, char *module_path,
|
|
||||||
size_t module_path_len,void **pc_offset), false);
|
|
||||||
EXT_FUNC(__sanitizer_reset_coverage, void, (), true);
|
|
||||||
EXT_FUNC(__sanitizer_set_death_callback, void, (void (*)(void)), true);
|
|
||||||
EXT_FUNC(__sanitizer_set_report_fd, void, (void*), false);
|
|
||||||
EXT_FUNC(__sanitizer_update_counter_bitset_and_clear_counters, uintptr_t,
|
|
||||||
(uint8_t*), false);
|
|
||||||
@@ -1,35 +0,0 @@
|
|||||||
//===- FuzzerExtFunctions.h - Interface to external functions ---*- C++ -* ===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// Defines an interface to (possibly optional) functions.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
|
|
||||||
#ifndef LLVM_FUZZER_EXT_FUNCTIONS_H
|
|
||||||
#define LLVM_FUZZER_EXT_FUNCTIONS_H
|
|
||||||
|
|
||||||
#include <stddef.h>
|
|
||||||
#include <stdint.h>
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
struct ExternalFunctions {
|
|
||||||
// Initialize function pointers. Functions that are not available will be set
|
|
||||||
// to nullptr. Do not call this constructor before ``main()`` has been
|
|
||||||
// entered.
|
|
||||||
ExternalFunctions();
|
|
||||||
|
|
||||||
#define EXT_FUNC(NAME, RETURN_TYPE, FUNC_SIG, WARN) \
|
|
||||||
RETURN_TYPE(*NAME) FUNC_SIG = nullptr
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.def"
|
|
||||||
|
|
||||||
#undef EXT_FUNC
|
|
||||||
};
|
|
||||||
} // namespace fuzzer
|
|
||||||
|
|
||||||
#endif
|
|
||||||
@@ -1,52 +0,0 @@
|
|||||||
//===- FuzzerExtFunctionsDlsym.cpp - Interface to external functions ------===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// Implementation for operating systems that support dlsym(). We only use it on
|
|
||||||
// Apple platforms for now. We don't use this approach on Linux because it
|
|
||||||
// requires that clients of LibFuzzer pass ``--export-dynamic`` to the linker.
|
|
||||||
// That is a complication we don't wish to expose to clients right now.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
#include "FuzzerDefs.h"
|
|
||||||
#if LIBFUZZER_APPLE
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.h"
|
|
||||||
#include "FuzzerIO.h"
|
|
||||||
#include <dlfcn.h>
|
|
||||||
|
|
||||||
using namespace fuzzer;
|
|
||||||
|
|
||||||
template <typename T>
|
|
||||||
static T GetFnPtr(const char *FnName, bool WarnIfMissing) {
|
|
||||||
dlerror(); // Clear any previous errors.
|
|
||||||
void *Fn = dlsym(RTLD_DEFAULT, FnName);
|
|
||||||
if (Fn == nullptr) {
|
|
||||||
if (WarnIfMissing) {
|
|
||||||
const char *ErrorMsg = dlerror();
|
|
||||||
Printf("WARNING: Failed to find function \"%s\".", FnName);
|
|
||||||
if (ErrorMsg)
|
|
||||||
Printf(" Reason %s.", ErrorMsg);
|
|
||||||
Printf("\n");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return reinterpret_cast<T>(Fn);
|
|
||||||
}
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
ExternalFunctions::ExternalFunctions() {
|
|
||||||
#define EXT_FUNC(NAME, RETURN_TYPE, FUNC_SIG, WARN) \
|
|
||||||
this->NAME = GetFnPtr<decltype(ExternalFunctions::NAME)>(#NAME, WARN)
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.def"
|
|
||||||
|
|
||||||
#undef EXT_FUNC
|
|
||||||
}
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
|
|
||||||
#endif // LIBFUZZER_APPLE
|
|
||||||
@@ -1,53 +0,0 @@
|
|||||||
//===- FuzzerExtFunctionsWeak.cpp - Interface to external functions -------===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// Implementation for Linux. This relies on the linker's support for weak
|
|
||||||
// symbols. We don't use this approach on Apple platforms because it requires
|
|
||||||
// clients of LibFuzzer to pass ``-U _<symbol_name>`` to the linker to allow
|
|
||||||
// weak symbols to be undefined. That is a complication we don't want to expose
|
|
||||||
// to clients right now.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
#include "FuzzerDefs.h"
|
|
||||||
#if LIBFUZZER_LINUX
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.h"
|
|
||||||
#include "FuzzerIO.h"
|
|
||||||
|
|
||||||
extern "C" {
|
|
||||||
// Declare these symbols as weak to allow them to be optionally defined.
|
|
||||||
#define EXT_FUNC(NAME, RETURN_TYPE, FUNC_SIG, WARN) \
|
|
||||||
__attribute__((weak)) RETURN_TYPE NAME FUNC_SIG
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.def"
|
|
||||||
|
|
||||||
#undef EXT_FUNC
|
|
||||||
}
|
|
||||||
|
|
||||||
using namespace fuzzer;
|
|
||||||
|
|
||||||
static void CheckFnPtr(void *FnPtr, const char *FnName, bool WarnIfMissing) {
|
|
||||||
if (FnPtr == nullptr && WarnIfMissing) {
|
|
||||||
Printf("WARNING: Failed to find function \"%s\".\n", FnName);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
ExternalFunctions::ExternalFunctions() {
|
|
||||||
#define EXT_FUNC(NAME, RETURN_TYPE, FUNC_SIG, WARN) \
|
|
||||||
this->NAME = ::NAME; \
|
|
||||||
CheckFnPtr((void *)::NAME, #NAME, WARN);
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.def"
|
|
||||||
|
|
||||||
#undef EXT_FUNC
|
|
||||||
}
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
|
|
||||||
#endif // LIBFUZZER_LINUX
|
|
||||||
@@ -1,56 +0,0 @@
|
|||||||
//===- FuzzerExtFunctionsWeakAlias.cpp - Interface to external functions --===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// Implementation using weak aliases. Works for Windows.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
#include "FuzzerDefs.h"
|
|
||||||
#if LIBFUZZER_WINDOWS
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.h"
|
|
||||||
#include "FuzzerIO.h"
|
|
||||||
|
|
||||||
using namespace fuzzer;
|
|
||||||
|
|
||||||
extern "C" {
|
|
||||||
// Declare these symbols as weak to allow them to be optionally defined.
|
|
||||||
#define EXT_FUNC(NAME, RETURN_TYPE, FUNC_SIG, WARN) \
|
|
||||||
RETURN_TYPE NAME##Def FUNC_SIG { \
|
|
||||||
Printf("ERROR: Function \"%s\" not defined.\n", #NAME); \
|
|
||||||
exit(1); \
|
|
||||||
} \
|
|
||||||
RETURN_TYPE NAME FUNC_SIG __attribute__((weak, alias(#NAME "Def")));
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.def"
|
|
||||||
|
|
||||||
#undef EXT_FUNC
|
|
||||||
}
|
|
||||||
|
|
||||||
template <typename T>
|
|
||||||
static T *GetFnPtr(T *Fun, T *FunDef, const char *FnName, bool WarnIfMissing) {
|
|
||||||
if (Fun == FunDef) {
|
|
||||||
if (WarnIfMissing)
|
|
||||||
Printf("WARNING: Failed to find function \"%s\".\n", FnName);
|
|
||||||
return nullptr;
|
|
||||||
}
|
|
||||||
return Fun;
|
|
||||||
}
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
ExternalFunctions::ExternalFunctions() {
|
|
||||||
#define EXT_FUNC(NAME, RETURN_TYPE, FUNC_SIG, WARN) \
|
|
||||||
this->NAME = GetFnPtr<decltype(::NAME)>(::NAME, ::NAME##Def, #NAME, WARN);
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.def"
|
|
||||||
|
|
||||||
#undef EXT_FUNC
|
|
||||||
}
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
|
|
||||||
#endif // LIBFUZZER_WINDOWS
|
|
||||||
@@ -1,115 +0,0 @@
|
|||||||
//===- FuzzerFlags.def - Run-time flags -------------------------*- C++ -* ===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// Flags. FUZZER_FLAG_INT/FUZZER_FLAG_STRING macros should be defined at the
|
|
||||||
// point of inclusion. We are not using any flag parsing library for better
|
|
||||||
// portability and independence.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
FUZZER_FLAG_INT(verbosity, 1, "Verbosity level.")
|
|
||||||
FUZZER_FLAG_UNSIGNED(seed, 0, "Random seed. If 0, seed is generated.")
|
|
||||||
FUZZER_FLAG_INT(runs, -1,
|
|
||||||
"Number of individual test runs (-1 for infinite runs).")
|
|
||||||
FUZZER_FLAG_INT(max_len, 0, "Maximum length of the test input. "
|
|
||||||
"If 0, libFuzzer tries to guess a good value based on the corpus "
|
|
||||||
"and reports it. ")
|
|
||||||
FUZZER_FLAG_INT(cross_over, 1, "If 1, cross over inputs.")
|
|
||||||
FUZZER_FLAG_INT(mutate_depth, 5,
|
|
||||||
"Apply this number of consecutive mutations to each input.")
|
|
||||||
FUZZER_FLAG_INT(shuffle, 1, "Shuffle inputs at startup")
|
|
||||||
FUZZER_FLAG_INT(prefer_small, 1,
|
|
||||||
"If 1, always prefer smaller inputs during the corpus shuffle.")
|
|
||||||
FUZZER_FLAG_INT(
|
|
||||||
timeout, 1200,
|
|
||||||
"Timeout in seconds (if positive). "
|
|
||||||
"If one unit runs more than this number of seconds the process will abort.")
|
|
||||||
FUZZER_FLAG_INT(error_exitcode, 77, "When libFuzzer itself reports a bug "
|
|
||||||
"this exit code will be used.")
|
|
||||||
FUZZER_FLAG_INT(timeout_exitcode, 77, "When libFuzzer reports a timeout "
|
|
||||||
"this exit code will be used.")
|
|
||||||
FUZZER_FLAG_INT(max_total_time, 0, "If positive, indicates the maximal total "
|
|
||||||
"time in seconds to run the fuzzer.")
|
|
||||||
FUZZER_FLAG_INT(help, 0, "Print help.")
|
|
||||||
FUZZER_FLAG_INT(merge, 0, "If 1, the 2-nd, 3-rd, etc corpora will be "
|
|
||||||
"merged into the 1-st corpus. Only interesting units will be taken. "
|
|
||||||
"This flag can be used to minimize a corpus.")
|
|
||||||
FUZZER_FLAG_STRING(merge_control_file, "internal flag")
|
|
||||||
FUZZER_FLAG_INT(minimize_crash, 0, "If 1, minimizes the provided"
|
|
||||||
" crash input. Use with -runs=N or -max_total_time=N to limit "
|
|
||||||
"the number attempts")
|
|
||||||
FUZZER_FLAG_INT(minimize_crash_internal_step, 0, "internal flag")
|
|
||||||
FUZZER_FLAG_INT(use_counters, 1, "Use coverage counters")
|
|
||||||
FUZZER_FLAG_INT(use_indir_calls, 1, "Use indirect caller-callee counters")
|
|
||||||
FUZZER_FLAG_INT(use_memcmp, 1,
|
|
||||||
"Use hints from intercepting memcmp, strcmp, etc")
|
|
||||||
FUZZER_FLAG_INT(use_memmem, 1,
|
|
||||||
"Use hints from intercepting memmem, strstr, etc")
|
|
||||||
FUZZER_FLAG_INT(use_value_profile, 0,
|
|
||||||
"Experimental. Use value profile to guide fuzzing.")
|
|
||||||
FUZZER_FLAG_INT(use_cmp, 1, "Use CMP traces to guide mutations")
|
|
||||||
FUZZER_FLAG_INT(shrink, 0, "Experimental. Try to shrink corpus elements.")
|
|
||||||
FUZZER_FLAG_UNSIGNED(jobs, 0, "Number of jobs to run. If jobs >= 1 we spawn"
|
|
||||||
" this number of jobs in separate worker processes"
|
|
||||||
" with stdout/stderr redirected to fuzz-JOB.log.")
|
|
||||||
FUZZER_FLAG_UNSIGNED(workers, 0,
|
|
||||||
"Number of simultaneous worker processes to run the jobs."
|
|
||||||
" If zero, \"min(jobs,NumberOfCpuCores()/2)\" is used.")
|
|
||||||
FUZZER_FLAG_INT(reload, 1,
|
|
||||||
"Reload the main corpus every <N> seconds to get new units"
|
|
||||||
" discovered by other processes. If 0, disabled")
|
|
||||||
FUZZER_FLAG_INT(report_slow_units, 10,
|
|
||||||
"Report slowest units if they run for more than this number of seconds.")
|
|
||||||
FUZZER_FLAG_INT(only_ascii, 0,
|
|
||||||
"If 1, generate only ASCII (isprint+isspace) inputs.")
|
|
||||||
FUZZER_FLAG_STRING(dict, "Experimental. Use the dictionary file.")
|
|
||||||
FUZZER_FLAG_STRING(artifact_prefix, "Write fuzzing artifacts (crash, "
|
|
||||||
"timeout, or slow inputs) as "
|
|
||||||
"$(artifact_prefix)file")
|
|
||||||
FUZZER_FLAG_STRING(exact_artifact_path,
|
|
||||||
"Write the single artifact on failure (crash, timeout) "
|
|
||||||
"as $(exact_artifact_path). This overrides -artifact_prefix "
|
|
||||||
"and will not use checksum in the file name. Do not "
|
|
||||||
"use the same path for several parallel processes.")
|
|
||||||
FUZZER_FLAG_INT(output_csv, 0, "Enable pulse output in CSV format.")
|
|
||||||
FUZZER_FLAG_INT(print_pcs, 0, "If 1, print out newly covered PCs.")
|
|
||||||
FUZZER_FLAG_INT(print_final_stats, 0, "If 1, print statistics at exit.")
|
|
||||||
FUZZER_FLAG_INT(print_corpus_stats, 0,
|
|
||||||
"If 1, print statistics on corpus elements at exit.")
|
|
||||||
FUZZER_FLAG_INT(print_coverage, 0, "If 1, print coverage information at exit."
|
|
||||||
" Experimental, only with trace-pc-guard")
|
|
||||||
FUZZER_FLAG_INT(dump_coverage, 0, "If 1, dump coverage information at exit."
|
|
||||||
" Experimental, only with trace-pc-guard")
|
|
||||||
FUZZER_FLAG_INT(handle_segv, 1, "If 1, try to intercept SIGSEGV.")
|
|
||||||
FUZZER_FLAG_INT(handle_bus, 1, "If 1, try to intercept SIGSEGV.")
|
|
||||||
FUZZER_FLAG_INT(handle_abrt, 1, "If 1, try to intercept SIGABRT.")
|
|
||||||
FUZZER_FLAG_INT(handle_ill, 1, "If 1, try to intercept SIGILL.")
|
|
||||||
FUZZER_FLAG_INT(handle_fpe, 1, "If 1, try to intercept SIGFPE.")
|
|
||||||
FUZZER_FLAG_INT(handle_int, 1, "If 1, try to intercept SIGINT.")
|
|
||||||
FUZZER_FLAG_INT(handle_term, 1, "If 1, try to intercept SIGTERM.")
|
|
||||||
FUZZER_FLAG_INT(close_fd_mask, 0, "If 1, close stdout at startup; "
|
|
||||||
"if 2, close stderr; if 3, close both. "
|
|
||||||
"Be careful, this will also close e.g. asan's stderr/stdout.")
|
|
||||||
FUZZER_FLAG_INT(detect_leaks, 1, "If 1, and if LeakSanitizer is enabled "
|
|
||||||
"try to detect memory leaks during fuzzing (i.e. not only at shut down).")
|
|
||||||
FUZZER_FLAG_INT(trace_malloc, 0, "If >= 1 will print all mallocs/frees. "
|
|
||||||
"If >= 2 will also print stack traces.")
|
|
||||||
FUZZER_FLAG_INT(rss_limit_mb, 2048, "If non-zero, the fuzzer will exit upon"
|
|
||||||
"reaching this limit of RSS memory usage.")
|
|
||||||
FUZZER_FLAG_STRING(exit_on_src_pos, "Exit if a newly found PC originates"
|
|
||||||
" from the given source location. Example: -exit_on_src_pos=foo.cc:123. "
|
|
||||||
"Used primarily for testing libFuzzer itself.")
|
|
||||||
FUZZER_FLAG_STRING(exit_on_item, "Exit if an item with a given sha1 sum"
|
|
||||||
" was added to the corpus. "
|
|
||||||
"Used primarily for testing libFuzzer itself.")
|
|
||||||
|
|
||||||
FUZZER_DEPRECATED_FLAG(exit_on_first)
|
|
||||||
FUZZER_DEPRECATED_FLAG(save_minimized_corpus)
|
|
||||||
FUZZER_DEPRECATED_FLAG(sync_command)
|
|
||||||
FUZZER_DEPRECATED_FLAG(sync_timeout)
|
|
||||||
FUZZER_DEPRECATED_FLAG(test_single_input)
|
|
||||||
FUZZER_DEPRECATED_FLAG(drill)
|
|
||||||
FUZZER_DEPRECATED_FLAG(truncate_units)
|
|
||||||
@@ -1,117 +0,0 @@
|
|||||||
//===- FuzzerIO.cpp - IO utils. -------------------------------------------===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// IO functions.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
|
|
||||||
#include "FuzzerIO.h"
|
|
||||||
#include "FuzzerDefs.h"
|
|
||||||
#include "FuzzerExtFunctions.h"
|
|
||||||
#include <algorithm>
|
|
||||||
#include <cstdarg>
|
|
||||||
#include <fstream>
|
|
||||||
#include <iterator>
|
|
||||||
#include <sys/stat.h>
|
|
||||||
#include <sys/types.h>
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
static FILE *OutputFile = stderr;
|
|
||||||
|
|
||||||
long GetEpoch(const std::string &Path) {
|
|
||||||
struct stat St;
|
|
||||||
if (stat(Path.c_str(), &St))
|
|
||||||
return 0; // Can't stat, be conservative.
|
|
||||||
return St.st_mtime;
|
|
||||||
}
|
|
||||||
|
|
||||||
Unit FileToVector(const std::string &Path, size_t MaxSize, bool ExitOnError) {
|
|
||||||
std::ifstream T(Path);
|
|
||||||
if (ExitOnError && !T) {
|
|
||||||
Printf("No such directory: %s; exiting\n", Path.c_str());
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
|
|
||||||
T.seekg(0, T.end);
|
|
||||||
size_t FileLen = T.tellg();
|
|
||||||
if (MaxSize)
|
|
||||||
FileLen = std::min(FileLen, MaxSize);
|
|
||||||
|
|
||||||
T.seekg(0, T.beg);
|
|
||||||
Unit Res(FileLen);
|
|
||||||
T.read(reinterpret_cast<char *>(Res.data()), FileLen);
|
|
||||||
return Res;
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string FileToString(const std::string &Path) {
|
|
||||||
std::ifstream T(Path);
|
|
||||||
return std::string((std::istreambuf_iterator<char>(T)),
|
|
||||||
std::istreambuf_iterator<char>());
|
|
||||||
}
|
|
||||||
|
|
||||||
void CopyFileToErr(const std::string &Path) {
|
|
||||||
Printf("%s", FileToString(Path).c_str());
|
|
||||||
}
|
|
||||||
|
|
||||||
void WriteToFile(const Unit &U, const std::string &Path) {
|
|
||||||
// Use raw C interface because this function may be called from a sig handler.
|
|
||||||
FILE *Out = fopen(Path.c_str(), "w");
|
|
||||||
if (!Out) return;
|
|
||||||
fwrite(U.data(), sizeof(U[0]), U.size(), Out);
|
|
||||||
fclose(Out);
|
|
||||||
}
|
|
||||||
|
|
||||||
void ReadDirToVectorOfUnits(const char *Path, std::vector<Unit> *V,
|
|
||||||
long *Epoch, size_t MaxSize, bool ExitOnError) {
|
|
||||||
long E = Epoch ? *Epoch : 0;
|
|
||||||
std::vector<std::string> Files;
|
|
||||||
ListFilesInDirRecursive(Path, Epoch, &Files, /*TopDir*/true);
|
|
||||||
size_t NumLoaded = 0;
|
|
||||||
for (size_t i = 0; i < Files.size(); i++) {
|
|
||||||
auto &X = Files[i];
|
|
||||||
if (Epoch && GetEpoch(X) < E) continue;
|
|
||||||
NumLoaded++;
|
|
||||||
if ((NumLoaded & (NumLoaded - 1)) == 0 && NumLoaded >= 1024)
|
|
||||||
Printf("Loaded %zd/%zd files from %s\n", NumLoaded, Files.size(), Path);
|
|
||||||
auto S = FileToVector(X, MaxSize, ExitOnError);
|
|
||||||
if (!S.empty())
|
|
||||||
V->push_back(S);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string DirPlusFile(const std::string &DirPath,
|
|
||||||
const std::string &FileName) {
|
|
||||||
return DirPath + GetSeparator() + FileName;
|
|
||||||
}
|
|
||||||
|
|
||||||
void DupAndCloseStderr() {
|
|
||||||
int OutputFd = DuplicateFile(2);
|
|
||||||
if (OutputFd > 0) {
|
|
||||||
FILE *NewOutputFile = OpenFile(OutputFd, "w");
|
|
||||||
if (NewOutputFile) {
|
|
||||||
OutputFile = NewOutputFile;
|
|
||||||
if (EF->__sanitizer_set_report_fd)
|
|
||||||
EF->__sanitizer_set_report_fd(reinterpret_cast<void *>(OutputFd));
|
|
||||||
CloseFile(2);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void CloseStdout() {
|
|
||||||
CloseFile(1);
|
|
||||||
}
|
|
||||||
|
|
||||||
void Printf(const char *Fmt, ...) {
|
|
||||||
va_list ap;
|
|
||||||
va_start(ap, Fmt);
|
|
||||||
vfprintf(OutputFile, Fmt, ap);
|
|
||||||
va_end(ap);
|
|
||||||
fflush(OutputFile);
|
|
||||||
}
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
@@ -1,64 +0,0 @@
|
|||||||
//===- FuzzerIO.h - Internal header for IO utils ----------------*- C++ -* ===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// IO interface.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
|
|
||||||
#ifndef LLVM_FUZZER_IO_H
|
|
||||||
#define LLVM_FUZZER_IO_H
|
|
||||||
|
|
||||||
#include "FuzzerDefs.h"
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
long GetEpoch(const std::string &Path);
|
|
||||||
|
|
||||||
Unit FileToVector(const std::string &Path, size_t MaxSize = 0,
|
|
||||||
bool ExitOnError = true);
|
|
||||||
|
|
||||||
std::string FileToString(const std::string &Path);
|
|
||||||
|
|
||||||
void CopyFileToErr(const std::string &Path);
|
|
||||||
|
|
||||||
void WriteToFile(const Unit &U, const std::string &Path);
|
|
||||||
|
|
||||||
void ReadDirToVectorOfUnits(const char *Path, std::vector<Unit> *V,
|
|
||||||
long *Epoch, size_t MaxSize, bool ExitOnError);
|
|
||||||
|
|
||||||
// Returns "Dir/FileName" or equivalent for the current OS.
|
|
||||||
std::string DirPlusFile(const std::string &DirPath,
|
|
||||||
const std::string &FileName);
|
|
||||||
|
|
||||||
// Returns the name of the dir, similar to the 'dirname' utility.
|
|
||||||
std::string DirName(const std::string &FileName);
|
|
||||||
|
|
||||||
void DupAndCloseStderr();
|
|
||||||
|
|
||||||
void CloseStdout();
|
|
||||||
|
|
||||||
void Printf(const char *Fmt, ...);
|
|
||||||
|
|
||||||
// Platform specific functions:
|
|
||||||
bool IsFile(const std::string &Path);
|
|
||||||
|
|
||||||
void ListFilesInDirRecursive(const std::string &Dir, long *Epoch,
|
|
||||||
std::vector<std::string> *V, bool TopDir);
|
|
||||||
|
|
||||||
char GetSeparator();
|
|
||||||
|
|
||||||
FILE* OpenFile(int Fd, const char *Mode);
|
|
||||||
|
|
||||||
int CloseFile(int Fd);
|
|
||||||
|
|
||||||
int DuplicateFile(int Fd);
|
|
||||||
|
|
||||||
void RemoveFile(const std::string &Path);
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
|
|
||||||
#endif // LLVM_FUZZER_IO_H
|
|
||||||
@@ -1,88 +0,0 @@
|
|||||||
//===- FuzzerIOPosix.cpp - IO utils for Posix. ----------------------------===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// IO functions implementation using Posix API.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
#include "FuzzerDefs.h"
|
|
||||||
#if LIBFUZZER_POSIX
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.h"
|
|
||||||
#include "FuzzerIO.h"
|
|
||||||
#include <cstdarg>
|
|
||||||
#include <cstdio>
|
|
||||||
#include <dirent.h>
|
|
||||||
#include <fstream>
|
|
||||||
#include <iterator>
|
|
||||||
#include <libgen.h>
|
|
||||||
#include <sys/stat.h>
|
|
||||||
#include <sys/types.h>
|
|
||||||
#include <unistd.h>
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
bool IsFile(const std::string &Path) {
|
|
||||||
struct stat St;
|
|
||||||
if (stat(Path.c_str(), &St))
|
|
||||||
return false;
|
|
||||||
return S_ISREG(St.st_mode);
|
|
||||||
}
|
|
||||||
|
|
||||||
void ListFilesInDirRecursive(const std::string &Dir, long *Epoch,
|
|
||||||
std::vector<std::string> *V, bool TopDir) {
|
|
||||||
auto E = GetEpoch(Dir);
|
|
||||||
if (Epoch)
|
|
||||||
if (E && *Epoch >= E) return;
|
|
||||||
|
|
||||||
DIR *D = opendir(Dir.c_str());
|
|
||||||
if (!D) {
|
|
||||||
Printf("No such directory: %s; exiting\n", Dir.c_str());
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
while (auto E = readdir(D)) {
|
|
||||||
std::string Path = DirPlusFile(Dir, E->d_name);
|
|
||||||
if (E->d_type == DT_REG || E->d_type == DT_LNK)
|
|
||||||
V->push_back(Path);
|
|
||||||
else if (E->d_type == DT_DIR && *E->d_name != '.')
|
|
||||||
ListFilesInDirRecursive(Path, Epoch, V, false);
|
|
||||||
}
|
|
||||||
closedir(D);
|
|
||||||
if (Epoch && TopDir)
|
|
||||||
*Epoch = E;
|
|
||||||
}
|
|
||||||
|
|
||||||
char GetSeparator() {
|
|
||||||
return '/';
|
|
||||||
}
|
|
||||||
|
|
||||||
FILE* OpenFile(int Fd, const char* Mode) {
|
|
||||||
return fdopen(Fd, Mode);
|
|
||||||
}
|
|
||||||
|
|
||||||
int CloseFile(int fd) {
|
|
||||||
return close(fd);
|
|
||||||
}
|
|
||||||
|
|
||||||
int DuplicateFile(int Fd) {
|
|
||||||
return dup(Fd);
|
|
||||||
}
|
|
||||||
|
|
||||||
void RemoveFile(const std::string &Path) {
|
|
||||||
unlink(Path.c_str());
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string DirName(const std::string &FileName) {
|
|
||||||
char *Tmp = new char[FileName.size() + 1];
|
|
||||||
memcpy(Tmp, FileName.c_str(), FileName.size() + 1);
|
|
||||||
std::string Res = dirname(Tmp);
|
|
||||||
delete [] Tmp;
|
|
||||||
return Res;
|
|
||||||
}
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
|
|
||||||
#endif // LIBFUZZER_POSIX
|
|
||||||
@@ -1,282 +0,0 @@
|
|||||||
//===- FuzzerIOWindows.cpp - IO utils for Windows. ------------------------===//
|
|
||||||
//
|
|
||||||
// The LLVM Compiler Infrastructure
|
|
||||||
//
|
|
||||||
// This file is distributed under the University of Illinois Open Source
|
|
||||||
// License. See LICENSE.TXT for details.
|
|
||||||
//
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
// IO functions implementation for Windows.
|
|
||||||
//===----------------------------------------------------------------------===//
|
|
||||||
#include "FuzzerDefs.h"
|
|
||||||
#if LIBFUZZER_WINDOWS
|
|
||||||
|
|
||||||
#include "FuzzerExtFunctions.h"
|
|
||||||
#include "FuzzerIO.h"
|
|
||||||
#include <cstdarg>
|
|
||||||
#include <cstdio>
|
|
||||||
#include <fstream>
|
|
||||||
#include <io.h>
|
|
||||||
#include <iterator>
|
|
||||||
#include <sys/stat.h>
|
|
||||||
#include <sys/types.h>
|
|
||||||
#include <windows.h>
|
|
||||||
|
|
||||||
namespace fuzzer {
|
|
||||||
|
|
||||||
static bool IsFile(const std::string &Path, const DWORD &FileAttributes) {
|
|
||||||
|
|
||||||
if (FileAttributes & FILE_ATTRIBUTE_NORMAL)
|
|
||||||
return true;
|
|
||||||
|
|
||||||
if (FileAttributes & FILE_ATTRIBUTE_DIRECTORY)
|
|
||||||
return false;
|
|
||||||
|
|
||||||
HANDLE FileHandle(
|
|
||||||
CreateFileA(Path.c_str(), 0, FILE_SHARE_READ, NULL, OPEN_EXISTING,
|
|
||||||
FILE_FLAG_BACKUP_SEMANTICS, 0));
|
|
||||||
|
|
||||||
if (FileHandle == INVALID_HANDLE_VALUE) {
|
|
||||||
Printf("CreateFileA() failed for \"%s\" (Error code: %lu).\n", Path.c_str(),
|
|
||||||
GetLastError());
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
DWORD FileType = GetFileType(FileHandle);
|
|
||||||
|
|
||||||
if (FileType == FILE_TYPE_UNKNOWN) {
|
|
||||||
Printf("GetFileType() failed for \"%s\" (Error code: %lu).\n", Path.c_str(),
|
|
||||||
GetLastError());
|
|
||||||
CloseHandle(FileHandle);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (FileType != FILE_TYPE_DISK) {
|
|
||||||
CloseHandle(FileHandle);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
CloseHandle(FileHandle);
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool IsFile(const std::string &Path) {
|
|
||||||
DWORD Att = GetFileAttributesA(Path.c_str());
|
|
||||||
|
|
||||||
if (Att == INVALID_FILE_ATTRIBUTES) {
|
|
||||||
Printf("GetFileAttributesA() failed for \"%s\" (Error code: %lu).\n",
|
|
||||||
Path.c_str(), GetLastError());
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
return IsFile(Path, Att);
|
|
||||||
}
|
|
||||||
|
|
||||||
void ListFilesInDirRecursive(const std::string &Dir, long *Epoch,
|
|
||||||
std::vector<std::string> *V, bool TopDir) {
|
|
||||||
auto E = GetEpoch(Dir);
|
|
||||||
if (Epoch)
|
|
||||||
if (E && *Epoch >= E) return;
|
|
||||||
|
|
||||||
std::string Path(Dir);
|
|
||||||
assert(!Path.empty());
|
|
||||||
if (Path.back() != '\\')
|
|
||||||
Path.push_back('\\');
|
|
||||||
Path.push_back('*');
|
|
||||||
|
|
||||||
// Get the first directory entry.
|
|
||||||
WIN32_FIND_DATAA FindInfo;
|
|
||||||
HANDLE FindHandle(FindFirstFileA(Path.c_str(), &FindInfo));
|
|
||||||
if (FindHandle == INVALID_HANDLE_VALUE)
|
|
||||||
{
|
|
||||||
Printf("No file found in: %s.\n", Dir.c_str());
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
do {
|
|
||||||
std::string FileName = DirPlusFile(Dir, FindInfo.cFileName);
|
|
||||||
|
|
||||||
if (FindInfo.dwFileAttributes & FILE_ATTRIBUTE_DIRECTORY) {
|
|
||||||
size_t FilenameLen = strlen(FindInfo.cFileName);
|
|
||||||
if ((FilenameLen == 1 && FindInfo.cFileName[0] == '.') ||
|
|
||||||
(FilenameLen == 2 && FindInfo.cFileName[0] == '.' &&
|
|
||||||
FindInfo.cFileName[1] == '.'))
|
|
||||||
continue;
|
|
||||||
|
|
||||||
ListFilesInDirRecursive(FileName, Epoch, V, false);
|
|
||||||
}
|
|
||||||
else if (IsFile(FileName, FindInfo.dwFileAttributes))
|
|
||||||
V->push_back(FileName);
|
|
||||||
} while (FindNextFileA(FindHandle, &FindInfo));
|
|
||||||
|
|
||||||
DWORD LastError = GetLastError();
|
|
||||||
if (LastError != ERROR_NO_MORE_FILES)
|
|
||||||
Printf("FindNextFileA failed (Error code: %lu).\n", LastError);
|
|
||||||
|
|
||||||
FindClose(FindHandle);
|
|
||||||
|
|
||||||
if (Epoch && TopDir)
|
|
||||||
*Epoch = E;
|
|
||||||
}
|
|
||||||
|
|
||||||
char GetSeparator() {
|
|
||||||
return '\\';
|
|
||||||
}
|
|
||||||
|
|
||||||
FILE* OpenFile(int Fd, const char* Mode) {
|
|
||||||
return _fdopen(Fd, Mode);
|
|
||||||
}
|
|
||||||
|
|
||||||
int CloseFile(int Fd) {
|
|
||||||
return _close(Fd);
|
|
||||||
}
|
|
||||||
|
|
||||||
int DuplicateFile(int Fd) {
|
|
||||||
return _dup(Fd);
|
|
||||||
}
|
|
||||||
|
|
||||||
void RemoveFile(const std::string &Path) {
|
|
||||||
_unlink(Path.c_str());
|
|
||||||
}
|
|
||||||
|
|
||||||
static bool IsSeparator(char C) {
|
|
||||||
return C == '\\' || C == '/';
|
|
||||||
}
|
|
||||||
|
|
||||||
// Parse disk designators, like "C:\". If Relative == true, also accepts: "C:".
|
|
||||||
// Returns number of characters considered if successful.
|
|
||||||
static size_t ParseDrive(const std::string &FileName, const size_t Offset,
|
|
||||||
bool Relative = true) {
|
|
||||||
if (Offset + 1 >= FileName.size() || FileName[Offset + 1] != ':')
|
|
||||||
return 0;
|
|
||||||
if (Offset + 2 >= FileName.size() || !IsSeparator(FileName[Offset + 2])) {
|
|
||||||
if (!Relative) // Accept relative path?
|
|
||||||
return 0;
|
|
||||||
else
|
|
||||||
return 2;
|
|
||||||
}
|
|
||||||
return 3;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Parse a file name, like: SomeFile.txt
|
|
||||||
// Returns number of characters considered if successful.
|
|
||||||
static size_t ParseFileName(const std::string &FileName, const size_t Offset) {
|
|
||||||
size_t Pos = Offset;
|
|
||||||
const size_t End = FileName.size();
|
|
||||||
for(; Pos < End && !IsSeparator(FileName[Pos]); ++Pos)
|
|
||||||
;
|
|
||||||
return Pos - Offset;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Parse a directory ending in separator, like: SomeDir\
|
|
||||||
// Returns number of characters considered if successful.
|
|
||||||
static size_t ParseDir(const std::string &FileName, const size_t Offset) {
|
|
||||||
size_t Pos = Offset;
|
|
||||||
const size_t End = FileName.size();
|
|
||||||
if (Pos >= End || IsSeparator(FileName[Pos]))
|
|
||||||
return 0;
|
|
||||||
for(; Pos < End && !IsSeparator(FileName[Pos]); ++Pos)
|
|
||||||
;
|
|
||||||
if (Pos >= End)
|
|
||||||
return 0;
|
|
||||||
++Pos; // Include separator.
|
|
||||||
return Pos - Offset;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Parse a servername and share, like: SomeServer\SomeShare\
|
|
||||||
// Returns number of characters considered if successful.
|
|
||||||
static size_t ParseServerAndShare(const std::string &FileName,
|
|
||||||
const size_t Offset) {
|
|
||||||
size_t Pos = Offset, Res;
|
|
||||||
if (!(Res = ParseDir(FileName, Pos)))
|
|
||||||
return 0;
|
|
||||||
Pos += Res;
|
|
||||||
if (!(Res = ParseDir(FileName, Pos)))
|
|
||||||
return 0;
|
|
||||||
Pos += Res;
|
|
||||||
return Pos - Offset;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Parse the given Ref string from the position Offset, to exactly match the given
|
|
||||||
// string Patt.
|
|
||||||
// Returns number of characters considered if successful.
|
|
||||||
static size_t ParseCustomString(const std::string &Ref, size_t Offset,
|
|
||||||
const char *Patt) {
|
|
||||||
size_t Len = strlen(Patt);
|
|
||||||
if (Offset + Len > Ref.size())
|
|
||||||
return 0;
|
|
||||||
return Ref.compare(Offset, Len, Patt) == 0 ? Len : 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Parse a location, like:
|
|
||||||
// \\?\UNC\Server\Share\ \\?\C:\ \\Server\Share\ \ C:\ C:
|
|
||||||
// Returns number of characters considered if successful.
|
|
||||||
static size_t ParseLocation(const std::string &FileName) {
|
|
||||||
size_t Pos = 0, Res;
|
|
||||||
|
|
||||||
if ((Res = ParseCustomString(FileName, Pos, R"(\\?\)"))) {
|
|
||||||
Pos += Res;
|
|
||||||
if ((Res = ParseCustomString(FileName, Pos, R"(UNC\)"))) {
|
|
||||||
Pos += Res;
|
|
||||||
if ((Res = ParseServerAndShare(FileName, Pos)))
|
|
||||||
return Pos + Res;
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
if ((Res = ParseDrive(FileName, Pos, false)))
|
|
||||||
return Pos + Res;
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (Pos < FileName.size() && IsSeparator(FileName[Pos])) {
|
|
||||||
++Pos;
|
|
||||||
if (Pos < FileName.size() && IsSeparator(FileName[Pos])) {
|
|
||||||
++Pos;
|
|
||||||
if ((Res = ParseServerAndShare(FileName, Pos)))
|
|
||||||
return Pos + Res;
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
return Pos;
|
|
||||||
}
|
|
||||||
|
|
||||||
if ((Res = ParseDrive(FileName, Pos)))
|
|
||||||
return Pos + Res;
|
|
||||||
|
|
||||||
return Pos;
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string DirName(const std::string &FileName) {
|
|
||||||
size_t LocationLen = ParseLocation(FileName);
|
|
||||||
size_t DirLen = 0, Res;
|
|
||||||
while ((Res = ParseDir(FileName, LocationLen + DirLen)))
|
|
||||||
DirLen += Res;
|
|
||||||
size_t FileLen = ParseFileName(FileName, LocationLen + DirLen);
|
|
||||||
|
|
||||||
if (LocationLen + DirLen + FileLen != FileName.size()) {
|
|
||||||
Printf("DirName() failed for \"%s\", invalid path.\n", FileName.c_str());
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (DirLen) {
|
|
||||||
--DirLen; // Remove trailing separator.
|
|
||||||
if (!FileLen) { // Path ended in separator.
|
|
||||||
assert(DirLen);
|
|
||||||
// Remove file name from Dir.
|
|
||||||
while (DirLen && !IsSeparator(FileName[LocationLen + DirLen - 1]))
|
|
||||||
--DirLen;
|
|
||||||
if (DirLen) // Remove trailing separator.
|
|
||||||
--DirLen;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!LocationLen) { // Relative path.
|
|
||||||
if (!DirLen)
|
|
||||||
return ".";
|
|
||||||
return std::string(".\\").append(FileName, 0, DirLen);
|
|
||||||
}
|
|
||||||
|
|
||||||
return FileName.substr(0, LocationLen + DirLen);
|
|
||||||
}
|
|
||||||
|
|
||||||
} // namespace fuzzer
|
|
||||||
|
|
||||||
#endif // LIBFUZZER_WINDOWS
|
|
||||||