diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index 537095324..16d808485 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -2,6 +2,7 @@ - [ ] The changes are described in detail, both the what and why. - [ ] If applicable, an [existing issue](https://github.com/nlohmann/json/issues) is referenced. +- [ ] If applicable, a fixed [OSS-Fuzz](https://issues.oss-fuzz.com) issue is referenced as `OSS-Fuzz: ` (see [fuzz testing](https://github.com/nlohmann/json/blob/develop/tests/fuzzing.md#handling-oss-fuzz-reports)). - [ ] The [Code coverage](https://coveralls.io/github/nlohmann/json) remained at 100%. A test case for every new line of code. - [ ] If applicable, the [documentation](https://json.nlohmann.me) is updated. - [ ] The source code is amalgamated by running `make amalgamate`. diff --git a/BUILD.bazel b/BUILD.bazel index b13e62c22..db59095cd 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -71,7 +71,6 @@ cc_library( ], includes = ["include"], visibility = ["//visibility:public"], - alwayslink = True, ) cc_library( diff --git a/cmake/ci.cmake b/cmake/ci.cmake index a99788633..752bdc6f8 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -300,10 +300,10 @@ add_custom_target(ci_test_skiplibraryversioncheck # Disable thread-local storage. ############################################################################### -# Without thread-local storage, the copy constructor cannot bound its descent -# and copies every object and array without the call stack. That path is -# otherwise only reached by values nested deeper than the bound, so this target -# is what runs the whole test suite through it. +# Without thread-local storage, copying and comparing cannot bound their +# descent and handle every object and array without the call stack. Those paths +# are otherwise only reached by values nested deeper than the bound, so this +# target is what runs the whole test suite through them. add_custom_target(ci_test_no_thread_local COMMAND ${CMAKE_COMMAND} -DCMAKE_BUILD_TYPE=Debug -GNinja diff --git a/cmake/scripts/gen_bazel_build_file.cmake b/cmake/scripts/gen_bazel_build_file.cmake index 3c7db9493..a781a2e3f 100644 --- a/cmake/scripts/gen_bazel_build_file.cmake +++ b/cmake/scripts/gen_bazel_build_file.cmake @@ -42,7 +42,6 @@ string(APPEND CONTENT [=[ ], includes = ["include"], visibility = ["//visibility:public"], - alwayslink = True, ) cc_library( diff --git a/docs/mkdocs/docs/api/basic_json/get_to.md b/docs/mkdocs/docs/api/basic_json/get_to.md index 8334ba5cf..c50c077b0 100644 --- a/docs/mkdocs/docs/api/basic_json/get_to.md +++ b/docs/mkdocs/docs/api/basic_json/get_to.md @@ -21,6 +21,10 @@ This overload is chosen if: - `ValueType` is not `basic_json`, - `json_serializer` has a `from_json()` method of the form `void from_json(const basic_json&, ValueType&)` +`v` must not be `const`. Passing a `const` object is a compile-time error. For types such as arithmetic types, enums, +and C arrays, the error is a `static_assert` that names the problem. For other types, the overload is not viable, and +the compiler reports that no matching `get_to` was found. + ## Template parameters `ValueType` @@ -67,3 +71,4 @@ Depends on the `json_serializer::from_json()` implementation. ## Version history - Since version 3.3.0. +- Added a `static_assert` with a clear message for `const` arguments in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/sax_parse.md b/docs/mkdocs/docs/api/basic_json/sax_parse.md index fc8ce07b6..bf61ea9eb 100644 --- a/docs/mkdocs/docs/api/basic_json/sax_parse.md +++ b/docs/mkdocs/docs/api/basic_json/sax_parse.md @@ -69,7 +69,9 @@ The SAX event lister must follow the interface of [`json_sax`](../json_sax/index [`input_format_t`](input_format_t.md) for more information `strict` (in) -: whether the input has to be consumed completely (optional, `#!cpp true` by default) +: whether the input has to be consumed completely (optional, `#!cpp true` by default); when `#!cpp false` and the + input is a `#!cpp std::istream`, the character that terminates a number is consumed unless + [`JSON_PRECISE_STREAM_POSITION`](../macros/json_precise_stream_position.md) is defined to `1`; see [`operator>>`](../operator_gtgt.md#notes) `ignore_comments` (in) : whether comments should be ignored and treated like whitespace (`#!cpp true`) or yield a parse error @@ -136,6 +138,8 @@ A UTF-8 byte order mark is silently ignored. - Added `ignore_trailing_commas` in version 3.13.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave a `#!cpp std::istream` positioned right + after the parsed value when `strict` is `#!cpp false`. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index 70f02a7a6..bf773b5c4 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -16,6 +16,8 @@ header. See also the [macro overview page](../../features/macros.md). ## Parsing +- [**JSON_PRECISE_STREAM_POSITION**](json_precise_stream_position.md) - opt in to leaving an input stream positioned + right after a parsed number - [**JSON_STRICT_NUL_HANDLING**](json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of treating it as end of input diff --git a/docs/mkdocs/docs/api/macros/json_no_thread_local.md b/docs/mkdocs/docs/api/macros/json_no_thread_local.md index 126116ec3..0a001ac36 100644 --- a/docs/mkdocs/docs/api/macros/json_no_thread_local.md +++ b/docs/mkdocs/docs/api/macros/json_no_thread_local.md @@ -7,16 +7,16 @@ When defined, the library does not use `#!cpp thread_local` storage. This is relevant for the few environments whose toolchain does not support it. -The copy constructor copies the first levels of a value by copying the containers, which copy their elements, and -completes whatever is nested deeper than that without the call stack, so that copying a value cannot exhaust the stack -however deeply it is nested. It counts the levels it has descended into in a `#!cpp thread_local` variable, as a counter -shared between threads would be raced. +Copying a value and comparing two values both descend into the first levels by letting the containers copy or compare +themselves, and finish whatever is nested deeper than that without the call stack, so that neither can exhaust the stack +however deeply the values are nested. Each counts the levels it has descended into in a `#!cpp thread_local` variable, as +a counter shared between threads would be raced. -Without that counter, no descent can be bounded safely, so objects and arrays are copied without the call stack right -away. Copying keeps working exactly as it does otherwise - the same values come out, and deeply nested values are copied -just as safely - but copying is slower, because the containers no longer copy themselves. Copying the benchmark -documents takes 9% (`canada.json`) to 34% (`twitter.json`) longer; values built mostly from objects are affected the -most. +Without those counters, no descent can be bounded safely, so objects and arrays are copied and compared without the call +stack right away. Both keep working exactly as they do otherwise - the same values come out, the same comparisons hold, +and deeply nested values are handled just as safely - but both are slower, because the containers no longer copy or +compare themselves. Copying the benchmark documents takes 9% (`canada.json`) to 34% (`twitter.json`) longer, and +comparing two equal ones 10% (`citm_catalog.json`) to 90% (`canada.json`) longer. ## Default definition @@ -28,6 +28,7 @@ By default, `#!cpp JSON_NO_THREAD_LOCAL` is not defined. The library defines it by itself for Clang targeting MinGW, which does not survive the `#!cpp thread_local` storage: copying a value segfaults there, with both old and current Clang versions, while GCC targeting MinGW is unaffected. +Copying and comparing fall back to working without the call stack there, as they do whenever the macro is defined. ## Examples diff --git a/docs/mkdocs/docs/api/macros/json_precise_stream_position.md b/docs/mkdocs/docs/api/macros/json_precise_stream_position.md new file mode 100644 index 000000000..5710e9975 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_precise_stream_position.md @@ -0,0 +1,131 @@ +# JSON_PRECISE_STREAM_POSITION + +```cpp +#define JSON_PRECISE_STREAM_POSITION /* value */ +``` + +When defined to `1`, [`operator>>`](../operator_gtgt.md) and [`sax_parse`](../basic_json/sax_parse.md) with +`strict = false` leave a `#!cpp std::istream` positioned right after the parsed value for every value type. By default, +the character that terminates a number is consumed as well. + +The macro only affects reading from a `#!cpp std::istream` when the rest of the stream is not required to be consumed. +[`parse`](../basic_json/parse.md), [`accept`](../basic_json/accept.md), and all other inputs (strings, iterators, +containers, `#!cpp FILE*`) are never affected. + +## Default definition + +The default value is `0` (disabled — existing behavior is preserved). + +```cpp +#define JSON_PRECISE_STREAM_POSITION 0 +``` + +## Notes + +!!! note "Background" + + A number is the only JSON value whose end can be detected solely by reading the character that follows it. By + default, that character is consumed and not put back, so the stream is left one byte too far after a number, and + only after a number: + + ```cpp + std::istringstream input("1true"); + json j; + input >> j; // j == 1, but the stream now starts at "rue" + ``` + + With this macro, the character is only looked at and left in the stream, so the stream starts at `true`. This + does not require the stream buffer to support putting a character back. + + This was not changed unconditionally, because code can depend on the consumed character, even unknowingly (see + [#5340](https://github.com/nlohmann/json/issues/5340)). Both of the following work by default only because the + character after each number is swallowed, and behave differently with this macro: + + ```cpp + std::istringstream input("1,2,3"); + json j1, j2, j3; + input >> j1 >> j2 >> j3; // default: 1, 2, 3 + // with the macro: throws parse_error.101 at the ',' + ``` + + ```cpp + std::istringstream input("42\nfoo"); + json j; + std::string line; + input >> j; + std::getline(input, line); // default: "foo" + // with the macro: "" (like after reading an int with >>) + ``` + + In both cases, the behavior with the macro is what you already get today when the value is not a number: `"a","b"` + fails at the `,`, and `std::getline` after `{}` returns an empty string. This macro offers an opt-in path to + the consistent behavior ahead of version 4.0.0, where it is planned to become the default. + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no + effect. + +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_psp`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + +!!! tip "Workaround without the macro" + + Separate the values in the stream with whitespace. The character consumed after a number is then the separator, + and whitespace before the next value is skipped anyway. + +## Examples + +??? example "Default behavior (macro not defined)" + + Without the macro, the character after a number is consumed: + + ```cpp + #include + #include + #include + + using json = nlohmann::json; + + int main() + { + std::istringstream input("1true"); + json j1, j2; + input >> j1; // j1 == 1 + input >> j2; // throws parse_error.101: the stream now starts at "rue" + } + ``` + +??? example "Opt-in precise stream position (macro defined to 1)" + + With the macro, the stream is positioned right after the number: + + ```cpp + #define JSON_PRECISE_STREAM_POSITION 1 + #include + #include + #include + + using json = nlohmann::json; + + int main() + { + std::istringstream input("1true"); + json j1, j2; + input >> j1; // j1 == 1 + input >> j2; // j2 == true + } + ``` + +## See also + +- [**operator>>**](../operator_gtgt.md) - deserialize from stream +- [**sax_parse**](../basic_json/sax_parse.md) - generate SAX events + +## Version history + +- Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/api/operator_gtgt.md b/docs/mkdocs/docs/api/operator_gtgt.md index 3e60d5236..0173b9fb3 100644 --- a/docs/mkdocs/docs/api/operator_gtgt.md +++ b/docs/mkdocs/docs/api/operator_gtgt.md @@ -67,7 +67,9 @@ input >> j2; // parses the next value Only numbers are affected. Values ending in a self-delimiting character do not read past themselves, so `truefalse`, `[1][2]`, `{"a":1}{"b":2}`, and `"a""b"` can be read back to back without a separator. - This is tracked in [#5340](https://github.com/nlohmann/json/issues/5340). + Define [`JSON_PRECISE_STREAM_POSITION`](macros/json_precise_stream_position.md) to `1` to leave the terminating character in the stream + instead, so that the stream is positioned right after the value for every value type and no separator is + needed. This is tracked in [#5340](https://github.com/nlohmann/json/issues/5340). Note that reading concatenated values does **not** work for [JSON Lines](../features/parsing/json_lines.md) (newline-delimited JSON) input -- see that page for why and for the recommended alternative. @@ -107,9 +109,12 @@ being read. - [parse](basic_json/parse.md) - deserialize from a compatible input - [`JSON_STRICT_NUL_HANDLING`](macros/json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of treating it as end of input +- [`JSON_PRECISE_STREAM_POSITION`](macros/json_precise_stream_position.md) - opt in to leaving the stream positioned right after a number ## Version history - Added in version 1.0.0. - `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating it as end of input; planned to become the default in version 4.0.0. +- `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave the character that terminates a number in + the stream; planned to become the default in version 4.0.0. diff --git a/docs/mkdocs/docs/community/assurance_case.md b/docs/mkdocs/docs/community/assurance_case.md new file mode 100644 index 000000000..87d8d6f14 --- /dev/null +++ b/docs/mkdocs/docs/community/assurance_case.md @@ -0,0 +1,70 @@ +# Assurance case + +This page argues why the library meets its security requirements. It describes the threats the library faces, where the +trust boundaries lie, and how the library's design and the [quality assurance](quality_assurance.md) counter these +threats. To report a vulnerability, see the [security policy](security_policy.md). + +## Threat model + +The library parses, stores, and serializes JSON values in memory. It does not open network connections, does not open +files (it only reads from streams or `std::FILE*` handles that the caller has already opened), does not read environment +variables, and does not implement cryptography or handle credentials. + +The primary threat is therefore **untrusted input**: JSON text or binary data (BJData, BSON, CBOR, MessagePack, UBJSON) +that an attacker controls, passed to [`parse`](../api/basic_json/parse.md), [`accept`](../api/basic_json/accept.md), +[`sax_parse`](../api/basic_json/sax_parse.md), or one of the `from_*` functions such as +[`from_cbor`](../api/basic_json/from_cbor.md). Such input may try to + +- make the library read or write out of bounds (malformed lengths, truncated input, invalid UTF-8), +- trigger undefined behavior (integer overflow in sizes or numbers, invalid casts), +- exhaust memory (huge announced sizes), or +- exhaust the call stack (deeply nested arrays and objects). + +## Trust boundaries + +- **Untrusted:** all serialized input read by the parser, the SAX interface, and the binary readers. The library must + handle every possible input by either producing a value or throwing a [`parse_error`](../home/exceptions.md#parse-errors) + (or returning `false` when exceptions are disabled for the call). +- **Trusted:** the C++ code that calls the library. Calling a function with violated preconditions, for instance + accessing an array with [`operator[]`](../api/basic_json/operator%5B%5D.md) out of range, is a programming error and + not a security boundary. Such preconditions are checked with [runtime assertions](../features/assertions.md) in debug + builds; functions such as [`at`](../api/basic_json/at.md) offer checked access with exceptions. + +## Secure design + +- **Strict parsing.** The parser accepts exactly the JSON grammar of [RFC 8259](https://datatracker.ietf.org/doc/html/rfc8259). + Extensions such as [comments](../features/comments.md) and [trailing commas](../features/trailing_commas.md) must be + enabled explicitly. Invalid UTF-8 is rejected. +- **Errors are reported, not ignored.** Malformed input results in a [`parse_error`](../home/exceptions.md#parse-errors) + with the byte position of the error. Binary readers do not trust announced sizes: strings and binary values grow + only as bytes are actually read, arrays reserve at most a fixed number of elements up front, and sizes that no + container can hold are rejected. +- **Memory is owned by values.** Each `basic_json` value owns its content, and there is no manual memory management in + user code. The destructor does not recurse, so destroying a deeply nested value does not exhaust the stack. +- **Bounded recursion.** The JSON parser and the binary readers keep their state in explicit stacks instead of + recursing per nesting level. Operations that walk a value, such as [`dump`](../api/basic_json/dump.md), copying, + hashing, and [`merge_patch`](../api/basic_json/merge_patch.md), recurse only up to a fixed depth and continue with an + explicit stack below it. Some operations, such as comparison, [`diff`](../api/basic_json/diff.md), + [`flatten`](../api/basic_json/flatten.md), and the binary writers, still recurse once per nesting level; work on them + is in progress. Applications that process untrusted input can limit its nesting depth with a + [parser callback](../features/parsing/parser_callbacks.md). +- **Invariants are checked.** The class invariant (for instance, that the pointer for the stored type is never null) is + checked with runtime assertions throughout the test suite. + +## Common weaknesses + +The following table maps the relevant classes of the [Common Weakness Enumeration](https://cwe.mitre.org) to the +measures that counter them. The measures are described in detail in [Quality assurance](quality_assurance.md). + +| Weakness | Countermeasures | +|---------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------| +| Out-of-bounds read/write ([CWE-125](https://cwe.mitre.org/data/definitions/125.html), [CWE-787](https://cwe.mitre.org/data/definitions/787.html)) | bounds checks on all reads from the input; AddressSanitizer and Valgrind on the test suite; OSS-Fuzz | +| Integer overflow ([CWE-190](https://cwe.mitre.org/data/definitions/190.html)) | UndefinedBehaviorSanitizer with integer overflow detection; Clang-Tidy; Cppcheck | +| Use after free, double free ([CWE-416](https://cwe.mitre.org/data/definitions/416.html), [CWE-415](https://cwe.mitre.org/data/definitions/415.html)) | ownership of all memory by values; AddressSanitizer and Valgrind; Clang Static Analyzer | +| Memory leaks ([CWE-401](https://cwe.mitre.org/data/definitions/401.html)) | Valgrind (Memcheck) on the test suite | +| Uncontrolled recursion ([CWE-674](https://cwe.mitre.org/data/definitions/674.html)) | iterative parser, binary readers, and destructor; bounded recursion in value operations; tests with deeply nested inputs | +| Uncontrolled resource consumption ([CWE-400](https://cwe.mitre.org/data/definitions/400.html)) | allocations based on announced sizes are capped; OSS-Fuzz with memory limits | +| Undefined behavior in general ([CWE-758](https://cwe.mitre.org/data/definitions/758.html)) | UndefinedBehaviorSanitizer; runtime assertions; Clang-Tidy, Cppcheck, Clang Static Analyzer, Infer | + +In addition, every line of the library is covered by the unit tests, and all parsers are fuzz-tested around the clock +by [OSS-Fuzz](https://github.com/google/oss-fuzz/tree/master/projects/json). diff --git a/docs/mkdocs/docs/community/index.md b/docs/mkdocs/docs/community/index.md index 50baeab25..7b7f5c07a 100644 --- a/docs/mkdocs/docs/community/index.md +++ b/docs/mkdocs/docs/community/index.md @@ -5,4 +5,6 @@ - [Contribution Guidelines](contribution_guidelines.md) - guidelines how to contribute to this project - [Governance](governance.md) - the governance model of this project - [Quality Assurance](quality_assurance.md) - how the quality of this project is assured +- [Roadmap](roadmap.md) - what the project will and will not do - [Security Policy](security_policy.md) - the security policy of the project +- [Assurance Case](assurance_case.md) - why the library meets its security requirements diff --git a/docs/mkdocs/docs/community/quality_assurance.md b/docs/mkdocs/docs/community/quality_assurance.md index 4196f3532..bd35516b8 100644 --- a/docs/mkdocs/docs/community/quality_assurance.md +++ b/docs/mkdocs/docs/community/quality_assurance.md @@ -164,6 +164,9 @@ Note: Some modern features (like C++20 ranges or filesystem support) may be disa - [x] The parser is tested against extensive correctness suites for JSON compliance. - [x] In addition, the library is continuously fuzz-tested at [OSS-Fuzz](https://google.github.io/oss-fuzz/) where the library is checked against billions of inputs. +- [x] Every crash reported by OSS-Fuzz is fixed together with a unit test that reproduces it, and the fix references + the OSS-Fuzz issue. The round-trip checks of the fuzzer drivers are also part of the unit tests. See the + [fuzz testing documentation](https://github.com/nlohmann/json/blob/develop/tests/fuzzing.md#handling-oss-fuzz-reports). ## Static analysis diff --git a/docs/mkdocs/docs/community/roadmap.md b/docs/mkdocs/docs/community/roadmap.md new file mode 100644 index 000000000..e8c407d3f --- /dev/null +++ b/docs/mkdocs/docs/community/roadmap.md @@ -0,0 +1,43 @@ +# Roadmap + +This page describes what the project intends to do, and what it does not intend to do, over the next year. Concrete +work items are tracked in the [GitHub milestones](https://github.com/nlohmann/json/milestones) and the +[issue tracker](https://github.com/nlohmann/json/issues). + +## What the project will do + +- **Keep the C++11 baseline.** The library will continue to compile with every + [supported C++11 compiler](https://github.com/nlohmann/json/blob/develop/README.md#supported-compilers). Features of + later standards are only used when they are guarded by the `JSON_HAS_CPP_*` macros. +- **Stay conformant to JSON.** The parser and serializer follow [RFC 8259](https://datatracker.ietf.org/doc/html/rfc8259). + Extensions such as [comments](../features/comments.md) or [trailing commas](../features/trailing_commas.md) remain + opt-in. +- **Keep the 3.x public API stable.** Releases follow [semantic versioning](https://semver.org). Changes that would + break existing code are only added behind a feature macro, so users can opt in and test their code before a next + major release. +- **Support a broad range of compilers and platforms.** The [CI](quality_assurance.md) keeps testing old and new + versions of GCC, Clang, MSVC, and other compilers on Linux, macOS, and Windows. +- **Keep the quality assurance up.** Every change keeps the test coverage at 100%, passes the static and dynamic + analysis, and is fuzz-tested by OSS-Fuzz, see [Quality assurance](quality_assurance.md). +- **Harden the library against hostile input.** Handling deeply nested values without exhausting the call stack is + ongoing work. +- **Fix bugs and security issues** reported through the issue tracker and the [security policy](security_policy.md). + +## What the project will not do + +- **Break the public API of version 3.x.** See the + [contribution guidelines](https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#break-the-public-api) + for what counts as a breaking change. +- **Require a newer C++ standard than C++11.** +- **Break JSON conformance** or enable non-standard extensions by default. +- **Add dependencies** or require a build step. The library remains header-only, and the single header + `json.hpp` remains a complete distribution. +- **Trade simplicity for speed or memory efficiency.** Performance improvements are welcome, but the library is not + meant to compete with the fastest JSON libraries, see [Design goals](../home/design_goals.md). + +## Version 4.0 + +There is no decision yet on whether or when a version 4.0 with breaking changes will be released. Proposals that need +a major version, for instance stricter type conversions, are collected in issue +[#3453](https://github.com/nlohmann/json/issues/3453). Until then, such changes are only added as opt-in behavior +behind feature macros. diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index 4cb61053f..a0c84edaf 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -208,6 +208,16 @@ The library maps BJData types to JSON value types as follows: The mapping is **complete** in the sense that any BJData value can be converted to a JSON value. +!!! info "Round trips" + + A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with + [`to_bjdata`](../../api/basic_json/to_bjdata.md) using any combination of options and parsed back into an equal + value, and serializing that value again with the same options produces the same bytes. The exception is binary + values: they are only written as an optimized binary array (`[$B`) if Draft 3 is enabled and both `use_size` and + `use_type` are set. Otherwise, they are written as arrays of integers and parsed back as such (see the notes on + binary values above), and serializing such an array again may choose different, but equally valid, type markers. + The bytes can then differ, but parsing them again yields the same value. + ??? example ```cpp diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index e7baba0ae..2bc8a4a5b 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -93,11 +93,21 @@ See [full documentation of `JSON_NO_IO`](../api/macros/json_no_io.md). ## `JSON_NO_THREAD_LOCAL` -When defined, the library does not use `#!cpp thread_local` storage. Copying a value then always avoids the call stack -rather than descending into a bounded number of levels first, which is slower but yields the same values. +When defined, the library does not use `#!cpp thread_local` storage. Copying a value and comparing two values then +always avoid the call stack rather than descending into a bounded number of levels first, which is slower but yields the +same values and the same comparisons. See [full documentation of `JSON_NO_THREAD_LOCAL`](../api/macros/json_no_thread_local.md). +## `JSON_PRECISE_STREAM_POSITION` + +When defined to `1`, [`operator>>`](../api/operator_gtgt.md) and non-strict +[`sax_parse`](../api/basic_json/sax_parse.md) leave an input stream positioned right after the parsed value, instead of +also consuming the character that terminates a number. The default value is `0`, which preserves the existing behavior; +this is planned to become the default in version 4.0.0. + +See [full documentation of `JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md). + ## `JSON_SKIP_LIBRARY_VERSION_CHECK` When defined, the library will not create a compiler warning when a different version of the library was already diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 09e53f3a2..5eb4a76a9 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -18,6 +18,7 @@ The complete default namespace name is derived as follows: - [`JSON_DIAGNOSTIC_POSITIONS`](../api/macros/json_diagnostic_positions.md) defined non-zero appends `_dp`. - [`JSON_BRACE_INIT_COPY_SEMANTICS`](../api/macros/json_brace_init_copy_semantics.md) defined non-zero appends `_bics`. + - [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) defined non-zero appends `_psp`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/docs/mkdocs/docs/features/parsing/index.md b/docs/mkdocs/docs/features/parsing/index.md index 17624b8a2..476f024fa 100644 --- a/docs/mkdocs/docs/features/parsing/index.md +++ b/docs/mkdocs/docs/features/parsing/index.md @@ -41,7 +41,8 @@ document followed by trailing bytes" is accepted rather than rejected. If you ar reject any input that is not exactly one JSON document, prefer `parse`. When using `operator>>` to read several concatenated values this way, a value that is a number must be followed by -whitespace, because `operator>>` consumes the character that terminates a number — see the +whitespace, because `operator>>` consumes the character that terminates a number, unless +[`JSON_PRECISE_STREAM_POSITION`](../../api/macros/json_precise_stream_position.md) is defined to `1` — see the [`operator>>` notes](../../api/operator_gtgt.md#notes) for details and examples. ## SAX vs. DOM parsing diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index aba2be580..6c0892f11 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -1,34 +1,125 @@ # Architecture -!!! info - - This page is still under construction. Its goal is to provide a high-level overview of the library's architecture. - This should help new contributors to get an idea of the used concepts and where to make changes. +This page gives a high-level overview of the library's architecture. It should help new contributors to get an idea of +the used concepts and where to make changes. ## Overview -The main structure is class [nlohmann::basic_json](../api/basic_json/index.md). +The library is built around a single class template, [`nlohmann::basic_json`](../api/basic_json/index.md). A +`basic_json` value is a node in a tree of JSON values. All other components either create such a tree from an input +(parsing), write a tree to an output (serialization), or give access to it (iterators, JSON Pointer, conversions). -- public API -- container interface -- iterators +```mermaid +flowchart LR + input[/"input
(string, stream,
iterator range, file)"/] + ia["input adapter"] + lexer["lexer"] + parser["parser"] + breader["binary_reader"] + sax["SAX interface"] + value[("basic_json
value tree")] + serializer["serializer"] + bwriter["binary_writer"] + oa["output adapter"] + output[/"output
(string, stream,
vector)"/] -## Template specializations + input --> ia + ia --> lexer --> parser --> sax + ia --> breader --> sax + sax --> value + value --> serializer --> oa + value --> bwriter --> oa + oa --> output +``` -- describe template parameters of `basic_json` -- [`json`](../api/json.md) -- [`ordered_json`](../api/ordered_json.md) via [`ordered_map`](../api/ordered_map.md) +- **JSON text** is read by an [input adapter](#input-adapters), tokenized by the lexer, and turned into SAX events by + the parser. +- **Binary formats** (BJData, BSON, CBOR, MessagePack, UBJSON) are read by an input adapter and turned into the same SAX + events by the `binary_reader`. +- A [SAX consumer](#sax-interface) receives the events. The one used by [`parse`](../api/basic_json/parse.md) builds a + `basic_json` value tree. +- The `serializer` (JSON text) or the `binary_writer` (binary formats) writes a value tree to an + [output adapter](#output-adapters). + +## Source layout + +The public headers are in [`include/nlohmann`](https://github.com/nlohmann/json/tree/develop/include/nlohmann): + +- [`json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json.hpp) defines class [`basic_json`](../api/basic_json/index.md). +- [`json_fwd.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json_fwd.hpp) contains forward declarations. +- [`adl_serializer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/adl_serializer.hpp), [`byte_container_with_subtype.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/byte_container_with_subtype.hpp), and [`ordered_map.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/ordered_map.hpp) define + [`adl_serializer`](../api/adl_serializer/index.md), + [`byte_container_with_subtype`](../api/byte_container_with_subtype/index.md), and + [`ordered_map`](../api/ordered_map.md). + +Everything else lives in [`detail/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail) and namespace `nlohmann::detail`, which is not part of the public API. Paths +below are relative to `include/nlohmann`. + +| Component | Location | +|-----------|----------| +| Value type enumeration | [`detail/value_t.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/value_t.hpp) | +| Input adapters | [`detail/input/input_adapters.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/input_adapters.hpp) | +| Lexer | [`detail/input/lexer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/lexer.hpp), [`detail/input/number_parse.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/number_parse.hpp), [`detail/input/string_scan.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/string_scan.hpp) | +| Parser | [`detail/input/parser.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/parser.hpp) | +| SAX interface and DOM builders | [`detail/input/json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp) | +| Binary format readers | [`detail/input/binary_reader.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/binary_reader.hpp) | +| JSON serializer | [`detail/output/serializer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/serializer.hpp), [`detail/conversions/to_chars.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/to_chars.hpp) | +| Binary format writers | [`detail/output/binary_writer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/binary_writer.hpp) | +| Output adapters | [`detail/output/output_adapters.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/output_adapters.hpp) | +| Iterators | [`detail/iterators/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail/iterators) | +| Conversions from/to arbitrary types | [`detail/conversions/from_json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/from_json.hpp), [`detail/conversions/to_json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/to_json.hpp) | +| JSON Pointer | [`detail/json_pointer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/json_pointer.hpp) | +| Exceptions | [`detail/exceptions.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/exceptions.hpp) | +| Type traits and C++ feature backports | [`detail/meta/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail/meta) | +| Macros | [`detail/macro_scope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_scope.hpp), [`detail/macro_unscope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_unscope.hpp), [`detail/abi_macros.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/abi_macros.hpp) | + +The single-header version [`single_include/nlohmann/json.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json.hpp) +is generated from these files with `make amalgamate` and must not be edited by hand. + +## Template parameters + +[`basic_json`](../api/basic_json/index.md) is parameterized by the types it uses to store values and to convert from and to other types: + +| Template parameter | Default | Used for | +|----------------------|-----------------------------|-------------------------------------------------------------------| +| `ObjectType` | `std::map` | objects, see [`object_t`](../api/basic_json/object_t.md) | +| `ArrayType` | `std::vector` | arrays, see [`array_t`](../api/basic_json/array_t.md) | +| `StringType` | `std::string` | strings and object keys, see [`string_t`](../api/basic_json/string_t.md) | +| `BooleanType` | `bool` | Booleans, see [`boolean_t`](../api/basic_json/boolean_t.md) | +| `NumberIntegerType` | `std::int64_t` | signed integers, see [`number_integer_t`](../api/basic_json/number_integer_t.md) | +| `NumberUnsignedType` | `std::uint64_t` | unsigned integers, see [`number_unsigned_t`](../api/basic_json/number_unsigned_t.md) | +| `NumberFloatType` | `double` | floating-point numbers, see [`number_float_t`](../api/basic_json/number_float_t.md) | +| `AllocatorType` | `std::allocator` | allocating objects, arrays, strings, and binary values | +| `JSONSerializer` | `adl_serializer` | conversions from/to other types, see [`adl_serializer`](../api/adl_serializer/index.md) | +| `BinaryType` | `std::vector` | binary values, see [`binary_t`](../api/basic_json/binary_t.md) | +| `CustomBaseClass` | `void` | an optional base class, see [`json_base_class_t`](../api/basic_json/json_base_class_t.md) | + +The library provides two specializations: + +- [`json`](../api/json.md) uses all default template arguments. +- [`ordered_json`](../api/ordered_json.md) uses [`ordered_map`](../api/ordered_map.md) as `ObjectType` to keep the + insertion order of object keys. + +The requirements on the template arguments are listed in +[Template Parameter Requirements](../features/types/template_parameters.md). ## Value storage -Values are stored as a tagged union of [value_t](../api/basic_json/value_t.md) and json_value. +Each [`basic_json`](../api/basic_json/index.md) value stores its content as a tagged union: an enumeration [`value_t`](../api/basic_json/value_t.md) +names the type of the value, and a union `json_value` holds the value itself. Both are members of the nested struct +`data`, which is the only data member `m_data` of `basic_json`: ```cpp -/// the type of the current element -value_t m_type = value_t::null; +struct data +{ + /// the type of the current element + value_t m_type = value_t::null; -/// the value of the current element -json_value m_value = {}; + /// the value of the current element + json_value m_value = {}; +}; + +data m_data = {}; ``` with @@ -68,42 +159,83 @@ union json_value { }; ``` -## Parsing inputs (deserialization) +Objects, arrays, strings, and binary values are allocated on the heap with `AllocatorType`, and the union only stores a +pointer to them. This keeps a `basic_json` value small: one pointer-sized union and one byte for the type. The class +maintains the invariant that the pointer matching `m_type` is never null; `assert_invariant()` checks it with +[runtime assertions](../features/assertions.md). -Input is read via **input adapters** that abstract a source with a common interface: +## Input adapters + +Input is read via **input adapters** that abstract a source. Every input adapter provides this interface: ```cpp -/// read a single character -std::char_traits::int_type get_character() noexcept; +/// the type of the characters in the input +using char_type = ...; -/// read multiple characters to a destination buffer and -/// returns the number of characters successfully read +/// read a single character; returns std::char_traits::eof() at the end of the input +typename std::char_traits::int_type get_character(); + +/// read up to count * sizeof(T) bytes into dest and return the number of bytes read +/// (used by the binary readers) template std::size_t get_elements(T* dest, std::size_t count = 1); ``` -List examples of input adapters. +The lexer detects two optional extensions at compile time. Only `iterator_input_adapter` provides them, and only for +random-access input of single-byte characters: -## SAX Interface +- `supports_seek`, `get_consumed_count()`, and `copy_consumed_range()` let the lexer reconstruct already consumed input + for error messages instead of copying every character it reads. +- `supports_bulk_scan`, `bulk_data()`, `bulk_remaining()`, and `bulk_skip()` let the lexer scan strings directly in + contiguous memory, several bytes at a time. -TODO +The function `input_adapter` picks the right adapter for the argument passed to `parse`, `accept`, `sax_parse`, or the +`from_*` functions: -## Writing outputs (serialization) +- `iterator_input_adapter` reads from an iterator range, which also covers strings, containers, and pointers. +- `wide_string_input_adapter` reads from ranges of `wchar_t`, `char16_t`, or `char32_t` and converts them to UTF-8. + It cannot be used for binary formats; its `get_elements()` throws. +- `input_stream_adapter` reads from a `std::istream`. +- `file_input_adapter` reads from a `std::FILE*`. + +## SAX interface + +The parser does not build values itself. It reports what it reads as events to a [SAX](../features/parsing/sax_interface.md) +consumer, which implements the interface [`json_sax`](../api/json_sax/index.md): `null`, `boolean`, `number_integer`, +`number_unsigned`, `number_float`, `string`, `binary`, `start_object`, `key`, `end_object`, `start_array`, `end_array`, +and `parse_error`. + +The library comes with two consumers in `detail/input/json_sax.hpp`: + +- `json_sax_dom_parser` builds a [`basic_json`](../api/basic_json/index.md) value tree. [`parse`](../api/basic_json/parse.md) uses it. +- `json_sax_dom_callback_parser` does the same, but calls a [parser callback](../features/parsing/parser_callbacks.md) + for each event, which can skip values. `parse` uses it when a callback is given. + +The `binary_reader` emits the same events for binary formats, so [`sax_parse`](../api/basic_json/sax_parse.md) works +with a user-defined consumer for JSON and for all binary formats alike. + +## Output adapters Output is written via **output adapters**: ```cpp -template void write_character(CharType c); -template void write_characters(const CharType* s, std::size_t length); ``` -List examples of output adapters. +The `serializer` (used by [`dump`](../api/basic_json/dump.md) and [`operator<<`](../api/operator_ltlt.md)) and the +`binary_writer` (used by the `to_*` functions) write to one of these adapters: + +- `output_vector_adapter` appends to a `std::vector`. +- `output_stream_adapter` writes to a `std::ostream`. +- `output_string_adapter` appends to a string. ## Value conversion +Values are converted from and to other types with the `JSONSerializer` template parameter. The default, +[`adl_serializer`](../api/adl_serializer/index.md), calls the free functions + ```cpp template void to_json(basic_json& j, const T& t); @@ -112,13 +244,23 @@ template void from_json(const basic_json& j, T& t); ``` +found by argument-dependent lookup. The library defines them for standard types in `detail/conversions`; users add them +for their own types, see [Arbitrary Type Conversions](../features/arbitrary_types.md). The +[serialization macros](../features/macros.md) generate these functions. + ## Additional features -- JSON Pointers -- Binary formats -- Custom base class -- Conversion macros +- [JSON Pointer](../features/json_pointer.md) (class `json_pointer`) addresses values inside a tree. It is also the + basis of [JSON Patch](../features/json_patch.md). +- [Binary formats](../features/binary_formats/index.md) are read by `binary_reader` and written by `binary_writer`. +- A [custom base class](../api/basic_json/json_base_class_t.md) can add members to every [`basic_json`](../api/basic_json/index.md) value. +- [Serialization macros](../features/macros.md) generate `to_json` and `from_json` functions for user-defined types. ## Details namespace -- C++ feature backports +Namespace `nlohmann::detail` contains all implementation details. It is not part of the public API and may change in any +release. Besides the components above, it contains: + +- type traits to detect the capabilities of user-defined types (`detail/meta/type_traits.hpp`), +- backports of C++14/17 features to C++11 (`detail/meta/cpp_future.hpp`), and +- helpers such as `string_concat` and `string_escape`. diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 8f92a838f..a05ce2dff 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -293,6 +293,7 @@ nav: - 'JSON_NOEXCEPTION': api/macros/json_noexception.md - 'JSON_NO_IO': api/macros/json_no_io.md - 'JSON_NO_THREAD_LOCAL': api/macros/json_no_thread_local.md + - 'JSON_PRECISE_STREAM_POSITION': api/macros/json_precise_stream_position.md - 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md - 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md - 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md @@ -317,7 +318,9 @@ nav: - community/contribution_guidelines.md - community/quality_assurance.md - community/governance.md + - community/roadmap.md - community/security_policy.md + - community/assurance_case.md # Extras extra: diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index cca04e8ec..a6666c66e 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -38,6 +38,10 @@ #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif +#ifndef JSON_PRECISE_STREAM_POSITION + #define JSON_PRECISE_STREAM_POSITION 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -62,21 +66,28 @@ #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS #endif +#if JSON_PRECISE_STREAM_POSITION + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION _psp +#else + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ - NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index 775a8398c..317b38806 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -101,6 +101,11 @@ class input_stream_adapter // maintain ifstream flags, except eof if (is != nullptr) { +#if JSON_PRECISE_STREAM_POSITION + // consume the character last returned by get_character() unless it + // was given back with release_lookahead() + commit_lookahead(); +#endif is->clear(is->rdstate() & std::ios::eofbit); } } @@ -114,6 +119,58 @@ class input_stream_adapter input_stream_adapter& operator=(input_stream_adapter&) = delete; input_stream_adapter& operator=(input_stream_adapter&&) = delete; +#if JSON_PRECISE_STREAM_POSITION + input_stream_adapter(input_stream_adapter&& rhs) noexcept + : is(rhs.is), sb(rhs.sb), lookahead(rhs.lookahead) + { + rhs.is = nullptr; + rhs.sb = nullptr; + rhs.lookahead = false; + } + + // Whether the character last returned by get_character() can be given back + // to the input with release_lookahead(). + static constexpr bool supports_lookahead = true; + + // std::istream/std::streambuf use std::char_traits::to_int_type, to + // ensure that std::char_traits::eof() and the character 0xFF do not + // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is peeked rather than consumed: it is only stepped over + // once the next character is requested, or when the adapter is destroyed. + // Until then, release_lookahead() can leave it in the input. + std::char_traits::int_type get_character() + { + if (lookahead) + { + // step over the character returned by the previous call + sb->sbumpc(); + } + + auto res = sb->sgetc(); + // set eof manually, as we don't use the istream interface. + if (JSON_HEDLEY_UNLIKELY(res == std::char_traits::eof())) + { + // there is nothing to step over next time + lookahead = false; + is->clear(is->rdstate() | std::ios::eofbit); + } + else + { + lookahead = true; + } + return res; + } + + // Leave the character last returned by get_character() in the input, so + // that the next read from the stream - by this adapter or by the caller + // once parsing is done - sees it again. Unlike putting a consumed + // character back, this cannot fail. + void release_lookahead() noexcept + { + lookahead = false; + } +#else input_stream_adapter(input_stream_adapter&& rhs) noexcept : is(rhs.is), sb(rhs.sb) { @@ -124,6 +181,9 @@ class input_stream_adapter // std::istream/std::streambuf use std::char_traits::to_int_type, to // ensure that std::char_traits::eof() and the character 0xFF do not // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is consumed, so the character that terminates a number + // stays consumed after parsing; see JSON_PRECISE_STREAM_POSITION. std::char_traits::int_type get_character() { auto res = sb->sbumpc(); @@ -134,10 +194,14 @@ class input_stream_adapter } return res; } +#endif template std::size_t get_elements(T* dest, std::size_t count = 1) { +#if JSON_PRECISE_STREAM_POSITION + commit_lookahead(); +#endif auto res = static_cast(sb->sgetn(reinterpret_cast(dest), static_cast(count * sizeof(T)))); if (JSON_HEDLEY_UNLIKELY(res < count * sizeof(T))) { @@ -147,9 +211,27 @@ class input_stream_adapter } private: +#if JSON_PRECISE_STREAM_POSITION + // Step over the character last returned by get_character(). The character + // has already been peeked successfully, so for every streambuf with a get + // area this is a pointer increment that cannot fail. + void commit_lookahead() + { + if (lookahead) + { + lookahead = false; + sb->sbumpc(); + } + } +#endif + /// the associated input stream std::istream* is = nullptr; std::streambuf* sb = nullptr; +#if JSON_PRECISE_STREAM_POSITION + /// whether get_character() peeked a character that is not consumed yet + bool lookahead = false; +#endif }; #endif // JSON_NO_IO diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index fe85cd53d..98c0fd76a 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -127,6 +127,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/) return false; } +// Detect whether an input adapter reads with one character of lookahead that +// can be left in the input (see input_stream_adapter::supports_lookahead, +// which is only defined with JSON_PRECISE_STREAM_POSITION), detected like +// supports_seek above. +template +using detect_supports_lookahead = decltype(InputAdapterType::supports_lookahead); + +template +constexpr bool input_adapter_supports_lookahead(std::true_type /*detected*/) +{ + return InputAdapterType::supports_lookahead; +} + +template +constexpr bool input_adapter_supports_lookahead(std::false_type /*detected*/) +{ + return false; +} + // Detect whether an input adapter exposes a contiguous byte block that the // lexer can scan directly (see iterator_input_adapter::supports_bulk_scan). // Adapters without the flag - file, stream, wide-string, user-defined - fall @@ -167,6 +186,12 @@ class lexer : public lexer_base static constexpr bool lazy_token_string = input_adapter_supports_seek(is_detected {}); + /// whether a simulated unget can be passed on to the input adapter, which + /// then leaves the character in the input; see + /// input_adapter_supports_lookahead + static constexpr bool can_release_lookahead = + input_adapter_supports_lookahead(is_detected {}); + /// whether string scanning may bulk-consume runs of ordinary characters /// directly from a contiguous input buffer (SWAR fast path). This requires /// the token to be reconstructible lazily (lazy_token_string), so bypassing @@ -1898,6 +1923,21 @@ scan_number_done: uncapture_char(std::integral_constant {}); } + /// adapter without lookahead: nothing to do (see release_lookahead) + void release_lookahead_impl(std::false_type /*can_release*/) const noexcept {} + + /// adapter with lookahead: leave the character in the input instead + void release_lookahead_impl(std::true_type /*can_release*/) + { + if (next_unget) + { + // the character is read from the input again rather than replayed + // from current, so the adapter must not step over it + next_unget = false; + ia.release_lookahead(); + } + } + /// seekable adapter: nothing was captured, so nothing to undo void uncapture_char(std::true_type /*lazy*/) const noexcept {} @@ -1961,6 +2001,31 @@ scan_number_done: return position; } + /*! + @brief pass a pending simulated unget on to the input + + unget() only rewinds the lexer's own bookkeeping, so the character that + terminated the last token (e.g. the character after a number) would still + be stepped over when the input adapter is done. Callers that hand the + input back to the user afterwards - operator>> and non-strict sax_parse - + call this once when scanning is done, so that the input is positioned + right after the value. + + Adapters without lookahead (see input_adapter_supports_lookahead) are not + handed back to the user, so this is a no-op for them. Without + JSON_PRECISE_STREAM_POSITION, no adapter has lookahead, so this is always a + no-op and the terminating character stays consumed. + + Scanning may continue after this call: @a next_unget is cleared, and the + character is read from the input again instead of being replayed from + @a current. A pending unget of EOF needs no special case, because reaching + EOF leaves no lookahead to release. + */ + void release_lookahead() + { + release_lookahead_impl(std::integral_constant {}); + } + #if JSON_DIAGNOSTIC_POSITIONS /// return the offset of the first character of the last read token; unlike /// the token's parsed value, this accounts for escape sequences diff --git a/include/nlohmann/detail/input/parser.hpp b/include/nlohmann/detail/input/parser.hpp index a45ee4a0a..5fec57a70 100644 --- a/include/nlohmann/detail/input/parser.hpp +++ b/include/nlohmann/detail/input/parser.hpp @@ -100,13 +100,22 @@ class parser json_sax_dom_callback_parser sdp(result, callback, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), - exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), + exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -128,12 +137,20 @@ class parser json_sax_dom_parser sdp(result, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // see above + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -166,12 +183,24 @@ class parser (void)detail::is_sax_static_asserts {}; const bool result = sax_parse_internal(sax); - // strict mode: next byte must be EOF - if (result && strict && (get_token() != token_type::end_of_input)) + if (result) { - return sax->parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + if (strict) + { + // strict mode: next byte must be EOF + if (get_token() != token_type::end_of_input) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); + } } return result; diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index afcbfc38b..6ace4cf4a 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -45,6 +45,7 @@ #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS + #undef JSON_PRECISE_STREAM_POSITION #endif #include diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 4f41c865c..c578fd11a 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -924,6 +924,31 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #endif + /*! + @brief whether a descent must stop here and finish without the call stack + + @a may_descend says whether the operator descends at all; it is a constant + at every call site, and is passed rather than tested by the caller so that + the test does not become a constant condition there, which MSVC reports as + C4127. + + The comparison operators use this rather than @ref nesting_depth_guard::okay, + because they are written as a macro and a macro cannot use the preprocessor + the way the guard's constructor does; @ref copy_structured, which can, asks + the guard instead and never calls this. + */ + static bool nesting_depth_exhausted(bool may_descend = true) noexcept + { +#ifdef JSON_NO_THREAD_LOCAL + // without a count of its own per thread, a descent cannot be bounded + // without racing another one, so none is made + static_cast(may_descend); + return true; +#else + return !may_descend || nesting_depth() >= nesting_depth_limit(); +#endif + } + /*! @brief counts one level of a bounded descent for as long as it runs, and reports whether the descent was still within the limit when it began @@ -1244,6 +1269,274 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// the result of comparing two values, including values that cannot be + /// ordered at all, such as a discarded value or a NaN + enum class compare_result { less, equal, greater, unordered }; + +#if JSON_HAS_THREE_WAY_COMPARISON + /// @brief the ordering that @a result stands for + static std::partial_ordering to_partial_ordering(compare_result result) noexcept // *NOPAD* + { + switch (result) + { + case compare_result::less: + return std::partial_ordering::less; + case compare_result::greater: + return std::partial_ordering::greater; + case compare_result::equal: + return std::partial_ordering::equivalent; + case compare_result::unordered: + default: + return std::partial_ordering::unordered; + } + } +#endif + + /*! + @brief compare two values that are not both an array or both an object + + Such a pair is compared by the operators themselves, which cannot descend + into it and therefore cannot recurse. + + That holds for a pair whose types differ as much as for a pair of leaves: an + array and an object are told apart by their types alone, because an operator + only ever descends into two values of the same type. So `==` reports them as + unequal without looking inside either, and an ordering falls back to the + order of the types - an object sorts before an array - exactly as it does + for a value that is not nested deeply enough to get here. + */ + template + static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept + { + if (lhs == rhs) + { + return compare_result::equal; + } + + return order_leaves(lhs, rhs, std::integral_constant {}); + } + + /*! + @brief compare two object keys + + An object compares its entries as pairs of a key and a value, so its keys + are compared exactly as std::pair compares them: with < where the objects + are being ordered, and with == where they are only checked for equality. + Note that this is not the object's own comparator, which for a vector-backed + object type such as nlohmann::ordered_map tells equality rather than order. + */ + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::true_type /*ordered*/) + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::equal; + } + + /// @brief check two object keys for equality + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::false_type /*ordered*/) + { + return lhs == rhs ? compare_result::equal : compare_result::unordered; + } + + /// @brief tell apart two values that are not equal + /// @note only instantiated where the values are being ordered, as a key or + /// string type is not required to be ordered to be compared for equality + static compare_result order_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::unordered; + } + + /// @brief report two values as not equal without ordering them + static compare_result order_leaves(const_reference /*lhs*/, const_reference /*rhs*/, std::false_type /*ordered*/) noexcept + { + return compare_result::unordered; + } + + /*! + @brief compare @a lhs and @a rhs without descending into them + + Reached once a comparison has descended @ref nesting_depth_limit levels, so + that comparing values cannot exhaust the call stack however deeply they are + nested. The two values are walked in lockstep on an explicit stack and + compared lexicographically, element by element in the order the containers + enumerate them - which is how the container types this library ships compare + themselves: a std::map enumerates its entries in key order, and + nlohmann::ordered_map in insertion order. An object type that enumerates its + entries in an unspecified order, such as std::unordered_map, compares them + pairwise instead; the difference could only ever show below the bound. + + Note that the stack this walks with is allocated, while the comparison + operators are noexcept and the container comparison this replaces allocated + nothing. Failing that allocation therefore ends the process rather than + throwing. It only arises for values nested past the bound, and only when + memory has run out - where the same comparison used to exhaust the call + stack instead - but it is a way to fail that the operators did not have. + */ + template + static compare_result compare_iteratively(const_reference lhs, const_reference rhs, + const bool unordered_compares_equal) noexcept + { + /// a pair of containers being compared in lockstep + struct frame + { + const basic_json* lhs_value{nullptr}; + const basic_json* rhs_value{nullptr}; + typename array_t::const_iterator lhs_array_it{}; + typename array_t::const_iterator rhs_array_it{}; + typename object_t::const_iterator lhs_object_it{}; + typename object_t::const_iterator rhs_object_it{}; + }; + + std::vector stack; + const basic_json* left = &lhs; + const basic_json* right = &rhs; + + for (;;) + { + const auto type = left->m_data.m_type; + + if (type == right->m_data.m_type && (type == value_t::array || type == value_t::object)) + { + // descend: the elements decide, and are compared further down + stack.emplace_back(); + frame& pushed = stack.back(); + pushed.lhs_value = left; + pushed.rhs_value = right; + + if (type == value_t::array) + { + pushed.lhs_array_it = left->m_data.m_value.array->cbegin(); + pushed.rhs_array_it = right->m_data.m_value.array->cbegin(); + } + else + { + pushed.lhs_object_it = left->m_data.m_value.object->cbegin(); + pushed.rhs_object_it = right->m_data.m_value.object->cbegin(); + } + } + else + { + const compare_result result = compare_leaves(*left, *right); + + // Values that cannot be ordered - a NaN, say - end an ordered + // comparison for std::lexicographical_compare_three_way, but + // std::lexicographical_compare treats them as equivalent and + // carries on with the next element. Both are reproduced here, + // so that a value nested too deeply to descend into compares + // exactly as one that is not. + if (result != compare_result::equal && + !(unordered_compares_equal && result == compare_result::unordered)) + { + return result; + } + } + + // walk back up past the containers that are exhausted, then take the + // next pair of elements from the innermost one that is not + for (;;) + { + if (stack.empty()) + { + return compare_result::equal; + } + + frame& current = stack.back(); + const bool is_object = current.lhs_value->m_data.m_type == value_t::object; + + const bool lhs_done = is_object + ? current.lhs_object_it == current.lhs_value->m_data.m_value.object->cend() + : current.lhs_array_it == current.lhs_value->m_data.m_value.array->cend(); + const bool rhs_done = is_object + ? current.rhs_object_it == current.rhs_value->m_data.m_value.object->cend() + : current.rhs_array_it == current.rhs_value->m_data.m_value.array->cend(); + + if (lhs_done || rhs_done) + { + // whichever ran out first holds the smaller container; if + // both did, they are equal and the container above decides + if (lhs_done != rhs_done) + { + return lhs_done ? compare_result::less : compare_result::greater; + } + + stack.pop_back(); + continue; + } + + if (is_object) + { + // an entry is a key and a value, and the key decides first + const compare_result key_result = + compare_keys(current.lhs_object_it->first, current.rhs_object_it->first, + std::integral_constant {}); + + if (key_result != compare_result::equal) + { + return key_result; + } + + left = &(current.lhs_object_it->second); + right = &(current.rhs_object_it->second); + ++current.lhs_object_it; + ++current.rhs_object_it; + } + else + { + left = &(*current.lhs_array_it); + right = &(*current.rhs_array_it); + ++current.lhs_array_it; + ++current.rhs_array_it; + } + + break; + } + } + } + + + /// @brief restore the parent pointers after erasing from an object + /// ordered_json keeps its members in a vector, and erasing a member + /// re-constructs every member after it in place, which resets their + /// parent pointers + void set_parents_after_object_erase() + { +#if JSON_DIAGNOSTICS +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) + { + set_parents(); + } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif +#endif + } + public: ////////////////////////// // JSON parser callback // @@ -2261,6 +2554,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec ValueType & get_to(ValueType& v) const noexcept(noexcept( JSONSerializer::from_json(std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } @@ -2286,6 +2580,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec noexcept(noexcept(JSONSerializer::from_json( std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } @@ -2932,6 +3227,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -3004,6 +3300,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { result.m_it.object_iterator = m_data.m_value.object->erase(first.m_it.object_iterator, last.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -3034,7 +3331,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - return m_data.m_value.object->erase(std::forward(key)); + const auto erased = m_data.m_value.object->erase(std::forward(key)); + set_parents_after_object_erase(); + return erased; } template < typename KeyType, detail::enable_if_t < @@ -3051,6 +3350,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); + set_parents_after_object_erase(); return 1; } return 0; @@ -3962,16 +4262,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); -#if JSON_DIAGNOSTICS - it2->second.set_parents(); -#endif continue; } } - m_data.m_value.object->operator[](it.key()) = it.value(); -#if JSON_DIAGNOSTICS - m_data.m_value.object->operator[](it.key()).m_parent = this; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + set_parent(m_data.m_value.object->operator[](it.key()) = it.value()); } } @@ -4000,9 +4296,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // a nested object is merged: continue with its parent -#if JSON_DIAGNOSTICS - target->set_parents(); -#endif target = stack.back().target; first = stack.back().position; last = stack.back().last; @@ -4024,10 +4317,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec continue; } } - target->m_data.m_value.object->operator[](first.key()) = first.value(); -#if JSON_DIAGNOSTICS - target->m_data.m_value.object->operator[](first.key()).m_parent = target; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + target->set_parent(target->m_data.m_value.object->operator[](first.key()) = first.value()); ++first; } } @@ -4166,7 +4458,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // because any negative signed value is smaller than any unsigned value. // Otherwise, the non-negative signed value is cast to unsigned before the // comparison to avoid wraparound. -#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result) \ +#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result, deep_result, may_descend) \ const auto lhs_type = lhs.type(); \ const auto rhs_type = rhs.type(); \ \ @@ -4175,11 +4467,25 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec switch (lhs_type) \ { \ case value_t::array: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.array) op (*rhs.m_data.m_value.array); \ - \ + } \ + \ case value_t::object: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.object) op (*rhs.m_data.m_value.object); \ - \ + } \ + \ case value_t::null: \ return (null_result); \ \ @@ -4279,7 +4585,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif const_reference lhs = *this; - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -4304,7 +4611,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_IMPLEMENT_OPERATOR(<=>, // *NOPAD* std::partial_ordering::equivalent, std::partial_ordering::unordered, - lhs_type <=> rhs_type) // *NOPAD* + lhs_type <=> rhs_type, // *NOPAD* + to_partial_ordering(compare_iteratively(lhs, rhs, false)), true) } /// @brief comparison: 3-way @@ -4371,7 +4679,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -4427,7 +4736,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // default_result is used if we cannot compare values. In that case, // we compare types. Note we have to call the operator explicitly, // because MSVC has problems otherwise. - JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type)) + JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type), + compare_iteratively(lhs, rhs, true) == compare_result::less, false) } /// @brief comparison: less than diff --git a/nlohmann_json.natvis b/nlohmann_json.natvis index 2eccbe17c..8f4eec31a 100644 --- a/nlohmann_json.natvis +++ b/nlohmann_json.natvis @@ -335,6 +335,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -515,6 +575,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -635,6 +755,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -695,6 +875,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -815,6 +1115,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -875,6 +1235,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -935,6 +1415,186 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -995,4 +1655,304 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 37734fdc7..399f922e0 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -95,6 +95,10 @@ #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif +#ifndef JSON_PRECISE_STREAM_POSITION + #define JSON_PRECISE_STREAM_POSITION 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -119,21 +123,28 @@ #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS #endif +#if JSON_PRECISE_STREAM_POSITION + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION _psp +#else + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ - NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -7419,6 +7430,11 @@ class input_stream_adapter // maintain ifstream flags, except eof if (is != nullptr) { +#if JSON_PRECISE_STREAM_POSITION + // consume the character last returned by get_character() unless it + // was given back with release_lookahead() + commit_lookahead(); +#endif is->clear(is->rdstate() & std::ios::eofbit); } } @@ -7432,6 +7448,58 @@ class input_stream_adapter input_stream_adapter& operator=(input_stream_adapter&) = delete; input_stream_adapter& operator=(input_stream_adapter&&) = delete; +#if JSON_PRECISE_STREAM_POSITION + input_stream_adapter(input_stream_adapter&& rhs) noexcept + : is(rhs.is), sb(rhs.sb), lookahead(rhs.lookahead) + { + rhs.is = nullptr; + rhs.sb = nullptr; + rhs.lookahead = false; + } + + // Whether the character last returned by get_character() can be given back + // to the input with release_lookahead(). + static constexpr bool supports_lookahead = true; + + // std::istream/std::streambuf use std::char_traits::to_int_type, to + // ensure that std::char_traits::eof() and the character 0xFF do not + // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is peeked rather than consumed: it is only stepped over + // once the next character is requested, or when the adapter is destroyed. + // Until then, release_lookahead() can leave it in the input. + std::char_traits::int_type get_character() + { + if (lookahead) + { + // step over the character returned by the previous call + sb->sbumpc(); + } + + auto res = sb->sgetc(); + // set eof manually, as we don't use the istream interface. + if (JSON_HEDLEY_UNLIKELY(res == std::char_traits::eof())) + { + // there is nothing to step over next time + lookahead = false; + is->clear(is->rdstate() | std::ios::eofbit); + } + else + { + lookahead = true; + } + return res; + } + + // Leave the character last returned by get_character() in the input, so + // that the next read from the stream - by this adapter or by the caller + // once parsing is done - sees it again. Unlike putting a consumed + // character back, this cannot fail. + void release_lookahead() noexcept + { + lookahead = false; + } +#else input_stream_adapter(input_stream_adapter&& rhs) noexcept : is(rhs.is), sb(rhs.sb) { @@ -7442,6 +7510,9 @@ class input_stream_adapter // std::istream/std::streambuf use std::char_traits::to_int_type, to // ensure that std::char_traits::eof() and the character 0xFF do not // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is consumed, so the character that terminates a number + // stays consumed after parsing; see JSON_PRECISE_STREAM_POSITION. std::char_traits::int_type get_character() { auto res = sb->sbumpc(); @@ -7452,10 +7523,14 @@ class input_stream_adapter } return res; } +#endif template std::size_t get_elements(T* dest, std::size_t count = 1) { +#if JSON_PRECISE_STREAM_POSITION + commit_lookahead(); +#endif auto res = static_cast(sb->sgetn(reinterpret_cast(dest), static_cast(count * sizeof(T)))); if (JSON_HEDLEY_UNLIKELY(res < count * sizeof(T))) { @@ -7465,9 +7540,27 @@ class input_stream_adapter } private: +#if JSON_PRECISE_STREAM_POSITION + // Step over the character last returned by get_character(). The character + // has already been peeked successfully, so for every streambuf with a get + // area this is a pointer increment that cannot fail. + void commit_lookahead() + { + if (lookahead) + { + lookahead = false; + sb->sbumpc(); + } + } +#endif + /// the associated input stream std::istream* is = nullptr; std::streambuf* sb = nullptr; +#if JSON_PRECISE_STREAM_POSITION + /// whether get_character() peeked a character that is not consumed yet + bool lookahead = false; +#endif }; #endif // JSON_NO_IO @@ -8879,6 +8972,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/) return false; } +// Detect whether an input adapter reads with one character of lookahead that +// can be left in the input (see input_stream_adapter::supports_lookahead, +// which is only defined with JSON_PRECISE_STREAM_POSITION), detected like +// supports_seek above. +template +using detect_supports_lookahead = decltype(InputAdapterType::supports_lookahead); + +template +constexpr bool input_adapter_supports_lookahead(std::true_type /*detected*/) +{ + return InputAdapterType::supports_lookahead; +} + +template +constexpr bool input_adapter_supports_lookahead(std::false_type /*detected*/) +{ + return false; +} + // Detect whether an input adapter exposes a contiguous byte block that the // lexer can scan directly (see iterator_input_adapter::supports_bulk_scan). // Adapters without the flag - file, stream, wide-string, user-defined - fall @@ -8919,6 +9031,12 @@ class lexer : public lexer_base static constexpr bool lazy_token_string = input_adapter_supports_seek(is_detected {}); + /// whether a simulated unget can be passed on to the input adapter, which + /// then leaves the character in the input; see + /// input_adapter_supports_lookahead + static constexpr bool can_release_lookahead = + input_adapter_supports_lookahead(is_detected {}); + /// whether string scanning may bulk-consume runs of ordinary characters /// directly from a contiguous input buffer (SWAR fast path). This requires /// the token to be reconstructible lazily (lazy_token_string), so bypassing @@ -10650,6 +10768,21 @@ scan_number_done: uncapture_char(std::integral_constant {}); } + /// adapter without lookahead: nothing to do (see release_lookahead) + void release_lookahead_impl(std::false_type /*can_release*/) const noexcept {} + + /// adapter with lookahead: leave the character in the input instead + void release_lookahead_impl(std::true_type /*can_release*/) + { + if (next_unget) + { + // the character is read from the input again rather than replayed + // from current, so the adapter must not step over it + next_unget = false; + ia.release_lookahead(); + } + } + /// seekable adapter: nothing was captured, so nothing to undo void uncapture_char(std::true_type /*lazy*/) const noexcept {} @@ -10713,6 +10846,31 @@ scan_number_done: return position; } + /*! + @brief pass a pending simulated unget on to the input + + unget() only rewinds the lexer's own bookkeeping, so the character that + terminated the last token (e.g. the character after a number) would still + be stepped over when the input adapter is done. Callers that hand the + input back to the user afterwards - operator>> and non-strict sax_parse - + call this once when scanning is done, so that the input is positioned + right after the value. + + Adapters without lookahead (see input_adapter_supports_lookahead) are not + handed back to the user, so this is a no-op for them. Without + JSON_PRECISE_STREAM_POSITION, no adapter has lookahead, so this is always a + no-op and the terminating character stays consumed. + + Scanning may continue after this call: @a next_unget is cleared, and the + character is read from the input again instead of being replayed from + @a current. A pending unget of EOF needs no special case, because reaching + EOF leaves no lookahead to release. + */ + void release_lookahead() + { + release_lookahead_impl(std::integral_constant {}); + } + #if JSON_DIAGNOSTIC_POSITIONS /// return the offset of the first character of the last read token; unlike /// the token's parsed value, this accounts for escape sequences @@ -15960,13 +16118,22 @@ class parser json_sax_dom_callback_parser sdp(result, callback, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), - exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), + exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -15988,12 +16155,20 @@ class parser json_sax_dom_parser sdp(result, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // see above + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -16026,12 +16201,24 @@ class parser (void)detail::is_sax_static_asserts {}; const bool result = sax_parse_internal(sax); - // strict mode: next byte must be EOF - if (result && strict && (get_token() != token_type::end_of_input)) + if (result) { - return sax->parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + if (strict) + { + // strict mode: next byte must be EOF + if (get_token() != token_type::end_of_input) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); + } } return result; @@ -25382,6 +25569,31 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #endif + /*! + @brief whether a descent must stop here and finish without the call stack + + @a may_descend says whether the operator descends at all; it is a constant + at every call site, and is passed rather than tested by the caller so that + the test does not become a constant condition there, which MSVC reports as + C4127. + + The comparison operators use this rather than @ref nesting_depth_guard::okay, + because they are written as a macro and a macro cannot use the preprocessor + the way the guard's constructor does; @ref copy_structured, which can, asks + the guard instead and never calls this. + */ + static bool nesting_depth_exhausted(bool may_descend = true) noexcept + { +#ifdef JSON_NO_THREAD_LOCAL + // without a count of its own per thread, a descent cannot be bounded + // without racing another one, so none is made + static_cast(may_descend); + return true; +#else + return !may_descend || nesting_depth() >= nesting_depth_limit(); +#endif + } + /*! @brief counts one level of a bounded descent for as long as it runs, and reports whether the descent was still within the limit when it began @@ -25702,6 +25914,274 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// the result of comparing two values, including values that cannot be + /// ordered at all, such as a discarded value or a NaN + enum class compare_result { less, equal, greater, unordered }; + +#if JSON_HAS_THREE_WAY_COMPARISON + /// @brief the ordering that @a result stands for + static std::partial_ordering to_partial_ordering(compare_result result) noexcept // *NOPAD* + { + switch (result) + { + case compare_result::less: + return std::partial_ordering::less; + case compare_result::greater: + return std::partial_ordering::greater; + case compare_result::equal: + return std::partial_ordering::equivalent; + case compare_result::unordered: + default: + return std::partial_ordering::unordered; + } + } +#endif + + /*! + @brief compare two values that are not both an array or both an object + + Such a pair is compared by the operators themselves, which cannot descend + into it and therefore cannot recurse. + + That holds for a pair whose types differ as much as for a pair of leaves: an + array and an object are told apart by their types alone, because an operator + only ever descends into two values of the same type. So `==` reports them as + unequal without looking inside either, and an ordering falls back to the + order of the types - an object sorts before an array - exactly as it does + for a value that is not nested deeply enough to get here. + */ + template + static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept + { + if (lhs == rhs) + { + return compare_result::equal; + } + + return order_leaves(lhs, rhs, std::integral_constant {}); + } + + /*! + @brief compare two object keys + + An object compares its entries as pairs of a key and a value, so its keys + are compared exactly as std::pair compares them: with < where the objects + are being ordered, and with == where they are only checked for equality. + Note that this is not the object's own comparator, which for a vector-backed + object type such as nlohmann::ordered_map tells equality rather than order. + */ + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::true_type /*ordered*/) + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::equal; + } + + /// @brief check two object keys for equality + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::false_type /*ordered*/) + { + return lhs == rhs ? compare_result::equal : compare_result::unordered; + } + + /// @brief tell apart two values that are not equal + /// @note only instantiated where the values are being ordered, as a key or + /// string type is not required to be ordered to be compared for equality + static compare_result order_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::unordered; + } + + /// @brief report two values as not equal without ordering them + static compare_result order_leaves(const_reference /*lhs*/, const_reference /*rhs*/, std::false_type /*ordered*/) noexcept + { + return compare_result::unordered; + } + + /*! + @brief compare @a lhs and @a rhs without descending into them + + Reached once a comparison has descended @ref nesting_depth_limit levels, so + that comparing values cannot exhaust the call stack however deeply they are + nested. The two values are walked in lockstep on an explicit stack and + compared lexicographically, element by element in the order the containers + enumerate them - which is how the container types this library ships compare + themselves: a std::map enumerates its entries in key order, and + nlohmann::ordered_map in insertion order. An object type that enumerates its + entries in an unspecified order, such as std::unordered_map, compares them + pairwise instead; the difference could only ever show below the bound. + + Note that the stack this walks with is allocated, while the comparison + operators are noexcept and the container comparison this replaces allocated + nothing. Failing that allocation therefore ends the process rather than + throwing. It only arises for values nested past the bound, and only when + memory has run out - where the same comparison used to exhaust the call + stack instead - but it is a way to fail that the operators did not have. + */ + template + static compare_result compare_iteratively(const_reference lhs, const_reference rhs, + const bool unordered_compares_equal) noexcept + { + /// a pair of containers being compared in lockstep + struct frame + { + const basic_json* lhs_value{nullptr}; + const basic_json* rhs_value{nullptr}; + typename array_t::const_iterator lhs_array_it{}; + typename array_t::const_iterator rhs_array_it{}; + typename object_t::const_iterator lhs_object_it{}; + typename object_t::const_iterator rhs_object_it{}; + }; + + std::vector stack; + const basic_json* left = &lhs; + const basic_json* right = &rhs; + + for (;;) + { + const auto type = left->m_data.m_type; + + if (type == right->m_data.m_type && (type == value_t::array || type == value_t::object)) + { + // descend: the elements decide, and are compared further down + stack.emplace_back(); + frame& pushed = stack.back(); + pushed.lhs_value = left; + pushed.rhs_value = right; + + if (type == value_t::array) + { + pushed.lhs_array_it = left->m_data.m_value.array->cbegin(); + pushed.rhs_array_it = right->m_data.m_value.array->cbegin(); + } + else + { + pushed.lhs_object_it = left->m_data.m_value.object->cbegin(); + pushed.rhs_object_it = right->m_data.m_value.object->cbegin(); + } + } + else + { + const compare_result result = compare_leaves(*left, *right); + + // Values that cannot be ordered - a NaN, say - end an ordered + // comparison for std::lexicographical_compare_three_way, but + // std::lexicographical_compare treats them as equivalent and + // carries on with the next element. Both are reproduced here, + // so that a value nested too deeply to descend into compares + // exactly as one that is not. + if (result != compare_result::equal && + !(unordered_compares_equal && result == compare_result::unordered)) + { + return result; + } + } + + // walk back up past the containers that are exhausted, then take the + // next pair of elements from the innermost one that is not + for (;;) + { + if (stack.empty()) + { + return compare_result::equal; + } + + frame& current = stack.back(); + const bool is_object = current.lhs_value->m_data.m_type == value_t::object; + + const bool lhs_done = is_object + ? current.lhs_object_it == current.lhs_value->m_data.m_value.object->cend() + : current.lhs_array_it == current.lhs_value->m_data.m_value.array->cend(); + const bool rhs_done = is_object + ? current.rhs_object_it == current.rhs_value->m_data.m_value.object->cend() + : current.rhs_array_it == current.rhs_value->m_data.m_value.array->cend(); + + if (lhs_done || rhs_done) + { + // whichever ran out first holds the smaller container; if + // both did, they are equal and the container above decides + if (lhs_done != rhs_done) + { + return lhs_done ? compare_result::less : compare_result::greater; + } + + stack.pop_back(); + continue; + } + + if (is_object) + { + // an entry is a key and a value, and the key decides first + const compare_result key_result = + compare_keys(current.lhs_object_it->first, current.rhs_object_it->first, + std::integral_constant {}); + + if (key_result != compare_result::equal) + { + return key_result; + } + + left = &(current.lhs_object_it->second); + right = &(current.rhs_object_it->second); + ++current.lhs_object_it; + ++current.rhs_object_it; + } + else + { + left = &(*current.lhs_array_it); + right = &(*current.rhs_array_it); + ++current.lhs_array_it; + ++current.rhs_array_it; + } + + break; + } + } + } + + + /// @brief restore the parent pointers after erasing from an object + /// ordered_json keeps its members in a vector, and erasing a member + /// re-constructs every member after it in place, which resets their + /// parent pointers + void set_parents_after_object_erase() + { +#if JSON_DIAGNOSTICS +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) + { + set_parents(); + } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif +#endif + } + public: ////////////////////////// // JSON parser callback // @@ -26719,6 +27199,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec ValueType & get_to(ValueType& v) const noexcept(noexcept( JSONSerializer::from_json(std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } @@ -26744,6 +27225,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec noexcept(noexcept(JSONSerializer::from_json( std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } @@ -27390,6 +27872,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -27462,6 +27945,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { result.m_it.object_iterator = m_data.m_value.object->erase(first.m_it.object_iterator, last.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -27492,7 +27976,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - return m_data.m_value.object->erase(std::forward(key)); + const auto erased = m_data.m_value.object->erase(std::forward(key)); + set_parents_after_object_erase(); + return erased; } template < typename KeyType, detail::enable_if_t < @@ -27509,6 +27995,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); + set_parents_after_object_erase(); return 1; } return 0; @@ -28420,16 +28907,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); -#if JSON_DIAGNOSTICS - it2->second.set_parents(); -#endif continue; } } - m_data.m_value.object->operator[](it.key()) = it.value(); -#if JSON_DIAGNOSTICS - m_data.m_value.object->operator[](it.key()).m_parent = this; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + set_parent(m_data.m_value.object->operator[](it.key()) = it.value()); } } @@ -28458,9 +28941,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // a nested object is merged: continue with its parent -#if JSON_DIAGNOSTICS - target->set_parents(); -#endif target = stack.back().target; first = stack.back().position; last = stack.back().last; @@ -28482,10 +28962,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec continue; } } - target->m_data.m_value.object->operator[](first.key()) = first.value(); -#if JSON_DIAGNOSTICS - target->m_data.m_value.object->operator[](first.key()).m_parent = target; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + target->set_parent(target->m_data.m_value.object->operator[](first.key()) = first.value()); ++first; } } @@ -28624,7 +29103,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // because any negative signed value is smaller than any unsigned value. // Otherwise, the non-negative signed value is cast to unsigned before the // comparison to avoid wraparound. -#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result) \ +#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result, deep_result, may_descend) \ const auto lhs_type = lhs.type(); \ const auto rhs_type = rhs.type(); \ \ @@ -28633,11 +29112,25 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec switch (lhs_type) \ { \ case value_t::array: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.array) op (*rhs.m_data.m_value.array); \ - \ + } \ + \ case value_t::object: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.object) op (*rhs.m_data.m_value.object); \ - \ + } \ + \ case value_t::null: \ return (null_result); \ \ @@ -28737,7 +29230,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif const_reference lhs = *this; - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -28762,7 +29256,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_IMPLEMENT_OPERATOR(<=>, // *NOPAD* std::partial_ordering::equivalent, std::partial_ordering::unordered, - lhs_type <=> rhs_type) // *NOPAD* + lhs_type <=> rhs_type, // *NOPAD* + to_partial_ordering(compare_iteratively(lhs, rhs, false)), true) } /// @brief comparison: 3-way @@ -28829,7 +29324,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -28885,7 +29381,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // default_result is used if we cannot compare values. In that case, // we compare types. Note we have to call the operator explicitly, // because MSVC has problems otherwise. - JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type)) + JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type), + compare_iteratively(lhs, rhs, true) == compare_result::less, false) } /// @brief comparison: less than @@ -30748,6 +31245,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS + #undef JSON_PRECISE_STREAM_POSITION #endif // #include diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index 281c05efa..af776d652 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -56,6 +56,10 @@ #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif +#ifndef JSON_PRECISE_STREAM_POSITION + #define JSON_PRECISE_STREAM_POSITION 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -80,21 +84,28 @@ #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS #endif +#if JSON_PRECISE_STREAM_POSITION + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION _psp +#else + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ - NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 7a9bc5471..ea6da107c 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -80,7 +80,10 @@ target_compile_options(test_main PUBLIC # std::allocator_traits<...>::construct for the custom-base-class # test's map type past VS2015's limit. The name is only used for # debug info, so truncation does not affect the build. - $<$:/W4;/wd4566;/wd4996;/wd4702;/wd4503> + # Disable warning C5285: cannot declare a specialization for 'std::tuple'; MSVC 19.51 + # reports the forward declarations of standard library + # templates in the vendored doctest.h + $<$:/W4;/wd4566;/wd4996;/wd4702;/wd4503;/wd5285> # https://github.com/nlohmann/json/issues/1114 $<$:/bigobj> $<$:-Wa,-mbig-obj> diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index d0b4ba54b..8f66dfbf1 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -36,6 +36,10 @@ TEST_CASE("default namespace") expected += "_bics"; #endif +#if JSON_PRECISE_STREAM_POSITION + expected += "_psp"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 789107181..12b7603b9 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -37,6 +37,10 @@ TEST_CASE("default namespace without version component") expected += "_bics"; #endif +#if JSON_PRECISE_STREAM_POSITION + expected += "_psp"; +#endif + expected += "::basic_json"; // fallback for Clang diff --git a/tests/fuzzing.md b/tests/fuzzing.md index cfbf4f249..b3bf90d5d 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -79,3 +79,26 @@ the same `fuzzers` target as above and also relies on the `FUZZER_ENGINE` variab [build script](https://github.com/google/oss-fuzz/blob/master/projects/json/build.sh) for more information. In case the build at OSS-Fuzz fails, an issue will be created automatically. + +### Handling OSS-Fuzz reports + +OSS-Fuzz files the crashes it finds in its own [issue tracker](https://issues.oss-fuzz.com), not on GitHub. So that +each report can be traced to the change that fixed it, and each fix to the report it answers, fixes follow these +conventions: + +- **Reference the OSS-Fuzz issue in the pull request**, next to any GitHub issue it closes, as `OSS-Fuzz: ` (for + example, `OSS-Fuzz: 563659413`), and in the commit message. The ID alone does not disclose the crash. If the report + was triaged into a GitHub issue, link the OSS-Fuzz issue there too. +- **Turn the reproducer into a unit test.** Download the testcase from the OSS-Fuzz report, reduce it if possible, and + add it as a regression test to the unit test of the affected format (e.g., `tests/src/unit-bjdata.cpp`), with a + comment naming the OSS-Fuzz issue. This way the input is checked by every CI run rather than only by OSS-Fuzz, and + it stays covered even if OSS-Fuzz later closes the report as not reproducible. +- **Keep the fuzzer drivers and the unit tests in sync.** The round-trip checks of the UBJSON and BJData drivers are + also run on a fixed corpus in the unit tests (see `tests/src/round_trip_corpus.hpp` and the "round-trip invariants" + test cases), so a regression shows up in CI first. When a driver's checks change, change the unit tests with them. +- **Record in the report whether the bug shipped.** OSS-Fuzz asks whether a crash was a short-lived regression or + affects a released version; answer it when the fix is merged, as it decides whether the fix needs a release note or + a security advisory (see the [security policy](../.github/SECURITY.md)). + +After the fix is merged, OSS-Fuzz re-runs the reproducer on its next build and marks the report as verified and +closed. If it does not, the fix is incomplete. diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index 41c51a311..d3c9e7a33 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -42,6 +42,9 @@ dump() serializes any non-finite double the same deterministic way (as JSON `null`, since JSON itself cannot represent NaN/Infinity), so comparing dumps is stable under exactly the same values that break operator==. +The unit tests run the same checks on a fixed corpus (see the "BJData round-trip +invariants" test case), so keep both in sync. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ diff --git a/tests/src/fuzzer-parse_ubjson.cpp b/tests/src/fuzzer-parse_ubjson.cpp index 20c20eda7..ebf775b59 100644 --- a/tests/src/fuzzer-parse_ubjson.cpp +++ b/tests/src/fuzzer-parse_ubjson.cpp @@ -21,6 +21,9 @@ array data, it performs the following steps: - j4 = from_ubjson(vec3) - assert(j1 == j4) +The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip +invariants" test case), so keep both in sync. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ diff --git a/tests/src/round_trip_corpus.hpp b/tests/src/round_trip_corpus.hpp new file mode 100644 index 000000000..41cdac3e6 --- /dev/null +++ b/tests/src/round_trip_corpus.hpp @@ -0,0 +1,213 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // nan +#include // size_t +#include // int32_t, int64_t, uint32_t, uint64_t +#include // numeric_limits +#include // mt19937 +#include // string, to_string +#include // move +#include // vector + +#include + +// Values for the round-trip property tests of the UBJSON and BJData writers. +// +// The fuzzer drivers (tests/src/fuzzer-parse_ubjson.cpp and +// fuzzer-parse_bjdata.cpp) check that anything the library parses can be +// serialized, parsed back, and serialized again without loss. Those checks +// only run at OSS-Fuzz, so a regression used to surface days later as an +// external report. The unit tests run the same checks on this corpus in CI. +// +// The corpus is deterministic: std::mt19937's output sequence is fixed by +// the standard, and it is used directly rather than through a distribution +// (whose results are implementation-defined). +namespace utils +{ + +class round_trip_corpus +{ + public: + using json = nlohmann::json; + + static std::vector values() + { + round_trip_corpus corpus; + return corpus.build(); + } + + // whether a value contains a binary value, which a BJData or UBJSON round + // trip may turn into an array of integers + static bool contains_binary(const json& j) + { + if (j.is_binary()) + { + return true; + } + if (j.is_structured()) + { + for (const auto& element : j) + { + if (contains_binary(element)) + { + return true; + } + } + } + return false; + } + + private: + std::vector atoms; + // a fixed seed is the point: the corpus must be the same in every run + std::mt19937 generator{42}; // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) + + round_trip_corpus() + : atoms + { + nullptr, true, false, + // integers at the boundaries of every UBJSON/BJData integer type + 0, 1, -1, 127, 128, 255, 256, -128, -129, + 32767, 32768, 65535, 65536, -32768, -32769, + (std::numeric_limits::min)(), (std::numeric_limits::max)(), + (std::numeric_limits::max)(), + (std::numeric_limits::min)(), (std::numeric_limits::max)(), + static_cast((std::numeric_limits::max)()) + 1u, + (std::numeric_limits::max)(), + // floating-point numbers, including non-finite ones + 0.0, -0.0, 1.5, -2.25, 3.4e38, (std::numeric_limits::max)(), + std::nan(""), std::numeric_limits::infinity(), -std::numeric_limits::infinity(), + // strings, including a non-ASCII one and one longer than 255 bytes + "", "a", "\xC3\xA4", std::string(300, 'x'), + // binary values with and without subtype + json::binary({}), json::binary({1, 2, 255}), json::binary({0x80, 0x7F}, 42), json::binary({1}, 0) + } + {} + + std::vector build() + { + std::vector result = atoms; + + // each atom inside containers, including homogeneous ones that the + // writers encode as optimized (typed) containers + result.emplace_back(json::array()); + result.emplace_back(json::object()); + for (const auto& atom : atoms) + { + result.push_back(json::array({atom})); + result.push_back(json::array({atom, atom, atom})); + result.push_back(json::array({json::array({atom})})); + result.push_back(json::object({{"key", atom}})); + } + result.push_back(json::array({1, 1.5})); + result.push_back(json::array({-1, 255})); + result.push_back(json::array({"a", "b"})); + + // deep, but well below any recursion or depth limit + json nested_array = 1; + json nested_object = 1; + for (int i = 0; i < 300; ++i) + { + nested_array = json::array({nested_array}); + nested_object = json::object({{"key", nested_object}}); + } + result.push_back(nested_array); + result.push_back(nested_object); + + add_annotated_arrays(result); + add_random_values(result); + return result; + } + + // objects in the JData annotated array format, which the BJData writer + // encodes as ND-arrays when the annotation describes a packed array, and + // as plain objects otherwise (see #5398, #5399, #5403, #5404, and #5542) + static void add_annotated_arrays(std::vector& result) + { + const std::vector types = + { + "uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", + "single", "double", "char", "byte", "bool", "unknown", 5, nullptr + }; + const std::vector sizes = + { + json::array(), {3}, {1, 3}, {3, 1}, {2, 3}, {2, 0}, {0, 2}, {2, 2, 2}, {-1, 2}, {2, 1.5}, + "3", 3, nullptr, json::binary({}) + }; + const std::vector data = + { + nullptr, 5, "s", json::object({{"a", 1}}), json::array(), + {1, 2, 3}, {1, 2, 3, 4, 5, 6}, {1, 2, 3, 4, 5, 6, 7, 8}, + {1.5, 2.5, 3.5, 4.5, 5.5, 6.5}, {300, -300, 70000, -70000, 1, 2}, + {"a", "b", "c", "d", "e", "f"}, {json::array({1, 2, 3}), json::array({4, 5, 6})} + }; + + for (const auto& type : types) + { + for (const auto& size : sizes) + { + for (const auto& d : data) + { + result.push_back({{"_ArrayType_", type}, {"_ArraySize_", size}, {"_ArrayData_", d}}); + } + } + } + + // incomplete annotations and annotations with an extra key + result.push_back({{"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}, {"extra", 1}}); + } + + // random containers of atoms, both homogeneous and mixed + void add_random_values(std::vector& result) + { + for (int i = 0; i < 1000; ++i) + { + result.push_back(random_value(0)); + } + } + + std::size_t random_below(std::size_t bound) + { + return generator() % bound; + } + + json random_value(int depth) + { + const auto kind = random_below(10); + if (depth > 3 || kind < 5) + { + return atoms[random_below(atoms.size())]; + } + + json result = kind < 8 ? json::array() : json::object(); + const auto count = random_below(5); + const bool homogeneous = random_below(2) == 0; + const json fixed = atoms[random_below(atoms.size())]; + for (std::size_t i = 0; i < count; ++i) + { + json element = homogeneous ? fixed : random_value(depth + 1); + if (result.is_array()) + { + result.push_back(std::move(element)); + } + else + { + result[std::to_string(i)] = std::move(element); + } + } + return result; + } +}; + +} // namespace utils diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 9bba141d2..ebbbbfaf6 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "round_trip_corpus.hpp" #include "test_utils.hpp" namespace @@ -2867,6 +2868,21 @@ TEST_CASE("BJData") const auto out_num = json::to_bjdata(j_num); CHECK(out_num.at(0) == '{'); CHECK(json::from_bjdata(out_num) == j_num); + + // OSS-Fuzz issue 474400817: an empty object _ArraySize_ was + // written as the ND-array header length, which from_bjdata() + // could not read back + const std::vector input = + { + '[', '{', 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'D', 'a', 't', 'a', '_', 'Z', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'T', 'y', 'p', 'e', '_', 'S', 'i', 5, 'i', 'n', 't', '1', '6', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'S', 'i', 'z', 'e', '_', '{', '}', '}', ']' + }; + const json j1 = json::from_bjdata(input); + CHECK(j1 == json::parse(R"([{"_ArrayType_":"int16","_ArraySize_":{},"_ArrayData_":null}])")); + json j2; + CHECK_NOTHROW(j2 = json::from_bjdata(json::to_bjdata(j1, false, false))); + CHECK(j2 == j1); } SECTION("ndarray with out-of-range _ArrayData_ elements stays as object") @@ -4273,6 +4289,93 @@ TEST_CASE("BJData use_type requires use_size") } } +TEST_CASE("BJData round-trip invariants") +{ + // This checks what the parse_bjdata_fuzzer driver checks (see + // tests/src/fuzzer-parse_bjdata.cpp), so that a regression shows up in CI + // rather than as an OSS-Fuzz report: every value from_bjdata() returns + // (j1) can be serialized with any combination of options, the result can + // be parsed back (j2), and serializing j2 again with the same options + // yields a value-equal result. + // + // Beyond the driver, this also checks that j2 equals j1 and that + // serializing j2 reproduces the exact bytes, both except for values that + // contain a binary value: a binary value is only written as a binary + // value with Draft 3's optimized binary array, and otherwise read back as + // an array of integers, for which the writer may choose different (but + // equally valid) type markers when it is serialized again (see #5494). + // + // Values are compared with dump() rather than operator==, because a NaN + // never compares equal to itself. + struct options + { + bool use_size; + bool use_type; + json::bjdata_version_t version; + }; + const std::vector all_options = + { + {false, false, json::bjdata_version_t::draft2}, + {true, false, json::bjdata_version_t::draft2}, + {true, true, json::bjdata_version_t::draft2}, + {false, false, json::bjdata_version_t::draft3}, + {true, false, json::bjdata_version_t::draft3}, + {true, true, json::bjdata_version_t::draft3}, + }; + + for (const auto& j0 : utils::round_trip_corpus::values()) + { + // turn the corpus value into a value as from_bjdata() returns it + for (const auto& initial : all_options) + { + const json j1 = json::from_bjdata(json::to_bjdata(j0, initial.use_size, initial.use_type, initial.version)); + const bool has_binary = utils::round_trip_corpus::contains_binary(j1); + + for (const auto& o : all_options) + { + INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type + << ", draft3 = " << (o.version == json::bjdata_version_t::draft3)); + + const std::vector vec = json::to_bjdata(j1, o.use_size, o.use_type, o.version); + json j2; + // anything the library writes must be parsable by the library + REQUIRE_NOTHROW(j2 = json::from_bjdata(vec)); + const std::vector vec2 = json::to_bjdata(j2, o.use_size, o.use_type, o.version); + CHECK(json::from_bjdata(vec2).dump() == j2.dump()); + + if (!has_binary) + { + CHECK(j2.dump() == j1.dump()); + CHECK(vec2 == vec); + } + } + } + } +} + +TEST_CASE("BJData round trip of a binary value is value-stable, not byte-stable") +{ + // OSS-Fuzz issue 474480402: a Draft 3 optimized binary array is read as a + // binary value, which to_bjdata() writes in the default Draft 2 mode as a + // plain array of uint8 numbers. That is read back as an array of numbers, + // for which the writer then picks the smallest type marker, int8 ('i'), + // so re-serializing changes the bytes, but not the value. This is the + // exception described in the "Round trips" note of the BJData + // documentation, and why the fuzzer checks value stability (see #5494). + const std::vector input = {'[', '$', 'B', '#', 'U', 1, 0x20}; + const json j1 = json::from_bjdata(input); + CHECK(j1 == json::binary({0x20})); + + const std::vector vec = json::to_bjdata(j1, false, false); + CHECK(vec == std::vector({'[', 'U', 0x20, ']'})); + const json j2 = json::from_bjdata(vec); + CHECK(j2 == json::array({0x20})); + + const std::vector vec2 = json::to_bjdata(j2, false, false); + CHECK(vec2 == std::vector({'[', 'i', 0x20, ']'})); + CHECK(json::from_bjdata(vec2) == j2); +} + TEST_CASE("BJData roundtrips" * doctest::skip()) { SECTION("input from self-generated BJData files") diff --git a/tests/src/unit-deserialization.cpp b/tests/src/unit-deserialization.cpp index 0dbfdd15c..28359804b 100644 --- a/tests/src/unit-deserialization.cpp +++ b/tests/src/unit-deserialization.cpp @@ -25,6 +25,7 @@ using nlohmann::json; #include #include #include +#include #include #if defined(_WIN32) @@ -1233,6 +1234,57 @@ TEST_CASE("deserialization") } } + SECTION("stream position after extraction without JSON_PRECISE_STREAM_POSITION (#5340)") + { + // By default, the character that terminates a number is consumed, so + // the stream is left one byte too far after a number (and only after a + // number). JSON_PRECISE_STREAM_POSITION changes this; see + // unit-precise-stream-position.cpp. These checks pin the default. + const auto remaining = [](std::istream & is) + { + return std::string(std::istreambuf_iterator(is), std::istreambuf_iterator()); + }; + + SECTION("the character after a number is consumed") + { + std::istringstream ss("1true"); + json j; + ss >> j; + CHECK(j == 1); + CHECK(remaining(ss) == "rue"); + } + + SECTION("the character after other values is not consumed") + { + std::istringstream ss("[1]true"); + json j; + ss >> j; + CHECK(j == json::parse("[1]")); + CHECK(remaining(ss) == "true"); + } + + SECTION("comma-separated numbers can be read one by one") + { + std::istringstream ss("1,2,3"); + json j1, j2, j3; + ss >> j1 >> j2 >> j3; + CHECK(j1 == 1); + CHECK(j2 == 2); + CHECK(j3 == 3); + } + + SECTION("std::getline after a number skips the line break") + { + std::istringstream ss("42\nfoo"); + json j; + std::string line; + ss >> j; + std::getline(ss, line); + CHECK(j == 42); + CHECK(line == "foo"); + } + } + // build with C++20 // JSON_HAS_CPP_20 #if defined(__cpp_char8_t) diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index cdb6185b3..3ae649e5b 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -361,6 +361,106 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK(p == o); } } + + SECTION("Regression test - erase() and update() must keep JSON_DIAGNOSTICS parent pointers of ordered_json members") + { + // ordered_json keeps its members in a vector: erasing a member + // re-constructs all members after it in place, and adding a key may + // reallocate the vector; both reset the parent pointers of the members + // that were moved + using nlohmann::ordered_json; + + const auto check_parents = [](const ordered_json & j) + { + // const access, so operator[] cannot repair the parent pointers + CHECK_THROWS_WITH_AS(j["z"]["x"].at(0), "[json.exception.type_error.304] (/z/x) cannot use at() with number", ordered_json::type_error); + + // must not trigger assert_invariant() in a debug/assert-enabled build + ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + }; + + // erase(key) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + CHECK(j.erase("a") == 1); + check_parents(j); + } + + // erase(iterator) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + j.erase(j.begin()); + check_parents(j); + } + + // erase(iterator, iterator) + { + ordered_json j = {{"a", 1}, {"b", 2}, {"z", {{"x", 1}}}}; + j.erase(j.begin(), j.find("z")); + check_parents(j); + } + + // patch() removes via erase(iterator) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + j.patch_inplace(ordered_json::parse(R"([{"op": "remove", "path": "/a"}])")); + check_parents(j); + } + + // update(j) + { + ordered_json j = {{"z", {{"x", 1}}}}; + j.update({{"a", 1}, {"b", 2}}); + check_parents(j); + } + + // update(j, true), the outer and the nested vector both grow + { + ordered_json j = {{"z", {{"x", 1}}}}; + j.update({{"z", {{"y", 2}}}, {"a", 1}}, true); + check_parents(j); + } + + // update(j, true) around its descent bound, where the nested vectors + // grow while the objects are merged without recursing + for (const std::size_t depth : + { + nlohmann::detail::recursion_depth_limit() - 1, nlohmann::detail::recursion_depth_limit(), nlohmann::detail::recursion_depth_limit() + 2 + }) + { + ordered_json j = {{"z", {{"x", 1}}}}; + ordered_json patch = {{"a", 1}, {"b", 2}, {"c", {{"d", 3}}}}; + for (std::size_t i = 0; i < depth; ++i) + { + j = ordered_json{{"k", 0}, {"n", std::move(j)}}; + patch = ordered_json{{"n", std::move(patch)}, {"l", 1}, {"m", 2}}; + } + j.update(patch, true); + + // must not trigger assert_invariant() on any level in a + // debug/assert-enabled build + ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + } + + // merge_patch() inserts "c" and removes "d" at /a/c, then inserts "e" + // at /a, which copies /a/c + { + auto j = ordered_json::parse(R"({"a": {"c": {"d": {}}}})"); + j.merge_patch(ordered_json::parse(R"({"a": {"c": {"c": "s", "d": null}, "e": "s"}})")); + CHECK(j.dump() == R"({"a":{"c":{"c":"s"},"e":"s"}})"); + + auto const& constJ = j; +#if JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) (bytes 18-21) cannot use at() with string", ordered_json::type_error); +#else + CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) cannot use at() with string", ordered_json::type_error); +#endif + ordered_json const copy = j; + CHECK(copy == j); + } + } } TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()") diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index f10c8be41..3734966d5 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -157,6 +157,54 @@ TEST_CASE("tests on deeply nested JSONs") CHECK(deep_depth == depth); } + SECTION("comparing") + { + // Comparing used to descend once per level, and an ordered + // comparison used to compare every pair of elements twice, once in + // each direction, which took exponentially long in the nesting + // depth. Both are gone: these finish in milliseconds, where the + // second used to take longer than anyone would wait even for a + // value nested only a few dozen levels deep. + const std::string text = std::string(depth, '[') + '0' + std::string(depth, ']'); + const json j = json::parse(text); + const json same = json::parse(text); + const json larger = json::parse(std::string(depth, '[') + '1' + std::string(depth, ']')); + + CHECK(j == same); + CHECK_FALSE(j == larger); + CHECK(j != larger); + + CHECK(j < larger); + CHECK_FALSE(larger < j); + CHECK(larger > j); + CHECK(j <= same); + CHECK(j >= same); + + // a value that ends earlier is the smaller one + const json shorter = json::parse(std::string(depth - 1, '[') + '0' + std::string(depth - 1, ']')); + CHECK_FALSE(j == shorter); + } + + SECTION("comparing objects") + { + std::string text; + text.reserve((6 * depth) + 1); + for (std::size_t i = 0; i < depth; ++i) + { + text += "{\"a\":"; + } + text += '1'; + text.append(depth, '}'); + + const json j = json::parse(text); + const json same = json::parse(text); + + CHECK(j == same); + CHECK_FALSE(j != same); + CHECK(j <= same); + CHECK(j >= same); + } + SECTION("the copy is independent of the original") { const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); diff --git a/tests/src/unit-precise-stream-position.cpp b/tests/src/unit-precise-stream-position.cpp new file mode 100644 index 000000000..5b6bff682 --- /dev/null +++ b/tests/src/unit-precise-stream-position.cpp @@ -0,0 +1,237 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// This file tests the opt-in JSON_PRECISE_STREAM_POSITION, so it defines the +// macro itself rather than relying on a -D flag, and runs in every build. The +// default behavior is pinned in unit-deserialization.cpp. +#ifdef JSON_PRECISE_STREAM_POSITION + #undef JSON_PRECISE_STREAM_POSITION +#endif + +#define JSON_PRECISE_STREAM_POSITION 1 + +#include +using nlohmann::json; + +#include +#include +#include +#include +#include +#include + +#define STRINGIZE_EX(x) #x +#define STRINGIZE(x) STRINGIZE_EX(x) + +namespace +{ +// A streambuf that keeps no get area at all and refuses every putback: with an +// empty get area, sungetc() always ends up in pbackfail(). Used to check that +// the character terminating a number is left in the input without relying on +// the streambuf being able to put a consumed character back. +class no_putback_streambuf : public std::streambuf +{ + public: + explicit no_putback_streambuf(std::string s) : m_data(std::move(s)) {} + + protected: + // peek at the next character without consuming it + int_type underflow() override + { + if (m_pos >= m_data.size()) + { + return traits_type::eof(); + } + return traits_type::to_int_type(m_data[m_pos]); + } + + // consume the next character + int_type uflow() override + { + if (m_pos >= m_data.size()) + { + return traits_type::eof(); + } + return traits_type::to_int_type(m_data[m_pos++]); + } + + int_type pbackfail(int_type /*c*/) override + { + return traits_type::eof(); + } + + private: + std::string m_data; + std::size_t m_pos = 0; +}; + +// read the characters that are left in a stream +std::string remaining(std::istream& is) +{ + std::string result; + char c = 0; + while (is.get(c)) + { + result += c; + } + return result; +} +} // namespace + +TEST_CASE("JSON_PRECISE_STREAM_POSITION") +{ + SECTION("the macro is part of the ABI tag") + { + const std::string ns = STRINGIZE(NLOHMANN_JSON_NAMESPACE); + // other tags may come before it, e.g. json_abi_diag_psp + CHECK(ns.find("_psp") != std::string::npos); + } + + SECTION("a number does not consume the character that terminates it") + { + // a number is only terminated by the character following it; that + // character must be given back so the stream is positioned right + // after the value + const std::vector> tests = + { + {"1true", "true"}, + {"1[2]", "[2]"}, + {"1{}", "{}"}, + {R"(1"a")", R"("a")"}, + {"1 true", " true"}, + {"12,", ","}, + {"-0.5e3x", "x"}, + {"1null", "null"} + }; + + for (const auto& test : tests) + { + CAPTURE(test.first); + std::istringstream ss(test.first); + json j; + ss >> j; + CHECK(j == json::parse(test.first.substr(0, test.first.size() - test.second.size()))); + CHECK(remaining(ss) == test.second); + } + } + + SECTION("values that are self-delimiting are unaffected") + { + const std::vector> tests = + { + {"truefalse", "false"}, + {"[1][2]", "[2]"}, + {R"({"a":1}{"b":2})", R"({"b":2})"}, + {R"("a""b")", R"("b")"}, + {"null null", " null"} + }; + + for (const auto& test : tests) + { + CAPTURE(test.first); + std::istringstream ss(test.first); + json j; + ss >> j; + CHECK(remaining(ss) == test.second); + } + } + + SECTION("a number at the end of the input leaves nothing behind") + { + for (const std::string s : + {"1", "12", "-3.5e2", " 7 " + }) + { + CAPTURE(s); + std::istringstream ss(s); + json j; + ss >> j; + CHECK(remaining(ss).find_first_not_of(" \t\n\r") == std::string::npos); + } + } + + SECTION("repeated extraction of concatenated values") + { + std::istringstream ss(R"(1true[2]3"x"{"a":4}5)"); + const std::vector expected = + { + json(1), json(true), json::parse("[2]"), json(3), + json("x"), json::parse(R"({"a":4})"), json(5) + }; + + for (const auto& e : expected) + { + json j; + ss >> j; + CHECK(j == e); + } + } + + SECTION("differences to the default behavior") + { + // both of these work by accident without the macro, because the + // character after a number is swallowed; see unit-deserialization.cpp + + SECTION("a separator after a number is not skipped") + { + std::istringstream ss("1,2"); + json j; + ss >> j; + CHECK(j == 1); + CHECK_THROWS_AS(ss >> j, json::parse_error&); + } + + SECTION("std::getline after a number sees the line break") + { + std::istringstream ss("42\nfoo"); + json j; + std::string line; + ss >> j; + std::getline(ss, line); + CHECK(j == 42); + CHECK(line.empty()); + std::getline(ss, line); + CHECK(line == "foo"); + } + } + + SECTION("sax_parse with strict == false") + { + std::istringstream ss("1true"); + json j; + nlohmann::detail::json_sax_dom_parser sdp(j, true); + CHECK(json::sax_parse(ss, &sdp, nlohmann::detail::input_format_t::json, false)); + CHECK(j == 1); + CHECK(remaining(ss) == "true"); + } + + SECTION("strict parsing still rejects trailing data") + { + std::istringstream ss("1true"); + json _; + CHECK_THROWS_WITH_AS(_ = json::parse(ss), + "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - unexpected true literal; expected end of input", json::parse_error&); + + std::istringstream ss2("1true"); + CHECK_FALSE(json::accept(ss2)); + } + + SECTION("a streambuf that cannot put back is not needed") + { + // the terminating character is never consumed, so no putback + // position is required + no_putback_streambuf buf("1true"); + std::istream is(&buf); + json j; + is >> j; + CHECK(j == json(1)); + CHECK(remaining(is) == "true"); + } +} diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index aafbbf5a4..c8458c44d 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -15,6 +15,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "round_trip_corpus.hpp" #include "test_utils.hpp" namespace @@ -2265,7 +2266,9 @@ TEST_CASE("UBJSON optimized arrays of a valueless type are bounded") SECTION("an excessive count is rejected") { - // 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value + // 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value; + // OSS-Fuzz reported this shape as a parse_ubjson_fuzzer timeout + // (testcase 6347769435193344, no issue filed) for (const auto marker : {'Z', 'T', 'F' }) @@ -2817,6 +2820,51 @@ TEST_CASE("UBJSON use_type requires use_size") } } +TEST_CASE("UBJSON round-trip invariants") +{ + // This checks what the parse_ubjson_fuzzer driver checks (see + // tests/src/fuzzer-parse_ubjson.cpp), so that a regression shows up in CI + // rather than as an OSS-Fuzz report: every value from_ubjson() returns + // (j1) can be serialized with any combination of options, the result can + // be parsed back (j2), and serializing j2 again with the same options + // reproduces the exact bytes. Beyond the driver, this also checks that j2 + // equals j1. Values are compared with dump() rather than operator==, + // because a NaN never compares equal to itself. + struct options + { + bool use_size; + bool use_type; + }; + const std::vector all_options = + { + {false, false}, + {true, false}, + {true, true}, + }; + + for (const auto& j0 : utils::round_trip_corpus::values()) + { + // turn the corpus value into a value as from_ubjson() returns it; this + // has no binary values, as UBJSON writes them as arrays of integers + for (const auto& initial : all_options) + { + const json j1 = json::from_ubjson(json::to_ubjson(j0, initial.use_size, initial.use_type)); + + for (const auto& o : all_options) + { + INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type); + + const std::vector vec = json::to_ubjson(j1, o.use_size, o.use_type); + json j2; + // anything the library writes must be parsable by the library + REQUIRE_NOTHROW(j2 = json::from_ubjson(vec)); + CHECK(j2.dump() == j1.dump()); + CHECK(json::to_ubjson(j2, o.use_size, o.use_type) == vec); + } + } + } +} + TEST_CASE("UBJSON roundtrips" * doctest::skip()) { SECTION("input from self-generated UBJSON files") diff --git a/tools/generate_natvis/generate_natvis.py b/tools/generate_natvis/generate_natvis.py index 968690abe..fb1210db1 100755 --- a/tools/generate_natvis/generate_natvis.py +++ b/tools/generate_natvis/generate_natvis.py @@ -20,7 +20,7 @@ if __name__ == '__main__': namespaces = ['nlohmann'] abi_prefix = 'json_abi' - abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics'] + abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics', '_psp'] version = '_v' + args.version.replace('.', '_') inline_namespaces = []