diff --git a/.github/workflows/stale.yml b/.github/workflows/stale.yml index 562286018..1dbf6a571 100644 --- a/.github/workflows/stale.yml +++ b/.github/workflows/stale.yml @@ -20,7 +20,7 @@ jobs: with: egress-policy: audit - - uses: actions/stale@1e223db275d687790206a7acac4d1a11bd6fe629 # v10.4.0 + - uses: actions/stale@4391f3da665fdf50b6810c1a66712fb9ba21aa93 # v11.0.0 with: stale-issue-label: 'state: stale' stale-pr-label: 'state: stale' diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index 86f3cddb7..798885a8c 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -142,7 +142,7 @@ jobs: name: code-coverage-report path: ${{ github.workspace }}/build/html - name: Publish report to Coveralls - uses: coverallsapp/github-action@5cbfd81b66ca5d10c19b062c04de0199c215fb6e # v2.3.7 + uses: coverallsapp/github-action@8d6379e14d29928660c4ba802d8e85393440b329 # v2.3.8 with: github-token: ${{ secrets.GITHUB_TOKEN }} path-to-lcov: ${{ github.workspace }}/build/json.info.filtered.noexcept diff --git a/README.md b/README.md index 31e80e908..d4d2394a2 100644 --- a/README.md +++ b/README.md @@ -1801,13 +1801,13 @@ The library itself consists of a single header file licensed under the MIT licen - [**amalgamate.py - Amalgamate C source and header files**](https://github.com/edlund/amalgamate) to create a single header file - [**American fuzzy lop**](https://lcamtuf.coredump.cx/afl/) for fuzz testing - [**AppVeyor**](https://www.appveyor.com) for [continuous integration](https://ci.appveyor.com/project/nlohmann/json) on Windows -- [**Artistic Style**](http://astyle.sourceforge.net) for automatic source code indentation +- [**Artistic Style**](https://astyle.sourceforge.net) for automatic source code indentation - [**Clang**](https://clang.llvm.org) for compilation with code sanitizers - [**CMake**](https://cmake.org) for build automation - [**Codacy**](https://www.codacy.com) for further [code analysis](https://app.codacy.com/gh/nlohmann/json/dashboard) - [**Coveralls**](https://coveralls.io) to measure [code coverage](https://coveralls.io/github/nlohmann/json) - [**Coverity Scan**](https://scan.coverity.com) for [static analysis](https://scan.coverity.com/projects/nlohmann-json) -- [**cppcheck**](http://cppcheck.sourceforge.net) for static analysis +- [**cppcheck**](https://cppcheck.sourceforge.io) for static analysis - [**doctest**](https://github.com/onqtam/doctest) for the unit tests - [**GitHub Changelog Generator**](https://github.com/skywinder/github-changelog-generator) to generate the [ChangeLog](https://github.com/nlohmann/json/blob/develop/ChangeLog.md) - [**Google Benchmark**](https://github.com/google/benchmark) to implement the benchmarks @@ -1822,6 +1822,15 @@ The library itself consists of a single header file licensed under the MIT licen ## Notes +### Standards compliance + +The library targets strict conformance with [RFC 8259](https://tools.ietf.org/html/rfc8259.html). Both the original [JSONTestSuite](https://github.com/nst/JSONTestSuite) and its updated revision are exercised in CI; their test data is downloaded from [`nlohmann/json_test_data`](https://github.com/nlohmann/json_test_data) at configure time rather than committed to this repository (see [`tests/src/unit-testsuites.cpp`](https://github.com/nlohmann/json/blob/develop/tests/src/unit-testsuites.cpp)): + +- The updated revision runs all mandatory `y_` (must-accept) and `n_` (must-reject) cases through the strict [`parse()`](https://json.nlohmann.me/api/basic_json/parse/) entry point; the original suite runs its `n_` cases through `parse()` and its `y_` cases through [`operator>>`](https://json.nlohmann.me/api/operator_gtgt/). +- The `i_` (implementation-defined) cases are, by RFC 8259, free to be accepted *or* rejected, so "passing all `i_` cases" is not a meaningful conformance metric. The library makes deliberate, documented choices there: nesting depth is not artificially limited, a leading UTF-8 byte order mark is silently ignored, [Unicode noncharacters](https://www.unicode.org/faq/private_use.html#nonchar1) are forwarded unchanged, invalid UTF-8 and lone/unpaired UTF-16 surrogates are rejected (stricter than required), and a number that cannot be stored without becoming `NaN`/`INF` raises [`out_of_range.406`](https://json.nlohmann.me/home/exceptions/#jsonexceptionout_of_range406). + +One behavioral nuance is worth calling out, because a superficial test often misreads it as non-compliance: [`parse()`](https://json.nlohmann.me/api/basic_json/parse/) is strict and rejects trailing data after a value, whereas [`operator>>`](https://json.nlohmann.me/api/operator_gtgt/) follows relaxed iostream semantics — it parses a single value and leaves the stream positioned right after it. Feeding "a valid document followed by trailing bytes" through `operator>>` reports success; the same input through `parse()` is rejected. This is a documented two-API design, not a conformance gap. See [**parsing**](https://json.nlohmann.me/features/parsing/) for details. + ### Character encoding The library supports **Unicode input** as follows: diff --git a/cmake/clang_flags.cmake b/cmake/clang_flags.cmake index aea17820c..0619545ec 100644 --- a/cmake/clang_flags.cmake +++ b/cmake/clang_flags.cmake @@ -5,6 +5,9 @@ # -Wno-extra-semi-stmt The library uses assert which triggers this warning. # -Wno-padded We do not care about padding warnings. # -Wno-covered-switch-default All switches list all cases and a default case. +# -Wno-c2y-extensions Clang 22.1 diagnoses __COUNTER__ as a C2y extension, also in +# C++ mode. The library does not use __COUNTER__; the warnings +# all come from vendored Doctest (SECTION/TEST_CASE macros). # -Wno-unsafe-buffer-usage Pervasive: the library's own low-level numeric/buffer code # (to_chars, serializer, lexer, binary reader/writer, input # adapters, json_pointer) plus vendored Doctest itself (~208 @@ -20,5 +23,6 @@ set(CLANG_CXXFLAGS -Wno-extra-semi-stmt -Wno-padded -Wno-covered-switch-default + -Wno-c2y-extensions -Wno-unsafe-buffer-usage ) diff --git a/docs/mkdocs/docs/api/basic_json/parser_callback_t.md b/docs/mkdocs/docs/api/basic_json/parser_callback_t.md index 460f69cbb..da23e9bc2 100644 --- a/docs/mkdocs/docs/api/basic_json/parser_callback_t.md +++ b/docs/mkdocs/docs/api/basic_json/parser_callback_t.md @@ -29,7 +29,14 @@ Discarding a value (i.e., returning `#!cpp false`) has different effects dependi called: - Discarded values in structured types are skipped. That is, the parser will behave as if the discarded value was never - read. + read. This holds for every value type and for both kinds of parent: a discarded element is removed from the + surrounding array, and a discarded member is removed from the surrounding object together with its key. +- Arrays and objects can be discarded either at their `parse_event_t::array_start`/`parse_event_t::object_start` event + or at their `parse_event_t::array_end`/`parse_event_t::object_end` event, and both remove the whole value. Discarding + it at the start event also means the callback is called neither for the content of the value nor for its matching end + event. +- Discarding a `parse_event_t::key` event discards the whole object member. The callback is still called for the + associated value, but its return value has no further effect. - In case a value outside a structured type is skipped, it is replaced with `null`. This case happens if the top-level element is skipped. @@ -49,7 +56,7 @@ called: ## Return value Whether the JSON value which called the function during parsing should be kept (`#!cpp true`) or not (`#!cpp false`). In -the latter case, it is either skipped completely or replaced by an empty discarded object. +the latter case, it is skipped completely, or replaced by `null` if it is the top-level value. ## Examples @@ -68,6 +75,21 @@ the latter case, it is either skipped completely or replaced by an empty discard --8<-- "examples/parse__string__parser_callback_t.output" ``` +??? example + + The example below shows where discarded values are removed. The array and the number are discarded in different + ways, but in each case the parse result contains neither the value nor its key. + + ```cpp + --8<-- "examples/parser_callback_t.cpp" + ``` + + Output: + + ```json + --8<-- "examples/parser_callback_t.output" + ``` + ## See also - [parse](parse.md) deserialize from a compatible input @@ -76,3 +98,5 @@ the latter case, it is either skipped completely or replaced by an empty discard ## Version history - Added in version 1.0.0. +- Fixed in version 3.13.0 to also remove discarded values from a parent object; before, discarding an array or a value + stored under an object key left a discarded member behind, which made the parse result serialize to invalid JSON. diff --git a/docs/mkdocs/docs/api/ordered_json.md b/docs/mkdocs/docs/api/ordered_json.md index b28fe36f6..feea642cd 100644 --- a/docs/mkdocs/docs/api/ordered_json.md +++ b/docs/mkdocs/docs/api/ordered_json.md @@ -13,6 +13,12 @@ Therefore, adding object elements can yield a reallocation in which case all ite [`end()`](basic_json/end.md) iterator) and all references to the elements are invalidated. Also, any iterator or reference after the insertion point will point to the same index, which is now a different value. +## Complexity + +[`ordered_map`](ordered_map.md) has no lookup index: every key-based object operation is a linear scan, so building or +parsing an object of `n` keys costs O(n²) rather than O(n log n). See +[`ordered_map` complexity](ordered_map.md#complexity) for the per-operation table and for measured numbers. + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/ordered_map.md b/docs/mkdocs/docs/api/ordered_map.md index ca4934161..df21175d0 100644 --- a/docs/mkdocs/docs/api/ordered_map.md +++ b/docs/mkdocs/docs/api/ordered_map.md @@ -56,6 +56,48 @@ std::equal_to<> // since C++14 - **find** - **insert** +## Complexity + +Because the elements are stored in a `std::vector` in insertion order, there is no index to look a key up by. Every +key-based operation performs a **linear scan** over the stored elements. With `n` denoting the number of elements in the +container: + +| Operation | Complexity | Note | +|----------------------------------------|----------------|----------------------------------------------------------| +| **emplace** | O(n) | scans for an existing key, then appends (amortized O(1)) | +| **operator\[\]** | O(n) | delegates to **emplace** (non-const) or **at** (const) | +| **at** | O(n) | throws `#!cpp std::out_of_range` if the key is not found | +| **find** | O(n) | | +| **count** | O(n) | the result is always 0 or 1 | +| **erase(key)** | O(n) | scan, then move the remaining elements one position down | +| **erase(pos)**, **erase(first, last)** | O(n) | moves all elements after the erased range | +| **insert(value)** | O(n) | equivalent to **emplace** | +| **insert(first, last)** | O((n + m) * m) | for `m` inserted elements | + +This differs from `#!cpp std::map`, where the same operations are O(log n). + +!!! warning "Quadratic cost of building large objects" + + Because every insertion scans all elements inserted so far, building an object of `n` distinct keys costs + **O(n²)** in total. This applies to filling an [`ordered_json`](ordered_json.md) object key by key as well as to + parsing one, since the parser inserts each key as it is read. + + The cost is negligible for the object sizes typically found in configuration files or API payloads, but it grows + steeply for machine-generated objects with many thousands of keys. Measured with `-O2 -DNDEBUG` for parsing a flat + object of `n` keys, relative to `#!cpp nlohmann::json` (which uses `#!cpp std::map`): + + | `n` | `json` | `ordered_json` | factor | + |--------|--------|----------------|--------| + | 2000 | 0.7 ms | 3.6 ms | 5× | + | 4000 | 0.8 ms | 14.0 ms | 19× | + | 8000 | 1.6 ms | 67.8 ms | 43× | + | 16 000 | 3.3 ms | 181.6 ms | 54× | + + If key order matters for objects of that size, consider a container with a lookup index, such as + [`tsl::ordered_map`](https://github.com/Tessil/ordered-map) + ([integration](https://github.com/nlohmann/json/issues/546#issuecomment-304447518)), as the object type -- see + [object order](../features/object_order.md). + ## Examples ??? example diff --git a/docs/mkdocs/docs/examples/parser_callback_t.cpp b/docs/mkdocs/docs/examples/parser_callback_t.cpp new file mode 100644 index 000000000..45794cca6 --- /dev/null +++ b/docs/mkdocs/docs/examples/parser_callback_t.cpp @@ -0,0 +1,47 @@ +#include +#include + +using json = nlohmann::json; + +int main() +{ + // a JSON text with an array and a number inside an object + auto text = R"({"IDs": [116, 943], "Width": 800})"; + + // discard the array when the parser reads its opening bracket + json j_array_start = json::parse(text, [](int /*depth*/, json::parse_event_t event, json & /*parsed*/) + { + return event != json::parse_event_t::array_start; + }); + + // discard the same array when the parser reads its closing bracket + json j_array_end = json::parse(text, [](int /*depth*/, json::parse_event_t event, json & /*parsed*/) + { + return event != json::parse_event_t::array_end; + }); + + // discard the number, but keep its key + json j_value = json::parse(text, [](int /*depth*/, json::parse_event_t event, json & parsed) + { + return !(event == json::parse_event_t::value && parsed == json(800)); + }); + + // discard the key of the number + json j_key = json::parse(text, [](int /*depth*/, json::parse_event_t event, json & parsed) + { + return !(event == json::parse_event_t::key && parsed == json("Width")); + }); + + // discard the top-level object + json j_root = json::parse(text, [](int /*depth*/, json::parse_event_t event, json & /*parsed*/) + { + return event != json::parse_event_t::object_end; + }); + + // in every case, the discarded value is removed together with its key + std::cout << j_array_start << '\n' + << j_array_end << '\n' + << j_value << '\n' + << j_key << '\n' + << j_root << '\n'; +} diff --git a/docs/mkdocs/docs/examples/parser_callback_t.output b/docs/mkdocs/docs/examples/parser_callback_t.output new file mode 100644 index 000000000..00c19a939 --- /dev/null +++ b/docs/mkdocs/docs/examples/parser_callback_t.output @@ -0,0 +1,5 @@ +{"Width":800} +{"Width":800} +{"IDs":[116,943]} +{"IDs":[116,943]} +null diff --git a/docs/mkdocs/docs/features/conversions.md b/docs/mkdocs/docs/features/conversions.md index be85f5db6..765c610ab 100644 --- a/docs/mkdocs/docs/features/conversions.md +++ b/docs/mkdocs/docs/features/conversions.md @@ -66,8 +66,14 @@ auto t = j.get>(); // {1.0, "hello", 42} std::get<1>(refs) = "world"; // modifies j[1] in place ``` - A referenced type must be one the library actually stores (or an arithmetic type it can convert to/from); - otherwise this is a compile error. + A referenced element must name the type the library actually *stores* — one of [`boolean_t`](../api/basic_json/boolean_t.md), + [`number_integer_t`](../api/basic_json/number_integer_t.md), [`number_unsigned_t`](../api/basic_json/number_unsigned_t.md), + [`number_float_t`](../api/basic_json/number_float_t.md), [`string_t`](../api/basic_json/string_t.md), + [`binary_t`](../api/basic_json/binary_t.md), [`array_t`](../api/basic_json/array_t.md), or + [`object_t`](../api/basic_json/object_t.md). There is nothing else to refer to, so a reference to any other type is a + compile error even when a conversion would exist: `#!cpp std::tuple` is rejected, because the library stores a + `#!cpp number_integer_t` (`#!cpp std::int64_t` by default) and not an `#!cpp int`. This restriction applies only to + reference elements — a plain `#!cpp std::tuple` converts by value as usual. ## Implicit conversions @@ -116,17 +122,34 @@ which forces the explicit `get` form and can catch unintended conversions at com with a custom `adl_serializer>` specialization. Prefer `get>()`/`get_to()` over `static_cast` for optional types. -!!! warning "Converting to a fixed-size `std::array` does not check length" +!!! warning "Converting to a fixed-size destination does not check the array size" - Converting a JSON array to `#!cpp std::array` does not check that the JSON array's size matches `N`: - if the JSON array is longer, the extra elements are silently dropped; if it is shorter, the remaining - `std::array` elements are left default-constructed. No exception is thrown in either case. + Some destination types have a size that is fixed by their C++ type rather than by the JSON value: + `#!cpp std::pair`, `#!cpp std::tuple`, `#!cpp std::array`, C arrays `#!cpp T[N]`, and + `#!cpp std::map`/`#!cpp std::unordered_map` with a non-string key type (which is read from an array of + two-element arrays). All of them read exactly as many elements as they need via + [`at`](../api/basic_json/at.md) and **never compare the JSON array's size to that number**. The two + mismatch directions therefore behave differently: + + - The JSON array has **too many** elements: the surplus is **silently discarded**, and no exception is + thrown. + - The JSON array has **too few** elements: `at` throws + [`out_of_range.401`](../home/exceptions.md#jsonexceptionout_of_range401) for the first missing index -- + an out-of-range error, not a [`type_error`](../home/exceptions.md#type-errors), even though the cause + is a shape mismatch. ```cpp json j = {1, 2, 3, 4, 5}; - auto a = j.get>(); // {1, 2, 3} -- elements 4 and 5 silently dropped + + auto a = j.get>(); // {1, 2, 3} -- elements 4 and 5 silently dropped + auto p = j.get>(); // (1, 2) -- elements 3, 4, and 5 silently dropped + + json k = {1}; + auto q = k.get>(); // ❌ throws out_of_range.401 ``` + If a size mismatch is an error in your application, check the size yourself before converting. + ## Omitting a field when serializing `std::optional` By default, `to_json` for `std::optional` writes either the value or `#!json null` -- there is no built-in way diff --git a/docs/mkdocs/docs/features/object_order.md b/docs/mkdocs/docs/features/object_order.md index ac5eb7beb..f62474efd 100644 --- a/docs/mkdocs/docs/features/object_order.md +++ b/docs/mkdocs/docs/features/object_order.md @@ -53,6 +53,12 @@ If you do want to preserve the **insertion order**, you can use the type [`nlohm Alternatively, you can use a more sophisticated ordered map like [`tsl::ordered_map`](https://github.com/Tessil/ordered-map) ([integration](https://github.com/nlohmann/json/issues/546#issuecomment-304447518)) or [`nlohmann::fifo_map`](https://github.com/nlohmann/fifo_map) ([integration](https://github.com/nlohmann/json/issues/485#issuecomment-333652309)). +The [`ordered_map`](../api/ordered_map.md) behind `nlohmann::ordered_json` is deliberately minimal and has no lookup +index, so every key access is a linear scan and building an object of `n` keys costs O(n²). This is unnoticeable at +typical object sizes but becomes significant for objects with many thousands of keys; see +[`ordered_map` complexity](../api/ordered_map.md#complexity). The alternatives above keep a lookup index and do not +have this cost. + ### Notes on parsing Note that you also need to call the right [`parse`](../api/basic_json/parse.md) function when reading from a file. diff --git a/docs/mkdocs/docs/features/parsing/index.md b/docs/mkdocs/docs/features/parsing/index.md index c3cb88f63..17624b8a2 100644 --- a/docs/mkdocs/docs/features/parsing/index.md +++ b/docs/mkdocs/docs/features/parsing/index.md @@ -28,6 +28,22 @@ Inputs consisting of multiple values separated by newlines are handled by the [J By default, the library rejects comments and trailing commas. Both can be enabled with parameters of the `parse` function — see [comments](../comments.md) and [trailing commas](../trailing_commas.md). +## Strictness and trailing data + +[`parse`](../../api/basic_json/parse.md) reads a single JSON value and requires the whole input to be consumed: any +non-whitespace data after the value is reported as a parse error. Use it when you want to guarantee that an input is +exactly one complete JSON document. + +[`operator>>`](../../api/operator_gtgt.md) follows relaxed `#!cpp std::istream` semantics instead: it parses one JSON +value and leaves the stream positioned right after it, without requiring the rest of the stream to be consumed. This is +what makes it possible to read several concatenated values from the same stream, but it also means that "a valid +document followed by trailing bytes" is accepted rather than rejected. If you are validating conformance, or need to +reject any input that is not exactly one JSON document, prefer `parse`. + +When using `operator>>` to read several concatenated values this way, a value that is a number must be followed by +whitespace, because `operator>>` consumes the character that terminates a number — see the +[`operator>>` notes](../../api/operator_gtgt.md#notes) for details and examples. + ## SAX vs. DOM parsing The library offers two parsing models: diff --git a/docs/mkdocs/docs/features/parsing/json_lines.md b/docs/mkdocs/docs/features/parsing/json_lines.md index cda6f37c7..fb1481819 100644 --- a/docs/mkdocs/docs/features/parsing/json_lines.md +++ b/docs/mkdocs/docs/features/parsing/json_lines.md @@ -49,4 +49,5 @@ JSON Lines input with more than one value is treated as invalid JSON by the [`pa with a JSON Lines input does not work, because the parser will try to parse one value after the last one. This is different from parsing a stream of *concatenated* (non-newline-delimited) JSON values, for which - `operator>>` does work -- see its [notes](../../api/operator_gtgt.md#notes) for details. + `operator>>` does work, provided that a value that is a number is followed by whitespace -- see its + [notes](../../api/operator_gtgt.md#notes) for details. diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 235744009..f75872505 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -291,9 +291,10 @@ A JSON Pointer array index must be a number. ### json.exception.parse_error.110 -When parsing CBOR or MessagePack, the byte vector ends before the complete value has been read. +When parsing a [binary format](../features/binary_formats/index.md), the byte vector ends before the complete value has +been read. -!!! failure "Example message" +!!! failure "Example messages" ``` [json.exception.parse_error.110] parse error at byte 5: syntax error while parsing CBOR string: unexpected end of input @@ -301,6 +302,9 @@ When parsing CBOR or MessagePack, the byte vector ends before the complete value ``` [json.exception.parse_error.110] parse error at byte 2: syntax error while parsing UBJSON value: expected end of input; last byte: 0x5A ``` + ``` + [json.exception.parse_error.110] parse error at byte 8: syntax error while parsing BSON number: unexpected end of input + ``` ### json.exception.parse_error.112 @@ -329,10 +333,14 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde ``` [json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow ``` + ``` + [json.exception.parse_error.112] parse error at byte 5: syntax error while parsing BSON document: document size 6 does not match the number of bytes read (5) + ``` ### json.exception.parse_error.113 -While parsing a map key, a value that is not a string has been read. +A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a +string was read where one was required (for instance as a map key), or the string's length specification is invalid. !!! failure "Example messages" @@ -345,6 +353,9 @@ While parsing a map key, a value that is not a string has been read. ``` [json.exception.parse_error.113] parse error at byte 2: syntax error while parsing UBJSON char: byte after 'C' must be in range 0x00..0x7F; last byte: 0x82 ``` + ``` + [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData string: string length must not be negative + ``` ### json.exception.parse_error.114 @@ -853,13 +864,21 @@ and this exception no longer occurs. ### json.exception.out_of_range.408 -The size (following `#`) of an UBJSON array or object exceeds the maximal capacity. +The size of an array or object in a [binary format](../features/binary_formats/index.md) exceeds the maximal capacity: +the size following `#` for [UBJSON](../features/binary_formats/ubjson.md)/[BJData](../features/binary_formats/bjdata.md), +or the encoded length for [CBOR](../features/binary_formats/cbor.md). -!!! failure "Example message" +!!! failure "Example messages" ``` excessive array size: 8658170730974374167 ``` + ``` + [json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive array size + ``` + ``` + [json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive map size + ``` ### json.exception.out_of_range.409 diff --git a/include/nlohmann/detail/input/json_sax.hpp b/include/nlohmann/detail/input/json_sax.hpp index ea4132232..4c5f24a7b 100644 --- a/include/nlohmann/detail/input/json_sax.hpp +++ b/include/nlohmann/detail/input/json_sax.hpp @@ -626,14 +626,7 @@ class json_sax_dom_callback_parser if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_structured()) { // remove discarded value - for (auto it = ref_stack.back()->begin(); it != ref_stack.back()->end(); ++it) - { - if (it->is_discarded()) - { - ref_stack.back()->erase(it); - break; - } - } + remove_discarded_value(*ref_stack.back()); } return true; @@ -674,8 +667,9 @@ class json_sax_dom_callback_parser bool end_array() { bool keep = true; + const bool stored = ref_stack.back() != nullptr; - if (ref_stack.back()) + if (stored) { keep = callback(static_cast(ref_stack.size()) - 1, parse_event_t::array_end, *ref_stack.back()); if (keep) @@ -709,9 +703,19 @@ class json_sax_dom_callback_parser keep_stack.pop_back(); // remove discarded value - if (!keep && !ref_stack.empty() && ref_stack.back()->is_array()) + if (!ref_stack.empty() && ref_stack.back()) { - ref_stack.back()->m_data.m_value.array->pop_back(); + if (!keep && ref_stack.back()->is_array()) + { + ref_stack.back()->m_data.m_value.array->pop_back(); + } + else if ((!keep || !stored) && ref_stack.back()->is_object()) + { + // the array is either still stored under its key or was never + // stored, leaving the placeholder key() wrote; both show up as + // a discarded member of the parent object + remove_discarded_value(*ref_stack.back()); + } } return true; @@ -801,6 +805,19 @@ class json_sax_dom_callback_parser } #endif + /// remove the discarded value the callback rejected from its parent + static void remove_discarded_value(BasicJsonType& parent) + { + for (auto it = parent.begin(); it != parent.end(); ++it) + { + if (it->is_discarded()) + { + parent.erase(it); + break; + } + } + } + /*! @param[in] v value to add to the JSON value we build during parsing @param[in] skip_callback whether we should skip calling the callback @@ -841,6 +858,18 @@ class json_sax_dom_callback_parser // do not handle this value if we just learnt it shall be discarded if (!keep) { + // if the value was to become an object member, key() already + // stored a placeholder for it that has to be removed again + if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_object()) + { + JSON_ASSERT(!key_keep_stack.empty()); + const bool placeholder_stored = key_keep_stack.back(); + key_keep_stack.pop_back(); + if (placeholder_stored) + { + remove_discarded_value(*ref_stack.back()); + } + } return {false, nullptr}; } diff --git a/include/nlohmann/detail/meta/type_traits.hpp b/include/nlohmann/detail/meta/type_traits.hpp index 7036a4ef0..ebf6a2c26 100644 --- a/include/nlohmann/detail/meta/type_traits.hpp +++ b/include/nlohmann/detail/meta/type_traits.hpp @@ -231,7 +231,9 @@ struct char_traits : std::char_traits // Redefine to_int_type function static int_type to_int_type(char_type c) noexcept { - return static_cast(c); + // cast via unsigned char: sign-extending a negative char_type would make + // byte 0xFF indistinguishable from eof() + return static_cast(static_cast(c)); } static char_type to_char_type(int_type i) noexcept @@ -699,21 +701,35 @@ struct is_json_pointer_of> : std::true_type {}; template struct is_json_pointer_of&> : std::true_type {}; -// checks if A and B are comparable using Compare functor +// checks if A and B are comparable using Compare functor, assuming that +// neither A nor B is a json_pointer type (that case is handled by +// is_comparable below, which never instantiates this helper otherwise) template -struct is_comparable : std::false_type {}; +struct is_comparable_no_json_pointer : std::false_type {}; -// We exclude json_pointer here, because the checks using Compare(A, B) will -// use json_pointer::operator string_t() which triggers a deprecation warning -// for GCC. See https://github.com/nlohmann/json/issues/4621. The call to -// is_json_pointer_of can be removed once the deprecated function has been -// removed. template -struct is_comparable < Compare, A, B, enable_if_t < !is_json_pointer_of::value -&& std::is_constructible ()(std::declval(), std::declval()))>::value +struct is_comparable_no_json_pointer < Compare, A, B, enable_if_t < +std::is_constructible ()(std::declval(), std::declval()))>::value && std::is_constructible ()(std::declval(), std::declval()))>::value >> : std::true_type {}; +// checks if A and B are comparable using Compare functor +// We dispatch on is_json_pointer_of as a plain bool (rather than folding it +// into a single enable_if_t condition together with the checks below) so +// that the Compare(A, B) checks are only ever written - and thus only ever +// instantiated - when A/B are not a json_pointer/string pair. Those checks +// use json_pointer::operator string_t() (GCC, see #4621) resp. the +// deprecated json_pointer/string operator== (Clang, see #5288), and merely +// naming them as later operands of a plain && chain is not sufficient to +// avoid their instantiation on all compilers, even when the first operand +// is false. The dispatch on is_json_pointer_of can be removed once the +// deprecated json_pointer comparison operators have been removed. +template::value> +struct is_comparable : std::false_type {}; + +template +struct is_comparable : is_comparable_no_json_pointer {}; + template using detect_is_transparent = typename T::is_transparent; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index ae7b1c18c..883b7ad78 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -3994,7 +3994,9 @@ struct char_traits : std::char_traits // Redefine to_int_type function static int_type to_int_type(char_type c) noexcept { - return static_cast(c); + // cast via unsigned char: sign-extending a negative char_type would make + // byte 0xFF indistinguishable from eof() + return static_cast(static_cast(c)); } static char_type to_char_type(int_type i) noexcept @@ -4462,21 +4464,35 @@ struct is_json_pointer_of> : std::true_type {}; template struct is_json_pointer_of&> : std::true_type {}; -// checks if A and B are comparable using Compare functor +// checks if A and B are comparable using Compare functor, assuming that +// neither A nor B is a json_pointer type (that case is handled by +// is_comparable below, which never instantiates this helper otherwise) template -struct is_comparable : std::false_type {}; +struct is_comparable_no_json_pointer : std::false_type {}; -// We exclude json_pointer here, because the checks using Compare(A, B) will -// use json_pointer::operator string_t() which triggers a deprecation warning -// for GCC. See https://github.com/nlohmann/json/issues/4621. The call to -// is_json_pointer_of can be removed once the deprecated function has been -// removed. template -struct is_comparable < Compare, A, B, enable_if_t < !is_json_pointer_of::value -&& std::is_constructible ()(std::declval(), std::declval()))>::value +struct is_comparable_no_json_pointer < Compare, A, B, enable_if_t < +std::is_constructible ()(std::declval(), std::declval()))>::value && std::is_constructible ()(std::declval(), std::declval()))>::value >> : std::true_type {}; +// checks if A and B are comparable using Compare functor +// We dispatch on is_json_pointer_of as a plain bool (rather than folding it +// into a single enable_if_t condition together with the checks below) so +// that the Compare(A, B) checks are only ever written - and thus only ever +// instantiated - when A/B are not a json_pointer/string pair. Those checks +// use json_pointer::operator string_t() (GCC, see #4621) resp. the +// deprecated json_pointer/string operator== (Clang, see #5288), and merely +// naming them as later operands of a plain && chain is not sufficient to +// avoid their instantiation on all compilers, even when the first operand +// is false. The dispatch on is_json_pointer_of can be removed once the +// deprecated json_pointer comparison operators have been removed. +template::value> +struct is_comparable : std::false_type {}; + +template +struct is_comparable : is_comparable_no_json_pointer {}; + template using detect_is_transparent = typename T::is_transparent; @@ -10125,14 +10141,7 @@ class json_sax_dom_callback_parser if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_structured()) { // remove discarded value - for (auto it = ref_stack.back()->begin(); it != ref_stack.back()->end(); ++it) - { - if (it->is_discarded()) - { - ref_stack.back()->erase(it); - break; - } - } + remove_discarded_value(*ref_stack.back()); } return true; @@ -10173,8 +10182,9 @@ class json_sax_dom_callback_parser bool end_array() { bool keep = true; + const bool stored = ref_stack.back() != nullptr; - if (ref_stack.back()) + if (stored) { keep = callback(static_cast(ref_stack.size()) - 1, parse_event_t::array_end, *ref_stack.back()); if (keep) @@ -10208,9 +10218,19 @@ class json_sax_dom_callback_parser keep_stack.pop_back(); // remove discarded value - if (!keep && !ref_stack.empty() && ref_stack.back()->is_array()) + if (!ref_stack.empty() && ref_stack.back()) { - ref_stack.back()->m_data.m_value.array->pop_back(); + if (!keep && ref_stack.back()->is_array()) + { + ref_stack.back()->m_data.m_value.array->pop_back(); + } + else if ((!keep || !stored) && ref_stack.back()->is_object()) + { + // the array is either still stored under its key or was never + // stored, leaving the placeholder key() wrote; both show up as + // a discarded member of the parent object + remove_discarded_value(*ref_stack.back()); + } } return true; @@ -10300,6 +10320,19 @@ class json_sax_dom_callback_parser } #endif + /// remove the discarded value the callback rejected from its parent + static void remove_discarded_value(BasicJsonType& parent) + { + for (auto it = parent.begin(); it != parent.end(); ++it) + { + if (it->is_discarded()) + { + parent.erase(it); + break; + } + } + } + /*! @param[in] v value to add to the JSON value we build during parsing @param[in] skip_callback whether we should skip calling the callback @@ -10340,6 +10373,18 @@ class json_sax_dom_callback_parser // do not handle this value if we just learnt it shall be discarded if (!keep) { + // if the value was to become an object member, key() already + // stored a placeholder for it that has to be removed again + if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_object()) + { + JSON_ASSERT(!key_keep_stack.empty()); + const bool placeholder_stored = key_keep_stack.back(); + key_keep_stack.pop_back(); + if (placeholder_stored) + { + remove_discarded_value(*ref_stack.back()); + } + } return {false, nullptr}; } diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index f4c28b267..8b3ea660e 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -1440,6 +1440,13 @@ TEST_CASE("parser class") ] )"; + const auto* structured_object = R"( + { + "foo": [1, 2], + "bar": 3 + } + )"; + SECTION("filter nothing") { const json j_object = json::parse(s_object, [](int /*unused*/, json::parse_event_t /*unused*/, const json& /*unused*/) noexcept @@ -1515,6 +1522,48 @@ TEST_CASE("parser class") CHECK (j_filtered2 == json({1})); } + SECTION("filter array in object") + { + // the array is discarded once it is already stored under its key + const json j_filtered1 = json::parse(structured_object, [](int /*unused*/, json::parse_event_t e, const json& /*parsed*/) noexcept + { + return e != json::parse_event_t::array_end; + }); + + CHECK (j_filtered1 == json({{"bar", 3}})); + + // the array is discarded before it is stored, leaving the + // placeholder the key event wrote + const json j_filtered2 = json::parse(structured_object, [](int /*unused*/, json::parse_event_t e, const json& /*parsed*/) noexcept + { + return e != json::parse_event_t::array_start; + }); + + CHECK (j_filtered2 == json({{"bar", 3}})); + } + + SECTION("filter value in object") + { + // the value is discarded after its key was kept, leaving the + // placeholder the key event wrote + const json j_filtered1 = json::parse(structured_object, [](int /*unused*/, json::parse_event_t e, const json & parsed) noexcept + { + return !(e == json::parse_event_t::value && parsed == json(3)); + }); + + CHECK (j_filtered1 == json({{"foo", {1, 2}}})); + + // the same value is discarded together with its key, so no + // placeholder was stored for it + const json j_filtered2 = json::parse(structured_object, [](int /*unused*/, json::parse_event_t e, const json & parsed) noexcept + { + return !((e == json::parse_event_t::key && parsed == json("bar")) || + (e == json::parse_event_t::value && parsed == json(3))); + }); + + CHECK (j_filtered2 == json({{"foo", {1, 2}}})); + } + SECTION("filter specific events") { SECTION("first closing event") diff --git a/tests/src/unit-deserialization.cpp b/tests/src/unit-deserialization.cpp index 08c440cbc..5c737e872 100644 --- a/tests/src/unit-deserialization.cpp +++ b/tests/src/unit-deserialization.cpp @@ -484,6 +484,30 @@ TEST_CASE("deserialization") CHECK(l.events == std::vector({"boolean(true)"})); } + SECTION("from std::vector") + { + std::vector const v = {'t', 'r', 'u', 'e'}; + CHECK(json::parse(v) == json(true)); + CHECK(json::accept(v)); + + SaxEventLogger l; + CHECK(json::sax_parse(v, &l)); + CHECK(l.events.size() == 1); + CHECK(l.events == std::vector({"boolean(true)"})); + + // bytes outside ASCII are negative here and must not be sign-extended; + // 0xC3 and 0xA9 do not fit in signed char (MSVC C4309), so spell them as negative values + std::vector const umlaut = {'"', static_cast(0xC3 - 0x100), static_cast(0xA9 - 0x100), '"'}; + CHECK(json::parse(umlaut) == json("\xC3\xA9")); + CHECK(json::accept(umlaut)); + + // 0xFF (spelled as -1 to stay in range) must not be reported as end of input + std::vector const trailing = {'t', 'r', 'u', 'e', static_cast(0xFF - 0x100)}; + json _; + CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'true\xFF'; expected end of input", json::parse_error&); + CHECK(!json::accept(trailing)); + } + SECTION("from std::array") { std::array const v { {'t', 'r', 'u', 'e'} }; diff --git a/tests/src/unit-type_traits.cpp b/tests/src/unit-type_traits.cpp index d083293dd..6dc166d03 100644 --- a/tests/src/unit-type_traits.cpp +++ b/tests/src/unit-type_traits.cpp @@ -53,4 +53,34 @@ TEST_CASE("type traits") // NOLINTEND(hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-avoid-c-arrays) } } + + SECTION("char_traits") + { + SECTION("to_int_type does not sign-extend") + { + using unsigned_traits = nlohmann::detail::char_traits; + using signed_traits = nlohmann::detail::char_traits; + + CHECK(unsigned_traits::to_int_type(static_cast(0x7F)) == 0x7F); + CHECK(unsigned_traits::to_int_type(static_cast(0x80)) == 0x80); + CHECK(unsigned_traits::to_int_type(static_cast(0xFF)) == 0xFF); + + CHECK(signed_traits::to_int_type(static_cast(0x7F)) == 0x7F); + // 0x80 and 0xFF do not fit in signed char (MSVC C4309), so spell them as negative values + CHECK(signed_traits::to_int_type(static_cast(0x80 - 0x100)) == 0x80); + CHECK(signed_traits::to_int_type(static_cast(0xFF - 0x100)) == 0xFF); + } + + SECTION("no byte value collides with eof") + { + using unsigned_traits = nlohmann::detail::char_traits; + using signed_traits = nlohmann::detail::char_traits; + + for (int i = 0; i < 256; ++i) + { + CHECK(unsigned_traits::to_int_type(static_cast(i)) != unsigned_traits::eof()); + CHECK(signed_traits::to_int_type(static_cast(i)) != signed_traits::eof()); + } + } + } }