diff --git a/.gitignore b/.gitignore index 474b62347..9475ed596 100644 --- a/.gitignore +++ b/.gitignore @@ -47,3 +47,6 @@ nlohmann_json.spdx # Bazel-related MODULE.bazel.lock + +# GCC module cache +/gcm.cache/ diff --git a/README.md b/README.md index 754e18583..a1491b0ac 100644 --- a/README.md +++ b/README.md @@ -1402,7 +1402,7 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I - The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2008-2009 [Björn Hoehrmann](https://bjoern.hoehrmann.de/) - The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/) -- The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) +- The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, including the conversion of the digits to text by Xiang JunBo and the SIMD instruction sequence of Dougall Johnson, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) - The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/). - The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0). - The class contains an adapted version of the Eisel-Lemire algorithm, its table of powers of five, and its digit comparison for long numbers from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors diff --git a/docs/docset/docSet.sql b/docs/docset/docSet.sql index 008b36a43..d00ae8a76 100644 --- a/docs/docset/docSet.sql +++ b/docs/docset/docSet.sql @@ -184,7 +184,6 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::items', 'Met INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::materialize', 'Method', 'api/basic_json_view/materialize/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::number_format', 'Enum', 'api/basic_json_view/number_format/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::number_token', 'Method', 'api/basic_json_view/number_token/index.html'); -INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator bool', 'Method', 'api/basic_json_view/operator_bool/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator<<', 'Operator', 'api/basic_json_view/operator_ltlt/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator[]', 'Operator', 'api/basic_json_view/operator[]/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator==', 'Operator', 'api/basic_json_view/operator_eq/index.html'); diff --git a/docs/mkdocs/docs/api/basic_json/number_float_t.md b/docs/mkdocs/docs/api/basic_json/number_float_t.md index aa41f33f0..1cde41960 100644 --- a/docs/mkdocs/docs/api/basic_json/number_float_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_float_t.md @@ -23,10 +23,10 @@ type to use. ## Template parameters `NumberFloatType` -: the type to store floating-point numbers. The parser converts `#!cpp float`, `#!cpp double`, and a - `#!cpp long double` that is IEEE 754 binary64 itself and other `#!cpp long double` formats with - `#!cpp std::from_chars` or `#!cpp std::strtold`, and serialization falls back to `#!cpp std::snprintf`, so the - type must be `#!cpp float`, `#!cpp double`, or `#!cpp long double`. The +: the type to store floating-point numbers. The type must be `#!cpp float`, `#!cpp double`, or + `#!cpp long double`. The parser converts `#!cpp float`, `#!cpp double`, and a `#!cpp long double` that is IEEE 754 + binary64 itself. It converts other `#!cpp long double` formats with `#!cpp std::from_chars` where available, or + with `#!cpp std::strtold` otherwise. Serialization falls back to `#!cpp std::snprintf`. The [binary formats](../../features/binary_formats/index.md) additionally require `#!cpp float` or `#!cpp double`, because they have no encoding for `#!cpp long double`. See [Template Parameter Requirements](../../features/types/template_parameters.md#numberfloattype). diff --git a/docs/mkdocs/docs/api/basic_json_document/accept.md b/docs/mkdocs/docs/api/basic_json_document/accept.md index 5aa1fc93d..322350130 100644 --- a/docs/mkdocs/docs/api/basic_json_document/accept.md +++ b/docs/mkdocs/docs/api/basic_json_document/accept.md @@ -43,6 +43,11 @@ input's own copy (for inputs that are always read into a buffer) throws. Linear in the length of the input. +## Notes + +An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp accept(ptr, len)`, does not +compile; see [`parse`](parse.md#notes). + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json_document/parse.md b/docs/mkdocs/docs/api/basic_json_document/parse.md index a3c44f4e0..972c63752 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse.md @@ -19,24 +19,33 @@ static basic_json_document parse(IteratorType first, IteratorType last, 1. Deserialize from a compatible input, borrowing or owning it depending on its value category and type (see Notes). 2. Deserialize from a pair of input iterators. -Both overloads accept exactly what [`BasicJsonType::parse()`](../basic_json/parse.md) accepts, with the same +Both overloads accept the same JSON text as [`BasicJsonType::parse()`](../basic_json/parse.md), with the same `ignore_comments`/`ignore_trailing_commas` options, but build a [`basic_json_document`](index.md) (a flat index into -the input) instead of a tree of `BasicJsonType` values. +the input) instead of a tree of `BasicJsonType` values. The input must be byte-oriented (see the template parameters +below): not every input type of `BasicJsonType::parse()` is supported. ## Template parameters `InputType` -: A compatible input, for instance: +: A byte-oriented input, one of: - - a `#!cpp std::string`, `#!cpp std::string_view`, or a C-style array of characters - - a pointer to a null-terminated string of single byte characters + - a `#!cpp std::string`, `#!cpp std::string_view`, or a C-style array of single-byte characters + - a pointer to a null-terminated string of single-byte characters (`#!cpp char`, `#!cpp signed char`, + `#!cpp unsigned char`, `#!cpp std::uint8_t`) - a container for which `#!cpp obj.data()` and `#!cpp obj.size()` give contiguous single-byte access, e.g. `#!cpp std::vector` or `#!cpp std::vector` - - an `#!cpp std::istream` object, or anything else [`BasicJsonType::parse()`](../basic_json/parse.md) accepts + - an `#!cpp std::istream` object + - a wide string object (`#!cpp std::wstring`, `#!cpp std::u16string`, `#!cpp std::u32string`), which is converted + to UTF-8 + + Other inputs are not supported: a `#!cpp FILE*`, and pointers to or arrays of wide characters (`#!cpp wchar_t`, + `#!cpp char16_t`, `#!cpp char32_t`) are rejected at compile time by a `#!cpp static_assert`. (Use + [`BasicJsonType::parse()`](../basic_json/parse.md) for these.) `IteratorType` -: a compatible iterator type, for instance a pair of pointers such as `ptr` and `ptr + len`, or a pair of - `#!cpp std::string::iterator` +: an input iterator type, for instance a pair of pointers such as `ptr` and `ptr + len`, or a pair of + `#!cpp std::string::iterator`; the iterators of single-byte characters are borrowed or read like the byte inputs + above, those of wide characters are converted to UTF-8 ## Parameters @@ -70,8 +79,8 @@ discarded; see [`is_discarded`](is_discarded.md). Throws the same exception [`BasicJsonType::parse()`](../basic_json/parse.md) throws for the same input and options -- the same exception id, message, and position -- because on a failing input the library's own parser is run on the same bytes to produce the diagnostic. Additionally throws -[`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the input is 4 GiB or larger, a size -[`BasicJsonType::parse()`](../basic_json/parse.md) does not reject. +[`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the input is 4294967280 bytes (4 GiB +minus 16 bytes) or larger, a size [`BasicJsonType::parse()`](../basic_json/parse.md) does not reject. ## Complexity @@ -84,8 +93,8 @@ Linear in the length of the input. | `input` | ownership | |--------------------------------------------------------------------------------------|--------------------------------------------------------------| | lvalue byte container (`std::string`, `std::vector`, ...), `std::string_view`, C string, character array | **borrowed** -- `input` must outlive the document | -| rvalue `#!cpp std::string` | **owned**, moved in without a copy | -| rvalue byte container other than `#!cpp std::string` | **owned**, copied | +| non-const rvalue `#!cpp std::string` | **owned**, moved in without a copy | +| other rvalue byte container (including a `#!cpp const` rvalue `#!cpp std::string`) | **owned**, copied | | stream, wide string, or anything else read through the general input adapter | **owned**, read into a buffer (a stream is read to its end) | For overload (2), a pair of pointers to single-byte integers (e.g. `#!cpp const char*`, `#!cpp std::uint8_t*`) is @@ -100,6 +109,12 @@ See [`owns_source`](owns_source.md) to check which happened after a call, and th **Numbers.** As for [`BasicJsonType::parse()`](../basic_json/parse.md), an integer literal too large for the 64-bit integer type becomes a floating-point value. +**No lengths.** An integer argument that is not a `#!cpp bool` where the flags are expected -- for example +`#!cpp parse(ptr, len)` -- does not compile (the overload is deleted). Such a call would convert `len` to +`allow_exceptions` and read `ptr` as a null-terminated string, past the end of a buffer that has none. To parse a +buffer of a given length, pass a pair of pointers: `#!cpp parse(ptr, ptr + len)`. The same holds for +[`parse_copy`](parse_copy.md), [`accept`](accept.md), and [`read`](read.md). + ## Examples ??? example "Example: (1) borrowed vs. owned input, and errors identical to `BasicJsonType::parse()`" diff --git a/docs/mkdocs/docs/api/basic_json_document/parse_copy.md b/docs/mkdocs/docs/api/basic_json_document/parse_copy.md index ed4714d22..7cb51ad02 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse_copy.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse_copy.md @@ -51,6 +51,9 @@ Linear in the length of the input. only differs in that the input is always copied rather than sometimes borrowed. Prefer [`parse()`](parse.md) when the input's lifetime already covers the document's, since it avoids the copy for borrowed inputs. +An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp parse_copy(ptr, len)`, does not +compile; see [`parse`](parse.md#notes). + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json_document/read.md b/docs/mkdocs/docs/api/basic_json_document/read.md index 2b8e0fc48..f68774085 100644 --- a/docs/mkdocs/docs/api/basic_json_document/read.md +++ b/docs/mkdocs/docs/api/basic_json_document/read.md @@ -54,6 +54,9 @@ page at a time, and every page costs a page fault the first time it is written. a 55 MB document into a reused document took about 40 % less time than parsing it into a fresh one. Programs that parse many documents of similar size should therefore keep one document and call `read()`. +An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp read(ptr, len)`, does not +compile; see [`parse`](parse.md#notes). + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json_document/root.md b/docs/mkdocs/docs/api/basic_json_document/root.md index 85682dbcd..3ccedeb4a 100644 --- a/docs/mkdocs/docs/api/basic_json_document/root.md +++ b/docs/mkdocs/docs/api/basic_json_document/root.md @@ -1,10 +1,15 @@ # nlohmann::basic_json_document::root ```cpp -view_type root() const noexcept; +// (1) +view_type root() const& noexcept; + +// (2) +view_type root() const&& = delete; ``` -Returns a view of the root value of the document. +1. Returns a view of the root value of the document. +2. Deleted: the view of a temporary document would dangle. ## Return value @@ -21,6 +26,16 @@ Constant. ## Notes +**Lifetime.** A view refers into the document, so the document must outlive it. `root()` can therefore only be called +on a document that has a name (an lvalue); calling it on a temporary does not compile: + +```cpp +auto v = json_document::parse(text).root(); // error: the document is destroyed at the end of the statement + +auto doc = json_document::parse(text); // OK: keep the document alive +auto v = doc.root(); +``` + `root()` is a cheap handle into the document's index, not a copy of anything; call it as often as needed. The returned view is valid under the same conditions as any other view of the document -- see [Object inspection](../basic_json_view/index.md) -- in particular, it is invalidated by the next diff --git a/docs/mkdocs/docs/api/basic_json_view/at.md b/docs/mkdocs/docs/api/basic_json_view/at.md index 589e42b7b..e8bd8f253 100644 --- a/docs/mkdocs/docs/api/basic_json_view/at.md +++ b/docs/mkdocs/docs/api/basic_json_view/at.md @@ -8,15 +8,18 @@ basic_json_view at(const string_t& key) const; // (2) basic_json_view at(size_type idx) const; -basic_json_view at(int idx) const; +template +basic_json_view at(IntegerType idx) const; // (3) basic_json_view at(const json_pointer& ptr) const; ``` -1. Returns the value of the object member with key `key` -- the first one, should the key occur more than once (see +1. Returns the value of the object member with key `key` -- the last one, should the key occur more than once (see [Notes on duplicate keys](operator[].md#notes)). -2. Returns the array element at index `idx`. +2. Returns the array element at index `idx`. The template accepts every integer type except `#!cpp bool` and + `#!cpp std::size_t` and forwards to the `size_type` overload, as for [`operator[]`](operator[].md); a negative + `idx` is out of range. 3. Returns the value a JSON pointer `ptr` refers to, starting at this value. ## Parameters @@ -32,7 +35,7 @@ basic_json_view at(const json_pointer& ptr) const; ## Return value -1. the value of the first member with key `key` +1. the value of the last member with key `key` 2. the element at index `idx` 3. the value `ptr` resolves to, starting at this value @@ -72,8 +75,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics ## Complexity 1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after - another, in document order, stopping at the first match. Each comparison first checks the key's length -- - already known from the index, without reading the key bytes -- before comparing its content. + another, in document order, scanning all of them, since the last match is wanted. Each comparison first checks the + key's length -- already known from the index, without reading the key bytes -- before comparing its content. Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on average. 2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the diff --git a/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md b/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md index 031c3b29d..df1e939fa 100644 --- a/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md +++ b/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md @@ -4,8 +4,8 @@ basic_json_view() noexcept = default; ``` -Creates an invalid (discarded) view: [`type()`](type.md) is `#!cpp value_t::discarded`, -[`is_discarded()`](is_discarded.md) is `#!cpp true`, and `#!cpp explicit operator bool()` is `#!cpp false`. +Creates an invalid (discarded) view: [`type()`](type.md) is `#!cpp value_t::discarded` and +[`is_discarded()`](is_discarded.md) is `#!cpp true`. This is the only constructor a caller can use directly. Every other view is obtained from a [`basic_json_document`](../basic_json_document/index.md), via [`root()`](../basic_json_document/root.md) or by @@ -44,7 +44,6 @@ placeholder for "no value yet" and later be assigned a real view. ## See also - [is_discarded](is_discarded.md) - return whether the view is invalid -- [operator bool](operator_bool.md) - return whether the view refers to a value - [root](../basic_json_document/root.md) - the view of a document's root value ## Version history diff --git a/docs/mkdocs/docs/api/basic_json_view/begin.md b/docs/mkdocs/docs/api/basic_json_view/begin.md index ae2665ee9..f9c43802f 100644 --- a/docs/mkdocs/docs/api/basic_json_view/begin.md +++ b/docs/mkdocs/docs/api/basic_json_view/begin.md @@ -24,7 +24,7 @@ Constant. For an object, iteration visits **every** member, including all occurrences of a duplicate key -- unlike [`operator[]`](operator[].md), [`at`](at.md), [`find`](find.md), [`contains`](contains.md), and [`count`](count.md), -which all resolve to the *first* member with a given key. See the +which all resolve to the *last* member with a given key. See the [Notes on duplicate keys](operator[].md#notes) of `operator[]`. Because objects are iterated in document order rather than sorted by key, the order seen here can differ from what diff --git a/docs/mkdocs/docs/api/basic_json_view/contains.md b/docs/mkdocs/docs/api/basic_json_view/contains.md index a48395854..f44021f71 100644 --- a/docs/mkdocs/docs/api/basic_json_view/contains.md +++ b/docs/mkdocs/docs/api/basic_json_view/contains.md @@ -33,8 +33,8 @@ No-throw guarantee: this function never throws exceptions. ## Complexity 1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after - another, in document order, stopping at the first match. Each comparison first checks the key's length -- already - known from the index, without reading the key bytes -- before comparing its content. + another, in document order, scanning all of them, since the last match is wanted. Each comparison first checks the + key's length -- already known from the index, without reading the key bytes -- before comparing its content. Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on average. 2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at diff --git a/docs/mkdocs/docs/api/basic_json_view/count.md b/docs/mkdocs/docs/api/basic_json_view/count.md index 9cdfe3dee..a20f3aaee 100644 --- a/docs/mkdocs/docs/api/basic_json_view/count.md +++ b/docs/mkdocs/docs/api/basic_json_view/count.md @@ -23,9 +23,9 @@ No-throw guarantee: this function never throws exceptions. ## Complexity -Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after -another, in document order, stopping at the first match. Each comparison first checks the key's length -- already -known from the index, without reading the key bytes -- before comparing its content. +Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, +in document order, scanning all of them, since the last match is wanted. Each comparison first checks the key's length +-- already known from the index, without reading the key bytes -- before comparing its content. Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on average. @@ -38,7 +38,7 @@ Unlike [`BasicJsonType::count()`](../basic_json/count.md), whose return value ca an `ObjectType` that allows multiple entries per key, `count()` here never does: it is exactly [`contains()`](contains.md) as `#!cpp 0`/`#!cpp 1`. This holds even if the source text has a duplicate key -- see the [Notes on duplicate keys](operator[].md#notes) of `operator[]` -- because a `#!cpp count() > 1` result would require -counting every member with a matching key, not just finding the first one. +counting every member with a matching key (the lookup functions resolve to the *last* one). ## Examples diff --git a/docs/mkdocs/docs/api/basic_json_view/find.md b/docs/mkdocs/docs/api/basic_json_view/find.md index 04ee0476b..fb0561c10 100644 --- a/docs/mkdocs/docs/api/basic_json_view/find.md +++ b/docs/mkdocs/docs/api/basic_json_view/find.md @@ -6,7 +6,7 @@ iterator find(const char* key) const; iterator find(const string_t& key) const; ``` -Finds a member with key `key` -- the first one, should the key occur more than once (see +Finds a member with key `key` -- the last one, should the key occur more than once (see [Notes on duplicate keys](operator[].md#notes)). If the value is not an object, or no member has this key, [`end()`](end.md) is returned. @@ -25,9 +25,9 @@ No-throw guarantee: this function never throws exceptions. ## Complexity -Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after -another, in document order, stopping at the first match. Each comparison first checks the key's length -- already -known from the index, without reading the key bytes -- before comparing its content. +Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, +in document order, scanning all of them, since the last match is wanted. Each comparison first checks the key's length +-- already known from the index, without reading the key bytes -- before comparing its content. Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on average. diff --git a/docs/mkdocs/docs/api/basic_json_view/get.md b/docs/mkdocs/docs/api/basic_json_view/get.md index d8ec333c9..c3bb85790 100644 --- a/docs/mkdocs/docs/api/basic_json_view/get.md +++ b/docs/mkdocs/docs/api/basic_json_view/get.md @@ -87,9 +87,9 @@ exception thrown while converting through `materialize()` (the last bullet) is d !!! info "Duplicate keys" `#!cpp std::map`/`#!cpp std::unordered_map` conversions keep the *last* value of a repeated key, like - [`materialize()`](materialize.md) and [`BasicJsonType::parse()`](../basic_json/parse.md) do. This is the opposite - of [`operator[]`](operator[].md)/[`at`](at.md)/[`find`](find.md)/[`contains`](contains.md), which resolve to the - *first* occurrence (see the [Notes on duplicate keys](operator[].md#notes)). + [`materialize()`](materialize.md) and [`BasicJsonType::parse()`](../basic_json/parse.md) do. This is the member + [`operator[]`](operator[].md)/[`at`](at.md)/[`find`](find.md)/[`contains`](contains.md) resolve to, too (see the + [Notes on duplicate keys](operator[].md#notes)). !!! info "No pointers, references, or implicit conversion" diff --git a/docs/mkdocs/docs/api/basic_json_view/index.md b/docs/mkdocs/docs/api/basic_json_view/index.md index 98f8ce9eb..a81c6d161 100644 --- a/docs/mkdocs/docs/api/basic_json_view/index.md +++ b/docs/mkdocs/docs/api/basic_json_view/index.md @@ -85,7 +85,6 @@ still refers to it -- including ones taken before the change -- reads the new va - [**is_primitive**](is_primitive.md) - return whether the type is primitive - [**is_structured**](is_structured.md) - return whether the type is structured - [**is_discarded**](is_discarded.md) - return whether the view is invalid -- [**operator bool**](operator_bool.md) - return whether the view refers to a value ### Element access diff --git a/docs/mkdocs/docs/api/basic_json_view/is_discarded.md b/docs/mkdocs/docs/api/basic_json_view/is_discarded.md index 4cc11077d..4bd87a3e3 100644 --- a/docs/mkdocs/docs/api/basic_json_view/is_discarded.md +++ b/docs/mkdocs/docs/api/basic_json_view/is_discarded.md @@ -9,6 +9,10 @@ view (see [(constructor)](basic_json_view.md)), and for [`root()`](../basic_json that is itself [discarded](../basic_json_document/is_discarded.md) -- in particular, the root of a failed [`parse()`](../basic_json_document/parse.md) with `allow_exceptions` set to `#!cpp false`. +A discarded view is also what [`operator[]`](operator[].md) returns for a missing key, an index out of range, or a +JSON pointer that cannot be resolved, and for any access on a view that is itself discarded (so a chain such as +`#!cpp v["a"]["b"]` is safe). [`at`](at.md) throws instead. + ## Return value `#!cpp true` if the view is discarded, `#!cpp false` otherwise. @@ -23,8 +27,9 @@ Constant. ## Notes -`#!cpp v.is_discarded()` and `#!cpp !static_cast(v)` are equivalent; use whichever reads better at the call -site. +A `basic_json_view` is not convertible to `#!cpp bool`: such a conversion would mean "refers to a value", whereas +`basic_json` converts to the `#!cpp bool` it holds, so the same code would silently behave differently. Test +`#!cpp !v.is_discarded()` explicitly. ## Examples @@ -45,7 +50,7 @@ site. ## See also -- [operator bool](operator_bool.md) - return whether the view refers to a value +- [operator[]](operator[].md) - access specified element; yields a discarded view where an element is missing - [(constructor)](basic_json_view.md) - the default constructor creates a discarded view - [is_discarded (basic_json_document)](../basic_json_document/is_discarded.md) - return whether the last parse failed - [`BasicJsonType::is_discarded`](../basic_json/is_discarded.md) - the corresponding function of `basic_json` diff --git a/docs/mkdocs/docs/api/basic_json_view/items.md b/docs/mkdocs/docs/api/basic_json_view/items.md index d54f8d723..8e13852da 100644 --- a/docs/mkdocs/docs/api/basic_json_view/items.md +++ b/docs/mkdocs/docs/api/basic_json_view/items.md @@ -48,7 +48,7 @@ Constant. As for [`begin()`](begin.md)/[`end()`](end.md), `items()` visits **every** member of an object, including all occurrences of a duplicate key -- unlike [`operator[]`](operator[].md), [`at`](at.md), [`find`](find.md), -[`contains`](contains.md), and [`count`](count.md), which resolve to the *first* member with a given key. See the +[`contains`](contains.md), and [`count`](count.md), which resolve to the *last* member with a given key. See the [Notes on duplicate keys](operator[].md#notes) of `operator[]`. !!! danger "Lifetime issues" @@ -63,8 +63,8 @@ occurrences of a duplicate key -- unlike [`operator[]`](operator[].md), [`at`](a The example below shows a settings object whose source text records every update to a key as a duplicate member, in the order they happened. `items()` walks all of them, so the update history is visible, while - [`operator[]`](operator[].md) only ever sees the *first* one and [`materialize()`](materialize.md) -- like - [`BasicJsonType::parse()`](../basic_json/parse.md) -- keeps only the *last*. + [`operator[]`](operator[].md) sees the *last* one, and so does [`materialize()`](materialize.md) -- like + [`BasicJsonType::parse()`](../basic_json/parse.md). ```cpp --8<-- "examples/basic_json_view__items.cpp" diff --git a/docs/mkdocs/docs/api/basic_json_view/operator[].md b/docs/mkdocs/docs/api/basic_json_view/operator[].md index 835800a0a..4337d95eb 100644 --- a/docs/mkdocs/docs/api/basic_json_view/operator[].md +++ b/docs/mkdocs/docs/api/basic_json_view/operator[].md @@ -8,16 +8,19 @@ basic_json_view operator[](const string_t& key) const; // (2) basic_json_view operator[](size_type idx) const; -basic_json_view operator[](int idx) const; +template +basic_json_view operator[](IntegerType idx) const; // (3) basic_json_view operator[](const json_pointer& ptr) const; ``` -1. Returns the value of the object member with key `key` -- the first one, should the key occur more than once (see +1. Returns the value of the object member with key `key` -- the last one, should the key occur more than once (see the [Notes](#notes) below) -- or a [discarded](is_discarded.md) view if there is no such member. -2. Returns the array element at index `idx`, or a [discarded](is_discarded.md) view if `idx` is out of range. (The - `#!cpp int` overload only exists so that an integer literal is not ambiguous between this overload and 1.) +2. Returns the array element at index `idx`, or a [discarded](is_discarded.md) view if `idx` is out of range. The + template accepts every integer type except `#!cpp bool` and `#!cpp std::size_t` (`#!cpp int`, `#!cpp unsigned`, + `#!cpp long`, `#!cpp std::int64_t`, ...) and forwards to the `size_type` overload, so that an integer argument is + not ambiguous between that overload and 1; a negative `idx` is out of range. 3. Returns the value a JSON pointer `ptr` refers to, starting at this value, or a [discarded](is_discarded.md) view wherever resolving it further is not possible without inserting into or extending the document (see [Return value](#return-value) and [Exceptions](#exceptions) below). @@ -35,10 +38,12 @@ basic_json_view operator[](const json_pointer& ptr) const; ## Return value -1. the value of the first member with key `key`, or a discarded view if `#!cpp is_object()` is `#!cpp false` or no - member has this key -2. the element at index `idx`, or a discarded view if `#!cpp is_array()` is `#!cpp false` or `#!cpp idx >= size()` -3. the value `ptr` resolves to, starting at this value, or a discarded view for exactly the reference tokens where the +1. the value of the last member with key `key`, or a discarded view if no member has this key (or if this view is + [discarded](is_discarded.md)) +2. the element at index `idx`, or a discarded view if `#!cpp idx >= size()` or `idx` is negative (or if this view is + [discarded](is_discarded.md)) +3. the value `ptr` resolves to, starting at this value, or a discarded view (also if this view is + [discarded](is_discarded.md)) for exactly the reference tokens where the **const** overload of [`BasicJsonType::operator[]`](../basic_json/operator%5B%5D.md) invokes undefined behavior for the same pointer and the same document: an object member that does not exist, or an array index that is out of range @@ -49,10 +54,12 @@ Strong exception safety: if an exception is thrown, there are no changes to the ## Exceptions -1. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an object -- +1. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an object and + not [discarded](is_discarded.md) -- the same exception, with the same message, that the **const** overload of [`BasicJsonType::operator[]`](../basic_json/operator%5B%5D.md) throws for a string argument on a non-object value. -2. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an array -- +2. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an array and + not [discarded](is_discarded.md) -- the same exception, with the same message, that the **const** overload of [`BasicJsonType::operator[]`](../basic_json/operator%5B%5D.md) throws for a numeric argument on a non-array value. 3. Throws the same exceptions, with the same messages, that the **const** overload of @@ -73,9 +80,9 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics ## Complexity 1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after - another, in document order, stopping at the first match. Each comparison first checks the key's length -- - already known from the index, without reading the key bytes -- before comparing its content, so a key of a - different length than `key` is rejected without touching the source text. + another, in document order, scanning all of them, since the last match is wanted. Each comparison first checks the + key's length -- already known from the index, without reading the key bytes -- before comparing its content, so a + key of a different length than `key` is rejected without touching the source text. Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on average. 2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the @@ -86,20 +93,30 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics ## Notes Unlike `BasicJsonType::operator[]`, which is undefined behavior (guarded by a -[runtime assertion](../../features/assertions.md)) for a missing key on a **const** value, this operator always -returns a safe, testable result: a [discarded](is_discarded.md) view, which is `#!cpp false` in a boolean context. +[runtime assertion](../../features/assertions.md)) for a missing key on a **const** value, this operator returns a +safe, testable result for a missing key or an index out of range: a [discarded](is_discarded.md) view, which is +tested with [`is_discarded`](is_discarded.md). There is also no non-const overload that inserts a missing key or extends an array -- a view never modifies the document. +!!! info "Chained access" + + `#!cpp operator[]` on a [discarded](is_discarded.md) view returns a discarded view and does not throw, so a chain + like `#!cpp v["a"]["b"][0]` is safe even if `"a"` or `"b"` is missing: the first missing step makes the whole + result discarded, which is tested once at the end. Type errors on values that are *not* discarded still throw: a + key on an array or a primitive, or an index on an object or a primitive, is `type_error.305` as for + `BasicJsonType`. [`at`](at.md) still throws for a discarded view, as it does for a missing key. + !!! info "Duplicate keys" If the source text has an object with a duplicate key, `#!cpp operator[]` (and [`at`](at.md), [`find`](find.md), - [`contains`](contains.md), [`count`](count.md)) all resolve to the *first* member with that key, because a - lookup can stop as soon as it finds a match. This is different from - [`materialize()`](materialize.md) (and [`BasicJsonType::parse()`](../basic_json/parse.md)), which replay every - member in order and so end up keeping the *last* value for a repeated key -- there is no reason for them to stop - early. [`begin()`](begin.md)/[`end()`](end.md) and [`items()`](items.md) iterate over *all* members, including - duplicates, in document order. See the example below and [`size()`](size.md#notes). + [`contains`](contains.md), [`count`](count.md), [`value`](value.md), and JSON pointer resolution) all resolve to + the *last* member with that key. This is the member [`materialize()`](materialize.md) (and + [`BasicJsonType::parse()`](../basic_json/parse.md)) keeps, so a lookup in the view and in the materialized value + agree. [`begin()`](begin.md)/[`end()`](end.md) and [`items()`](items.md) iterate over *all* members, including + duplicates, in document order. A lookup in an object without a hash index scans all members for this: it cannot + stop at the first match. The hash index of a larger object (128 members or more) leads to the last member of a key + as well. See the example below and [`size()`](size.md#notes). !!! info "JSON pointer resolution" diff --git a/docs/mkdocs/docs/api/basic_json_view/operator_bool.md b/docs/mkdocs/docs/api/basic_json_view/operator_bool.md deleted file mode 100644 index bc13c100c..000000000 --- a/docs/mkdocs/docs/api/basic_json_view/operator_bool.md +++ /dev/null @@ -1,46 +0,0 @@ -# nlohmann::basic_json_view::operator bool - -```cpp -explicit operator bool() const noexcept; -``` - -Returns whether this view refers to a value, i.e. the negation of [`is_discarded()`](is_discarded.md). Being -`#!cpp explicit`, this conversion is only considered in a boolean context (`#!cpp if (v)`, `#!cpp !v`, `#!cpp v && -...`), not for implicit conversions to other types. - -## Return value - -`#!cpp true` if the view refers to a value, `#!cpp false` if it is [discarded](is_discarded.md). - -## Exception safety - -No-throw guarantee: this function never throws exceptions. - -## Complexity - -Constant. - -## Examples - -??? example - - The example below classifies several parsed documents by the type of their root value, without materializing any - of them into a `BasicJsonType` value. - - ```cpp - --8<-- "examples/basic_json_view__type_predicates.cpp" - ``` - - Output: - - ```json - --8<-- "examples/basic_json_view__type_predicates.output" - ``` - -## See also - -- [is_discarded](is_discarded.md) - return whether the view is invalid - -## Version history - -- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/value.md b/docs/mkdocs/docs/api/basic_json_view/value.md index 06d5372e5..7ec353cd5 100644 --- a/docs/mkdocs/docs/api/basic_json_view/value.md +++ b/docs/mkdocs/docs/api/basic_json_view/value.md @@ -12,7 +12,7 @@ T value(const json_pointer& ptr, const T& default_value) const; string_t value(const json_pointer& ptr, const char* default_value) const; ``` -1. Returns the value of the object member with key `key` -- the first one, should the key occur more than once (see +1. Returns the value of the object member with key `key` -- the last one, should the key occur more than once (see [Notes on duplicate keys](operator[].md#notes)) -- converted to `T`, or `default_value` if there is no such member. 2. Returns the value a JSON pointer `ptr` refers to, starting at this value, converted to `T`, or `default_value` if `ptr` cannot be resolved. @@ -39,7 +39,7 @@ equivalent) deduce `string_t`, not `const char*`, for their return type and for ## Return value -1. the first member with key `key`, converted to `T`, or `default_value` +1. the last member with key `key`, converted to `T`, or `default_value` 2. the value `ptr` resolves to, converted to `T`, or `default_value` ## Exception safety @@ -68,8 +68,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics ## Complexity 1. Linear in the number of members: as for [`operator[]`](operator[].md#complexity), members are compared one after - another, in document order, stopping at the first match. Plus the complexity of converting the found member to - `T` (see [`get`](get.md)). + another, in document order, scanning all of them, since the last match is wanted. Plus the complexity of converting + the found member to `T` (see [`get`](get.md)). Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on average. 2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at diff --git a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp index af08f22e0..3d813717d 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp @@ -8,10 +8,10 @@ int main() // the default constructor is the only public one: it creates an invalid // (discarded) view, useful as a "no value yet" placeholder nlohmann::json_view v; - std::cout << static_cast(v) << ' ' << v.is_discarded() << '\n'; + std::cout << v.is_discarded() << '\n'; // views are trivially copyable handles (two pointers); the document owns // the actual data nlohmann::json_view copy = v; - std::cout << static_cast(copy) << '\n'; + std::cout << copy.is_discarded() << '\n'; } diff --git a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output index 8192d6145..bb101b641 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output +++ b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output @@ -1,2 +1,2 @@ -false true -false +true +true diff --git a/docs/mkdocs/docs/examples/basic_json_view__items.cpp b/docs/mkdocs/docs/examples/basic_json_view__items.cpp index 3f7779c19..6357c7e4b 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__items.cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__items.cpp @@ -7,8 +7,8 @@ int main() { // a settings object whose source text records every update to a key as // a duplicate member. items() visits all of them, in document order, so - // the update history is visible; operator[] only ever sees the first - // one, and materialize() -- like basic_json::parse() -- keeps the last + // the update history is visible; operator[] and materialize() -- like + // basic_json::parse() -- see the last one json_document updates = json_document::parse(R"({"retries": 1, "timeout": 30, "retries": 5})"); const auto settings = updates.root(); @@ -17,6 +17,6 @@ int main() std::cout << item.key() << '=' << item.value().materialize().dump() << '\n'; } - std::cout << "first \"retries\" seen by operator[]: " << settings["retries"].materialize().dump() << '\n'; + std::cout << "last \"retries\" seen by operator[]: " << settings["retries"].materialize().dump() << '\n'; std::cout << "last \"retries\" kept by materialize(): " << settings.materialize()["retries"].dump() << '\n'; } diff --git a/docs/mkdocs/docs/examples/basic_json_view__items.output b/docs/mkdocs/docs/examples/basic_json_view__items.output index 78609c8ee..8f51dbf18 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__items.output +++ b/docs/mkdocs/docs/examples/basic_json_view__items.output @@ -1,5 +1,5 @@ retries=1 timeout=30 retries=5 -first "retries" seen by operator[]: 1 +last "retries" seen by operator[]: 5 last "retries" kept by materialize(): 5 diff --git a/docs/mkdocs/docs/examples/basic_json_view__operator[].cpp b/docs/mkdocs/docs/examples/basic_json_view__operator[].cpp index e17919a9d..1919e06be 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__operator[].cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__operator[].cpp @@ -22,17 +22,19 @@ int main() std::cout << user["name"].materialize().dump(); // operator[] on a missing object key gives a discarded view -- test - // it with a plain "if". The const overload of json::operator[] + // it with is_discarded(). The const overload of json::operator[] // would instead be undefined behavior (guarded by an assertion) for // a missing key - if (const auto email = user["email"]) + const auto email = user["email"]; + if (!email.is_discarded()) { std::cout << " <" << email.materialize().dump() << ">"; } // the same holds for an array index past the end: a discarded view, // not undefined behavior - if (const auto first_tag = user["tags"][0]) + const auto first_tag = user["tags"][0]; + if (!first_tag.is_discarded()) { std::cout << " #" << first_tag.materialize().dump(); } diff --git a/docs/mkdocs/docs/examples/basic_json_view__operator[]_json_pointer.cpp b/docs/mkdocs/docs/examples/basic_json_view__operator[]_json_pointer.cpp index 15f37555d..2ea8130a6 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__operator[]_json_pointer.cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__operator[]_json_pointer.cpp @@ -25,7 +25,8 @@ int main() // a missing key or an out-of-range index along the path gives a // discarded view, exactly where const json::operator[] would be // undefined behavior for the same pointer - if (const auto missing = root[json_pointer("/region/servers/5/metrics/cpu")]) + const auto missing = root[json_pointer("/region/servers/5/metrics/cpu")]; + if (!missing.is_discarded()) { std::cout << missing.materialize().dump() << '\n'; } diff --git a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp index 1e3ed62e2..edc37ff52 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp @@ -35,8 +35,8 @@ int main() // parse without exceptions, are both discarded nlohmann::json_view invalid; json_document failed = json_document::parse("not json", /* allow_exceptions */ false); - std::cout << static_cast(invalid) << ' ' << invalid.is_discarded() << '\n'; - std::cout << static_cast(failed.root()) << ' ' << failed.root().is_discarded() << '\n'; + std::cout << invalid.is_discarded() << '\n'; + std::cout << failed.root().is_discarded() << '\n'; // type() returns the same value_t enumeration as basic_json::type() std::cout << (d_object.root().type() == nlohmann::json::value_t::object) << '\n'; diff --git a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output index 285d31ca0..99e264447 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output +++ b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output @@ -7,6 +7,6 @@ true true true true false false -false true -false true +true +true true diff --git a/docs/mkdocs/docs/features/json_view.md b/docs/mkdocs/docs/features/json_view.md index 8a275db4a..877b48dff 100644 --- a/docs/mkdocs/docs/features/json_view.md +++ b/docs/mkdocs/docs/features/json_view.md @@ -81,6 +81,9 @@ Moving the document itself is fine and does **not** invalidate its views: the in that keeps its address across the move. Take a fresh view from [`root()`](../api/basic_json_document/root.md) whenever any of the other conditions above was not met. +Because a view dies with its document, [`root()`](../api/basic_json_document/root.md) is not callable on a temporary +document: `#!cpp auto v = json_document::parse(text).root();` does not compile. Give the document a name first. + ??? example "Example: borrowed and owned documents, and when views become invalid" ```cpp @@ -113,7 +116,7 @@ whenever any of the other conditions above was not met. - **Only 64-bit integers.** `basic_json_document` requires `BasicJsonType::number_integer_t` and `number_unsigned_t` to both be 64 bits wide; this is a compile-time `#!cpp static_assert`. -- **A 4 GiB input limit.** An input of 4 GiB or more throws +- **A 4 GiB input limit.** An input of 4294967280 bytes (4 GiB minus 16 bytes) or more throws [`out_of_range.416`](../home/exceptions.md#jsonexceptionout_of_range416), a limit `#!cpp basic_json::parse()` does not have. - **A stream is always read to its end.** There is no partial/streaming read of an `#!cpp std::istream`. @@ -126,14 +129,21 @@ whenever any of the other conditions above was not met. members in the order they appear in the source text. `basic_json`'s default `object_t` is a `std::map`, which sorts by key, so iterating a [`materialize()`](../api/basic_json_view/materialize.md)d value can print members in a different order than iterating the view they came from. +- **Chained access is safe.** [`operator[]`](../api/basic_json_view/operator%5B%5D.md) with a missing key, an index + out of range, or an unresolvable JSON pointer returns a [discarded](../api/basic_json_view/is_discarded.md) view, and + `operator[]` on a discarded view returns a discarded view without throwing: `#!cpp v["a"]["b"][0]` can be tested + once at the end. Type errors on values that exist (a key on an array, an index on an object) still throw, and + [`at`](../api/basic_json_view/at.md) throws for every missing value. - **Duplicate keys are visible.** If an object in the source text repeats a key, [`begin()`](../api/basic_json_view/begin.md)/[`end()`](../api/basic_json_view/end.md) and [`items()`](../api/basic_json_view/items.md) visit *every* occurrence (and [`size()`](../api/basic_json_view/size.md) counts all of them), while [`operator[]`](../api/basic_json_view/operator%5B%5D.md), [`at`](../api/basic_json_view/at.md), [`find`](../api/basic_json_view/find.md), [`contains`](../api/basic_json_view/contains.md), and [`count`](../api/basic_json_view/count.md) resolve to the - *first* occurrence, since a lookup can stop as soon as it finds a match. `basic_json::parse()` (and so - [`materialize()`](../api/basic_json_view/materialize.md)) instead keeps only the *last* value for a repeated key. + *last* occurrence -- the one `basic_json::parse()` (and so + [`materialize()`](../api/basic_json_view/materialize.md)) keeps for a repeated key -- which makes a lookup scan all + members instead of stopping at a match (objects with 128 members or more get a hash index that leads to the last + occurrence directly). See the [Notes on duplicate keys](../api/basic_json_view/operator%5B%5D.md#notes) of `operator[]`. - **No [`JSON_DIAGNOSTICS`](../api/macros/json_diagnostics.md) path.** Exceptions thrown by `basic_json_view`'s own element access and lookup functions never carry the JSON Pointer path `JSON_DIAGNOSTICS` would otherwise add: the diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md index b2560f372..162cfbad5 100644 --- a/docs/mkdocs/docs/features/types/template_parameters.md +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -353,16 +353,21 @@ using array_t = ArrayType>; ### Always required - A member type `value_type` that is one byte wide and `char`-compatible. The library stores and processes UTF-8 - encoded `char` data and passes `data()` to functions that take a `#!cpp const char*`, such as `#!cpp std::strtod`. + encoded `char` data and passes `data()` to functions that take a `#!cpp const char*`, such as `#!cpp std::strtold` + (only used to parse a `#!cpp long double` that is not IEEE 754 binary64, see + [`NumberFloatType`](#numberfloattype)). `#!cpp std::wstring`, `#!cpp std::u16string`, and `#!cpp std::u32string` are **not** valid choices; see the FAQ on [wide string handling](../../home/faq.md#wide-string-handling). - Constructors: default, copy, move, from `#!cpp const char*` (which must not be `#!cpp explicit`), from `#!cpp (const char*, size_type)`, and from `#!cpp (size_type, char)`; and copy or move assignment. - Member functions `size()`, `clear()`, `resize(n, c)`, `data()`, `push_back(char)`, and `operator[]` (const and non-const, returning references). `c_str()` and `back()` are **not** required. -- `data()` must return a pointer to a contiguous, **null-terminated** buffer -- the parser may hand it to - `#!cpp std::strtod`, which reads up to the null character. A type whose `data()` is not null-terminated does not - fail to compile; it can silently misparse floating-point numbers. +- `data()` must return a pointer to a contiguous, **null-terminated** buffer. `#!cpp float`, `#!cpp double`, and a + `#!cpp long double` that is IEEE 754 binary64 are converted by the library itself and do not depend on this. For any + other `NumberFloatType` (a `#!cpp long double` of another format), the parser falls back to `#!cpp std::strtold` when + `#!cpp std::from_chars` is not available or declines the token, and `std::strtold` reads up to the null character. A type whose `data()` + is not null-terminated does not fail to compile; with such a `NumberFloatType` it can silently misparse + floating-point numbers. - `append(const char*, size_type)`, used by [`dump`](../../api/basic_json/dump.md), and `append(const StringType&)`, used by the CBOR reader for indefinite-length strings. The library's internal string concatenation additionally has to append a `#!cpp char` and a `#!cpp const char*`; for each it selects between `append(arg)`, `#!cpp operator+=`, diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index 3329998f5..9352e1d7d 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -210,12 +210,13 @@ packet-beta - **Navigation** needs no pointers: the elements of an array or object follow its node, and the node after a value's subtree is `next` nodes further for an array or object, and the next node otherwise (`document_data::after`). Views step from element to element this way and skip whole subtrees in constant time. -- **Offsets** are 32 bits wide, so a document is limited to 4 GiB (`out_of_range.416`). +- **Offsets** are 32 bits wide, so a document is limited to 4294967279 bytes, 4 GiB minus 16 bytes (a margin below + 2^32 for positions one scanner step past the end of the text; `out_of_range.416`). - **Large objects** (128 members or more) get a hash index after parsing ([`detail/view/object_index.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/object_index.hpp)): an open-addressing table whose slots hold the distance from the object's node to a key's node, so that a lookup does - not compare every key. The object's `extra` holds the number of its table. Only 65,535 tables fit into `extra`; - objects beyond them are searched linearly. + not compare every key. Of duplicate keys the table leads to the last, as a linear search does. The object's `extra` + holds the number of its table. Only 65,535 tables fit into `extra`; objects beyond them are searched linearly. For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an array or object past its subtree: diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 3f5ed1ea9..c2d2a3733 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -1067,7 +1067,7 @@ MessagePack's ext type and BSON's binary subtype are each stored in a single byt [`basic_json_document::parse()`](../api/basic_json_document/parse.md) and the other parsing functions of [`basic_json_document`](../api/basic_json_document/index.md) index a value's position in the source text in 32 bits, -so they do not support an input of 4 GiB or more. The same 32-bit limit applies to an **editable** document's own +so they do not support an input of 4294967280 bytes (4 GiB minus 16 bytes) or more. The same 32-bit limit applies to an **editable** document's own storage: [`set`](../api/basic_json_document/set.md) and [`push_back`](../api/basic_json_document/push_back.md) throw this exception once the strings and number tokens written by edits reach 4 GiB in total, or once more than 4294967295 arrays/objects have had an element set or appended to them. @@ -1075,7 +1075,7 @@ this exception once the strings and number tokens written by edits reach 4 GiB i !!! failure "Example messages" ``` - [json.exception.out_of_range.416] input of 4 GiB or more is not supported by json_document + [json.exception.out_of_range.416] input of 4294967280 bytes or more is not supported by json_document ``` ``` [json.exception.out_of_range.416] edits of 4 GiB or more are not supported by json_document diff --git a/docs/mkdocs/docs/home/license.md b/docs/mkdocs/docs/home/license.md index 53b54fc82..714e971ed 100644 --- a/docs/mkdocs/docs/home/license.md +++ b/docs/mkdocs/docs/home/license.md @@ -18,7 +18,7 @@ The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed und The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/) -The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) +The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, including the conversion of the digits to text by Xiang JunBo and the SIMD instruction sequence of Dougall Johnson, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/). diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index aab394c72..4dee3df3d 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -287,7 +287,6 @@ nav: - 'materialize': api/basic_json_view/materialize.md - 'number_format': api/basic_json_view/number_format.md - 'number_token': api/basic_json_view/number_token.md - - 'operator bool': api/basic_json_view/operator_bool.md - 'operator<<': api/basic_json_view/operator_ltlt.md - 'operator[]': api/basic_json_view/operator[].md - 'operator==': api/basic_json_view/operator_eq.md diff --git a/gcm.cache/nlohmann.json.gcm b/gcm.cache/nlohmann.json.gcm deleted file mode 100644 index b5ca610d3..000000000 Binary files a/gcm.cache/nlohmann.json.gcm and /dev/null differ diff --git a/include/nlohmann/detail/abi_config.hpp b/include/nlohmann/detail/abi_config.hpp index 0e99c81ad..510a67385 100644 --- a/include/nlohmann/detail/abi_config.hpp +++ b/include/nlohmann/detail/abi_config.hpp @@ -15,10 +15,11 @@ namespace detail { /*! -@brief the configuration macros that change the library's behavior +@brief the configuration macros that json_view.hpp reads json.hpp undefines these macros at its end (see macro_unscope.hpp), so code -that builds on the library after it (json_view.hpp) reads them here. Like the +that builds on the library after it (json_view.hpp) reads them here. A macro +is added when the view starts to depend on it. Like the macros, they are part of the ABI namespace, so they always match the basic_json they are used with. */ @@ -26,8 +27,6 @@ struct abi_config { /// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0; - /// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON - static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0; }; } // namespace detail diff --git a/include/nlohmann/detail/bit_ops.hpp b/include/nlohmann/detail/bit_ops.hpp index 9da853faf..42a09b68e 100644 --- a/include/nlohmann/detail/bit_ops.hpp +++ b/include/nlohmann/detail/bit_ops.hpp @@ -9,8 +9,9 @@ #pragma once #include // uint64_t -#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) - #include // __umulh, _umul128 +#include // memcpy +#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) && (!defined(__SIZEOF_INT128__) || (!defined(__GNUC__) && !defined(__clang__))) + #include // __umulh, _umul128, _BitScanForward64, _BitScanReverse64 #endif #include // JSON_HEDLEY_ALWAYS_INLINE, NLOHMANN_JSON_NAMESPACE_BEGIN @@ -28,6 +29,10 @@ inline int count_leading_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_clzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanReverse64(&index, x); + return 63 - static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -47,6 +52,10 @@ inline int count_trailing_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_ctzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanForward64(&index, x); + return static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -94,15 +103,21 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc #endif } -/// eight bytes as a little-endian word (compilers fold this into one load on -/// little-endian targets; always inlined, as GCC otherwise calls it in the -/// number loops) +/// eight bytes as a little-endian word (a single load on little-endian +/// targets; always inlined, as GCC otherwise calls it in the number loops) JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const unsigned char* b) noexcept { +#if defined(_MSC_VER) || defined(__x86_64__) || defined(__i386__) || (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__) + // the byte order already matches (all MSVC targets are little-endian) + std::uint64_t result = 0; + std::memcpy(&result, b, sizeof(result)); + return result; +#else return static_cast(b[0]) | (static_cast(b[1]) << 8u) | (static_cast(b[2]) << 16u) | (static_cast(b[3]) << 24u) | (static_cast(b[4]) << 32u) | (static_cast(b[5]) << 40u) | (static_cast(b[6]) << 48u) | (static_cast(b[7]) << 56u); +#endif } /// eight bytes as a little-endian word diff --git a/include/nlohmann/detail/conversions/to_chars.hpp b/include/nlohmann/detail/conversions/to_chars.hpp index 0e16152c3..db50795b5 100644 --- a/include/nlohmann/detail/conversions/to_chars.hpp +++ b/include/nlohmann/detail/conversions/to_chars.hpp @@ -4,6 +4,7 @@ // |_____|_____|_____|_|___| https://github.com/nlohmann/json // // SPDX-FileCopyrightText: 2009 Florian Loitsch +// SPDX-FileCopyrightText: 2025 Victor Zverovich // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT @@ -939,88 +940,6 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value) grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus); } -/*! -@brief the shortest digits of a positive finite float (other than double): Grisu2 -*/ -template -JSON_HEDLEY_NON_NULL(1) -void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value) -{ - grisu2(buf, len, decimal_exponent, value); -} - -/*! -@brief the shortest digits of a positive finite double: the conversion of -Zmij (see zmij.hpp), which always finds the shortest digits that read back as -the same value (Grisu2 does not for about one double in a thousand), and the -closest of them if there are several - -v = buf * 10^decimal_exponent, as for grisu2() -*/ -JSON_HEDLEY_NON_NULL(1) -inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value) -{ - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); - JSON_ASSERT(std::isfinite(value)); - JSON_ASSERT(value > 0); - - std::uint64_t bits = 0; - std::memcpy(&bits, &value, sizeof(bits)); - zmij::decimal d = zmij::to_decimal(bits); - // without trailing zeros (up to 16): 8, 4, 2, 1 at a time - while (d.significand % 100000000 == 0) - { - d.significand /= 100000000; - d.exponent += 8; - } - if (d.significand % 10000 == 0) - { - d.significand /= 10000; - d.exponent += 4; - } - if (d.significand % 100 == 0) - { - d.significand /= 100; - d.exponent += 2; - } - if (d.significand % 10 == 0) - { - d.significand /= 10; - d.exponent += 1; - } - // at most 17 digits, written from the back two at a time - static constexpr const char* pairs = - "00010203040506070809101112131415161718192021222324252627282930313233343536373839" - "40414243444546474849505152535455565758596061626364656667686970717273747576777879" - "8081828384858687888990919293949596979899"; - std::array digits{}; - std::size_t n = digits.size(); - while (d.significand >= 100) - { - const std::uint64_t two_digits = d.significand % 100; // a variable: GCC calls a cast of the remainder useless where std::uint64_t is std::size_t - const auto i = static_cast(two_digits) * 2; - d.significand /= 100; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - if (d.significand >= 10) - { - const auto i = static_cast(d.significand) * 2; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - else - { - digits[--n] = static_cast('0' + d.significand); - } - len = static_cast(digits.size() - n); - std::memcpy(buf, digits.data() + n, static_cast(len)); - decimal_exponent = d.exponent; -} - /*! @brief appends a decimal representation of e to buf @return a pointer to the element following the exponent. @@ -1423,53 +1342,30 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep return end + (three ? 5 : 4); } -/// the powers of ten up to 10^16 -inline const std::array& powers_of_ten_16() noexcept -{ - static const std::array powers = - { - { - 1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u, - 100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u - } - }; - return powers; -} - /*! -@brief digits * 10^exp, as write_decimal() writes it, for the digits of a -double that need no conversion (count digits, at most 15, the first not 0; -trailing zeros allowed): extended to 16 digits and written by write_shortest() +@brief whether FloatType is an IEEE 754 binary64 type (a double, or a long double +that has the same format, as with MSVC and on Apple's Arm CPUs) -@return a pointer past the text; up to 41 bytes at @a first are written - (some beyond the returned end) +These are the types the conversion of Zmij (see zmij.hpp) is used for; all +others (binary32, or a format the library does not know) use Grisu2. */ -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept +template +constexpr bool has_binary64_format() noexcept { - JSON_ASSERT(digits >= powers_of_ten_16()[static_cast(count - 1)] && count <= 15); - const int scale = 16 - count; - return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast(scale)], exp - scale - 1, 0, false}); + return std::numeric_limits::is_iec559 + && std::numeric_limits::digits == 53 + && std::numeric_limits::max_exponent == 1024 + && sizeof(FloatType) == sizeof(std::uint64_t); } -/// as write_short_decimal(), counting the digits (not 0, less than 10^15) -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept -{ - JSON_ASSERT(digits != 0 && digits < 1000000000000000u); - // floor(log10(2^bits)) + 1 digits, or one less - const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12; - const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast(log2_bound)] ? 1 : 0); - return write_short_decimal(first, digits, count, exp); -} +template +struct is_binary64 : std::integral_constant()> {}; -/// a positive finite float (other than double): Grisu2 and format_buffer() +/// a positive finite float (other than binary64): Grisu2 and format_buffer() template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -char* write_positive(char* first, const char* last, FloatType value) +char* write_positive_grisu2(char* first, const char* last, FloatType value) { JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); static_cast(last); // (only used in the assertion) @@ -1480,7 +1376,7 @@ char* write_positive(char* first, const char* last, FloatType value) // len is the length of the buffer, i.e., the number of decimal digits. int len = 0; int decimal_exponent = 0; - shortest_digits(first, len, decimal_exponent, value); + grisu2(first, len, decimal_exponent, value); JSON_ASSERT(len <= std::numeric_limits::max_digits10); @@ -1496,15 +1392,16 @@ char* write_positive(char* first, const char* last, FloatType value) return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp); } -/// a positive finite double: the shortest digits (Zmij), laid out by +/// a positive finite binary64 number: the shortest digits (Zmij), laid out by /// write_shortest() (through a local buffer if [first, last) is shorter than /// the 41 bytes it may write) +template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_positive(char* first, const char* last, double value) +char* write_positive_zmij(char* first, const char* last, FloatType value) { - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); + static_assert(is_binary64::value, + "internal error: the conversion of Zmij needs IEEE 754 binary64 numbers"); std::uint64_t bits = 0; std::memcpy(&bits, &value, sizeof(bits)); const zmij::shortest_decimal d = zmij::to_shortest(bits); @@ -1519,6 +1416,34 @@ inline char* write_positive(char* first, const char* last, double value) return first + len; } +/// a positive finite binary64 number: Zmij (as a long double has the format of +/// a double here, its bits are those of the double of the same value) +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::true_type /*is_binary64*/) +{ + return write_positive_zmij(first, last, value); +} + +/// a positive finite float of any other format: Grisu2 +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::false_type /*is_binary64*/) +{ + return write_positive_grisu2(first, last, value); +} + +/// a positive finite float: Zmij for binary64 numbers, Grisu2 otherwise +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value) +{ + return write_positive(first, last, value, is_binary64 {}); +} + } // namespace dtoa_impl /*! diff --git a/include/nlohmann/detail/conversions/zmij.hpp b/include/nlohmann/detail/conversions/zmij.hpp index ffff8380e..40e477f7b 100644 --- a/include/nlohmann/detail/conversions/zmij.hpp +++ b/include/nlohmann/detail/conversions/zmij.hpp @@ -36,13 +36,6 @@ computed from the compressed tables of Zmij beyond it. namespace zmij { -/// significand * 10^exponent -struct decimal -{ - std::uint64_t significand; - int exponent; -}; - /// the compressed powers of ten of Zmij inline const std::array& pow10_minor() noexcept { @@ -221,18 +214,6 @@ JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexc return shortest_decimal{integral, dec_exp, static_cast(digit), !round_up && !round_down}; } -/// The shortest decimal in the rounding interval of a positive finite double -/// given by its bits, as one number. The significand can end in zeros. -inline decimal to_decimal(std::uint64_t bits) noexcept -{ - const shortest_decimal d = to_shortest(bits); - if (d.has_digit) - { - return decimal{(d.integral * 10) + d.digit, d.exponent}; - } - return decimal{d.integral, d.exponent + 1}; -} - } // namespace zmij } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp index 6b2ee0616..1722dcc34 100644 --- a/include/nlohmann/detail/view/builder.hpp +++ b/include/nlohmann/detail/view/builder.hpp @@ -9,7 +9,7 @@ #pragma once -#include // find, find_if, max +#include // find, find_if, max, min #include // array #include // size_t, ptrdiff_t #include // int64_t, uint8_t, uint16_t, uint32_t, uint64_t @@ -202,8 +202,18 @@ class builder const std::uint64_t done = static_cast(at - b) + 1; const std::uint64_t guess = static_cast(n) * static_cast(e - b + 1) / done; const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t + // (n is below 2^32: the input is smaller than 4 GiB; the sum cannot wrap) + const std::uint64_t wanted = (std::max)(grown, static_cast(n) + (n / 2) + 64); + const std::uint64_t limit = document_data::max_nodes(); doc.tape_size = n; - doc.reserve((std::max)(static_cast(grown), n + (n / 2) + 64)); + // LCOV_EXCL_START (a node array that fills the address space) + if (NLOHMANN_VIEW_UNLIKELY(n >= limit)) + { + document_data::throw_bad_alloc(); // no room for another node + } + // LCOV_EXCL_STOP + // (a count beyond the limit is cut: the index does not grow beyond what can be addressed) + doc.reserve(static_cast((std::min)(wanted, limit))); return doc.tape; } @@ -816,7 +826,10 @@ indent_done: n->flags = flags; n->extra = extra; n->off = static_cast(off); - set_integer_bits(*n, second); + // len is the low half of the second word, next the high half + // (not a native word over both, which swaps them on big-endian) + n->len = static_cast(second); + n->next = static_cast(second >> 32); #endif return n; } diff --git a/include/nlohmann/detail/view/document_data.hpp b/include/nlohmann/detail/view/document_data.hpp index adcb9ce53..2e1eac3ef 100644 --- a/include/nlohmann/detail/view/document_data.hpp +++ b/include/nlohmann/detail/view/document_data.hpp @@ -13,9 +13,10 @@ #include // uint32_t #include // memcpy #include // less +#include // numeric_limits #include // map #include // unique_ptr -#include // operator new, placement new +#include // bad_alloc, operator new, placement new #include // string #include // vector @@ -121,13 +122,30 @@ struct document_data tape_cap = inline_cap; } - /// make room for n nodes; keeps the first tape_size nodes + /// the largest node count whose size in bytes fits a std::size_t + static constexpr std::size_t max_nodes() noexcept + { + return (std::numeric_limits::max)() / sizeof(node); + } + + [[noreturn]] NLOHMANN_VIEW_NOINLINE static void throw_bad_alloc() + { + NLOHMANN_VIEW_THROW(std::bad_alloc()); + } + + /// make room for n nodes; keeps the first tape_size nodes (throws + /// std::bad_alloc for a count that does not fit the address space, + /// instead of wrapping around in n * sizeof(node)) void reserve(std::size_t n) { if (n <= tape_cap) { return; } + if (NLOHMANN_VIEW_UNLIKELY(n > max_nodes())) + { + throw_bad_alloc(); + } node* fresh = static_cast(::operator new (n * sizeof(node))); if (tape_size != 0) { diff --git a/include/nlohmann/detail/view/errors.hpp b/include/nlohmann/detail/view/errors.hpp index 5f3dca477..e45956c8c 100644 --- a/include/nlohmann/detail/view/errors.hpp +++ b/include/nlohmann/detail/view/errors.hpp @@ -66,9 +66,8 @@ template { if (f.code == error_code::input_too_large) { - // LCOV_EXCL_START (4 GiB) - NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr)); - // LCOV_EXCL_STOP + // (the limit is detail::view::max_input_size: 4 GiB minus 16 bytes) + NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4294967280 bytes or more is not supported by json_document", nullptr)); } const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas); // LCOV_EXCL_START (only if parse() accepts what the view rejects: a bug) diff --git a/include/nlohmann/detail/view/input.hpp b/include/nlohmann/detail/view/input.hpp index e8a5e3148..ac3257ee3 100644 --- a/include/nlohmann/detail/view/input.hpp +++ b/include/nlohmann/detail/view/input.hpp @@ -9,7 +9,7 @@ #pragma once #include // basic_string, char_traits, string -#include // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference +#include // decay, integral_constant, is_array, is_const, is_integral, is_lvalue_reference, is_pointer, is_same, remove_reference #include // forward #include @@ -28,12 +28,12 @@ namespace view /// how a document takes its input enum class input_kind { - move_string, ///< rvalue std::string: owned without a copy + move_string, ///< non-const rvalue std::string: owned without a copy c_string, ///< const char* (NUL-terminated): borrowed char_array, ///< char array (e.g. a string literal): borrowed borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed - copy_range, ///< rvalue contiguous byte container: copied - adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer + copy_range, ///< rvalue contiguous byte container (a const rvalue std::string too): copied + adapter, ///< streams, wide strings, and the rest of what the library's input adapter reads: read into a buffer }; template @@ -52,13 +52,20 @@ struct classify_input static constexpr input_kind value = std::is_array::value ? input_kind::char_array : std::is_pointer::value ? input_kind::c_string - : (is_rvalue && std::is_same::value) ? input_kind::move_string + : (is_rvalue && !std::is_const::value && std::is_same::value) ? input_kind::move_string : (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range : is_bytes ? input_kind::copy_range : input_kind::adapter; // NOLINTEND(readability-avoid-nested-conditional-operator) }; +/// an integer type other than bool: a length passed where a flag is expected +template +struct is_integer_not_bool : std::is_integral {}; + +template<> +struct is_integer_not_bool : std::false_type {}; + /// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel) template struct is_std_string : std::false_type {}; diff --git a/include/nlohmann/detail/view/lookup.hpp b/include/nlohmann/detail/view/lookup.hpp index 718acc077..9af032071 100644 --- a/include/nlohmann/detail/view/lookup.hpp +++ b/include/nlohmann/detail/view/lookup.hpp @@ -11,6 +11,8 @@ #include // size_t #include // uint16_t, uint32_t, uint64_t #include // memcmp, memcpy +#include // numeric_limits +#include // integral_constant, is_integral, is_same #include #include @@ -82,8 +84,9 @@ class short_key std::uint64_t m_b = 0; }; -/// the key node of the first member of an object with the given key, or -/// nullptr; most keys are rejected by their length, from the index alone +/// the key node of the last member of an object with the given key, or +/// nullptr (the last one, as materialize() and parse() keep it); most keys are +/// rejected by their length, from the index alone template const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept { @@ -94,6 +97,7 @@ const node* find_member(const document_data& d, const node* object, const char* } const node* const end = nav::end(d, object); const auto* const k = reinterpret_cast(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const node* last = nullptr; if (NLOHMANN_VIEW_LIKELY(n <= 16)) { const short_key probe(k, n); @@ -101,19 +105,37 @@ const node* find_member(const document_data& d, const node* object, const char* { if (m->len == n && probe.matches(reinterpret_cast(d.str(*m)))) // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) { - return m; + last = m; } } - return nullptr; + return last; } for (const node* m = nav::first(d, object); m != end; m = document_data::after(m + 1)) { if (m->len == n && std::memcmp(d.str(*m), key, n) == 0) { - return m; + last = m; } } - return nullptr; + return last; +} + +/// whether an integer type is accepted as an array index by the view's +/// operator[] and at(): every integer type but bool and size_t, which has its +/// own overload +template +struct is_index_type : std::integral_constant < bool, + std::is_integral::value && !std::is_same::value && !std::is_same::value > +{}; + +/// an integer as an index: negative values, and values that do not fit a +/// size_t, map to the largest size_t (out of range for every array) +template +SizeType to_index(IntegerType idx) noexcept +{ + const IntegerType zero = 0; + const auto result = static_cast(idx); + return (idx < zero || static_cast(result) != idx) ? (std::numeric_limits::max)() : result; } /// the entry of the element of an array at an index below its size (a link diff --git a/include/nlohmann/detail/view/node.hpp b/include/nlohmann/detail/view/node.hpp index 35e83fbe5..4f3d1fe2b 100644 --- a/include/nlohmann/detail/view/node.hpp +++ b/include/nlohmann/detail/view/node.hpp @@ -29,6 +29,11 @@ static_assert(static_cast(value_t::null) == 0 && static_cast(value_t::number_unsigned) == 6 && static_cast(value_t::number_float) == 7, "the node format depends on the numbering of value_t"); +/// The largest input a document accepts, in bytes. Offsets and node counts are +/// 32 bits wide; the limit keeps 16 bytes (the width of the scanner's steps) +/// below 2^32, so that a position one step past the end of the text fits. +static constexpr std::size_t max_input_size = 0xFFFFFFEFu; + /// node flags struct node_flags { @@ -53,7 +58,7 @@ struct node { std::uint8_t kind; ///< value_t, or kind_link std::uint8_t flags; ///< node_flags - std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; objects: number of the hash index; otherwise 0 + std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; objects: number of the hash index (1-based, 0 = none); otherwise 0 std::uint32_t off; ///< source offset (string content, number token, literal, bracket); arena offset if escaped/edited; number of the element sequence if moved std::uint32_t len; ///< string: decoded bytes; float: token bytes; array/object: element count std::uint32_t next; ///< array/object: number of nodes of the subtree (its extent in the enclosing sequence) @@ -82,17 +87,27 @@ inline void make_link(node& n, node* target) noexcept std::memcpy(reinterpret_cast(&n) + 8, static_cast(&target), sizeof(const node*)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) } -/// the converted value of an integer node (stored in len/next) +/// the converted value of an integer node: len is its low half, next its high +/// half (on little-endian targets the two words are the value in memory) NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::uint64_t v = 0; std::memcpy(&v, reinterpret_cast(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) return v; +#else + return static_cast(n.len) | (static_cast(n.next) << 32); +#endif } NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::memcpy(reinterpret_cast(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +#else + n.len = static_cast(v); + n.next = static_cast(v >> 32); +#endif } /// token length of a number node diff --git a/include/nlohmann/detail/view/object_index.hpp b/include/nlohmann/detail/view/object_index.hpp index 27348d827..721b4d62b 100644 --- a/include/nlohmann/detail/view/object_index.hpp +++ b/include/nlohmann/detail/view/object_index.hpp @@ -22,8 +22,14 @@ // large objects). An object with document_data::index_min_members members or // more gets an open-addressing table after parsing; its node stores the // number of the table (1-based) in `extra`. A slot holds the offset of a key -// node from its object node (0: empty). Of duplicate keys, the first is kept, -// as for the linear search. +// node from its object node (0: empty). Of duplicate keys, the last is kept, +// as for the linear search, and as basic_json::parse() does. +// +// The hash is not seeded, so keys chosen to collide could make the build +// quadratic. A key therefore sits at most index_max_displacement slots away +// from its home slot; if a key would sit further away, the table is dropped +// and the object is searched linearly (like a small one). For the same +// reason, a lookup visits at most index_max_displacement + 1 slots. NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -31,6 +37,11 @@ namespace detail namespace view { +/// the farthest a key may sit from its home slot (a table with at most half of +/// its slots in use gives random keys a distance of about 50 for millions of +/// members; and every member costs at most this many steps while building) +constexpr std::size_t index_max_displacement = 64; + /// hash of a key: its bytes, eight at a time, in a fixed byte order inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept { @@ -68,27 +79,44 @@ inline void build_object_index(document_data& d, node* obj) d.index_slots.resize(start + cap, 0); std::uint32_t* const slots = d.index_slots.data() + start; const std::size_t mask = cap - 1; + bool degenerate = false; for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1)) { const char* const key = d.str(*k); const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t) std::size_t i = static_cast(hash) & mask; bool duplicate = false; + std::size_t distance = 0; while (slots[i] != 0) { const node* const other = obj + slots[i]; if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0)) { - duplicate = true; // keep the first + duplicate = true; // keep the last: the key's slot now leads to this member + slots[i] = static_cast(k - obj); + break; + } + if (++distance > index_max_displacement) + { + degenerate = true; // too many keys share a home region break; } i = (i + 1) & mask; } + if (degenerate) + { + break; + } if (!duplicate) { slots[i] = static_cast(k - obj); } } + if (degenerate) + { + d.index_slots.resize(start); // no table: the object is searched linearly + return; + } d.indexes.push_back(document_data::object_index{start, static_cast(mask)}); obj->extra = static_cast(d.indexes.size()); } @@ -102,7 +130,7 @@ inline void build_object_indexes(document_data& d) } } -/// the key node of the first member with this key of an indexed object, or +/// the key node of the last member with this key of an indexed object, or /// nullptr inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept { @@ -110,7 +138,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c const std::uint32_t* const slots = d.index_slots.data() + ix.start; const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t) std::size_t i = static_cast(hash) & ix.mask; - for (;;) + for (std::size_t distance = 0; distance <= index_max_displacement; ++distance) { const std::uint32_t s = slots[i]; if (s == 0) @@ -124,6 +152,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c } i = (i + 1) & ix.mask; } + return nullptr; // (no key sits further from its home slot) } } // namespace view diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp index 3c5addf11..737654aa1 100644 --- a/include/nlohmann/detail/view/serializer.hpp +++ b/include/nlohmann/detail/view/serializer.hpp @@ -44,7 +44,13 @@ class output_buffer void finish() { - m_out.resize(static_cast(m_pos - m_out.data())); + const auto size = static_cast(m_pos - m_out.data()); + m_out.resize(size); + // do not keep a buffer that was sized for a much larger output + if (m_out.capacity() > 1024 && m_out.capacity() / 2 > size) + { + m_out.shrink_to_fit(); + } } NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n) diff --git a/include/nlohmann/detail/view/simd.hpp b/include/nlohmann/detail/view/simd.hpp index 4c99c7072..13ad9c59c 100644 --- a/include/nlohmann/detail/view/simd.hpp +++ b/include/nlohmann/detail/view/simd.hpp @@ -35,7 +35,8 @@ #else #define NLOHMANN_VIEW_NEON 0 #endif -#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2)) +// (x86 only: other targets can define __SSE2__ as well, e.g., WebAssembly with -msse2, but have no ) +#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__x86_64__) || defined(__i386__) || defined(_M_X64) || defined(_M_IX86)) && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2)) #include #define NLOHMANN_VIEW_SSE2 1 #else diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 72e7a4397..977a2a124 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -26,7 +26,7 @@ #ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ -#include // all_of +#include // all_of, min #include // size_t #include // uint32_t #include // memcpy, strlen @@ -183,12 +183,6 @@ class basic_json_view return type() == value_t::discarded; } - /// false for discarded views - explicit operator bool() const noexcept - { - return m_node != nullptr; - } - /// the name of the type, as basic_json::type_name() const char* type_name() const noexcept { @@ -248,13 +242,18 @@ class basic_json_view // element access // //////////////////// - /// the value of the member with this key (the first one, should the key - /// occur more than once); a discarded view if there is none. Throws - /// type_error.305 if this is not an object. + /// the value of the member with this key (the last one, should the key + /// occur more than once); a discarded view if there is none, or if this + /// is a discarded view (so that v["a"]["b"] is safe). Throws type_error.305 + /// if this is any other value but an object. NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view operator[](string_view_t key) const { if (NLOHMANN_VIEW_UNLIKELY(!is_object())) { + if (is_discarded()) + { + return basic_json_view(); + } detail::view::throw_type_error(305, "cannot use operator[] with a string argument with ", type_name()); } return lookup(key); @@ -271,31 +270,43 @@ class basic_json_view } /// the element at this index; a discarded view if the index is out of - /// range. Throws type_error.305 if this is not an array. + /// range, or if this is a discarded view. Throws type_error.305 if this is + /// any other value but an array. basic_json_view operator[](size_type idx) const { if (NLOHMANN_VIEW_UNLIKELY(!is_array())) { + if (is_discarded()) + { + return basic_json_view(); + } detail::view::throw_type_error(305, "cannot use operator[] with a numeric argument with ", type_name()); } return idx < m_node->len ? basic_json_view(m_doc, navigation::value(detail::view::element_at(*m_doc, m_node, idx))) : basic_json_view(); } - /// (an int argument would be ambiguous between size_type and const char*) - basic_json_view operator[](int idx) const + /// any other integer type (int, unsigned, long, std::int64_t, ...; a + /// single overload for size_type alone would be ambiguous for all of them + /// and for const char*); negative values are out of range + template < typename IntegerType, typename std::enable_if < detail::view::is_index_type::value, int >::type = 0 > + basic_json_view operator[](IntegerType idx) const { - return operator[](static_cast(idx)); + return operator[](detail::view::to_index(idx)); } /// the value a JSON pointer refers to; a discarded view if a key is - /// missing or an index is out of range. Other errors throw what const - /// basic_json::operator[] throws. + /// missing or an index is out of range, or if this is a discarded view. + /// Other errors throw what const basic_json::operator[] throws. basic_json_view operator[](const json_pointer& ptr) const { + if (NLOHMANN_VIEW_UNLIKELY(is_discarded())) + { + return basic_json_view(); + } return detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::unchecked); } - /// the value of the member with this key (the first one, should the key + /// the value of the member with this key (the last one, should the key /// occur more than once). Throws type_error.304 if this is not an object, /// and out_of_range.403 if there is no such member. basic_json_view at(string_view_t key) const @@ -305,7 +316,7 @@ class basic_json_view detail::view::throw_type_error(304, "cannot use at() with ", type_name()); } const basic_json_view r = lookup(key); - if (NLOHMANN_VIEW_UNLIKELY(!r)) + if (NLOHMANN_VIEW_UNLIKELY(r.is_discarded())) { detail::view::throw_out_of_range(403, detail::concat("key '", std::string(key.data(), key.size()), "' not found")); } @@ -337,9 +348,12 @@ class basic_json_view return basic_json_view(m_doc, navigation::value(detail::view::element_at(*m_doc, m_node, idx))); } - basic_json_view at(int idx) const + /// any other integer type, see operator[]; negative values are out of + /// range + template < typename IntegerType, typename std::enable_if < detail::view::is_index_type::value, int >::type = 0 > + basic_json_view at(IntegerType idx) const { - return at(static_cast(idx)); + return at(detail::view::to_index(idx)); } /// the value a JSON pointer refers to; throws what basic_json::at() @@ -350,7 +364,7 @@ class basic_json_view } /// the member with this key converted to T, or the default value if there - /// is no such member (the first one, should the key occur more than + /// is no such member (the last one, should the key occur more than /// once). Throws type_error.306 if this is not an object. template < typename T, typename std::enable_if < !std::is_same::type, const char*>::value, int >::type = 0 > T value(string_view_t key, const T& default_value) const @@ -360,7 +374,7 @@ class basic_json_view detail::view::throw_type_error(306, "cannot use value() with ", type_name()); } const basic_json_view r = lookup(key); - return r ? r.template get() : default_value; + return r.is_discarded() ? default_value : r.template get(); } string_t value(string_view_t key, const char* default_value) const @@ -379,7 +393,7 @@ class basic_json_view detail::view::throw_type_error(306, "cannot use value() with ", type_name()); } const basic_json_view r = detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::value); - return r ? r.template get() : default_value; + return r.is_discarded() ? default_value : r.template get(); } string_t value(const json_pointer& ptr, const char* default_value) const @@ -415,7 +429,7 @@ class basic_json_view // lookup // //////////// - /// an iterator to the member with this key (the first one, should the + /// an iterator to the member with this key (the last one, should the /// key occur more than once), or end(); end() also for non-objects iterator find(string_view_t key) const { @@ -457,7 +471,7 @@ class basic_json_view /// basic_json::contains()) bool contains(const json_pointer& ptr) const { - return static_cast(detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains)); + return !detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains).is_discarded(); } /// 1 if this is an object with a member with this key, else 0 (duplicate @@ -714,7 +728,7 @@ class basic_json_view } /// the number of source bytes of this value (estimated for values with - /// decoded strings) + /// decoded strings); the estimate sizes the output buffer of dump() std::size_t source_extent() const noexcept { if (editable() && m_doc->edits != nullptr) @@ -722,20 +736,33 @@ class basic_json_view // positions of moved and new values are not source offsets return m_node == m_doc->tape ? m_doc->size + m_doc->edits->text_used : 64; } - const node* const next = document_data::after(m_node); - const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0; - if (!in_source) + const node* const end = m_doc->tape + m_doc->tape_size; + if ((m_node->flags & detail::view::node_flags::storage) != 0) { return m_node->len; } - if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off) + // the value ends where the next node in the source begins; nodes + // with decoded strings (their offset is in the arena) are skipped, + // but only a few of them, to keep the walk short + const node* next = document_data::after(m_node); + for (int skipped = 0; next != end && skipped < 16; ++skipped, ++next) { - return next->off - m_node->off; + if ((next->flags & detail::view::node_flags::storage) == 0) + { + return next->off >= m_node->off ? next->off - m_node->off : 0; + } } - return m_doc->size - m_node->off; + if (next == end) + { + return m_doc->size - m_node->off; + } + // the end is unknown: assume a few bytes per node, the output buffer + // grows should the value be larger + const auto nodes = static_cast(document_data::after(m_node) - m_node); + return (std::min)(m_doc->size - m_node->off, static_cast(1024) + nodes * 16); } - /// the value of the first member with this key, or a discarded view + /// the value of the last member with this key, or a discarded view /// (object required) NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view lookup(string_view_t key) const noexcept { @@ -869,6 +896,13 @@ class basic_json_document return d; } + /// parse(ptr, len) does not compile: len would convert to allow_exceptions + /// and ptr be read as a C string (as for the overloads of parse_copy, + /// accept, and read below) + template + static typename std::enable_if::value, basic_json_document>::type + parse(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse [first, last) template::iterator_category>::value, int>::type = 0> @@ -896,6 +930,10 @@ class basic_json_document return d; } + template + static typename std::enable_if::value, basic_json_document>::type + parse_copy(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// check whether the input is valid JSON (the result of basic_json::accept) template static bool accept(InputType&& input, const bool ignore_comments = false, const bool ignore_trailing_commas = false) @@ -905,6 +943,10 @@ class basic_json_document return !d.is_discarded(); } + template + static typename std::enable_if::value, bool>::type + accept(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse into this document, reusing its memory template // flawfinder: ignore (a member function, not POSIX read()) @@ -917,12 +959,17 @@ class basic_json_document std::integral_constant::value> {}); } + template + // flawfinder: ignore (a member function, not POSIX read()) + typename std::enable_if::value, void>::type + read(InputType&& input, IntegerType value, Flags&&... flags) = delete; + //////////// // access // //////////// /// the root value (discarded if parsing failed without exceptions) - view_type root() const noexcept + view_type root() const& noexcept { if (!m_data || m_data->discarded) { @@ -931,6 +978,9 @@ class basic_json_document return view_type(m_data.get(), m_data->tape); } + /// deleted: the view of a temporary document would dangle + view_type root() const&& = delete; + bool is_discarded() const noexcept { return !m_data || m_data->discarded; @@ -988,6 +1038,8 @@ class basic_json_document // (edits link to the nodes of the index, which then stays in place) const bool shrink_tape = d.tape != d.inline_tape && d.tape_size != d.tape_cap && d.edits == nullptr; const bool into_header = d.tape_size <= d.inline_cap; + std::vector indexes(d.indexes.capacity() > d.indexes.size() ? d.indexes : std::vector()); + std::vector index_slots(d.index_slots.capacity() > d.index_slots.size() ? d.index_slots : std::vector()); node* fresh = (shrink_tape && !into_header) ? static_cast(::operator new (d.tape_size * sizeof(node))) : d.inline_tape; if (shrink_tape) @@ -1002,6 +1054,14 @@ class basic_json_document d.arena.swap(arena); d.base[1] = d.arena.data(); } + if (d.indexes.capacity() > d.indexes.size()) + { + d.indexes.swap(indexes); + } + if (d.index_slots.capacity() > d.index_slots.size()) + { + d.index_slots.swap(index_slots); + } } /////////// @@ -1226,9 +1286,9 @@ class basic_json_document d.discarded = true; detail::view::parse_failure failure; bool ok = false; - if (NLOHMANN_VIEW_UNLIKELY(size >= 0xFFFFFFF0u)) + if (NLOHMANN_VIEW_UNLIKELY(size > detail::view::max_input_size)) { - failure.code = detail::view::error_code::input_too_large; // LCOV_EXCL_LINE (4 GiB) + failure.code = detail::view::error_code::input_too_large; } else { @@ -1239,9 +1299,11 @@ class basic_json_document d.base[0] = d.src; d.base[1] = d.arena.data(); detail::view::build_object_indexes(d); + std::vector().swap(d.large_objects); // (only needed while parsing) d.discarded = false; return; } + std::vector().swap(d.large_objects); if (allow_exceptions) { detail::view::throw_parse_failure(failure, src, size, comments, trailing_commas); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 72ad2e02a..fa4a46864 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7544,10 +7544,11 @@ namespace detail { /*! -@brief the configuration macros that change the library's behavior +@brief the configuration macros that json_view.hpp reads json.hpp undefines these macros at its end (see macro_unscope.hpp), so code -that builds on the library after it (json_view.hpp) reads them here. Like the +that builds on the library after it (json_view.hpp) reads them here. A macro +is added when the view starts to depend on it. Like the macros, they are part of the ABI namespace, so they always match the basic_json they are used with. */ @@ -7555,8 +7556,6 @@ struct abi_config { /// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0; - /// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON - static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0; }; } // namespace detail @@ -8863,8 +8862,9 @@ NLOHMANN_JSON_NAMESPACE_END #include // uint64_t -#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) - #include // __umulh, _umul128 +#include // memcpy +#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) && (!defined(__SIZEOF_INT128__) || (!defined(__GNUC__) && !defined(__clang__))) + #include // __umulh, _umul128, _BitScanForward64, _BitScanReverse64 #endif // #include @@ -8883,6 +8883,10 @@ inline int count_leading_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_clzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanReverse64(&index, x); + return 63 - static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -8902,6 +8906,10 @@ inline int count_trailing_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_ctzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanForward64(&index, x); + return static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -8949,15 +8957,21 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc #endif } -/// eight bytes as a little-endian word (compilers fold this into one load on -/// little-endian targets; always inlined, as GCC otherwise calls it in the -/// number loops) +/// eight bytes as a little-endian word (a single load on little-endian +/// targets; always inlined, as GCC otherwise calls it in the number loops) JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const unsigned char* b) noexcept { +#if defined(_MSC_VER) || defined(__x86_64__) || defined(__i386__) || (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__) + // the byte order already matches (all MSVC targets are little-endian) + std::uint64_t result = 0; + std::memcpy(&result, b, sizeof(result)); + return result; +#else return static_cast(b[0]) | (static_cast(b[1]) << 8u) | (static_cast(b[2]) << 16u) | (static_cast(b[3]) << 24u) | (static_cast(b[4]) << 32u) | (static_cast(b[5]) << 40u) | (static_cast(b[6]) << 48u) | (static_cast(b[7]) << 56u); +#endif } /// eight bytes as a little-endian word @@ -25172,6 +25186,7 @@ NLOHMANN_JSON_NAMESPACE_END // |_____|_____|_____|_|___| https://github.com/nlohmann/json // // SPDX-FileCopyrightText: 2009 Florian Loitsch +// SPDX-FileCopyrightText: 2025 Victor Zverovich // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT @@ -25247,13 +25262,6 @@ computed from the compressed tables of Zmij beyond it. namespace zmij { -/// significand * 10^exponent -struct decimal -{ - std::uint64_t significand; - int exponent; -}; - /// the compressed powers of ten of Zmij inline const std::array& pow10_minor() noexcept { @@ -25432,18 +25440,6 @@ JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexc return shortest_decimal{integral, dec_exp, static_cast(digit), !round_up && !round_down}; } -/// The shortest decimal in the rounding interval of a positive finite double -/// given by its bits, as one number. The significand can end in zeros. -inline decimal to_decimal(std::uint64_t bits) noexcept -{ - const shortest_decimal d = to_shortest(bits); - if (d.has_digit) - { - return decimal{(d.integral * 10) + d.digit, d.exponent}; - } - return decimal{d.integral, d.exponent + 1}; -} - } // namespace zmij } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -26351,88 +26347,6 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value) grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus); } -/*! -@brief the shortest digits of a positive finite float (other than double): Grisu2 -*/ -template -JSON_HEDLEY_NON_NULL(1) -void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value) -{ - grisu2(buf, len, decimal_exponent, value); -} - -/*! -@brief the shortest digits of a positive finite double: the conversion of -Zmij (see zmij.hpp), which always finds the shortest digits that read back as -the same value (Grisu2 does not for about one double in a thousand), and the -closest of them if there are several - -v = buf * 10^decimal_exponent, as for grisu2() -*/ -JSON_HEDLEY_NON_NULL(1) -inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value) -{ - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); - JSON_ASSERT(std::isfinite(value)); - JSON_ASSERT(value > 0); - - std::uint64_t bits = 0; - std::memcpy(&bits, &value, sizeof(bits)); - zmij::decimal d = zmij::to_decimal(bits); - // without trailing zeros (up to 16): 8, 4, 2, 1 at a time - while (d.significand % 100000000 == 0) - { - d.significand /= 100000000; - d.exponent += 8; - } - if (d.significand % 10000 == 0) - { - d.significand /= 10000; - d.exponent += 4; - } - if (d.significand % 100 == 0) - { - d.significand /= 100; - d.exponent += 2; - } - if (d.significand % 10 == 0) - { - d.significand /= 10; - d.exponent += 1; - } - // at most 17 digits, written from the back two at a time - static constexpr const char* pairs = - "00010203040506070809101112131415161718192021222324252627282930313233343536373839" - "40414243444546474849505152535455565758596061626364656667686970717273747576777879" - "8081828384858687888990919293949596979899"; - std::array digits{}; - std::size_t n = digits.size(); - while (d.significand >= 100) - { - const std::uint64_t two_digits = d.significand % 100; // a variable: GCC calls a cast of the remainder useless where std::uint64_t is std::size_t - const auto i = static_cast(two_digits) * 2; - d.significand /= 100; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - if (d.significand >= 10) - { - const auto i = static_cast(d.significand) * 2; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - else - { - digits[--n] = static_cast('0' + d.significand); - } - len = static_cast(digits.size() - n); - std::memcpy(buf, digits.data() + n, static_cast(len)); - decimal_exponent = d.exponent; -} - /*! @brief appends a decimal representation of e to buf @return a pointer to the element following the exponent. @@ -26835,53 +26749,30 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep return end + (three ? 5 : 4); } -/// the powers of ten up to 10^16 -inline const std::array& powers_of_ten_16() noexcept -{ - static const std::array powers = - { - { - 1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u, - 100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u - } - }; - return powers; -} - /*! -@brief digits * 10^exp, as write_decimal() writes it, for the digits of a -double that need no conversion (count digits, at most 15, the first not 0; -trailing zeros allowed): extended to 16 digits and written by write_shortest() +@brief whether FloatType is an IEEE 754 binary64 type (a double, or a long double +that has the same format, as with MSVC and on Apple's Arm CPUs) -@return a pointer past the text; up to 41 bytes at @a first are written - (some beyond the returned end) +These are the types the conversion of Zmij (see zmij.hpp) is used for; all +others (binary32, or a format the library does not know) use Grisu2. */ -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept +template +constexpr bool has_binary64_format() noexcept { - JSON_ASSERT(digits >= powers_of_ten_16()[static_cast(count - 1)] && count <= 15); - const int scale = 16 - count; - return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast(scale)], exp - scale - 1, 0, false}); + return std::numeric_limits::is_iec559 + && std::numeric_limits::digits == 53 + && std::numeric_limits::max_exponent == 1024 + && sizeof(FloatType) == sizeof(std::uint64_t); } -/// as write_short_decimal(), counting the digits (not 0, less than 10^15) -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept -{ - JSON_ASSERT(digits != 0 && digits < 1000000000000000u); - // floor(log10(2^bits)) + 1 digits, or one less - const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12; - const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast(log2_bound)] ? 1 : 0); - return write_short_decimal(first, digits, count, exp); -} +template +struct is_binary64 : std::integral_constant()> {}; -/// a positive finite float (other than double): Grisu2 and format_buffer() +/// a positive finite float (other than binary64): Grisu2 and format_buffer() template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -char* write_positive(char* first, const char* last, FloatType value) +char* write_positive_grisu2(char* first, const char* last, FloatType value) { JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); static_cast(last); // (only used in the assertion) @@ -26892,7 +26783,7 @@ char* write_positive(char* first, const char* last, FloatType value) // len is the length of the buffer, i.e., the number of decimal digits. int len = 0; int decimal_exponent = 0; - shortest_digits(first, len, decimal_exponent, value); + grisu2(first, len, decimal_exponent, value); JSON_ASSERT(len <= std::numeric_limits::max_digits10); @@ -26908,15 +26799,16 @@ char* write_positive(char* first, const char* last, FloatType value) return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp); } -/// a positive finite double: the shortest digits (Zmij), laid out by +/// a positive finite binary64 number: the shortest digits (Zmij), laid out by /// write_shortest() (through a local buffer if [first, last) is shorter than /// the 41 bytes it may write) +template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_positive(char* first, const char* last, double value) +char* write_positive_zmij(char* first, const char* last, FloatType value) { - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); + static_assert(is_binary64::value, + "internal error: the conversion of Zmij needs IEEE 754 binary64 numbers"); std::uint64_t bits = 0; std::memcpy(&bits, &value, sizeof(bits)); const zmij::shortest_decimal d = zmij::to_shortest(bits); @@ -26931,6 +26823,34 @@ inline char* write_positive(char* first, const char* last, double value) return first + len; } +/// a positive finite binary64 number: Zmij (as a long double has the format of +/// a double here, its bits are those of the double of the same value) +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::true_type /*is_binary64*/) +{ + return write_positive_zmij(first, last, value); +} + +/// a positive finite float of any other format: Grisu2 +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::false_type /*is_binary64*/) +{ + return write_positive_grisu2(first, last, value); +} + +/// a positive finite float: Zmij for binary64 numbers, Grisu2 otherwise +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value) +{ + return write_positive(first, last, value, is_binary64 {}); +} + } // namespace dtoa_impl /*! diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index fe0be75b0..2eaa2f224 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -26,7 +26,7 @@ #ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ -#include // all_of +#include // all_of, min #include // size_t #include // uint32_t #include // memcpy, strlen @@ -62,7 +62,7 @@ -#include // find, find_if, max +#include // find, find_if, max, min #include // array #include // size_t, ptrdiff_t #include // int64_t, uint8_t, uint16_t, uint32_t, uint64_t @@ -88,9 +88,10 @@ #include // uint32_t #include // memcpy #include // less +#include // numeric_limits #include // map #include // unique_ptr -#include // operator new, placement new +#include // bad_alloc, operator new, placement new #include // string #include // vector @@ -201,6 +202,11 @@ static_assert(static_cast(value_t::null) == 0 && static_cast(value_t::number_unsigned) == 6 && static_cast(value_t::number_float) == 7, "the node format depends on the numbering of value_t"); +/// The largest input a document accepts, in bytes. Offsets and node counts are +/// 32 bits wide; the limit keeps 16 bytes (the width of the scanner's steps) +/// below 2^32, so that a position one step past the end of the text fits. +static constexpr std::size_t max_input_size = 0xFFFFFFEFu; + /// node flags struct node_flags { @@ -225,7 +231,7 @@ struct node { std::uint8_t kind; ///< value_t, or kind_link std::uint8_t flags; ///< node_flags - std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; objects: number of the hash index; otherwise 0 + std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; objects: number of the hash index (1-based, 0 = none); otherwise 0 std::uint32_t off; ///< source offset (string content, number token, literal, bracket); arena offset if escaped/edited; number of the element sequence if moved std::uint32_t len; ///< string: decoded bytes; float: token bytes; array/object: element count std::uint32_t next; ///< array/object: number of nodes of the subtree (its extent in the enclosing sequence) @@ -254,17 +260,27 @@ inline void make_link(node& n, node* target) noexcept std::memcpy(reinterpret_cast(&n) + 8, static_cast(&target), sizeof(const node*)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) } -/// the converted value of an integer node (stored in len/next) +/// the converted value of an integer node: len is its low half, next its high +/// half (on little-endian targets the two words are the value in memory) NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::uint64_t v = 0; std::memcpy(&v, reinterpret_cast(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) return v; +#else + return static_cast(n.len) | (static_cast(n.next) << 32); +#endif } NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::memcpy(reinterpret_cast(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +#else + n.len = static_cast(v); + n.next = static_cast(v >> 32); +#endif } /// token length of a number node @@ -392,13 +408,30 @@ struct document_data tape_cap = inline_cap; } - /// make room for n nodes; keeps the first tape_size nodes + /// the largest node count whose size in bytes fits a std::size_t + static constexpr std::size_t max_nodes() noexcept + { + return (std::numeric_limits::max)() / sizeof(node); + } + + [[noreturn]] NLOHMANN_VIEW_NOINLINE static void throw_bad_alloc() + { + NLOHMANN_VIEW_THROW(std::bad_alloc()); + } + + /// make room for n nodes; keeps the first tape_size nodes (throws + /// std::bad_alloc for a count that does not fit the address space, + /// instead of wrapping around in n * sizeof(node)) void reserve(std::size_t n) { if (n <= tape_cap) { return; } + if (NLOHMANN_VIEW_UNLIKELY(n > max_nodes())) + { + throw_bad_alloc(); + } node* fresh = static_cast(::operator new (n * sizeof(node))); if (tape_size != 0) { @@ -572,7 +605,8 @@ NLOHMANN_JSON_NAMESPACE_END #else #define NLOHMANN_VIEW_NEON 0 #endif -#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2)) +// (x86 only: other targets can define __SSE2__ as well, e.g., WebAssembly with -msse2, but have no ) +#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__x86_64__) || defined(__i386__) || defined(_M_X64) || defined(_M_IX86)) && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2)) #include #define NLOHMANN_VIEW_SSE2 1 #else @@ -1265,8 +1299,18 @@ class builder const std::uint64_t done = static_cast(at - b) + 1; const std::uint64_t guess = static_cast(n) * static_cast(e - b + 1) / done; const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t + // (n is below 2^32: the input is smaller than 4 GiB; the sum cannot wrap) + const std::uint64_t wanted = (std::max)(grown, static_cast(n) + (n / 2) + 64); + const std::uint64_t limit = document_data::max_nodes(); doc.tape_size = n; - doc.reserve((std::max)(static_cast(grown), n + (n / 2) + 64)); + // LCOV_EXCL_START (a node array that fills the address space) + if (NLOHMANN_VIEW_UNLIKELY(n >= limit)) + { + document_data::throw_bad_alloc(); // no room for another node + } + // LCOV_EXCL_STOP + // (a count beyond the limit is cut: the index does not grow beyond what can be addressed) + doc.reserve(static_cast((std::min)(wanted, limit))); return doc.tape; } @@ -1879,7 +1923,10 @@ indent_done: n->flags = flags; n->extra = extra; n->off = static_cast(off); - set_integer_bits(*n, second); + // len is the low half of the second word, next the high half + // (not a native word over both, which swaps them on big-endian) + n->len = static_cast(second); + n->next = static_cast(second >> 32); #endif return n; } @@ -2560,9 +2607,8 @@ template { if (f.code == error_code::input_too_large) { - // LCOV_EXCL_START (4 GiB) - NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr)); - // LCOV_EXCL_STOP + // (the limit is detail::view::max_input_size: 4 GiB minus 16 bytes) + NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4294967280 bytes or more is not supported by json_document", nullptr)); } const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas); // LCOV_EXCL_START (only if parse() accepts what the view rejects: a bug) @@ -2853,6 +2899,8 @@ NLOHMANN_JSON_NAMESPACE_END #include // size_t #include // uint16_t, uint32_t, uint64_t #include // memcmp, memcpy +#include // numeric_limits +#include // integral_constant, is_integral, is_same // #include // #include @@ -2889,8 +2937,14 @@ NLOHMANN_JSON_NAMESPACE_END // large objects). An object with document_data::index_min_members members or // more gets an open-addressing table after parsing; its node stores the // number of the table (1-based) in `extra`. A slot holds the offset of a key -// node from its object node (0: empty). Of duplicate keys, the first is kept, -// as for the linear search. +// node from its object node (0: empty). Of duplicate keys, the last is kept, +// as for the linear search, and as basic_json::parse() does. +// +// The hash is not seeded, so keys chosen to collide could make the build +// quadratic. A key therefore sits at most index_max_displacement slots away +// from its home slot; if a key would sit further away, the table is dropped +// and the object is searched linearly (like a small one). For the same +// reason, a lookup visits at most index_max_displacement + 1 slots. NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -2898,6 +2952,11 @@ namespace detail namespace view { +/// the farthest a key may sit from its home slot (a table with at most half of +/// its slots in use gives random keys a distance of about 50 for millions of +/// members; and every member costs at most this many steps while building) +constexpr std::size_t index_max_displacement = 64; + /// hash of a key: its bytes, eight at a time, in a fixed byte order inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept { @@ -2935,27 +2994,44 @@ inline void build_object_index(document_data& d, node* obj) d.index_slots.resize(start + cap, 0); std::uint32_t* const slots = d.index_slots.data() + start; const std::size_t mask = cap - 1; + bool degenerate = false; for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1)) { const char* const key = d.str(*k); const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t) std::size_t i = static_cast(hash) & mask; bool duplicate = false; + std::size_t distance = 0; while (slots[i] != 0) { const node* const other = obj + slots[i]; if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0)) { - duplicate = true; // keep the first + duplicate = true; // keep the last: the key's slot now leads to this member + slots[i] = static_cast(k - obj); + break; + } + if (++distance > index_max_displacement) + { + degenerate = true; // too many keys share a home region break; } i = (i + 1) & mask; } + if (degenerate) + { + break; + } if (!duplicate) { slots[i] = static_cast(k - obj); } } + if (degenerate) + { + d.index_slots.resize(start); // no table: the object is searched linearly + return; + } d.indexes.push_back(document_data::object_index{start, static_cast(mask)}); obj->extra = static_cast(d.indexes.size()); } @@ -2969,7 +3045,7 @@ inline void build_object_indexes(document_data& d) } } -/// the key node of the first member with this key of an indexed object, or +/// the key node of the last member with this key of an indexed object, or /// nullptr inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept { @@ -2977,7 +3053,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c const std::uint32_t* const slots = d.index_slots.data() + ix.start; const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t) std::size_t i = static_cast(hash) & ix.mask; - for (;;) + for (std::size_t distance = 0; distance <= index_max_displacement; ++distance) { const std::uint32_t s = slots[i]; if (s == 0) @@ -2991,6 +3067,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c } i = (i + 1) & ix.mask; } + return nullptr; // (no key sits further from its home slot) } } // namespace view @@ -3062,8 +3139,9 @@ class short_key std::uint64_t m_b = 0; }; -/// the key node of the first member of an object with the given key, or -/// nullptr; most keys are rejected by their length, from the index alone +/// the key node of the last member of an object with the given key, or +/// nullptr (the last one, as materialize() and parse() keep it); most keys are +/// rejected by their length, from the index alone template const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept { @@ -3074,6 +3152,7 @@ const node* find_member(const document_data& d, const node* object, const char* } const node* const end = nav::end(d, object); const auto* const k = reinterpret_cast(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const node* last = nullptr; if (NLOHMANN_VIEW_LIKELY(n <= 16)) { const short_key probe(k, n); @@ -3081,19 +3160,37 @@ const node* find_member(const document_data& d, const node* object, const char* { if (m->len == n && probe.matches(reinterpret_cast(d.str(*m)))) // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) { - return m; + last = m; } } - return nullptr; + return last; } for (const node* m = nav::first(d, object); m != end; m = document_data::after(m + 1)) { if (m->len == n && std::memcmp(d.str(*m), key, n) == 0) { - return m; + last = m; } } - return nullptr; + return last; +} + +/// whether an integer type is accepted as an array index by the view's +/// operator[] and at(): every integer type but bool and size_t, which has its +/// own overload +template +struct is_index_type : std::integral_constant < bool, + std::is_integral::value && !std::is_same::value && !std::is_same::value > +{}; + +/// an integer as an index: negative values, and values that do not fit a +/// size_t, map to the largest size_t (out of range for every array) +template +SizeType to_index(IntegerType idx) noexcept +{ + const IntegerType zero = 0; + const auto result = static_cast(idx); + return (idx < zero || static_cast(result) != idx) ? (std::numeric_limits::max)() : result; } /// the entry of the element of an array at an index below its size (a link @@ -4008,7 +4105,7 @@ NLOHMANN_JSON_NAMESPACE_END #include // basic_string, char_traits, string -#include // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference +#include // decay, integral_constant, is_array, is_const, is_integral, is_lvalue_reference, is_pointer, is_same, remove_reference #include // forward // #include @@ -4028,12 +4125,12 @@ namespace view /// how a document takes its input enum class input_kind { - move_string, ///< rvalue std::string: owned without a copy + move_string, ///< non-const rvalue std::string: owned without a copy c_string, ///< const char* (NUL-terminated): borrowed char_array, ///< char array (e.g. a string literal): borrowed borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed - copy_range, ///< rvalue contiguous byte container: copied - adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer + copy_range, ///< rvalue contiguous byte container (a const rvalue std::string too): copied + adapter, ///< streams, wide strings, and the rest of what the library's input adapter reads: read into a buffer }; template @@ -4052,13 +4149,20 @@ struct classify_input static constexpr input_kind value = std::is_array::value ? input_kind::char_array : std::is_pointer::value ? input_kind::c_string - : (is_rvalue && std::is_same::value) ? input_kind::move_string + : (is_rvalue && !std::is_const::value && std::is_same::value) ? input_kind::move_string : (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range : is_bytes ? input_kind::copy_range : input_kind::adapter; // NOLINTEND(readability-avoid-nested-conditional-operator) }; +/// an integer type other than bool: a length passed where a flag is expected +template +struct is_integer_not_bool : std::is_integral {}; + +template<> +struct is_integer_not_bool : std::false_type {}; + /// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel) template struct is_std_string : std::false_type {}; @@ -4921,7 +5025,13 @@ class output_buffer void finish() { - m_out.resize(static_cast(m_pos - m_out.data())); + const auto size = static_cast(m_pos - m_out.data()); + m_out.resize(size); + // do not keep a buffer that was sized for a much larger output + if (m_out.capacity() > 1024 && m_out.capacity() / 2 > size) + { + m_out.shrink_to_fit(); + } } NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n) @@ -5644,12 +5754,6 @@ class basic_json_view return type() == value_t::discarded; } - /// false for discarded views - explicit operator bool() const noexcept - { - return m_node != nullptr; - } - /// the name of the type, as basic_json::type_name() const char* type_name() const noexcept { @@ -5709,13 +5813,18 @@ class basic_json_view // element access // //////////////////// - /// the value of the member with this key (the first one, should the key - /// occur more than once); a discarded view if there is none. Throws - /// type_error.305 if this is not an object. + /// the value of the member with this key (the last one, should the key + /// occur more than once); a discarded view if there is none, or if this + /// is a discarded view (so that v["a"]["b"] is safe). Throws type_error.305 + /// if this is any other value but an object. NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view operator[](string_view_t key) const { if (NLOHMANN_VIEW_UNLIKELY(!is_object())) { + if (is_discarded()) + { + return basic_json_view(); + } detail::view::throw_type_error(305, "cannot use operator[] with a string argument with ", type_name()); } return lookup(key); @@ -5732,31 +5841,43 @@ class basic_json_view } /// the element at this index; a discarded view if the index is out of - /// range. Throws type_error.305 if this is not an array. + /// range, or if this is a discarded view. Throws type_error.305 if this is + /// any other value but an array. basic_json_view operator[](size_type idx) const { if (NLOHMANN_VIEW_UNLIKELY(!is_array())) { + if (is_discarded()) + { + return basic_json_view(); + } detail::view::throw_type_error(305, "cannot use operator[] with a numeric argument with ", type_name()); } return idx < m_node->len ? basic_json_view(m_doc, navigation::value(detail::view::element_at(*m_doc, m_node, idx))) : basic_json_view(); } - /// (an int argument would be ambiguous between size_type and const char*) - basic_json_view operator[](int idx) const + /// any other integer type (int, unsigned, long, std::int64_t, ...; a + /// single overload for size_type alone would be ambiguous for all of them + /// and for const char*); negative values are out of range + template < typename IntegerType, typename std::enable_if < detail::view::is_index_type::value, int >::type = 0 > + basic_json_view operator[](IntegerType idx) const { - return operator[](static_cast(idx)); + return operator[](detail::view::to_index(idx)); } /// the value a JSON pointer refers to; a discarded view if a key is - /// missing or an index is out of range. Other errors throw what const - /// basic_json::operator[] throws. + /// missing or an index is out of range, or if this is a discarded view. + /// Other errors throw what const basic_json::operator[] throws. basic_json_view operator[](const json_pointer& ptr) const { + if (NLOHMANN_VIEW_UNLIKELY(is_discarded())) + { + return basic_json_view(); + } return detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::unchecked); } - /// the value of the member with this key (the first one, should the key + /// the value of the member with this key (the last one, should the key /// occur more than once). Throws type_error.304 if this is not an object, /// and out_of_range.403 if there is no such member. basic_json_view at(string_view_t key) const @@ -5766,7 +5887,7 @@ class basic_json_view detail::view::throw_type_error(304, "cannot use at() with ", type_name()); } const basic_json_view r = lookup(key); - if (NLOHMANN_VIEW_UNLIKELY(!r)) + if (NLOHMANN_VIEW_UNLIKELY(r.is_discarded())) { detail::view::throw_out_of_range(403, detail::concat("key '", std::string(key.data(), key.size()), "' not found")); } @@ -5798,9 +5919,12 @@ class basic_json_view return basic_json_view(m_doc, navigation::value(detail::view::element_at(*m_doc, m_node, idx))); } - basic_json_view at(int idx) const + /// any other integer type, see operator[]; negative values are out of + /// range + template < typename IntegerType, typename std::enable_if < detail::view::is_index_type::value, int >::type = 0 > + basic_json_view at(IntegerType idx) const { - return at(static_cast(idx)); + return at(detail::view::to_index(idx)); } /// the value a JSON pointer refers to; throws what basic_json::at() @@ -5811,7 +5935,7 @@ class basic_json_view } /// the member with this key converted to T, or the default value if there - /// is no such member (the first one, should the key occur more than + /// is no such member (the last one, should the key occur more than /// once). Throws type_error.306 if this is not an object. template < typename T, typename std::enable_if < !std::is_same::type, const char*>::value, int >::type = 0 > T value(string_view_t key, const T& default_value) const @@ -5821,7 +5945,7 @@ class basic_json_view detail::view::throw_type_error(306, "cannot use value() with ", type_name()); } const basic_json_view r = lookup(key); - return r ? r.template get() : default_value; + return r.is_discarded() ? default_value : r.template get(); } string_t value(string_view_t key, const char* default_value) const @@ -5840,7 +5964,7 @@ class basic_json_view detail::view::throw_type_error(306, "cannot use value() with ", type_name()); } const basic_json_view r = detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::value); - return r ? r.template get() : default_value; + return r.is_discarded() ? default_value : r.template get(); } string_t value(const json_pointer& ptr, const char* default_value) const @@ -5876,7 +6000,7 @@ class basic_json_view // lookup // //////////// - /// an iterator to the member with this key (the first one, should the + /// an iterator to the member with this key (the last one, should the /// key occur more than once), or end(); end() also for non-objects iterator find(string_view_t key) const { @@ -5918,7 +6042,7 @@ class basic_json_view /// basic_json::contains()) bool contains(const json_pointer& ptr) const { - return static_cast(detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains)); + return !detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains).is_discarded(); } /// 1 if this is an object with a member with this key, else 0 (duplicate @@ -6175,7 +6299,7 @@ class basic_json_view } /// the number of source bytes of this value (estimated for values with - /// decoded strings) + /// decoded strings); the estimate sizes the output buffer of dump() std::size_t source_extent() const noexcept { if (editable() && m_doc->edits != nullptr) @@ -6183,20 +6307,33 @@ class basic_json_view // positions of moved and new values are not source offsets return m_node == m_doc->tape ? m_doc->size + m_doc->edits->text_used : 64; } - const node* const next = document_data::after(m_node); - const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0; - if (!in_source) + const node* const end = m_doc->tape + m_doc->tape_size; + if ((m_node->flags & detail::view::node_flags::storage) != 0) { return m_node->len; } - if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off) + // the value ends where the next node in the source begins; nodes + // with decoded strings (their offset is in the arena) are skipped, + // but only a few of them, to keep the walk short + const node* next = document_data::after(m_node); + for (int skipped = 0; next != end && skipped < 16; ++skipped, ++next) { - return next->off - m_node->off; + if ((next->flags & detail::view::node_flags::storage) == 0) + { + return next->off >= m_node->off ? next->off - m_node->off : 0; + } } - return m_doc->size - m_node->off; + if (next == end) + { + return m_doc->size - m_node->off; + } + // the end is unknown: assume a few bytes per node, the output buffer + // grows should the value be larger + const auto nodes = static_cast(document_data::after(m_node) - m_node); + return (std::min)(m_doc->size - m_node->off, static_cast(1024) + nodes * 16); } - /// the value of the first member with this key, or a discarded view + /// the value of the last member with this key, or a discarded view /// (object required) NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view lookup(string_view_t key) const noexcept { @@ -6330,6 +6467,13 @@ class basic_json_document return d; } + /// parse(ptr, len) does not compile: len would convert to allow_exceptions + /// and ptr be read as a C string (as for the overloads of parse_copy, + /// accept, and read below) + template + static typename std::enable_if::value, basic_json_document>::type + parse(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse [first, last) template::iterator_category>::value, int>::type = 0> @@ -6357,6 +6501,10 @@ class basic_json_document return d; } + template + static typename std::enable_if::value, basic_json_document>::type + parse_copy(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// check whether the input is valid JSON (the result of basic_json::accept) template static bool accept(InputType&& input, const bool ignore_comments = false, const bool ignore_trailing_commas = false) @@ -6366,6 +6514,10 @@ class basic_json_document return !d.is_discarded(); } + template + static typename std::enable_if::value, bool>::type + accept(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse into this document, reusing its memory template // flawfinder: ignore (a member function, not POSIX read()) @@ -6378,12 +6530,17 @@ class basic_json_document std::integral_constant::value> {}); } + template + // flawfinder: ignore (a member function, not POSIX read()) + typename std::enable_if::value, void>::type + read(InputType&& input, IntegerType value, Flags&&... flags) = delete; + //////////// // access // //////////// /// the root value (discarded if parsing failed without exceptions) - view_type root() const noexcept + view_type root() const& noexcept { if (!m_data || m_data->discarded) { @@ -6392,6 +6549,9 @@ class basic_json_document return view_type(m_data.get(), m_data->tape); } + /// deleted: the view of a temporary document would dangle + view_type root() const&& = delete; + bool is_discarded() const noexcept { return !m_data || m_data->discarded; @@ -6449,6 +6609,8 @@ class basic_json_document // (edits link to the nodes of the index, which then stays in place) const bool shrink_tape = d.tape != d.inline_tape && d.tape_size != d.tape_cap && d.edits == nullptr; const bool into_header = d.tape_size <= d.inline_cap; + std::vector indexes(d.indexes.capacity() > d.indexes.size() ? d.indexes : std::vector()); + std::vector index_slots(d.index_slots.capacity() > d.index_slots.size() ? d.index_slots : std::vector()); node* fresh = (shrink_tape && !into_header) ? static_cast(::operator new (d.tape_size * sizeof(node))) : d.inline_tape; if (shrink_tape) @@ -6463,6 +6625,14 @@ class basic_json_document d.arena.swap(arena); d.base[1] = d.arena.data(); } + if (d.indexes.capacity() > d.indexes.size()) + { + d.indexes.swap(indexes); + } + if (d.index_slots.capacity() > d.index_slots.size()) + { + d.index_slots.swap(index_slots); + } } /////////// @@ -6687,9 +6857,9 @@ class basic_json_document d.discarded = true; detail::view::parse_failure failure; bool ok = false; - if (NLOHMANN_VIEW_UNLIKELY(size >= 0xFFFFFFF0u)) + if (NLOHMANN_VIEW_UNLIKELY(size > detail::view::max_input_size)) { - failure.code = detail::view::error_code::input_too_large; // LCOV_EXCL_LINE (4 GiB) + failure.code = detail::view::error_code::input_too_large; } else { @@ -6700,9 +6870,11 @@ class basic_json_document d.base[0] = d.src; d.base[1] = d.arena.data(); detail::view::build_object_indexes(d); + std::vector().swap(d.large_objects); // (only needed while parsing) d.discarded = false; return; } + std::vector().swap(d.large_objects); if (allow_exceptions) { detail::view::throw_parse_failure(failure, src, size, comments, trailing_commas); diff --git a/tests/fuzzing.md b/tests/fuzzing.md index de2d66716..ca305cf48 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -6,7 +6,10 @@ Each parser of the library (JSON, BJData, BON8, BSON, CBOR, MessagePack, and UBJ Additionally, `parse_json_view_fuzzer` (`tests/src/fuzzer-parse_json_view.cpp`) cross-checks `json_document`/`json_view` (the zero-copy, read-only view declared in `json_view.hpp`) against `basic_json` on the same JSON text: it asserts that `json_document::accept` agrees with `json::accept`, that an accepted input materializes to the same value `json::parse` -produces, and that a rejected input makes both parsers throw with an identical `what()`. It takes plain JSON text, so it +produces, and that a rejected input makes both parsers throw with an identical `what()`. It checks this for a +`std::string` input (borrowed, with a NUL after the last byte) and for an exact-size `std::vector` +(borrowed, with nothing after the last byte), and for the `ignore_comments` and `ignore_trailing_commas` options, which +are taken from the low bits of the first input byte (the byte stays part of the text). It takes plain JSON text, so it reuses the `corpus_json` corpus rather than a format of its own. ## What the fuzzers check diff --git a/tests/src/fuzzer-parse_json_view.cpp b/tests/src/fuzzer-parse_json_view.cpp index 59e32a556..9b9f5a02a 100644 --- a/tests/src/fuzzer-parse_json_view.cpp +++ b/tests/src/fuzzer-parse_json_view.cpp @@ -9,7 +9,9 @@ /* This file implements a parser test suitable for fuzz testing. It checks that json_document (the zero-copy, read-only view of a parsed JSON text declared in -json_view.hpp) agrees with basic_json on every input: +json_view.hpp) agrees with basic_json on every input, for the parse options +selected by the low bits of the first input byte (bit 0: ignore_comments, bit 1: +ignore_trailing_commas; the byte stays part of the text): - json_document::accept(data) must equal json::accept(data) - if the input is accepted, json_document::parse(data).root().materialize() @@ -18,12 +20,20 @@ json_view.hpp) agrees with basic_json on every input: enabled) must throw a json::parse_error or json::out_of_range whose what() is identical to the one json::parse(data) throws +This is checked for two kinds of input: a std::string, which the document +borrows and which ends in the NUL the parser uses as sentinel, and an +exact-size byte vector, which has no NUL after its last byte and takes the +parser's bounds-checked path (AddressSanitizer reports any read past the end). + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ #include +#include +#include #include +#include #include #include @@ -35,65 +45,110 @@ drivers. using json = nlohmann::json; using json_document = nlohmann::json_document; -// see http://llvm.org/docs/LibFuzzer.html -extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +namespace { - // json_document::accept only has a single-argument overload; wrap the raw - // bytes in a (borrowed) std::string so the same bytes can be handed to it - const std::string input(reinterpret_cast(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +// what json::parse does with a text: the value, or the message of the exception +struct reference_result +{ + bool accepted = false; + json value{}; + std::string what{}; // NOLINT(readability-redundant-member-init) +}; - const bool accepted_by_json = json::accept(data, data + size); - const bool accepted_by_view = json_document::accept(input); +reference_result parse_reference(const std::uint8_t* data, std::size_t size, bool comments, bool trailing_commas) +{ + reference_result r; + r.accepted = json::accept(data, data + size, comments, trailing_commas); + bool json_threw = false; + try + { + r.value = json::parse(data, data + size, nullptr, true, comments, trailing_commas); + } + catch (const json::parse_error& e) + { + r.what = e.what(); + json_threw = true; + } + catch (const json::out_of_range& e) + { + r.what = e.what(); + json_threw = true; + } + // json::accept and json::parse must agree + assert(json_threw == !r.accepted); + static_cast(json_threw); + return r; +} +// json_document must agree with the reference for this input (a container +// that json_document::parse borrows) +template +void check_input(const Input& input, const reference_result& expected, bool comments, bool trailing_commas) +{ // json_document::accept must agree with json::accept on every input - assert(accepted_by_json == accepted_by_view); + const bool accepted_by_view = json_document::accept(input, comments, trailing_commas); + assert(expected.accepted == accepted_by_view); + static_cast(accepted_by_view); - if (accepted_by_json) + if (expected.accepted) { // both parsers must agree on the resulting value - json const j1 = json::parse(data, data + size); - json_document const doc = json_document::parse(input); + json_document const doc = json_document::parse(input, true, comments, trailing_commas); + assert(!doc.is_discarded()); json const j2 = doc.root().materialize(); - assert(j1 == j2); + assert(expected.value == j2); + static_cast(j2); + + // (without exceptions, the same document) + json_document const quiet = json_document::parse(input, false, comments, trailing_commas); + assert(!quiet.is_discarded()); + assert(quiet.node_count() == doc.node_count()); } else { // both parsers must reject the input the same way when exceptions are used - std::string expected_what; - bool json_threw = false; - try - { - static_cast(json::parse(data, data + size)); - } - catch (const json::parse_error& e) - { - expected_what = e.what(); - json_threw = true; - } - catch (const json::out_of_range& e) - { - expected_what = e.what(); - json_threw = true; - } - assert(json_threw); - bool view_threw = false; try { - static_cast(json_document::parse(input)); + static_cast(json_document::parse(input, true, comments, trailing_commas)); } catch (const json::parse_error& e) { - assert(e.what() == expected_what); + assert(e.what() == expected.what); view_threw = true; } catch (const json::out_of_range& e) { - assert(e.what() == expected_what); + assert(e.what() == expected.what); view_threw = true; } assert(view_threw); + static_cast(view_threw); + + // and without exceptions, the document is discarded + json_document const quiet = json_document::parse(input, false, comments, trailing_commas); + assert(quiet.is_discarded()); + static_cast(quiet); } +} +} // namespace + +// see http://llvm.org/docs/LibFuzzer.html +extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +{ + // the parse options are taken from the low bits of the first byte + const bool comments = size > 0 && (data[0] & 1U) != 0; + const bool trailing_commas = size > 0 && (data[0] & 2U) != 0; + + const reference_result expected = parse_reference(data, size, comments, trailing_commas); + + // a std::string: borrowed, with the NUL of std::string as sentinel + const std::string input(reinterpret_cast(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + check_input(input, expected, comments, trailing_commas); + + // an exact-size byte vector: borrowed, with nothing after its last byte + const std::vector exact(data, data + size); + check_input(exact, expected, comments, trailing_commas); // return 0 - non-zero return values are reserved for future use return 0; diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index 82690a669..e5f82b53c 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -1792,4 +1792,11 @@ TEST_CASE("string scanning kernels") CHECK(nlohmann::detail::count_trailing_zeros(bit) == k); CHECK(nlohmann::detail::count_trailing_zeros(bit | (bit << 1u) | 0x8000000000000000u) == k); } + + // eight bytes as a little-endian word, at any alignment + const unsigned char bytes[16] = {0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF}; + CHECK(nlohmann::detail::read_eight_bytes(bytes) == 0x0807060504030201u); + CHECK(nlohmann::detail::read_eight_bytes(bytes + 1) == 0x0908070605040302u); + CHECK(nlohmann::detail::read_eight_bytes(bytes + 8) == 0xFF0F0E0D0C0B0A09u); + CHECK(nlohmann::detail::read_eight_bytes(reinterpret_cast(bytes) + 3) == 0x0B0A090807060504u); } diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 4acb6f81b..f381c6889 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -19,16 +19,19 @@ using nlohmann::ordered_json_view; #include #include #include +#include #include #include #include #include #include +#include #include #include #include #include #include +#include #include #include #include @@ -39,6 +42,55 @@ using nlohmann::ordered_json_view; namespace { +// the value of a text, through a named document: the views of a temporary +// document would dangle (root() of an rvalue document does not compile) +template +auto materialized(Args&& ... args) -> decltype(std::declval().materialize()) +{ + const Document d = Document::parse(std::forward(args)...); + return d.root().materialize(); +} + +template +auto materialized_copy(Input&& input) -> decltype(std::declval().materialize()) +{ + const Document d = Document::parse_copy(std::forward(input)); + return d.root().materialize(); +} + +// a "byte container" that claims to hold `size` bytes, to reach the limit on +// the size of the input without allocating gigabytes; nothing past the first +// bytes is ever read, because the size is checked before the parse starts +struct oversized_input +{ + using value_type = char; + std::size_t claimed; + + const char* data() const + { + return "[1]"; + } + + std::size_t size() const + { + return claimed; + } +}; + +// detection of calls that must not compile +template +using parse_call_t = decltype(json_document::parse(std::declval()...)); +template +using parse_copy_call_t = decltype(json_document::parse_copy(std::declval()...)); +template +using accept_call_t = decltype(json_document::accept(std::declval()...)); +template +using read_call_t = decltype(std::declval().read(std::declval()...)); +template +using root_call_t = decltype(std::declval().root()); +template +using bool_conversion_t = decltype(static_cast(std::declval())); + #if !defined(JSON_NOEXCEPTION) // the exception parse() throws for a text, or "" if it accepts it std::string parse_exception(const std::string& text, bool comments = false, bool trailing_commas = false) @@ -152,7 +204,6 @@ TEST_CASE("json_view") CHECK(v.is_primitive() == j.is_primitive()); CHECK(v.is_structured() == j.is_structured()); CHECK(!v.is_discarded()); - CHECK(static_cast(v)); CHECK(v.size() == j.size()); CHECK(v.empty() == j.empty()); CHECK(v.materialize() == j); @@ -160,7 +211,6 @@ TEST_CASE("json_view") const json_view invalid{}; CHECK(invalid.is_discarded()); - CHECK(!static_cast(invalid)); CHECK(invalid.type() == json::value_t::discarded); CHECK(invalid.size() == 0); CHECK(invalid.empty()); @@ -176,19 +226,19 @@ TEST_CASE("json_view") std::string text; g.value(text, 0); CAPTURE(text) - CHECK(json_document::parse(text).root().materialize() == json::parse(text)); + CHECK(materialized(text) == json::parse(text)); // member order as ordered_json::parse keeps it - CHECK(ordered_json_document::parse(text).root().materialize().dump() == ordered_json::parse(text).dump()); + CHECK(materialized(text).dump() == ordered_json::parse(text).dump()); } // duplicate keys: the last value, at the position of the first key - CHECK(json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize() == json::parse(R"({"a":1,"b":2,"a":3})")); - CHECK(ordered_json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize().dump() == R"({"a":3,"b":2})"); + CHECK(materialized(R"({"a":1,"b":2,"a":3})") == json::parse(R"({"a":1,"b":2,"a":3})")); + CHECK(materialized(R"({"a":1,"b":2,"a":3})").dump() == R"({"a":3,"b":2})"); // very deep nesting (iterative, as parse()) const std::string deep = std::string(100000, '[') + std::string(100000, ']'); - CHECK(json_document::parse(deep).root().materialize() == json::parse(deep)); + CHECK(materialized(deep) == json::parse(deep)); #if JSON_DIAGNOSTICS // the parents are set, so errors name the path - const json m = json_document::parse(R"({"a":{"b":[1]}})").root().materialize(); + const json m = materialized(R"({"a":{"b":[1]}})"); CHECK_THROWS_WITH_AS(m.at("a").at("b").at(0).at("x"), "[json.exception.type_error.304] (/a/b/0) cannot use at() with number", json::type_error&); #endif } @@ -261,7 +311,7 @@ TEST_CASE("json_view") CHECK(json_document::accept(text)); if (accepted) { - CHECK(float_document::parse(text).root().materialize() == json_float::parse(text)); + CHECK(materialized(text) == json_float::parse(text)); } } float_document f; @@ -275,7 +325,7 @@ TEST_CASE("json_view") CHECK(json_document::accept(with_nul) == json::accept(with_nul)); const std::string nul_in_comment("[1, // c\0\n2]", 12); CHECK(json_document::accept(nul_in_comment, true) == json::accept(nul_in_comment, true)); - CHECK(json_document::parse("\xEF\xBB\xBF[1]").root().materialize() == json::parse("\xEF\xBB\xBF[1]")); + CHECK(materialized("\xEF\xBB\xBF[1]") == json::parse("\xEF\xBB\xBF[1]")); #if !defined(JSON_NOEXCEPTION) CHECK(view_exception("\xEF\xBB") == parse_exception("\xEF\xBB")); #endif @@ -291,18 +341,18 @@ TEST_CASE("json_view") CHECK(!borrowed.owns_source()); CHECK(borrowed.source().data() == text.data()); CHECK(borrowed.root().materialize() == expected); - CHECK(json_document::parse(text.c_str()).root().materialize() == expected); - CHECK(json_document::parse(R"([1, "two", {"three": 3.5}])").root().materialize() == expected); - CHECK(json_document::parse(text.data(), text.data() + text.size()).root().materialize() == expected); + CHECK(materialized(text.c_str()) == expected); + CHECK(materialized(R"([1, "two", {"three": 3.5}])") == expected); + CHECK(materialized(text.data(), text.data() + text.size()) == expected); const std::vector chars(text.begin(), text.end()); CHECK(!json_document::parse(chars).owns_source()); - CHECK(json_document::parse(chars).root().materialize() == expected); + CHECK(materialized(chars) == expected); const std::vector bytes(text.begin(), text.end()); - CHECK(json_document::parse(bytes).root().materialize() == expected); + CHECK(materialized(bytes) == expected); #ifdef JSON_HAS_CPP_17 const std::string_view sv = text; CHECK(!json_document::parse(sv).owns_source()); - CHECK(json_document::parse(sv).root().materialize() == expected); + CHECK(materialized(sv) == expected); #endif // owned @@ -311,15 +361,28 @@ TEST_CASE("json_view") CHECK(from_rvalue.owns_source()); CHECK(from_rvalue.root().materialize() == expected); CHECK(json_document::parse(std::vector(text.begin(), text.end())).owns_source()); + // a const rvalue cannot be moved from, and is not borrowed (it may be a + // temporary): it is copied, as is a const rvalue of any container + const std::string const_text = text; + const json_document from_const_rvalue = json_document::parse(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) + CHECK(from_const_rvalue.owns_source()); + CHECK(from_const_rvalue.source().data() != const_text.data()); + CHECK(from_const_rvalue.root().materialize() == expected); + json_document read_const_rvalue; + read_const_rvalue.read(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) + CHECK(read_const_rvalue.owns_source()); + CHECK(read_const_rvalue.root().materialize() == expected); + const std::vector const_chars(text.begin(), text.end()); + CHECK(json_document::parse(std::move(const_chars)).owns_source()); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) CHECK(json_document::parse_copy(text).owns_source()); - CHECK(json_document::parse_copy(text).root().materialize() == expected); + CHECK(materialized_copy(text) == expected); std::istringstream stream(text); const json_document from_stream = json_document::parse(stream); CHECK(from_stream.owns_source()); CHECK(from_stream.root().materialize() == expected); const std::list list(text.begin(), text.end()); CHECK(json_document::parse(list.begin(), list.end()).owns_source()); - CHECK(json_document::parse(list.begin(), list.end()).root().materialize() == expected); + CHECK(materialized(list.begin(), list.end()) == expected); // iterator pairs: pointers are borrowed, and so are contiguous library // iterators where the input adapter detects them (C++20) @@ -330,14 +393,85 @@ TEST_CASE("json_view") CHECK((from_iterators.source().data() == chars.data()) == contiguous); CHECK(from_iterators.root().materialize() == expected); const std::string padded = "x" + text + "x"; - CHECK(json_document::parse(padded.begin() + 1, padded.end() - 1).root().materialize() == expected); + CHECK(materialized(padded.begin() + 1, padded.end() - 1) == expected); CHECK(json_document::parse(chars.cbegin(), chars.cbegin(), false).is_discarded()); const std::wstring wide = L"[\"\u00e4\u20ac\", 1]"; - CHECK(json_document::parse(wide).root().materialize() == json::parse(wide)); + CHECK(materialized(wide) == json::parse(wide)); CHECK(json_document::parse(static_cast(nullptr), false).is_discarded()); CHECK(json_document::parse("", false).is_discarded()); } + SECTION("integer arguments do not compile") + { + using nlohmann::detail::is_detected; + + // a length is not a flag: parse(ptr, len) would convert len to + // allow_exceptions and read ptr as a C string, which need not end + static_assert(is_detected::value, "parse(ptr, bool) is valid"); + static_assert(is_detected::value, "parse(ptr, bool, bool, bool) is valid"); + static_assert(is_detected::value, "parse(first, last) is valid"); + static_assert(is_detected::value, "parse(first, last, bool) is valid"); + static_assert(!is_detected::value, "parse(ptr, len) must not compile"); + static_assert(!is_detected::value, "parse(ptr, int) must not compile"); + static_assert(!is_detected::value, "parse(ptr, char) must not compile"); + static_assert(!is_detected::value, "parse(ptr, len, bool) must not compile"); + static_assert(!is_detected::value, "parse(string, len) must not compile"); + static_assert(!is_detected&, std::size_t>::value, "parse(vector, len) must not compile"); + + static_assert(is_detected::value, "parse_copy(ptr, bool) is valid"); + static_assert(!is_detected::value, "parse_copy(ptr, len) must not compile"); + + static_assert(is_detected::value, "accept(ptr, bool) is valid"); + static_assert(!is_detected::value, "accept(ptr, len) must not compile"); + + static_assert(is_detected::value, "read(ptr, bool) is valid"); + static_assert(!is_detected::value, "read(ptr, len) must not compile"); + + // json_view has no conversion to bool: unlike basic_json's, it would + // mean "exists", not "is not null"; use is_discarded() + static_assert(!is_detected::value, "json_view must not convert to bool"); + + // the valid calls still work + const char* const text = "[1]"; + CHECK(materialized(text, true) == json::parse(text)); + CHECK(json_document::accept(text, true, true)); + } + + SECTION("root of a temporary document does not compile") + { + using nlohmann::detail::is_detected; + + // the view would dangle: auto v = json_document::parse(text).root(); + static_assert(is_detected::value, "root() of an lvalue is valid"); + static_assert(is_detected::value, "root() of a const lvalue is valid"); + static_assert(!is_detected::value, "root() of an rvalue must not compile"); + static_assert(!is_detected < root_call_t, json_document && >::value, "root() of an rvalue must not compile"); + static_assert(!is_detected < root_call_t, const json_document && >::value, "root() of a const rvalue must not compile"); + static_assert(!is_detected::value, "root() of an rvalue must not compile"); + + // a named document is fine, also after a move + json_document d = json_document::parse("[1]"); + CHECK(d.root().size() == 1); + const json_document moved = std::move(d); + CHECK(moved.root().size() == 1); + } + + SECTION("input size limit") + { + // 32-bit offsets: the limit is 4 GiB minus 16 bytes (a margin below 2^32), + // which is what the exception message and the documentation say + const std::size_t limit = nlohmann::detail::view::max_input_size; + CHECK(limit == std::size_t{4294967279u}); + + const oversized_input input{limit + 1}; + CHECK(!json_document::accept(input)); + CHECK(json_document::parse(input, false).is_discarded()); +#if !defined(JSON_NOEXCEPTION) + json_document d; + CHECK_THROWS_WITH_AS(d = json_document::parse(input), "[json.exception.out_of_range.416] input of 4294967280 bytes or more is not supported by json_document", json::out_of_range&); +#endif + } + SECTION("document lifetime and reuse") { json_document d; @@ -443,7 +577,7 @@ std::string exception_of(F f) // compares a view with the ordered_json value materialize() gives for it: // types, sizes, elements and members (by index, key, and iteration), in -// document order; duplicate keys are found as their first occurrence +// document order; duplicate keys are found as their last occurrence void check_access(const ordered_json_view& v, const ordered_json& j) { REQUIRE(v.type() == j.type()); @@ -460,7 +594,7 @@ void check_access(const ordered_json_view& v, const ordered_json& j) ++i; } CHECK(i == v.size()); - CHECK(!v[v.size()]); + CHECK(v[v.size()].is_discarded()); std::size_t index = 0; for (const auto& item : v.items()) { @@ -484,11 +618,23 @@ void check_access(const ordered_json_view& v, const ordered_json& j) const std::string key(it.key().data(), it.key().size()); CHECK(v.contains(key)); CHECK(v.count(key) == 1); - if (std::find(keys.begin(), keys.end(), key) != keys.end()) + if (std::find(keys.begin(), keys.end(), key) == keys.end()) { - continue; // a duplicate: lookups find the first one + keys.push_back(key); + } + // lookups find the last member with the key, which is this one if + // there is no later one + auto next = it; + ++next; + bool is_last = true; + for (; next != v.end(); ++next) + { + is_last = is_last && next.key() != it.key(); + } + if (!is_last) + { + continue; } - keys.push_back(key); CHECK(v.find(key) == it); CHECK(v[key].materialize() == it->materialize()); CHECK(v.at(key).materialize() == it.value().materialize()); @@ -514,7 +660,7 @@ void check_access(const ordered_json_view& v, const ordered_json& j) CHECK(v.back().materialize() == j.back()); } } - CHECK(!v["not a key in the generated documents"]); + CHECK(v["not a key in the generated documents"].is_discarded()); CHECK(v.find("not a key in the generated documents") == v.end()); } else @@ -579,15 +725,35 @@ TEST_CASE("json_view element access and iteration") #endif } - SECTION("duplicate keys: lookups find the first member, iteration all") + SECTION("duplicate keys: lookups find the last member, iteration all") { const json_document d = json_document::parse(R"({"a":1,"b":2,"a":3})"); const json_view v = d.root(); CHECK(v.size() == 3); - CHECK(v["a"].materialize() == 1); - CHECK(v.at("a").materialize() == 1); - CHECK(v.find("a") == v.begin()); + CHECK(v["a"].materialize() == 3); + CHECK(v.at("a").materialize() == 3); + CHECK(v.find("a") == std::next(v.begin(), 2)); + CHECK(v.find("a").value().materialize() == 3); + CHECK(v.find("b") == std::next(v.begin())); CHECK(v.count("a") == 1); + CHECK(v.contains("a")); + CHECK(v.value("a", 0) == 3); + CHECK(v["a"].materialize() == v.materialize()["a"]); // as materialize() + // keys of every length class (the 16-byte short compare and memcmp) + for (const std::size_t n : + { + 0u, 1u, 3u, 7u, 8u, 15u, 16u, 17u, 40u + }) + { + const std::string key(n, 'k'); + const json_document dk = json_document::parse("{\"" + key + "\":1,\"" + key + "x\":2,\"" + key + "\":3,\"" + key + "\":4}"); + CAPTURE(n) + CHECK(dk.root()[key].materialize() == 4); + CHECK(dk.root().at(key).materialize() == 4); + CHECK(dk.root().find(key) == std::next(dk.root().begin(), 3)); + CHECK(dk.root().value(key, 0) == 4); + CHECK(dk.root()[key + "x"].materialize() == 2); + } std::string order; for (auto it = v.begin(); it != v.end(); ++it) { @@ -638,14 +804,131 @@ TEST_CASE("json_view element access and iteration") // where basic_json has undefined behavior, the view answers safely const json_document d = json_document::parse(R"({"a":[]})"); - CHECK(!d.root()["b"]); - CHECK(!d.root()["a"][0]); + CHECK(d.root()["b"].is_discarded()); + CHECK(d.root()["a"][0].is_discarded()); CHECK_THROWS_WITH_AS(d.root()["a"].front(), "[json.exception.invalid_iterator.214] cannot get value", json::invalid_iterator&); CHECK_THROWS_WITH_AS(d.root()["a"].back(), "[json.exception.invalid_iterator.214] cannot get value", json::invalid_iterator&); const json_view invalid{}; CHECK(invalid.begin() == invalid.end()); CHECK(std::string(invalid.type_name()) == "discarded"); - CHECK_THROWS_WITH_AS(invalid["a"], "[json.exception.type_error.305] cannot use operator[] with a string argument with discarded", json::type_error&); + } + + SECTION("chained access is safe: operator[] of a discarded view is discarded") + { + const json_document d = json_document::parse(R"({"a":{"b":[10,20]},"s":"str"})"); + const json_view v = d.root(); + // missing keys and indexes + CHECK(v["x"].is_discarded()); + CHECK(v["x"]["y"].is_discarded()); + CHECK(v["x"]["y"]["z"].is_discarded()); + CHECK(v["x"][0].is_discarded()); + CHECK(v["x"][0u][1L].is_discarded()); + CHECK(v["a"]["b"][2].is_discarded()); + CHECK(v["a"]["b"][2]["c"].is_discarded()); + CHECK(v["a"]["b"][2][json_view::json_pointer("/c")].is_discarded()); + CHECK(v["x"][json_view::json_pointer("/a/b")].is_discarded()); + CHECK(v["x"][json_view::json_pointer("")].is_discarded()); + CHECK(v["x"]["y"].is_discarded()); + CHECK(v["x"][std::string("y")].is_discarded()); + // a resolvable path still resolves + CHECK(v["a"]["b"][1].materialize() == 20); + CHECK(v[json_view::json_pointer("/a/b/1")].materialize() == 20); + // the discarded view of an unresolved pointer is discarded too + CHECK(v[json_view::json_pointer("/x/y")]["z"].is_discarded()); + CHECK(v[json_view::json_pointer("/a/b/5")][0].is_discarded()); +#if !defined(JSON_NOEXCEPTION) + // type errors on values that are not discarded stay + CHECK_THROWS_WITH_AS(v[0], "[json.exception.type_error.305] cannot use operator[] with a numeric argument with object", json::type_error&); + CHECK_THROWS_WITH_AS(v["a"]["b"]["c"], "[json.exception.type_error.305] cannot use operator[] with a string argument with array", json::type_error&); + CHECK_THROWS_WITH_AS(v["s"]["c"], "[json.exception.type_error.305] cannot use operator[] with a string argument with string", json::type_error&); + CHECK_THROWS_WITH_AS(v["s"][0], "[json.exception.type_error.305] cannot use operator[] with a numeric argument with string", json::type_error&); + CHECK_THROWS_AS(v["a"]["b"][0]["c"], json::type_error&); + CHECK_THROWS_AS(v["s"][json_view::json_pointer("/x")], json::out_of_range&); + // at() keeps throwing on a discarded view + const json_view invalid{}; + CHECK(invalid["a"].is_discarded()); + CHECK(invalid[0].is_discarded()); + CHECK(invalid[json_view::json_pointer("/a")].is_discarded()); + CHECK_THROWS_WITH_AS(invalid.at("a"), "[json.exception.type_error.304] cannot use at() with discarded", json::type_error&); + CHECK_THROWS_WITH_AS(invalid.at(0), "[json.exception.type_error.304] cannot use at() with discarded", json::type_error&); + CHECK_THROWS_AS(v.at("x").at("y"), json::out_of_range&); + CHECK_THROWS_AS(v["x"].at("y"), json::type_error&); + CHECK_THROWS_AS(v.at(json_view::json_pointer("/x/y")), json::out_of_range&); + CHECK_THROWS_AS(v["x"].at(json_view::json_pointer("/y")), json::out_of_range&); +#endif + } + + SECTION("integer types as array indexes") + { + const json_document d = json_document::parse("[10,20,30]"); + const json_view v = d.root(); + const json j = v.materialize(); + // (compile-time: no overload is ambiguous) + CHECK(v[0].materialize() == 10); + CHECK(v[1].materialize() == 20); + CHECK(v[0u].materialize() == 10); + CHECK(v[1u].materialize() == 20); + CHECK(v[1L].materialize() == 20); + CHECK(v[2UL].materialize() == 30); + CHECK(v[1LL].materialize() == 20); + CHECK(v[2ULL].materialize() == 30); + CHECK(v[static_cast(1)].materialize() == 20); + CHECK(v[static_cast(2)].materialize() == 30); + CHECK(v[static_cast(1)].materialize() == 20); + CHECK(v[static_cast(2)].materialize() == 30); + CHECK(v[std::int8_t(1)].materialize() == 20); + CHECK(v[std::int16_t(2)].materialize() == 30); + CHECK(v[std::int32_t(1)].materialize() == 20); + CHECK(v[std::int64_t(2)].materialize() == 30); + CHECK(v[std::uint32_t(0)].materialize() == 10); + CHECK(v[std::uint64_t(1)].materialize() == 20); + CHECK(v[std::size_t(2)].materialize() == 30); + CHECK(v[std::ptrdiff_t(1)].materialize() == 20); + CHECK(j[0u] == 10); // as basic_json + + CHECK(v.at(0).materialize() == 10); + CHECK(v.at(1u).materialize() == 20); + CHECK(v.at(1L).materialize() == 20); + CHECK(v.at(2LL).materialize() == 30); + CHECK(v.at(2ULL).materialize() == 30); + CHECK(v.at(static_cast(1)).materialize() == 20); + CHECK(v.at(static_cast(2)).materialize() == 30); + CHECK(v.at(std::int32_t(0)).materialize() == 10); + CHECK(v.at(std::uint32_t(0)).materialize() == 10); + CHECK(v.at(std::int64_t(0)).materialize() == 10); + CHECK(v.at(std::uint64_t(1)).materialize() == 20); + CHECK(v.at(std::size_t(2)).materialize() == 30); + CHECK(j.at(std::uint32_t(0)) == 10); // as basic_json + + // out of range, including negative values (no wrap-around) + CHECK(v[3].is_discarded()); + CHECK(v[3u].is_discarded()); + CHECK(v[3L].is_discarded()); + CHECK(v[-1].is_discarded()); + CHECK(v[-1L].is_discarded()); + CHECK(v[-1LL].is_discarded()); + CHECK(v[static_cast(-1)].is_discarded()); + CHECK(v[std::int64_t(-3)].is_discarded()); + CHECK(v[(std::numeric_limits::min)()].is_discarded()); + CHECK(v[(std::numeric_limits::max)()].is_discarded()); + CHECK(v[(std::numeric_limits::max)()].is_discarded()); + CHECK(v[std::numeric_limits::max()].is_discarded()); +#if !defined(JSON_NOEXCEPTION) + CHECK_THROWS_WITH_AS(v.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&); + CHECK_THROWS_WITH_AS(v.at(3u), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&); + CHECK_THROWS_WITH_AS(v.at(std::int64_t(3)), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&); + CHECK_THROWS_AS(v.at(-1), json::out_of_range&); + CHECK_THROWS_AS(v.at(-1L), json::out_of_range&); + CHECK_THROWS_AS(v.at(std::int64_t(-1)), json::out_of_range&); + CHECK_THROWS_AS(v.at((std::numeric_limits::min)()), json::out_of_range&); + CHECK_THROWS_AS(v.at((std::numeric_limits::max)()), json::out_of_range&); + // not an array + const json_document o = json_document::parse("{}"); + CHECK_THROWS_AS(o.root()[0u], json::type_error&); + CHECK_THROWS_AS(o.root()[1L], json::type_error&); + CHECK_THROWS_AS(o.root().at(std::uint32_t(0)), json::type_error&); + CHECK_THROWS_AS(o.root().at(std::int64_t(0)), json::type_error&); +#endif } SECTION("iterators") @@ -919,10 +1202,12 @@ TEST_CASE("json_view values") CAPTURE(token) const std::string text = "[" + token + "]"; const double b = json::parse(text)[0].get(); - CHECK(bits(json_document::parse(text).root()[0].get()) == bits(b)); + const json_document dd = json_document::parse(text); + CHECK(bits(dd.root()[0].get()) == bits(b)); if (std::abs(b) < 1e38) { - CHECK(bits(nlohmann::basic_json_document::parse(text).root()[0].get()) == bits(json_float::parse(text)[0].get())); + const nlohmann::basic_json_document df = nlohmann::basic_json_document::parse(text); + CHECK(bits(df.root()[0].get()) == bits(json_float::parse(text)[0].get())); } } } @@ -979,7 +1264,8 @@ TEST_CASE("json_view values") CHECK(count == 3); // a duplicate key: the last value, as parse() - CHECK((json_document::parse(R"({"a":1,"a":2})").root().get>() == std::map {{"a", 2}})); + const json_document dup = json_document::parse(R"({"a":1,"a":2})"); + CHECK((dup.root().get>() == std::map {{"a", 2}})); const json_view invalid{}; CHECK_THROWS_WITH_AS(invalid.get(), "[json.exception.type_error.302] type must be number, but is discarded", json::type_error&); @@ -1070,7 +1356,7 @@ TEST_CASE("json_view JSON pointers") else if (at_error.find("out_of_range.401") != std::string::npos || at_error.find("out_of_range.403") != std::string::npos) // NOLINT(abseil-string-find-str-contains) { // undefined behavior for const basic_json::operator[] - CHECK(!v[p]); + CHECK(v[p].is_discarded()); } else { @@ -1164,10 +1450,12 @@ TEST_CASE("json_view dump") } } many += ']'; - CHECK(json_document::parse(many).root().dump() == json::parse(many).dump()); + const json_document many_document = json_document::parse(many); + CHECK(many_document.root().dump() == json::parse(many).dump()); using json_float = nlohmann::basic_json; - CHECK(nlohmann::basic_json_document::parse("[0.1, 1.5e10, 3.4028235e38]").root().dump() == json_float::parse("[0.1, 1.5e10, 3.4028235e38]").dump()); + const nlohmann::basic_json_document float_document = nlohmann::basic_json_document::parse("[0.1, 1.5e10, 3.4028235e38]"); + CHECK(float_document.root().dump() == json_float::parse("[0.1, 1.5e10, 3.4028235e38]").dump()); } SECTION("members in document order, all of them") @@ -1180,7 +1468,42 @@ TEST_CASE("json_view dump") SECTION("deep nesting") { const std::string deep = std::string(100000, '[') + std::string(100000, ']'); - CHECK(json_document::parse(deep).root().dump() == deep); + const json_document deep_document = json_document::parse(deep); + CHECK(deep_document.root().dump() == deep); + } + + SECTION("the output buffer of a small value is small") + { + // an escaped key after the value: its node lies in the arena, so the + // source extent of the value cannot be read from the next node + const std::string big(100000, 'a'); + const std::string text = R"({"small":1,"list":[1,2,3],"k\n":")" + big + R"("})"; + const json_document d = json_document::parse(text); + const auto small = d.root()["small"].dump(); + CHECK(small == "1"); + CHECK(small.capacity() < 4096); + const auto list = d.root()["list"].dump(); + CHECK(list == "[1,2,3]"); + CHECK(list.capacity() < 4096); + CHECK(d.root()["list"].dump(2).capacity() < 4096); + + // the whole document and the large value are unaffected + CHECK(d.root().dump() == ordered_json::parse(text).dump()); + CHECK(d.root()["k\n"].dump() == "\"" + big + "\""); + } + + SECTION("output that outgrows the estimate") + { + // ensure_ascii writes six bytes for each two-byte character + std::string chars; + for (int i = 0; i < 5000; ++i) + { + chars += "\xC3\xA9"; + } + const json_document d = json_document::parse("{\"a\":\"" + chars + R"(","k\n":1})"); + const json expected = json::parse("\"" + chars + "\""); + CHECK(d.root()["a"].dump(-1, ' ', true) == expected.dump(-1, ' ', true)); + CHECK(d.root()["a"].dump(-1, ' ', true).size() == 2 + 5000 * 6); } SECTION("streams and discarded views") @@ -1242,7 +1565,9 @@ TEST_CASE("json_view comparison") { const auto same = [](const char* x, const char* y) { - return json_document::parse(x).root() == json_document::parse(y).root(); + const json_document dx = json_document::parse(x); + const json_document dy = json_document::parse(y); + return dx.root() == dy.root(); }; CHECK(same("1", "1.0")); CHECK(same("[1, -1, 2.5]", "[1.0, -1.0, 25e-1]")); @@ -1257,15 +1582,20 @@ TEST_CASE("json_view comparison") CHECK(same("\"\\u00e9\"", "\"\xc3\xa9\"")); CHECK(!same("null", "false")); CHECK(!same("[]", "{}")); - CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2, "a": 3})").root() == ordered_json_document::parse(R"({"a": 3, "b": 2})").root()); - CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2})").root() != ordered_json_document::parse(R"({"b": 2, "a": 1})").root()); + const ordered_json_document dup = ordered_json_document::parse(R"({"a": 1, "b": 2, "a": 3})"); + const ordered_json_document last = ordered_json_document::parse(R"({"a": 3, "b": 2})"); + CHECK(dup.root() == last.root()); + const ordered_json_document ab = ordered_json_document::parse(R"({"a": 1, "b": 2})"); + const ordered_json_document ba = ordered_json_document::parse(R"({"b": 2, "a": 1})"); + CHECK(ab.root() != ba.root()); // discarded values compare as basic_json's do const json discarded(json::value_t::discarded); CHECK((json_view() == json_view()) == (discarded == discarded)); // NOLINT(readability-container-size-empty): operator== is tested CHECK((json_view() == discarded) == (discarded == discarded)); - CHECK(!(json_view() == json_document::parse("null").root())); // NOLINT(readability-container-size-empty) - CHECK(!(json_document::parse("null").root() == discarded)); + const json_document null_document = json_document::parse("null"); + CHECK(!(json_view() == null_document.root())); // NOLINT(readability-container-size-empty) + CHECK(!(null_document.root() == discarded)); } SECTION("deep nesting") @@ -1276,7 +1606,8 @@ TEST_CASE("json_view comparison") CHECK(a.root() == b.root()); CHECK(a.root() == json::parse(deep)); const std::string other = std::string(100000, '[') + "1" + std::string(100000, ']'); - CHECK(a.root() != json_document::parse(other).root()); + const json_document c = json_document::parse(other); + CHECK(a.root() != c.root()); } } @@ -1301,20 +1632,223 @@ TEST_CASE("json_view large objects") for (std::size_t i = 0; i < members; ++i) { const std::string key = std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\n" : ""); - CHECK(v[key].get() == i); CHECK(v.contains(key)); CHECK(v.find(key).key() == key); - CHECK(v.at(key).get() == i); CHECK(!v.contains(key + "x")); + if (key == "k1") + { + continue; // repeated below: the last member wins + } + CHECK(v[key].get() == i); + CHECK(v.at(key).get() == i); } CHECK(v[""].get_string() == "empty key"); - CHECK(v["k1"].get() == 1); // the first of duplicate keys, as for small objects + CHECK(v["k1"].get_string() == "a duplicate of an earlier key"); // the last of duplicate keys, as for small objects + CHECK(v.at("k1").get_string() == "a duplicate of an earlier key"); CHECK(!v.contains("missing")); CHECK_THROWS_WITH_AS(v.at("missing"), "[json.exception.out_of_range.403] key 'missing' not found", json::out_of_range&); CHECK(v == j); CHECK(v.materialize() == j); } + SECTION("duplicate keys: the last member wins, with and without a table") + { + // an object of `total` members: the keys "k0".."k" in order, then + // three keys repeated twice more (one copy in the middle, one at the + // end), and two keys repeated once; the value of a member is its + // position, so that the last member of a key can be told apart + struct member + { + std::string key; + std::size_t position; + }; + const auto make_members = [](std::size_t total) + { + std::vector keys; + for (std::size_t i = 0; i + 8 < total; ++i) + { + keys.push_back("k" + std::to_string(i)); + } + const std::size_t n = keys.size(); + const std::array triple = {{3, 17, n - 1}}; + const std::array twice = {{5, n / 2}}; + std::vector ordered = keys; + for (const std::size_t i : triple) + { + ordered.insert(ordered.begin() + static_cast(ordered.size() / 2), keys[i]); + } + for (const std::size_t i : twice) + { + ordered.insert(ordered.begin() + static_cast(ordered.size() / 3), keys[i]); + } + for (const std::size_t i : triple) + { + ordered.push_back(keys[i]); + } + std::vector result; + for (std::size_t i = 0; i < ordered.size(); ++i) + { + result.push_back({ordered[i], i}); + } + return result; + }; + + // 100 and 127 members: no table; 128 members and more: a table + for (const std::size_t total : + { + 100u, 127u, 128u, 200u, 5000u + }) + { + CAPTURE(total) + const std::vector members = make_members(total); + REQUIRE(members.size() >= total); + REQUIRE(members.size() >= 100); + std::string text = "{"; + std::map last; + for (const member& m : members) + { + text += (text.size() > 1 ? ",\"" : "\"") + m.key + "\":" + std::to_string(m.position); + last[m.key] = m.position; + } + text += '}'; + REQUIRE(last.size() < members.size()); + + const json_document d = json_document::parse(text); + const json_view v = d.root(); + const json j = json::parse(text); + CHECK(v.size() == members.size()); // every occurrence is visited + for (const auto& entry : last) + { + CAPTURE(entry.first) + const std::size_t expected = entry.second; + CHECK(v[entry.first].get() == expected); + CHECK(v.at(entry.first).get() == expected); + CHECK(v.find(entry.first).value().get() == expected); + CHECK(v.value(entry.first, std::size_t{0}) == expected); + CHECK(v.contains(entry.first)); + CHECK(v.count(entry.first) == 1); + const std::string pointer = "/" + entry.first; + CHECK(v[json::json_pointer(pointer)].get() == expected); + CHECK(v.at(json::json_pointer(pointer)).get() == expected); + CHECK(v.value(json::json_pointer(pointer), std::size_t{0}) == expected); + CHECK(v.contains(json::json_pointer(pointer))); + CHECK(j[entry.first].get() == expected); // as materialize() and parse() keep it + } + CHECK(!v.contains("k")); + CHECK(v["missing"].is_discarded()); + CHECK(v.materialize() == j); + CHECK(v == j); + } + } + + SECTION("colliding keys") + { + // the hash is not seeded, so keys that all land in one place must not + // make the table build quadratic: such an object gets no table and is + // searched linearly + constexpr std::size_t members = 300; + constexpr std::size_t slots = 1024; // the table size for 300 members: the next power of two >= 600 + std::vector colliding; + std::vector spread; + for (std::uint64_t counter = 0; colliding.size() < members || spread.size() < members; ++counter) + { + std::string key(8, 'a'); + for (std::uint64_t x = counter, i = 0; i < 8; ++i, x /= 26) + { + key[i] = static_cast('a' + (x % 26)); + } + const bool lands_in_slot_zero = (nlohmann::detail::view::key_hash(key.data(), key.size()) & (slots - 1)) == 0; + if (lands_in_slot_zero && colliding.size() < members) + { + colliding.push_back(key); + } + else if (!lands_in_slot_zero && spread.size() < members) + { + spread.push_back(key); + } + } + + const auto make_text = [](const std::vector& keys) + { + std::string text = "{"; + for (std::size_t i = 0; i < keys.size(); ++i) + { + text += (i != 0 ? ",\"" : "\"") + keys[i] + "\":" + std::to_string(i % 10); + } + return text + "}"; + }; + const std::string colliding_text = make_text(colliding); + const std::string spread_text = make_text(spread); + + json_document with_collisions = json_document::parse(colliding_text); + json_document without_collisions = json_document::parse(spread_text); + for (const auto* pair : + { + &colliding, &spread + }) + { + const json_view v = (pair == &colliding ? with_collisions : without_collisions).root(); + for (std::size_t i = 0; i < members; ++i) + { + CAPTURE(i) + CHECK(v[(*pair)[i]].get() == i % 10); + CHECK(v.at((*pair)[i]).get() == i % 10); + CHECK(v.find((*pair)[i]).key() == (*pair)[i]); + CHECK(!v.contains((*pair)[i] + "x")); + } + CHECK(!v.contains("missing")); + } + CHECK(with_collisions.root() == json::parse(colliding_text)); + + // only the object with the spread keys got a table (both texts have the + // same length, so the tables are the only difference) + with_collisions.shrink_to_fit(); + without_collisions.shrink_to_fit(); + CHECK(without_collisions.memory_usage() >= with_collisions.memory_usage() + (slots * sizeof(std::uint32_t))); + } + + SECTION("shrink_to_fit releases the tables' spare capacity") + { + const auto make_text = [](int objects, int members) + { + std::string text = "["; + for (int object = 0; object < objects; ++object) + { + text += object != 0 ? ",{" : "{"; + for (int i = 0; i < members + object; ++i) + { + text += (i != 0 ? ",\"" : "\"") + std::to_string(i) + "\":" + std::to_string(i); + } + text += '}'; + } + return text + "]"; + }; + const std::string small_text = make_text(5, 150); + const std::string big_text = make_text(40, 400); + + // reading a big text, and then a small one, leaves the spare capacity + // of the big one: shrink_to_fit() brings the document to the size of + // one parsed from the small text alone + json_document d = json_document::parse(big_text); + const std::size_t big = d.memory_usage(); + d.read(small_text); + CHECK(d.memory_usage() >= big); + d.shrink_to_fit(); + json_document fresh = json_document::parse(small_text); + fresh.shrink_to_fit(); + CHECK(d.memory_usage() == fresh.memory_usage()); + CHECK(d.memory_usage() < big / 2); + CHECK(d.root() == json::parse(small_text)); + for (int object = 0; object < 5; ++object) + { + const json_view v = d.root()[static_cast(object)]; + for (int i = 0; i < 150 + object; ++i) + { + CHECK(v[std::to_string(i)].get() == i); + } + } + } + SECTION("nested, reused, and in arrays") { std::string inner = "{"; diff --git a/tests/src/unit-json_view_builder.cpp b/tests/src/unit-json_view_builder.cpp index 36f3d7c17..8b50a1599 100644 --- a/tests/src/unit-json_view_builder.cpp +++ b/tests/src/unit-json_view_builder.cpp @@ -20,8 +20,10 @@ using nlohmann::json; #include #include #include +#include #include #include +#include #include #include #include @@ -506,3 +508,89 @@ TEST_CASE("json_view builder: strings across vector blocks") } } } + +TEST_CASE("json_view node integer bits") +{ + using nlohmann::detail::view::integer_bits; + using nlohmann::detail::view::set_integer_bits; + + // an integer lives in len (low half) and next (high half), on any byte + // order; a big-endian target must not store the native word over both + SECTION("set_integer_bits and integer_bits") + { + node n = {}; + for (const std::uint64_t v : + { + std::uint64_t{0}, std::uint64_t{1}, std::uint64_t{0xFFFFFFFFu}, std::uint64_t{0x100000000u}, + std::uint64_t{0x0000000200000003u}, std::uint64_t{0x0123456789ABCDEFu}, std::uint64_t{0xFFFFFFFFFFFFFFFEu} + }) + { + CAPTURE(v) + n.kind = 0x5A; + n.flags = 0xA5; + n.extra = 0x1234; + n.off = 0x89ABCDEFu; + set_integer_bits(n, v); + CHECK(integer_bits(n) == v); + CHECK(n.len == static_cast(v)); + CHECK(n.next == static_cast(v >> 32)); + // the other fields are untouched + CHECK(n.kind == 0x5A); + CHECK(n.flags == 0xA5); + CHECK(n.extra == 0x1234); + CHECK(n.off == 0x89ABCDEFu); + } + } + + SECTION("parsed integers") + { + struct integer_case + { + const char* text; + std::uint64_t bits; + }; + for (const integer_case c : + { + integer_case{"[8589934595]", 0x0000000200000003u}, integer_case{"[4294967296]", 0x100000000u}, integer_case{"[4294967295]", 0xFFFFFFFFu}, + integer_case{"[-2]", 0xFFFFFFFFFFFFFFFEu}, integer_case{"[-4294967297]", 0xFFFFFFFEFFFFFFFFu}, integer_case{"[18446744073709551615]", 0xFFFFFFFFFFFFFFFFu}, + integer_case{"[7]", 7u} + }) + { + CAPTURE(c.text) + const built b = build(c.text, false, false, true); + REQUIRE(b.ok); + const node& n = b.data->tape[1]; + CHECK(integer_bits(n) == c.bits); + CHECK(n.len == static_cast(c.bits)); + CHECK(n.next == static_cast(c.bits >> 32)); + } + } +} + +TEST_CASE("json_view node array size limit") +{ + // a node array larger than the address space is refused, not wrapped to a + // small allocation (the size computation overflows on 32-bit targets, and + // for absurd counts everywhere) + std::unique_ptr d(document_data::create(0)); + d->reserve(8); + REQUIRE(d->tape_cap >= 8); + d->tape_size = 2; + const std::size_t cap = d->tape_cap; + node* const tape = d->tape; + +#if !defined(JSON_NOEXCEPTION) + const std::size_t too_many = document_data::max_nodes() + 1; + CHECK_THROWS_AS(d->reserve(too_many), std::bad_alloc&); + CHECK_THROWS_AS(d->reserve((std::numeric_limits::max)()), std::bad_alloc&); + // the array is unchanged + CHECK(d->tape == tape); + CHECK(d->tape_cap == cap); + CHECK(d->tape_size == 2); +#endif + + // the largest count that fits is not refused by the check (nothing is + // allocated for a count that is already there) + d->reserve(cap); + CHECK(d->tape == tape); +} diff --git a/tests/src/unit-to_chars.cpp b/tests/src/unit-to_chars.cpp index 8ced43ad0..d8eb44434 100644 --- a/tests/src/unit-to_chars.cpp +++ b/tests/src/unit-to_chars.cpp @@ -15,6 +15,7 @@ #include using nlohmann::detail::dtoa_impl::reinterpret_bits; +#include #include #include #include @@ -666,13 +667,24 @@ void check_shortest(double v) const std::string text(buf.data(), end); CAPTURE(text) CHECK(parse_double(text) == v); - // the layout is that of format_buffer() for the same digits + // the layout is that of format_buffer() for the digits of Zmij + const auto sd = nlohmann::detail::zmij::to_shortest(reinterpret_bits(v)); + const std::uint64_t significand = sd.has_digit ? (sd.integral * 10) + sd.digit : sd.integral; + int exponent = sd.has_digit ? sd.exponent : sd.exponent + 1; + std::string significand_digits = std::to_string(significand); + while (significand_digits.size() > 1 && significand_digits.back() == '0') + { + significand_digits.pop_back(); + ++exponent; + } std::array reference{}; - int len = 0; - int exponent = 0; - nlohmann::detail::dtoa_impl::shortest_digits(reference.data(), len, exponent, v); - const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), len, exponent, -4, 15); + std::copy(significand_digits.begin(), significand_digits.end(), reference.begin()); + const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), static_cast(significand_digits.size()), exponent, -4, 15); CHECK(text == std::string(reference.data(), static_cast(reference_end - reference.data()))); + // and write_positive() is what to_chars() calls + std::array positive{}; + const char* const positive_end = nlohmann::detail::dtoa_impl::write_positive(positive.data(), positive.data() + positive.size(), v); + CHECK(text == std::string(positive.data(), static_cast(positive_end - positive.data()))); const auto de = digits_and_exponent(text); const std::string& digits = de.first; if (digits.size() > 1) @@ -785,3 +797,58 @@ TEST_CASE("shortest digits of doubles") } } } + +TEST_CASE("choice of the conversion") +{ + using nlohmann::detail::dtoa_impl::is_binary64; + + SECTION("by the format of the type") + { + // Zmij needs binary64 numbers; everything else uses Grisu2 + static_assert(!is_binary64::value, "float is not binary64"); + static_assert(is_binary64::value == (std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53), + "double is binary64 where it is IEEE 754 with 53 digits"); + static_assert(!is_binary64::value, "integers are not binary64"); + static_assert(is_binary64::value == (std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53 && sizeof(long double) == 8), + "long double is binary64 where it has the format of a double"); + CHECK(!is_binary64::value); + CHECK(is_binary64::value); + } + + SECTION("float: Grisu2, double: Zmij") + { + // 5.3165205877497296e+16 is one of the doubles for which Grisu2 does not find the shortest digits + constexpr double value = 5.3165205877497296e+16; + std::array buf{}; + const char* const last = buf.data() + buf.size(); + + char* end = nlohmann::detail::dtoa_impl::write_positive(buf.data(), last, value); + CHECK(std::string(buf.data(), end) == "5.31652058774973e+16"); + end = nlohmann::detail::dtoa_impl::write_positive_grisu2(buf.data(), last, value); + CHECK(std::string(buf.data(), end) == "5.3165205877497296e+16"); + + constexpr float f = 1.1754944e-38f; + end = nlohmann::detail::dtoa_impl::write_positive(buf.data(), last, f); + const std::string dispatched(buf.data(), end); + end = nlohmann::detail::dtoa_impl::write_positive_grisu2(buf.data(), last, f); + CHECK(dispatched == std::string(buf.data(), end)); + } + + SECTION("long double with the format of a double: Zmij") + { + // (on platforms where long double is wider, Grisu2 does not apply either: the snprintf fallback does) + if (std::numeric_limits::digits == 53 && std::numeric_limits::is_iec559) + { + using long_double_json = nlohmann::json::with_float_t; + for (const double d : + { + 5.3165205877497296e+16, 1.0, 0.1, 123456.789, 2.2250738585072014e-308, 1.7976931348623157e+308, -5.3165205877497296e+16 + }) + { + CAPTURE(d) + CHECK(long_double_json(static_cast(d)).dump() == nlohmann::json(d).dump()); + } + CHECK(long_double_json(5.3165205877497296e+16L).dump() == "5.31652058774973e+16"); + } + } +}