Merge branch 'json-view/22-view-dump-fast' into json-view/15-view-bench

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann committed 2026-10-09 18:28:59 +02:00
commit 3e78b5664b
68 files changed
+3734 -964

No files matched your search

+3
View File
@@ -47,3 +47,6 @@ nlohmann_json.spdx
# Bazel-related
MODULE.bazel.lock
# GCC module cache
/gcm.cache/
+1 -1
View File
@@ -1402,7 +1402,7 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I
- The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright &copy; 2008-2009 [Björn Hoehrmann](https://bjoern.hoehrmann.de/) <bjoern@hoehrmann.de>
- The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright &copy; 2009 [Florian Loitsch](https://florian.loitsch.com/)
- The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright &copy; 2025 [Victor Zverovich](https://github.com/vitaut)
- The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, including the conversion of the digits to text by Xiang JunBo and the SIMD instruction sequence of Dougall Johnson, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright &copy; 2025 [Victor Zverovich](https://github.com/vitaut)
- The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/).
- The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0).
- The class contains an adapted version of the Eisel-Lemire algorithm, its table of powers of five, and its digit comparison for long numbers from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright &copy; 2021 The fast_float authors
-1
View File
@@ -186,7 +186,6 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::items', 'Met
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::materialize', 'Method', 'api/basic_json_view/materialize/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::number_format', 'Enum', 'api/basic_json_view/number_format/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::number_token', 'Method', 'api/basic_json_view/number_token/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator bool', 'Method', 'api/basic_json_view/operator_bool/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator<<', 'Operator', 'api/basic_json_view/operator_ltlt/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator[]', 'Operator', 'api/basic_json_view/operator[]/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator==', 'Operator', 'api/basic_json_view/operator_eq/index.html');
@@ -23,10 +23,10 @@ type to use.
## Template parameters
`NumberFloatType`
: the type to store floating-point numbers. The parser converts `#!cpp float`, `#!cpp double`, and a
`#!cpp long double` that is IEEE 754 binary64 itself and other `#!cpp long double` formats with
`#!cpp std::from_chars` or `#!cpp std::strtold`, and serialization falls back to `#!cpp std::snprintf`, so the
type must be `#!cpp float`, `#!cpp double`, or `#!cpp long double`. The
: the type to store floating-point numbers. The type must be `#!cpp float`, `#!cpp double`, or
`#!cpp long double`. The parser converts `#!cpp float`, `#!cpp double`, and a `#!cpp long double` that is IEEE 754
binary64 itself. It converts other `#!cpp long double` formats with `#!cpp std::from_chars` where available, or
with `#!cpp std::strtold` otherwise. Serialization falls back to `#!cpp std::snprintf`. The
[binary formats](../../features/binary_formats/index.md) additionally require `#!cpp float` or `#!cpp double`,
because they have no encoding for `#!cpp long double`. See
[Template Parameter Requirements](../../features/types/template_parameters.md#numberfloattype).
@@ -43,6 +43,11 @@ input's own copy (for inputs that are always read into a buffer) throws.
Linear in the length of the input.
## Notes
An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp accept(ptr, len)`, does not
compile; see [`parse`](parse.md#notes).
## Examples
??? example
@@ -86,8 +86,8 @@ view of a *different* document (overloads 1-2 only; overload 3 always starts fro
!!! info "Duplicate keys"
Overload 1. removes *every* member with `key`, not just the first -- unlike [`set`](set.md), which assigns the
first occurrence and drops the rest. This is why it returns a count rather than a single view: there may be
Overload 1. removes *every* member with `key`, not just the last one that lookups find -- unlike [`set`](set.md),
which assigns that member and drops the rest. This is why it returns a count rather than a single view: there may be
more than one member removed, or none.
Like [`set`](set.md) and [`push_back`](push_back.md), `erase` never moves an element's *value*: a view still
@@ -48,6 +48,7 @@ bookkeeping edits need, and calling any of them on one fails to compile (`#!cpp
- **view_type** - the type of view returned by [`root()`](root.md) (`#!cpp basic_json_view<BasicJsonType, Editable>`)
- **value_t** - the JSON type enumeration, see [`basic_json::value_t`](../basic_json/value_t.md)
- **image_check** - how [`load()`](load.md) validates an image (`full`, `bounds`, `none`), see [`image_check`](load.md#image_check)
## Member functions
@@ -85,9 +85,13 @@ Otherwise throws [`parse_error.116`](../../home/exceptions.md#jsonexceptionparse
## Complexity
Linear in the number of nodes, which are always copied into the document. With `#!cpp check == image_check::full`,
additionally linear in the combined length of the text and the decoded strings; `#!cpp image_check::bounds` and
`#!cpp image_check::none` do not read them.
Linear in the number of nodes, which are always copied into the document. `#!cpp image_check::none` does not read
the text or the decoded strings. `#!cpp image_check::bounds` reads one byte of the text for each float token (its sign,
to check the recorded digits against the token's length), and nothing else of the text or the decoded strings.
With `#!cpp image_check::full`, linear in the size of the image (the nodes, the text, and the decoded strings), plus
sorting the ranges of the strings and of the float tokens (at most one per node): the contents of each distinct range
are checked once, however many nodes refer to it.
## Notes
@@ -108,7 +112,7 @@ How thoroughly `load()` validates `image` before trusting it.
| value | checks | guarantees |
|----------|--------------------------------------------------------------------------------------------------------------|------------|
| `full` | everything the parser itself guarantees: structure and bounds; that every string is valid UTF-8 (and, for a string still in the source text, that it contains no quote, backslash, or control character); and that every number token is well-formed and matches the value stored for it | reading and serializing a checked image is safe and always produces valid JSON, exactly as for a parsed document |
| `full` | everything the parser itself guarantees: structure and bounds; that every string is valid UTF-8 (and, for a string still in the source text, that it contains no quote, backslash, or control character); and that every number token is well-formed and matches the value stored for it; strings and float tokens may share a range only if the ranges are identical (as nodes that share a value do), and never overlap otherwise | reading and serializing a checked image is safe and always produces valid JSON, exactly as for a parsed document |
| `bounds` | structure and bounds only -- that every offset and count in the node index stays inside the image | reading and serializing stay memory-safe, but a crafted image can hold strings that are not valid UTF-8 or that serialize to invalid JSON ([`dump()`](../basic_json_view/dump.md) writes them unchanged or throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316)), and numbers whose values differ from their text |
| `none` | nothing | images from a trusted source only -- reading a damaged image is undefined behavior |
@@ -136,6 +140,9 @@ so those stay safe on a damaged one. It does *not* guarantee that the image desc
that a `full` check would have rejected can make [`dump()`](../basic_json_view/dump.md) write invalid UTF-8 or invalid
JSON, or throw `type_error.316`, and a number can read back with a value that does not match how it is spelled.
Reserve `bounds` for images you already trust to be well-formed, and use it only to skip the extra scan.
An [editable document](../json_editable_document.md) does not take such a string over either:
[`set`](set.md), [`push_back`](push_back.md) and [`insert`](insert.md) throw `type_error.316` when they copy it from
a view of the loaded document, and leave the editable document unchanged.
## Examples
@@ -19,24 +19,33 @@ static basic_json_document parse(IteratorType first, IteratorType last,
1. Deserialize from a compatible input, borrowing or owning it depending on its value category and type (see Notes).
2. Deserialize from a pair of input iterators.
Both overloads accept exactly what [`BasicJsonType::parse()`](../basic_json/parse.md) accepts, with the same
Both overloads accept the same JSON text as [`BasicJsonType::parse()`](../basic_json/parse.md), with the same
`ignore_comments`/`ignore_trailing_commas` options, but build a [`basic_json_document`](index.md) (a flat index into
the input) instead of a tree of `BasicJsonType` values.
the input) instead of a tree of `BasicJsonType` values. The input must be byte-oriented (see the template parameters
below): not every input type of `BasicJsonType::parse()` is supported.
## Template parameters
`InputType`
: A compatible input, for instance:
: A byte-oriented input, one of:
- a `#!cpp std::string`, `#!cpp std::string_view`, or a C-style array of characters
- a pointer to a null-terminated string of single byte characters
- a `#!cpp std::string`, `#!cpp std::string_view`, or a C-style array of single-byte characters
- a pointer to a null-terminated string of single-byte characters (`#!cpp char`, `#!cpp signed char`,
`#!cpp unsigned char`, `#!cpp std::uint8_t`)
- a container for which `#!cpp obj.data()` and `#!cpp obj.size()` give contiguous single-byte access, e.g.
`#!cpp std::vector<char>` or `#!cpp std::vector<std::uint8_t>`
- an `#!cpp std::istream` object, or anything else [`BasicJsonType::parse()`](../basic_json/parse.md) accepts
- an `#!cpp std::istream` object
- a wide string object (`#!cpp std::wstring`, `#!cpp std::u16string`, `#!cpp std::u32string`), which is converted
to UTF-8
Other inputs are not supported: a `#!cpp FILE*`, and pointers to or arrays of wide characters (`#!cpp wchar_t`,
`#!cpp char16_t`, `#!cpp char32_t`) are rejected at compile time by a `#!cpp static_assert`. (Use
[`BasicJsonType::parse()`](../basic_json/parse.md) for these.)
`IteratorType`
: a compatible iterator type, for instance a pair of pointers such as `ptr` and `ptr + len`, or a pair of
`#!cpp std::string::iterator`
: an input iterator type, for instance a pair of pointers such as `ptr` and `ptr + len`, or a pair of
`#!cpp std::string::iterator`; the iterators of single-byte characters are borrowed or read like the byte inputs
above, those of wide characters are converted to UTF-8
## Parameters
@@ -70,8 +79,8 @@ discarded; see [`is_discarded`](is_discarded.md).
Throws the same exception [`BasicJsonType::parse()`](../basic_json/parse.md) throws for the same input and options --
the same exception id, message, and position -- because on a failing input the library's own parser is run on the
same bytes to produce the diagnostic. Additionally throws
[`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the input is 4 GiB or larger, a size
[`BasicJsonType::parse()`](../basic_json/parse.md) does not reject.
[`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the input is 4294967280 bytes (4 GiB
minus 16 bytes) or larger, a size [`BasicJsonType::parse()`](../basic_json/parse.md) does not reject.
## Complexity
@@ -84,8 +93,8 @@ Linear in the length of the input.
| `input` | ownership |
|--------------------------------------------------------------------------------------|--------------------------------------------------------------|
| lvalue byte container (`std::string`, `std::vector<char>`, ...), `std::string_view`, C string, character array | **borrowed** -- `input` must outlive the document |
| rvalue `#!cpp std::string` | **owned**, moved in without a copy |
| rvalue byte container other than `#!cpp std::string` | **owned**, copied |
| non-const rvalue `#!cpp std::string` | **owned**, moved in without a copy |
| other rvalue byte container (including a `#!cpp const` rvalue `#!cpp std::string`) | **owned**, copied |
| stream, wide string, or anything else read through the general input adapter | **owned**, read into a buffer (a stream is read to its end) |
For overload (2), a pair of pointers to single-byte integers (e.g. `#!cpp const char*`, `#!cpp std::uint8_t*`) is
@@ -100,6 +109,12 @@ See [`owns_source`](owns_source.md) to check which happened after a call, and th
**Numbers.** As for [`BasicJsonType::parse()`](../basic_json/parse.md), an integer literal too large for the 64-bit
integer type becomes a floating-point value.
**No lengths.** An integer argument that is not a `#!cpp bool` where the flags are expected -- for example
`#!cpp parse(ptr, len)` -- does not compile (the overload is deleted). Such a call would convert `len` to
`allow_exceptions` and read `ptr` as a null-terminated string, past the end of a buffer that has none. To parse a
buffer of a given length, pass a pair of pointers: `#!cpp parse(ptr, ptr + len)`. The same holds for
[`parse_copy`](parse_copy.md), [`accept`](accept.md), and [`read`](read.md).
## Examples
??? example "Example: (1) borrowed vs. owned input, and errors identical to `BasicJsonType::parse()`"
@@ -51,6 +51,9 @@ Linear in the length of the input.
only differs in that the input is always copied rather than sometimes borrowed. Prefer [`parse()`](parse.md) when the
input's lifetime already covers the document's, since it avoids the copy for borrowed inputs.
An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp parse_copy(ptr, len)`, does not
compile; see [`parse`](parse.md#notes).
## Examples
??? example
@@ -54,6 +54,9 @@ page at a time, and every page costs a page fault the first time it is written.
a 55 MB document into a reused document took about 40 % less time than parsing it into a fresh one. Programs that parse
many documents of similar size should therefore keep one document and call `read()`.
An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp read(ptr, len)`, does not
compile; see [`parse`](parse.md#notes).
## Examples
??? example
@@ -1,10 +1,15 @@
# <small>nlohmann::basic_json_document::</small>root
```cpp
view_type root() const noexcept;
// (1)
view_type root() const& noexcept;
// (2)
view_type root() const&& = delete;
```
Returns a view of the root value of the document.
1. Returns a view of the root value of the document.
2. Deleted: the view of a temporary document would dangle.
## Return value
@@ -21,6 +26,16 @@ Constant.
## Notes
**Lifetime.** A view refers into the document, so the document must outlive it. `root()` can therefore only be called
on a document that has a name (an lvalue); calling it on a temporary does not compile:
```cpp
auto v = json_document::parse(text).root(); // error: the document is destroyed at the end of the statement
auto doc = json_document::parse(text); // OK: keep the document alive
auto v = doc.root();
```
`root()` is a cheap handle into the document's index, not a copy of anything; call it as often as needed. The
returned view is valid under the same conditions as any other view of the document -- see
[Object inspection](../basic_json_view/index.md) -- in particular, it is invalidated by the next
+24 -13
View File
@@ -23,7 +23,8 @@ has `set`; calling it on a read-only `basic_json_document` fails to compile (`#!
1. Replaces the value `target` refers to with `value`.
2. Sets the member `key` of the object `object` to `value`: assigns it if `object` already has a member with this
key -- the first one, should the key occur more than once, and the later duplicates are then dropped (see the
key -- the last one, should the key occur more than once (the member
[`operator[]`](../basic_json_view/operator%5B%5D.md) returns), and the other duplicates are then dropped (see the
[Notes](#notes) below) -- or appends a new member at the end otherwise. A [null](../basic_json_view/is_null.md)
`object` first becomes an empty object.
3. Assigns `value` to the element at index `idx` of the array `array`, which must already exist (`#!cpp idx <
@@ -31,7 +32,10 @@ has `set`; calling it on a read-only `basic_json_document` fails to compile (`#!
4. Sets the value the JSON pointer `ptr` refers to, relative to [`root()`](root.md), to `value`. The *parent* of the
target must already exist: an object member is set as in 2. (added if it does not exist yet), an array element is
assigned as in 3., and a last reference token of `#!cpp "-"`, or equal to the size of the array, appends `value`
instead, exactly as [`push_back`](push_back.md) would. An empty `ptr` sets [`root()`](root.md) itself, as in 1.
instead, exactly as [`push_back`](push_back.md) would. A [null](../basic_json_view/is_null.md) parent becomes what
[`basic_json::operator[]`](../basic_json/operator%5B%5D.md) with a JSON pointer makes of it: an array if the last
reference token is `#!cpp "-"` or consists of digits only (for an index beyond 0, the array is first filled with
null values up to that index), an object otherwise. An empty `ptr` sets [`root()`](root.md) itself, as in 1.
In every overload, `value` is accepted three ways: a [`basic_json_view`](../basic_json_view/index.md) of *any*
document -- read-only or editable, and it does not have to be `target`'s/`object`'s/`array`'s own document -- which
@@ -85,7 +89,8 @@ invalid argument, or `#!cpp std::bad_alloc`) leaves the document completely unch
for the encoding that is not reclaimed. A failure of a later allocation -- while an edited array or object switches
from its parsed layout to a growable block, see [Notes](#notes) -- can still leave a partial effect, such as a
[null](../basic_json_view/is_null.md) `object`/`array` argument already turned into an empty object/array even
though `value` itself was not linked in.
though `value` itself was not linked in. Likewise, a failure of `value` in 4. leaves a null parent that is set with an
index beyond 0 already filled with the null values before the index.
## Exceptions
@@ -111,8 +116,10 @@ though `value` itself was not linked in.
[`parse_error.106`](../../home/exceptions.md#jsonexceptionparse_error106) (a leading `#!cpp '0'`),
[`parse_error.109`](../../home/exceptions.md#jsonexceptionparse_error109) (not a number),
[`out_of_range.410`](../../home/exceptions.md#jsonexceptionout_of_range410) (too large for `size_type`), or
[`out_of_range.404`](../../home/exceptions.md#jsonexceptionout_of_range404) (an empty token). Also throws what 1.
throws for `value`.
[`out_of_range.404`](../../home/exceptions.md#jsonexceptionout_of_range404) (an empty token); the same errors are
thrown for a null parent and a token of digits (the parent is not changed then), and
[`out_of_range.401`](../../home/exceptions.md#jsonexceptionout_of_range401) if the index is 4294967295 or more.
Also throws what 1. throws for `value`.
Every overload also throws [`type_error.319`](../../home/exceptions.md#jsonexceptiontype_error319) if `value` is (or
contains) a binary value -- `BasicJsonType` can hold one, but a `json_document` cannot -- and
@@ -125,24 +132,28 @@ document") if `target`/`object`/`array` is a [discarded](../basic_json_view/is_d
1. Linear in the size of `value` (encoding it into the document's storage): constant for a scalar, linear in the
number of nested values for an array or object. If `target` is itself an array or object that spans more than one
node in its parent's original, unedited layout, and `value` is a scalar, replacing it additionally costs time
linear in the number of elements of that parent, the *first* time -- see [Notes](#notes).
linear in the size of the document, the *first* time (the parent of `target` is looked up from
[`root()`](root.md), and then switches to links) -- see [Notes](#notes). Once the parent has links, `target` is
replaced in constant time: setting every element of a large array one after the other is linear overall. To avoid
the lookup altogether, use 3. (or 2. for an object), which know the parent.
2. Linear in the number of members of `object`, to find an existing member with `key`, plus the complexity of 1. for
`value`.
3. Constant, plus the complexity of 1. for `value`.
4. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
that level or the index into the array (as [`at`](../basic_json_view/at.md)), plus the complexity of 2. or 3. for
the last token.
the last token; for a null parent and an index, linear in the index.
## Notes
!!! info "Duplicate keys"
If `object` already has more than one member with `key` (2.), the *first* one is assigned `value` and every
later member with the same key is removed -- so that a lookup, an iteration, and
[`materialize()`](../basic_json_view/materialize.md) of `object` afterward all agree on a single value for
`key`, the same way [`operator[]`](../basic_json_view/operator%5B%5D.md) already picks the first occurrence of a
duplicate key for reading. See the [Notes on duplicate keys](../basic_json_view/operator%5B%5D.md#notes) of
`operator[]`.
If `object` already has more than one member with `key` (2.), `value` is assigned to the *last* one -- the member
[`operator[]`](../basic_json_view/operator%5B%5D.md), [`at`](../basic_json_view/at.md), and
[`find`](../basic_json_view/find.md) return for reading, so that a view taken from `object["key"]` before the call
shows `value` afterward -- and every other member with the same key is removed. The key stays at the
position of its *first* occurrence, where [`materialize()`](../basic_json_view/materialize.md) puts it as well. A
lookup, an iteration, and `materialize()` of `object` afterward therefore all agree on a single member for `key`. See the
[Notes on duplicate keys](../basic_json_view/operator%5B%5D.md#notes) of `operator[]`.
Setting a member (2.) or an element (3., through 4.) of an array or object whose elements have not been edited
before switches it from its parsed layout to a growable block holding links to its elements; a later
+9 -6
View File
@@ -8,15 +8,18 @@ basic_json_view at(const string_t& key) const;
// (2)
basic_json_view at(size_type idx) const;
basic_json_view at(int idx) const;
template<typename IntegerType>
basic_json_view at(IntegerType idx) const;
// (3)
basic_json_view at(const json_pointer& ptr) const;
```
1. Returns the value of the object member with key `key` -- the first one, should the key occur more than once (see
1. Returns the value of the object member with key `key` -- the last one, should the key occur more than once (see
[Notes on duplicate keys](operator[].md#notes)).
2. Returns the array element at index `idx`.
2. Returns the array element at index `idx`. The template accepts every integer type except `#!cpp bool` and
`#!cpp std::size_t` and forwards to the `size_type` overload, as for [`operator[]`](operator[].md); a negative
`idx` is out of range.
3. Returns the value a JSON pointer `ptr` refers to, starting at this value.
## Parameters
@@ -32,7 +35,7 @@ basic_json_view at(const json_pointer& ptr) const;
## Return value
1. the value of the first member with key `key`
1. the value of the last member with key `key`
2. the element at index `idx`
3. the value `ptr` resolves to, starting at this value
@@ -72,8 +75,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
## Complexity
1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
another, in document order, stopping at the first match. Each comparison first checks the key's length --
already known from the index, without reading the key bytes -- before comparing its content.
another, in document order, scanning all of them, since the last match is wanted. Each comparison first checks the
key's length -- already known from the index, without reading the key bytes -- before comparing its content.
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
on average.
2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the
@@ -4,8 +4,8 @@
basic_json_view() noexcept = default;
```
Creates an invalid (discarded) view: [`type()`](type.md) is `#!cpp value_t::discarded`,
[`is_discarded()`](is_discarded.md) is `#!cpp true`, and `#!cpp explicit operator bool()` is `#!cpp false`.
Creates an invalid (discarded) view: [`type()`](type.md) is `#!cpp value_t::discarded` and
[`is_discarded()`](is_discarded.md) is `#!cpp true`.
This is the only constructor a caller can use directly. Every other view is obtained from a
[`basic_json_document`](../basic_json_document/index.md), via [`root()`](../basic_json_document/root.md) or by
@@ -44,7 +44,6 @@ placeholder for "no value yet" and later be assigned a real view.
## See also
- [is_discarded](is_discarded.md) - return whether the view is invalid
- [operator bool](operator_bool.md) - return whether the view refers to a value
- [root](../basic_json_document/root.md) - the view of a document's root value
## Version history
@@ -24,7 +24,7 @@ Constant.
For an object, iteration visits **every** member, including all occurrences of a duplicate key -- unlike
[`operator[]`](operator[].md), [`at`](at.md), [`find`](find.md), [`contains`](contains.md), and [`count`](count.md),
which all resolve to the *first* member with a given key. See the
which all resolve to the *last* member with a given key. See the
[Notes on duplicate keys](operator[].md#notes) of `operator[]`.
Because objects are iterated in document order rather than sorted by key, the order seen here can differ from what
@@ -33,8 +33,8 @@ No-throw guarantee: this function never throws exceptions.
## Complexity
1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
known from the index, without reading the key bytes -- before comparing its content.
another, in document order, scanning all of them, since the last match is wanted. Each comparison first checks the
key's length -- already known from the index, without reading the key bytes -- before comparing its content.
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
on average.
2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
@@ -23,9 +23,9 @@ No-throw guarantee: this function never throws exceptions.
## Complexity
Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
known from the index, without reading the key bytes -- before comparing its content.
Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another,
in document order, scanning all of them, since the last match is wanted. Each comparison first checks the key's length
-- already known from the index, without reading the key bytes -- before comparing its content.
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on
average.
@@ -38,7 +38,7 @@ Unlike [`BasicJsonType::count()`](../basic_json/count.md), whose return value ca
an `ObjectType` that allows multiple entries per key, `count()` here never does: it is exactly
[`contains()`](contains.md) as `#!cpp 0`/`#!cpp 1`. This holds even if the source text has a duplicate key -- see the
[Notes on duplicate keys](operator[].md#notes) of `operator[]` -- because a `#!cpp count() > 1` result would require
counting every member with a matching key, not just finding the first one.
counting every member with a matching key (the lookup functions resolve to the *last* one).
## Examples
+4 -4
View File
@@ -6,7 +6,7 @@ iterator find(const char* key) const;
iterator find(const string_t& key) const;
```
Finds a member with key `key` -- the first one, should the key occur more than once (see
Finds a member with key `key` -- the last one, should the key occur more than once (see
[Notes on duplicate keys](operator[].md#notes)). If the value is not an object, or no member has this key,
[`end()`](end.md) is returned.
@@ -25,9 +25,9 @@ No-throw guarantee: this function never throws exceptions.
## Complexity
Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
known from the index, without reading the key bytes -- before comparing its content.
Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another,
in document order, scanning all of them, since the last match is wanted. Each comparison first checks the key's length
-- already known from the index, without reading the key bytes -- before comparing its content.
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on
average.
+3 -3
View File
@@ -87,9 +87,9 @@ exception thrown while converting through `materialize()` (the last bullet) is d
!!! info "Duplicate keys"
`#!cpp std::map`/`#!cpp std::unordered_map` conversions keep the *last* value of a repeated key, like
[`materialize()`](materialize.md) and [`BasicJsonType::parse()`](../basic_json/parse.md) do. This is the opposite
of [`operator[]`](operator[].md)/[`at`](at.md)/[`find`](find.md)/[`contains`](contains.md), which resolve to the
*first* occurrence (see the [Notes on duplicate keys](operator[].md#notes)).
[`materialize()`](materialize.md) and [`BasicJsonType::parse()`](../basic_json/parse.md) do. This is the member
[`operator[]`](operator[].md)/[`at`](at.md)/[`find`](find.md)/[`contains`](contains.md) resolve to, too (see the
[Notes on duplicate keys](operator[].md#notes)).
!!! info "No pointers, references, or implicit conversion"
@@ -85,7 +85,6 @@ still refers to it -- including ones taken before the change -- reads the new va
- [**is_primitive**](is_primitive.md) - return whether the type is primitive
- [**is_structured**](is_structured.md) - return whether the type is structured
- [**is_discarded**](is_discarded.md) - return whether the view is invalid
- [**operator bool**](operator_bool.md) - return whether the view refers to a value
### Element access
@@ -9,6 +9,10 @@ view (see [(constructor)](basic_json_view.md)), and for [`root()`](../basic_json
that is itself [discarded](../basic_json_document/is_discarded.md) -- in particular, the root of a failed
[`parse()`](../basic_json_document/parse.md) with `allow_exceptions` set to `#!cpp false`.
A discarded view is also what [`operator[]`](operator[].md) returns for a missing key, an index out of range, or a
JSON pointer that cannot be resolved, and for any access on a view that is itself discarded (so a chain such as
`#!cpp v["a"]["b"]` is safe). [`at`](at.md) throws instead.
## Return value
`#!cpp true` if the view is discarded, `#!cpp false` otherwise.
@@ -23,8 +27,9 @@ Constant.
## Notes
`#!cpp v.is_discarded()` and `#!cpp !static_cast<bool>(v)` are equivalent; use whichever reads better at the call
site.
A `basic_json_view` is not convertible to `#!cpp bool`: such a conversion would mean "refers to a value", whereas
`basic_json` converts to the `#!cpp bool` it holds, so the same code would silently behave differently. Test
`#!cpp !v.is_discarded()` explicitly.
## Examples
@@ -45,7 +50,7 @@ site.
## See also
- [operator bool](operator_bool.md) - return whether the view refers to a value
- [operator[]](operator[].md) - access specified element; yields a discarded view where an element is missing
- [(constructor)](basic_json_view.md) - the default constructor creates a discarded view
- [is_discarded (basic_json_document)](../basic_json_document/is_discarded.md) - return whether the last parse failed
- [`BasicJsonType::is_discarded`](../basic_json/is_discarded.md) - the corresponding function of `basic_json`
@@ -48,7 +48,7 @@ Constant.
As for [`begin()`](begin.md)/[`end()`](end.md), `items()` visits **every** member of an object, including all
occurrences of a duplicate key -- unlike [`operator[]`](operator[].md), [`at`](at.md), [`find`](find.md),
[`contains`](contains.md), and [`count`](count.md), which resolve to the *first* member with a given key. See the
[`contains`](contains.md), and [`count`](count.md), which resolve to the *last* member with a given key. See the
[Notes on duplicate keys](operator[].md#notes) of `operator[]`.
!!! danger "Lifetime issues"
@@ -63,8 +63,8 @@ occurrences of a duplicate key -- unlike [`operator[]`](operator[].md), [`at`](a
The example below shows a settings object whose source text records every update to a key as a duplicate
member, in the order they happened. `items()` walks all of them, so the update history is visible, while
[`operator[]`](operator[].md) only ever sees the *first* one and [`materialize()`](materialize.md) -- like
[`BasicJsonType::parse()`](../basic_json/parse.md) -- keeps only the *last*.
[`operator[]`](operator[].md) sees the *last* one, and so does [`materialize()`](materialize.md) -- like
[`BasicJsonType::parse()`](../basic_json/parse.md).
```cpp
--8<-- "examples/basic_json_view__items.cpp"
@@ -8,16 +8,19 @@ basic_json_view operator[](const string_t& key) const;
// (2)
basic_json_view operator[](size_type idx) const;
basic_json_view operator[](int idx) const;
template<typename IntegerType>
basic_json_view operator[](IntegerType idx) const;
// (3)
basic_json_view operator[](const json_pointer& ptr) const;
```
1. Returns the value of the object member with key `key` -- the first one, should the key occur more than once (see
1. Returns the value of the object member with key `key` -- the last one, should the key occur more than once (see
the [Notes](#notes) below) -- or a [discarded](is_discarded.md) view if there is no such member.
2. Returns the array element at index `idx`, or a [discarded](is_discarded.md) view if `idx` is out of range. (The
`#!cpp int` overload only exists so that an integer literal is not ambiguous between this overload and 1.)
2. Returns the array element at index `idx`, or a [discarded](is_discarded.md) view if `idx` is out of range. The
template accepts every integer type except `#!cpp bool` and `#!cpp std::size_t` (`#!cpp int`, `#!cpp unsigned`,
`#!cpp long`, `#!cpp std::int64_t`, ...) and forwards to the `size_type` overload, so that an integer argument is
not ambiguous between that overload and 1; a negative `idx` is out of range.
3. Returns the value a JSON pointer `ptr` refers to, starting at this value, or a [discarded](is_discarded.md) view
wherever resolving it further is not possible without inserting into or extending the document (see
[Return value](#return-value) and [Exceptions](#exceptions) below).
@@ -35,10 +38,12 @@ basic_json_view operator[](const json_pointer& ptr) const;
## Return value
1. the value of the first member with key `key`, or a discarded view if `#!cpp is_object()` is `#!cpp false` or no
member has this key
2. the element at index `idx`, or a discarded view if `#!cpp is_array()` is `#!cpp false` or `#!cpp idx >= size()`
3. the value `ptr` resolves to, starting at this value, or a discarded view for exactly the reference tokens where the
1. the value of the last member with key `key`, or a discarded view if no member has this key (or if this view is
[discarded](is_discarded.md))
2. the element at index `idx`, or a discarded view if `#!cpp idx >= size()` or `idx` is negative (or if this view is
[discarded](is_discarded.md))
3. the value `ptr` resolves to, starting at this value, or a discarded view (also if this view is
[discarded](is_discarded.md)) for exactly the reference tokens where the
**const** overload of [`BasicJsonType::operator[]`](../basic_json/operator%5B%5D.md) invokes undefined behavior for
the same pointer and the same document: an object member that does not exist, or an array index that is out of
range
@@ -49,10 +54,12 @@ Strong exception safety: if an exception is thrown, there are no changes to the
## Exceptions
1. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an object --
1. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an object and
not [discarded](is_discarded.md) --
the same exception, with the same message, that the **const** overload of
[`BasicJsonType::operator[]`](../basic_json/operator%5B%5D.md) throws for a string argument on a non-object value.
2. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an array --
2. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an array and
not [discarded](is_discarded.md) --
the same exception, with the same message, that the **const** overload of
[`BasicJsonType::operator[]`](../basic_json/operator%5B%5D.md) throws for a numeric argument on a non-array value.
3. Throws the same exceptions, with the same messages, that the **const** overload of
@@ -73,9 +80,9 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
## Complexity
1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
another, in document order, stopping at the first match. Each comparison first checks the key's length --
already known from the index, without reading the key bytes -- before comparing its content, so a key of a
different length than `key` is rejected without touching the source text.
another, in document order, scanning all of them, since the last match is wanted. Each comparison first checks the
key's length -- already known from the index, without reading the key bytes -- before comparing its content, so a
key of a different length than `key` is rejected without touching the source text.
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
on average.
2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the
@@ -86,20 +93,30 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
## Notes
Unlike `BasicJsonType::operator[]`, which is undefined behavior (guarded by a
[runtime assertion](../../features/assertions.md)) for a missing key on a **const** value, this operator always
returns a safe, testable result: a [discarded](is_discarded.md) view, which is `#!cpp false` in a boolean context.
[runtime assertion](../../features/assertions.md)) for a missing key on a **const** value, this operator returns a
safe, testable result for a missing key or an index out of range: a [discarded](is_discarded.md) view, which is
tested with [`is_discarded`](is_discarded.md).
There is also no non-const overload that inserts a missing key or extends an array -- a view never modifies the
document.
!!! info "Chained access"
`#!cpp operator[]` on a [discarded](is_discarded.md) view returns a discarded view and does not throw, so a chain
like `#!cpp v["a"]["b"][0]` is safe even if `"a"` or `"b"` is missing: the first missing step makes the whole
result discarded, which is tested once at the end. Type errors on values that are *not* discarded still throw: a
key on an array or a primitive, or an index on an object or a primitive, is `type_error.305` as for
`BasicJsonType`. [`at`](at.md) still throws for a discarded view, as it does for a missing key.
!!! info "Duplicate keys"
If the source text has an object with a duplicate key, `#!cpp operator[]` (and [`at`](at.md), [`find`](find.md),
[`contains`](contains.md), [`count`](count.md)) all resolve to the *first* member with that key, because a
lookup can stop as soon as it finds a match. This is different from
[`materialize()`](materialize.md) (and [`BasicJsonType::parse()`](../basic_json/parse.md)), which replay every
member in order and so end up keeping the *last* value for a repeated key -- there is no reason for them to stop
early. [`begin()`](begin.md)/[`end()`](end.md) and [`items()`](items.md) iterate over *all* members, including
duplicates, in document order. See the example below and [`size()`](size.md#notes).
[`contains`](contains.md), [`count`](count.md), [`value`](value.md), and JSON pointer resolution) all resolve to
the *last* member with that key. This is the member [`materialize()`](materialize.md) (and
[`BasicJsonType::parse()`](../basic_json/parse.md)) keeps, so a lookup in the view and in the materialized value
agree. [`begin()`](begin.md)/[`end()`](end.md) and [`items()`](items.md) iterate over *all* members, including
duplicates, in document order. A lookup in an object without a hash index scans all members for this: it cannot
stop at the first match. The hash index of a larger object (128 members or more) leads to the last member of a key
as well. See the example below and [`size()`](size.md#notes).
!!! info "JSON pointer resolution"
@@ -1,46 +0,0 @@
# <small>nlohmann::basic_json_view::</small>operator bool
```cpp
explicit operator bool() const noexcept;
```
Returns whether this view refers to a value, i.e. the negation of [`is_discarded()`](is_discarded.md). Being
`#!cpp explicit`, this conversion is only considered in a boolean context (`#!cpp if (v)`, `#!cpp !v`, `#!cpp v &&
...`), not for implicit conversions to other types.
## Return value
`#!cpp true` if the view refers to a value, `#!cpp false` if it is [discarded](is_discarded.md).
## Exception safety
No-throw guarantee: this function never throws exceptions.
## Complexity
Constant.
## Examples
??? example
The example below classifies several parsed documents by the type of their root value, without materializing any
of them into a `BasicJsonType` value.
```cpp
--8<-- "examples/basic_json_view__type_predicates.cpp"
```
Output:
```json
--8<-- "examples/basic_json_view__type_predicates.output"
```
## See also
- [is_discarded](is_discarded.md) - return whether the view is invalid
## Version history
- Added in version 3.13.0.
@@ -12,7 +12,7 @@ T value(const json_pointer& ptr, const T& default_value) const;
string_t value(const json_pointer& ptr, const char* default_value) const;
```
1. Returns the value of the object member with key `key` -- the first one, should the key occur more than once (see
1. Returns the value of the object member with key `key` -- the last one, should the key occur more than once (see
[Notes on duplicate keys](operator[].md#notes)) -- converted to `T`, or `default_value` if there is no such member.
2. Returns the value a JSON pointer `ptr` refers to, starting at this value, converted to `T`, or `default_value` if
`ptr` cannot be resolved.
@@ -39,7 +39,7 @@ equivalent) deduce `string_t`, not `const char*`, for their return type and for
## Return value
1. the first member with key `key`, converted to `T`, or `default_value`
1. the last member with key `key`, converted to `T`, or `default_value`
2. the value `ptr` resolves to, converted to `T`, or `default_value`
## Exception safety
@@ -68,8 +68,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
## Complexity
1. Linear in the number of members: as for [`operator[]`](operator[].md#complexity), members are compared one after
another, in document order, stopping at the first match. Plus the complexity of converting the found member to
`T` (see [`get`](get.md)).
another, in document order, scanning all of them, since the last match is wanted. Plus the complexity of converting
the found member to `T` (see [`get`](get.md)).
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
on average.
2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
@@ -8,10 +8,10 @@ int main()
// the default constructor is the only public one: it creates an invalid
// (discarded) view, useful as a "no value yet" placeholder
nlohmann::json_view v;
std::cout << static_cast<bool>(v) << ' ' << v.is_discarded() << '\n';
std::cout << v.is_discarded() << '\n';
// views are trivially copyable handles (two pointers); the document owns
// the actual data
nlohmann::json_view copy = v;
std::cout << static_cast<bool>(copy) << '\n';
std::cout << copy.is_discarded() << '\n';
}
@@ -1,2 +1,2 @@
false true
false
true
true
@@ -7,8 +7,8 @@ int main()
{
// a settings object whose source text records every update to a key as
// a duplicate member. items() visits all of them, in document order, so
// the update history is visible; operator[] only ever sees the first
// one, and materialize() -- like basic_json::parse() -- keeps the last
// the update history is visible; operator[] and materialize() -- like
// basic_json::parse() -- see the last one
json_document updates = json_document::parse(R"({"retries": 1, "timeout": 30, "retries": 5})");
const auto settings = updates.root();
@@ -17,6 +17,6 @@ int main()
std::cout << item.key() << '=' << item.value().materialize().dump() << '\n';
}
std::cout << "first \"retries\" seen by operator[]: " << settings["retries"].materialize().dump() << '\n';
std::cout << "last \"retries\" seen by operator[]: " << settings["retries"].materialize().dump() << '\n';
std::cout << "last \"retries\" kept by materialize(): " << settings.materialize()["retries"].dump() << '\n';
}
@@ -1,5 +1,5 @@
retries=1
timeout=30
retries=5
first "retries" seen by operator[]: 1
last "retries" seen by operator[]: 5
last "retries" kept by materialize(): 5
@@ -22,17 +22,19 @@ int main()
std::cout << user["name"].materialize().dump();
// operator[] on a missing object key gives a discarded view -- test
// it with a plain "if". The const overload of json::operator[]
// it with is_discarded(). The const overload of json::operator[]
// would instead be undefined behavior (guarded by an assertion) for
// a missing key
if (const auto email = user["email"])
const auto email = user["email"];
if (!email.is_discarded())
{
std::cout << " <" << email.materialize().dump() << ">";
}
// the same holds for an array index past the end: a discarded view,
// not undefined behavior
if (const auto first_tag = user["tags"][0])
const auto first_tag = user["tags"][0];
if (!first_tag.is_discarded())
{
std::cout << " #" << first_tag.materialize().dump();
}
@@ -25,7 +25,8 @@ int main()
// a missing key or an out-of-range index along the path gives a
// discarded view, exactly where const json::operator[] would be
// undefined behavior for the same pointer
if (const auto missing = root[json_pointer("/region/servers/5/metrics/cpu")])
const auto missing = root[json_pointer("/region/servers/5/metrics/cpu")];
if (!missing.is_discarded())
{
std::cout << missing.materialize().dump() << '\n';
}
@@ -35,8 +35,8 @@ int main()
// parse without exceptions, are both discarded
nlohmann::json_view invalid;
json_document failed = json_document::parse("not json", /* allow_exceptions */ false);
std::cout << static_cast<bool>(invalid) << ' ' << invalid.is_discarded() << '\n';
std::cout << static_cast<bool>(failed.root()) << ' ' << failed.root().is_discarded() << '\n';
std::cout << invalid.is_discarded() << '\n';
std::cout << failed.root().is_discarded() << '\n';
// type() returns the same value_t enumeration as basic_json::type()
std::cout << (d_object.root().type() == nlohmann::json::value_t::object) << '\n';
@@ -7,6 +7,6 @@ true
true true
true false
false
false true
false true
true
true
true
+13 -3
View File
@@ -81,6 +81,9 @@ Moving the document itself is fine and does **not** invalidate its views: the in
that keeps its address across the move. Take a fresh view from [`root()`](../api/basic_json_document/root.md)
whenever any of the other conditions above was not met.
Because a view dies with its document, [`root()`](../api/basic_json_document/root.md) is not callable on a temporary
document: `#!cpp auto v = json_document::parse(text).root();` does not compile. Give the document a name first.
??? example "Example: borrowed and owned documents, and when views become invalid"
```cpp
@@ -113,7 +116,7 @@ whenever any of the other conditions above was not met.
- **Only 64-bit integers.** `basic_json_document<BasicJsonType>` requires `BasicJsonType::number_integer_t` and
`number_unsigned_t` to both be 64 bits wide; this is a compile-time `#!cpp static_assert`.
- **A 4 GiB input limit.** An input of 4 GiB or more throws
- **A 4 GiB input limit.** An input of 4294967280 bytes (4 GiB minus 16 bytes) or more throws
[`out_of_range.416`](../home/exceptions.md#jsonexceptionout_of_range416), a limit
`#!cpp basic_json::parse()` does not have.
- **A stream is always read to its end.** There is no partial/streaming read of an `#!cpp std::istream`.
@@ -126,14 +129,21 @@ whenever any of the other conditions above was not met.
members in the order they appear in the source text. `basic_json`'s default `object_t` is a `std::map`, which
sorts by key, so iterating a [`materialize()`](../api/basic_json_view/materialize.md)d value can print members in
a different order than iterating the view they came from.
- **Chained access is safe.** [`operator[]`](../api/basic_json_view/operator%5B%5D.md) with a missing key, an index
out of range, or an unresolvable JSON pointer returns a [discarded](../api/basic_json_view/is_discarded.md) view, and
`operator[]` on a discarded view returns a discarded view without throwing: `#!cpp v["a"]["b"][0]` can be tested
once at the end. Type errors on values that exist (a key on an array, an index on an object) still throw, and
[`at`](../api/basic_json_view/at.md) throws for every missing value.
- **Duplicate keys are visible.** If an object in the source text repeats a key,
[`begin()`](../api/basic_json_view/begin.md)/[`end()`](../api/basic_json_view/end.md) and
[`items()`](../api/basic_json_view/items.md) visit *every* occurrence (and [`size()`](../api/basic_json_view/size.md)
counts all of them), while [`operator[]`](../api/basic_json_view/operator%5B%5D.md),
[`at`](../api/basic_json_view/at.md), [`find`](../api/basic_json_view/find.md),
[`contains`](../api/basic_json_view/contains.md), and [`count`](../api/basic_json_view/count.md) resolve to the
*first* occurrence, since a lookup can stop as soon as it finds a match. `basic_json::parse()` (and so
[`materialize()`](../api/basic_json_view/materialize.md)) instead keeps only the *last* value for a repeated key.
*last* occurrence -- the one `basic_json::parse()` (and so
[`materialize()`](../api/basic_json_view/materialize.md)) keeps for a repeated key -- which makes a lookup scan all
members instead of stopping at a match (objects with 128 members or more get a hash index that leads to the last
occurrence directly).
See the [Notes on duplicate keys](../api/basic_json_view/operator%5B%5D.md#notes) of `operator[]`.
- **No [`JSON_DIAGNOSTICS`](../api/macros/json_diagnostics.md) path.** Exceptions thrown by `basic_json_view`'s own
element access and lookup functions never carry the JSON Pointer path `JSON_DIAGNOSTICS` would otherwise add: the
@@ -353,16 +353,21 @@ using array_t = ArrayType<basic_json, AllocatorType<basic_json>>;
### Always required
- A member type `value_type` that is one byte wide and `char`-compatible. The library stores and processes UTF-8
encoded `char` data and passes `data()` to functions that take a `#!cpp const char*`, such as `#!cpp std::strtod`.
encoded `char` data and passes `data()` to functions that take a `#!cpp const char*`, such as `#!cpp std::strtold`
(only used to parse a `#!cpp long double` that is not IEEE 754 binary64, see
[`NumberFloatType`](#numberfloattype)).
`#!cpp std::wstring`, `#!cpp std::u16string`, and `#!cpp std::u32string` are **not** valid choices; see the FAQ on
[wide string handling](../../home/faq.md#wide-string-handling).
- Constructors: default, copy, move, from `#!cpp const char*` (which must not be `#!cpp explicit`), from
`#!cpp (const char*, size_type)`, and from `#!cpp (size_type, char)`; and copy or move assignment.
- Member functions `size()`, `clear()`, `resize(n, c)`, `data()`, `push_back(char)`, and `operator[]`
(const and non-const, returning references). `c_str()` and `back()` are **not** required.
- `data()` must return a pointer to a contiguous, **null-terminated** buffer -- the parser may hand it to
`#!cpp std::strtod`, which reads up to the null character. A type whose `data()` is not null-terminated does not
fail to compile; it can silently misparse floating-point numbers.
- `data()` must return a pointer to a contiguous, **null-terminated** buffer. `#!cpp float`, `#!cpp double`, and a
`#!cpp long double` that is IEEE 754 binary64 are converted by the library itself and do not depend on this. For any
other `NumberFloatType` (a `#!cpp long double` of another format), the parser falls back to `#!cpp std::strtold` when
`#!cpp std::from_chars` is not available or declines the token, and `std::strtold` reads up to the null character. A type whose `data()`
is not null-terminated does not fail to compile; with such a `NumberFloatType` it can silently misparse
floating-point numbers.
- `append(const char*, size_type)`, used by [`dump`](../../api/basic_json/dump.md), and `append(const StringType&)`,
used by the CBOR reader for indefinite-length strings. The library's internal string concatenation additionally has
to append a `#!cpp char` and a `#!cpp const char*`; for each it selects between `append(arg)`, `#!cpp operator+=`,
+4 -3
View File
@@ -210,12 +210,13 @@ packet-beta
- **Navigation** needs no pointers: the elements of an array or object follow its node, and the node after a value's
subtree is `next` nodes further for an array or object, and the next node otherwise (`document_data::after`). Views
step from element to element this way and skip whole subtrees in constant time.
- **Offsets** are 32 bits wide, so a document is limited to 4 GiB (`out_of_range.416`).
- **Offsets** are 32 bits wide, so a document is limited to 4294967279 bytes, 4 GiB minus 16 bytes (a margin below
2^32 for positions one scanner step past the end of the text; `out_of_range.416`).
- **Large objects** (128 members or more) get a hash index after parsing
([`detail/view/object_index.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/object_index.hpp)):
an open-addressing table whose slots hold the distance from the object's node to a key's node, so that a lookup does
not compare every key. The object's `extra` holds the number of its table. Only 65,535 tables fit into `extra`;
objects beyond them are searched linearly.
not compare every key. Of duplicate keys the table leads to the last, as a linear search does. The object's `extra`
holds the number of its table. Only 65,535 tables fit into `extra`; objects beyond them are searched linearly.
For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an
array or object past its subtree:
+5 -4
View File
@@ -824,8 +824,9 @@ does not list an enumerator and it is therefore converted like the first listed
### json.exception.type_error.319
[`basic_json_document::set`](../api/basic_json_document/set.md) and
[`basic_json_document::push_back`](../api/basic_json_document/push_back.md) can store any `basic_json` value except
[`basic_json_document::set`](../api/basic_json_document/set.md),
[`basic_json_document::push_back`](../api/basic_json_document/push_back.md), and
[`basic_json_document::insert`](../api/basic_json_document/insert.md) can store any `basic_json` value except
a binary one: a `json_document` has no representation for [binary values](../features/binary_values.md), which only
ever arise from parsing a binary format or from an explicit [`json::binary`](../api/basic_json/binary.md) value, not
from JSON text.
@@ -1103,7 +1104,7 @@ MessagePack's ext type and BSON's binary subtype are each stored in a single byt
[`basic_json_document::parse()`](../api/basic_json_document/parse.md) and the other parsing functions of
[`basic_json_document`](../api/basic_json_document/index.md) index a value's position in the source text in 32 bits,
so they do not support an input of 4 GiB or more. The same 32-bit limit applies to an **editable** document's own
so they do not support an input of 4294967280 bytes (4 GiB minus 16 bytes) or more. The same 32-bit limit applies to an **editable** document's own
storage: [`set`](../api/basic_json_document/set.md) and [`push_back`](../api/basic_json_document/push_back.md) throw
this exception once the strings and number tokens written by edits reach 4 GiB in total, or once more than
4294967295 arrays/objects have had an element set or appended to them. The same limit applies to an
@@ -1113,7 +1114,7 @@ count, the text, or the decoded strings it would write would individually reach
!!! failure "Example messages"
```
[json.exception.out_of_range.416] input of 4 GiB or more is not supported by json_document
[json.exception.out_of_range.416] input of 4294967280 bytes or more is not supported by json_document
```
```
[json.exception.out_of_range.416] edits of 4 GiB or more are not supported by json_document
+1 -1
View File
@@ -18,7 +18,7 @@ The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed und
The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright &copy; 2009 [Florian Loitsch](https://florian.loitsch.com/)
The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright &copy; 2025 [Victor Zverovich](https://github.com/vitaut)
The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, including the conversion of the digits to text by Xiang JunBo and the SIMD instruction sequence of Dougall Johnson, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright &copy; 2025 [Victor Zverovich](https://github.com/vitaut)
The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/).
-1
View File
@@ -289,7 +289,6 @@ nav:
- 'materialize': api/basic_json_view/materialize.md
- 'number_format': api/basic_json_view/number_format.md
- 'number_token': api/basic_json_view/number_token.md
- 'operator bool': api/basic_json_view/operator_bool.md
- 'operator<<': api/basic_json_view/operator_ltlt.md
- 'operator[]': api/basic_json_view/operator[].md
- 'operator==': api/basic_json_view/operator_eq.md
Binary file not shown.
+3 -4
View File
@@ -15,10 +15,11 @@ namespace detail
{
/*!
@brief the configuration macros that change the library's behavior
@brief the configuration macros that json_view.hpp reads
json.hpp undefines these macros at its end (see macro_unscope.hpp), so code
that builds on the library after it (json_view.hpp) reads them here. Like the
that builds on the library after it (json_view.hpp) reads them here. A macro
is added when the view starts to depend on it. Like the
macros, they are part of the ABI namespace, so they always match the
basic_json they are used with.
*/
@@ -26,8 +27,6 @@ struct abi_config
{
/// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input
static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0;
/// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0;
};
} // namespace detail
+20 -5
View File
@@ -9,8 +9,9 @@
#pragma once
#include <cstdint> // uint64_t
#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
#include <intrin0.h> // __umulh, _umul128
#include <cstring> // memcpy
#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) && (!defined(__SIZEOF_INT128__) || (!defined(__GNUC__) && !defined(__clang__)))
#include <intrin0.h> // __umulh, _umul128, _BitScanForward64, _BitScanReverse64
#endif
#include <nlohmann/detail/macro_scope.hpp> // JSON_HEDLEY_ALWAYS_INLINE, NLOHMANN_JSON_NAMESPACE_BEGIN
@@ -28,6 +29,10 @@ inline int count_leading_zeros(std::uint64_t x) noexcept
{
#if defined(__GNUC__) || defined(__clang__)
return __builtin_clzll(x);
#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
unsigned long index = 0;
_BitScanReverse64(&index, x);
return 63 - static_cast<int>(index);
#else
int n = 0;
for (int shift = 32; shift != 0; shift >>= 1)
@@ -47,6 +52,10 @@ inline int count_trailing_zeros(std::uint64_t x) noexcept
{
#if defined(__GNUC__) || defined(__clang__)
return __builtin_ctzll(x);
#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
unsigned long index = 0;
_BitScanForward64(&index, x);
return static_cast<int>(index);
#else
int n = 0;
for (int shift = 32; shift != 0; shift >>= 1)
@@ -94,15 +103,21 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc
#endif
}
/// eight bytes as a little-endian word (compilers fold this into one load on
/// little-endian targets; always inlined, as GCC otherwise calls it in the
/// number loops)
/// eight bytes as a little-endian word (a single load on little-endian
/// targets; always inlined, as GCC otherwise calls it in the number loops)
JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
{
#if defined(_MSC_VER) || defined(__x86_64__) || defined(__i386__) || (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__)
// the byte order already matches (all MSVC targets are little-endian)
std::uint64_t result = 0;
std::memcpy(&result, b, sizeof(result));
return result;
#else
return static_cast<std::uint64_t>(b[0]) | (static_cast<std::uint64_t>(b[1]) << 8u)
| (static_cast<std::uint64_t>(b[2]) << 16u) | (static_cast<std::uint64_t>(b[3]) << 24u)
| (static_cast<std::uint64_t>(b[4]) << 32u) | (static_cast<std::uint64_t>(b[5]) << 40u)
| (static_cast<std::uint64_t>(b[6]) << 48u) | (static_cast<std::uint64_t>(b[7]) << 56u);
#endif
}
/// eight bytes as a little-endian word
@@ -4,6 +4,7 @@
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2009 Florian Loitsch <https://florian.loitsch.com/>
// SPDX-FileCopyrightText: 2025 Victor Zverovich <https://github.com/vitaut/zmij>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
@@ -939,88 +940,6 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value)
grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus);
}
/*!
@brief the shortest digits of a positive finite float (other than double): Grisu2
*/
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1)
void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value)
{
grisu2(buf, len, decimal_exponent, value);
}
/*!
@brief the shortest digits of a positive finite double: the conversion of
Zmij (see zmij.hpp), which always finds the shortest digits that read back as
the same value (Grisu2 does not for about one double in a thousand), and the
closest of them if there are several
v = buf * 10^decimal_exponent, as for grisu2()
*/
JSON_HEDLEY_NON_NULL(1)
inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value)
{
static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
JSON_ASSERT(std::isfinite(value));
JSON_ASSERT(value > 0);
std::uint64_t bits = 0;
std::memcpy(&bits, &value, sizeof(bits));
zmij::decimal d = zmij::to_decimal(bits);
// without trailing zeros (up to 16): 8, 4, 2, 1 at a time
while (d.significand % 100000000 == 0)
{
d.significand /= 100000000;
d.exponent += 8;
}
if (d.significand % 10000 == 0)
{
d.significand /= 10000;
d.exponent += 4;
}
if (d.significand % 100 == 0)
{
d.significand /= 100;
d.exponent += 2;
}
if (d.significand % 10 == 0)
{
d.significand /= 10;
d.exponent += 1;
}
// at most 17 digits, written from the back two at a time
static constexpr const char* pairs =
"00010203040506070809101112131415161718192021222324252627282930313233343536373839"
"40414243444546474849505152535455565758596061626364656667686970717273747576777879"
"8081828384858687888990919293949596979899";
std::array<char, 20> digits{};
std::size_t n = digits.size();
while (d.significand >= 100)
{
const std::uint64_t two_digits = d.significand % 100; // a variable: GCC calls a cast of the remainder useless where std::uint64_t is std::size_t
const auto i = static_cast<std::size_t>(two_digits) * 2;
d.significand /= 100;
n -= 2;
digits[n] = pairs[i];
digits[n + 1] = pairs[i + 1];
}
if (d.significand >= 10)
{
const auto i = static_cast<std::size_t>(d.significand) * 2;
n -= 2;
digits[n] = pairs[i];
digits[n + 1] = pairs[i + 1];
}
else
{
digits[--n] = static_cast<char>('0' + d.significand);
}
len = static_cast<int>(digits.size() - n);
std::memcpy(buf, digits.data() + n, static_cast<std::size_t>(len));
decimal_exponent = d.exponent;
}
/*!
@brief appends a decimal representation of e to buf
@return a pointer to the element following the exponent.
@@ -1465,11 +1384,30 @@ inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noe
return write_short_decimal(first, digits, count, exp);
}
/// a positive finite float (other than double): Grisu2 and format_buffer()
/*!
@brief whether FloatType is an IEEE 754 binary64 type (a double, or a long double
that has the same format, as with MSVC and on Apple's Arm CPUs)
These are the types the conversion of Zmij (see zmij.hpp) is used for; all
others (binary32, or a format the library does not know) use Grisu2.
*/
template<typename FloatType>
constexpr bool has_binary64_format() noexcept
{
return std::numeric_limits<FloatType>::is_iec559
&& std::numeric_limits<FloatType>::digits == 53
&& std::numeric_limits<FloatType>::max_exponent == 1024
&& sizeof(FloatType) == sizeof(std::uint64_t);
}
template<typename FloatType>
struct is_binary64 : std::integral_constant<bool, has_binary64_format<FloatType>()> {};
/// a positive finite float (other than binary64): Grisu2 and format_buffer()
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value)
char* write_positive_grisu2(char* first, const char* last, FloatType value)
{
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10);
static_cast<void>(last); // (only used in the assertion)
@@ -1480,7 +1418,7 @@ char* write_positive(char* first, const char* last, FloatType value)
// len is the length of the buffer, i.e., the number of decimal digits.
int len = 0;
int decimal_exponent = 0;
shortest_digits(first, len, decimal_exponent, value);
grisu2(first, len, decimal_exponent, value);
JSON_ASSERT(len <= std::numeric_limits<FloatType>::max_digits10);
@@ -1496,15 +1434,16 @@ char* write_positive(char* first, const char* last, FloatType value)
return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp);
}
/// a positive finite double: the shortest digits (Zmij), laid out by
/// a positive finite binary64 number: the shortest digits (Zmij), laid out by
/// write_shortest() (through a local buffer if [first, last) is shorter than
/// the 41 bytes it may write)
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
inline char* write_positive(char* first, const char* last, double value)
char* write_positive_zmij(char* first, const char* last, FloatType value)
{
static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
static_assert(is_binary64<FloatType>::value,
"internal error: the conversion of Zmij needs IEEE 754 binary64 numbers");
std::uint64_t bits = 0;
std::memcpy(&bits, &value, sizeof(bits));
const zmij::shortest_decimal d = zmij::to_shortest(bits);
@@ -1519,6 +1458,34 @@ inline char* write_positive(char* first, const char* last, double value)
return first + len;
}
/// a positive finite binary64 number: Zmij (as a long double has the format of
/// a double here, its bits are those of the double of the same value)
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value, std::true_type /*is_binary64*/)
{
return write_positive_zmij(first, last, value);
}
/// a positive finite float of any other format: Grisu2
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value, std::false_type /*is_binary64*/)
{
return write_positive_grisu2(first, last, value);
}
/// a positive finite float: Zmij for binary64 numbers, Grisu2 otherwise
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value)
{
return write_positive(first, last, value, is_binary64<FloatType> {});
}
} // namespace dtoa_impl
/*!
@@ -36,13 +36,6 @@ computed from the compressed tables of Zmij beyond it.
namespace zmij
{
/// significand * 10^exponent
struct decimal
{
std::uint64_t significand;
int exponent;
};
/// the compressed powers of ten of Zmij
inline const std::array<std::uint64_t, 28>& pow10_minor() noexcept
{
@@ -221,18 +214,6 @@ JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexc
return shortest_decimal{integral, dec_exp, static_cast<unsigned char>(digit), !round_up && !round_down};
}
/// The shortest decimal in the rounding interval of a positive finite double
/// given by its bits, as one number. The significand can end in zeros.
inline decimal to_decimal(std::uint64_t bits) noexcept
{
const shortest_decimal d = to_shortest(bits);
if (d.has_digit)
{
return decimal{(d.integral * 10) + d.digit, d.exponent};
}
return decimal{d.integral, d.exponent + 1};
}
} // namespace zmij
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
+16 -3
View File
@@ -9,7 +9,7 @@
#pragma once
#include <algorithm> // find, find_if, max
#include <algorithm> // find, find_if, max, min
#include <array> // array
#include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // int64_t, uint8_t, uint16_t, uint32_t, uint64_t
@@ -202,8 +202,18 @@ class builder
const std::uint64_t done = static_cast<std::uint64_t>(at - b) + 1;
const std::uint64_t guess = static_cast<std::uint64_t>(n) * static_cast<std::uint64_t>(e - b + 1) / done;
const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t
// (n is below 2^32: the input is smaller than 4 GiB; the sum cannot wrap)
const std::uint64_t wanted = (std::max)(grown, static_cast<std::uint64_t>(n) + (n / 2) + 64);
const std::uint64_t limit = document_data::max_nodes();
doc.tape_size = n;
doc.reserve((std::max)(static_cast<std::size_t>(grown), n + (n / 2) + 64));
// LCOV_EXCL_START (a node array that fills the address space)
if (NLOHMANN_VIEW_UNLIKELY(n >= limit))
{
document_data::throw_bad_alloc(); // no room for another node
}
// LCOV_EXCL_STOP
// (a count beyond the limit is cut: the index does not grow beyond what can be addressed)
doc.reserve(static_cast<std::size_t>((std::min)(wanted, limit)));
return doc.tape;
}
@@ -816,7 +826,10 @@ indent_done:
n->flags = flags;
n->extra = extra;
n->off = static_cast<std::uint32_t>(off);
set_integer_bits(*n, second);
// len is the low half of the second word, next the high half
// (not a native word over both, which swaps them on big-endian)
n->len = static_cast<std::uint32_t>(second);
n->next = static_cast<std::uint32_t>(second >> 32);
#endif
return n;
}
+20 -2
View File
@@ -13,9 +13,10 @@
#include <cstdint> // uint8_t, uint32_t
#include <cstring> // memcpy
#include <functional> // less
#include <limits> // numeric_limits
#include <map> // map
#include <memory> // unique_ptr
#include <new> // operator new, placement new
#include <new> // bad_alloc, operator new, placement new
#include <string> // string
#include <vector> // vector
@@ -123,13 +124,30 @@ struct document_data
tape_cap = inline_cap;
}
/// make room for n nodes; keeps the first tape_size nodes
/// the largest node count whose size in bytes fits a std::size_t
static constexpr std::size_t max_nodes() noexcept
{
return (std::numeric_limits<std::size_t>::max)() / sizeof(node);
}
[[noreturn]] NLOHMANN_VIEW_NOINLINE static void throw_bad_alloc()
{
NLOHMANN_VIEW_THROW(std::bad_alloc());
}
/// make room for n nodes; keeps the first tape_size nodes (throws
/// std::bad_alloc for a count that does not fit the address space,
/// instead of wrapping around in n * sizeof(node))
void reserve(std::size_t n)
{
if (n <= tape_cap)
{
return;
}
if (NLOHMANN_VIEW_UNLIKELY(n > max_nodes()))
{
throw_bad_alloc();
}
node* fresh = static_cast<node*>(::operator new (n * sizeof(node)));
if (tape_size != 0)
{
+209 -84
View File
@@ -17,6 +17,7 @@
#include <string> // string, to_string
#include <type_traits> // decay, enable_if, integral_constant, is_arithmetic, is_convertible, is_floating_point, is_same, is_signed
#include <utility> // forward
#include <vector> // vector
#include <nlohmann/json.hpp>
#include <nlohmann/detail/view/document_data.hpp>
@@ -151,25 +152,23 @@ class editor
{
become_empty(o, value_t::object);
}
// an existing member: assign it (and drop later duplicates, so that
// lookups, iteration, and materialize() agree)
// an existing member: assign the one that lookups find (the last
// one, should the key occur more than once), and drop the others, so
// that lookups, iteration, and materialize() agree. The key stays at
// the position of its first occurrence, as materialize() puts it.
node* slot = nullptr;
bool duplicates = false;
std::size_t matches = 0;
for (const node* k = nav::first(m_doc, o), *end = nav::end(m_doc, o); k != end; k = document_data::after(k + 1))
{
if (key_equals(*k, key))
{
if (slot != nullptr)
{
duplicates = true;
break;
}
slot = const_cast<node*>(nav::value(k + 1)); // NOLINT(cppcoreguidelines-pro-type-const-cast): the nodes belong to this document
++matches;
}
}
if (slot != nullptr)
{
if (duplicates)
if (matches > 1)
{
erase_members(o, key, true);
}
@@ -315,27 +314,38 @@ class editor
return k.len == key.size() && (key.size() == 0 || std::memcmp(m_doc.str(k), key.data(), key.size()) == 0);
}
/// remove the members with this key (all, or all but the first) from an object
std::size_t erase_members(node* o, string_view_t key, bool keep_first)
/// Remove the members with this key from an object: all of them, or all
/// but one. That one stays where the first occurrence is, but holds the
/// value of the last (the one that lookups find, which views may refer to).
std::size_t erase_members(node* o, string_view_t key, bool keep_one)
{
node* const h = block_of(m_doc, o, 0);
node last_value{}; // the entry of the value of the last member
node* const end = h + h->next;
if (keep_one)
{
for (node* r = h + 1; r != end; r += 2)
{
if (key_equals(*r, key))
{
last_value = r[1];
}
}
}
node* w = h + 1;
std::size_t erased = 0;
bool kept = false;
for (node* r = h + 1, *end = h + h->next; r != end; r += 2)
for (node* r = h + 1; r != end; r += 2)
{
const bool match = key_equals(*r, key);
if (match && (kept || !keep_first))
if (match && (kept || !keep_one))
{
++erased;
continue;
}
w[0] = r[0];
w[1] = match ? last_value : r[1];
kept = kept || match;
if (w != r)
{
w[0] = r[0];
w[1] = r[1];
}
w += 2;
}
h->next = static_cast<std::uint32_t>(w - h);
@@ -347,9 +357,10 @@ class editor
/// turn a null into an empty array/object in place
static void become_empty(node* n, value_t k) noexcept
{
const std::uint8_t linked = n->flags & node_flags::linked;
*n = node{};
n->kind = static_cast<std::uint8_t>(k);
n->flags = node_flags::is_new;
n->flags = static_cast<std::uint8_t>(node_flags::is_new | linked);
n->next = 1;
}
@@ -357,13 +368,17 @@ class editor
/// include slot (if known).
void assign(node* slot, const encoded& e, node* parent, bool parent_known)
{
// an entry of a moved sequence links to the slot: it can take any extent
const std::uint8_t linked = slot->flags & node_flags::linked;
if (e.region == nullptr)
{
if (is_container(*slot) && slot->next > 1 && slot != m_doc.tape)
if (is_container(*slot) && slot->next > 1 && slot != m_doc.tape && linked == 0)
{
// The slot spans its old elements in the enclosing sequence, but
// a scalar is one node: the enclosing container first switches to
// links (then the extent of the slot no longer matters).
// links (then the extent of the slot no longer matters). Looking
// for the container is linear in the size of the document, so
// links (which are marked in the slot) avoid it.
node* const p = parent_known ? parent : find_parent(m_doc, slot);
if (p != nullptr && ((p->flags & node_flags::moved) == 0 || moved_capacity(m_doc, p) == 0))
{
@@ -371,6 +386,7 @@ class editor
}
}
*slot = e.scalar;
slot->flags = static_cast<std::uint8_t>(slot->flags | linked);
return;
}
// an array/object: the slot keeps its extent (so that the enclosing
@@ -379,11 +395,21 @@ class editor
const node* const r = e.region;
const std::uint32_t extent = is_container(*slot) ? slot->next : 1;
const bool was_moved = (slot->flags & node_flags::moved) != 0;
// Everything that can throw happens before the slot changes: a slot
// that is a container without the moved flag would show its old
// elements. reserve_moved() makes the set_moved() below, which sets
// the flag, safe; the entry of `regions` exists already (encode()
// added it), so that the assignment at the end does not allocate.
if (!was_moved)
{
reserve_moved(m_doc);
}
slot->kind = r->kind;
slot->extra = 0;
slot->len = r->len;
slot->next = extent;
slot->flags = was_moved ? static_cast<std::uint8_t>(node_flags::moved | node_flags::is_new) : std::uint8_t{0};
// (set_moved() adds the moved flag to a slot that does not have it yet)
slot->flags = static_cast<std::uint8_t>((was_moved ? node_flags::moved | node_flags::is_new : 0) | linked);
set_moved(m_doc, slot, e.region, 0);
edit_state_of(m_doc).regions[e.region] = slot;
}
@@ -600,6 +626,8 @@ class editor
switch (static_cast<value_t>(n.kind))
{
case value_t::string:
// (an editable document only holds valid UTF-8, whatever the check of the other document was)
check_utf8(from.str(n), n.len);
return string_node(from.str(n), n.len);
case value_t::number_integer:
case value_t::number_unsigned:
@@ -663,100 +691,197 @@ class editor
}
}
// The subtrees are walked with an explicit stack (as materialize() does):
// the nesting depth is limited by memory only, not by the call stack.
/// number of nodes of a subtree (containers, keys, scalars)
template<bool E>
static std::size_t count_nodes(const document_data& d, const node* n)
{
if (!is_container(*n))
using walk = navigation<E>;
struct frame
{
return 1;
}
const bool object = n->kind == static_cast<std::uint8_t>(value_t::object);
std::size_t r = 1;
for (const node* c = navigation<E>::first(d, n), *end = navigation<E>::end(d, n); c != end;)
const node* pos; ///< next element, or key of the next member
const node* end;
bool object;
};
std::vector<frame> open;
std::size_t r = 0;
for (;;)
{
const node* const v = object ? c + 1 : c;
r += (object ? 1 : 0) + count_nodes<E>(d, navigation<E>::value(v));
c = document_data::after(v);
++r;
if (is_container(*n))
{
open.push_back(frame{walk::first(d, n), walk::end(d, n), n->kind == static_cast<std::uint8_t>(value_t::object)});
}
// the next value: close finished containers, then step over the key
for (;;)
{
if (open.empty())
{
return r;
}
frame& f = open.back();
if (f.pos == f.end)
{
open.pop_back();
continue;
}
const node* v = f.pos;
if (f.object)
{
++r; // the key
++v;
}
f.pos = document_data::after(v);
n = walk::value(v);
break;
}
}
return r;
}
/// copy a subtree (of any document) as a contiguous sequence; returns its end
/// copy a subtree (of any document) as a contiguous sequence of
/// count_nodes() nodes
template<bool E>
node* fill_nodes(const document_data& d, const node* n, node* out)
void fill_nodes(const document_data& d, const node* n, node* out)
{
if (!is_container(*n))
using walk = navigation<E>;
struct frame
{
*out = copy_scalar(d, *n);
return out + 1;
}
node* const self = out++;
*self = plain_node(static_cast<value_t>(n->kind));
self->len = n->len;
const bool object = n->kind == static_cast<std::uint8_t>(value_t::object);
for (const node* c = navigation<E>::first(d, n), *end = navigation<E>::end(d, n); c != end;)
const node* pos; ///< next element, or key of the next member
const node* end;
bool object;
node* self; ///< the container in the copy
};
std::vector<frame> open;
for (;;)
{
if (object)
if (is_container(*n))
{
*out++ = copy_scalar(d, *c);
++c;
node* const self = out++;
*self = plain_node(static_cast<value_t>(n->kind));
self->len = n->len;
open.push_back(frame{walk::first(d, n), walk::end(d, n), n->kind == static_cast<std::uint8_t>(value_t::object), self});
}
else
{
*out++ = copy_scalar(d, *n);
}
// the next value: close finished containers, then copy the key
for (;;)
{
if (open.empty())
{
return;
}
frame& f = open.back();
if (f.pos == f.end)
{
f.self->next = static_cast<std::uint32_t>(out - f.self);
open.pop_back();
continue;
}
const node* v = f.pos;
if (f.object)
{
*out++ = copy_scalar(d, *v);
++v;
}
f.pos = document_data::after(v);
n = walk::value(v);
break;
}
out = fill_nodes<E>(d, navigation<E>::value(c), out);
c = document_data::after(c);
}
self->next = static_cast<std::uint32_t>(out - self);
return out;
}
static std::size_t count_nodes(const BasicJsonType& j)
{
std::size_t r = 1;
if (j.is_object())
using iterator = typename BasicJsonType::const_iterator;
struct frame
{
for (const auto& member : j.items())
iterator pos;
iterator end;
bool object;
};
std::vector<frame> open;
const BasicJsonType* n = &j;
std::size_t r = 0;
for (;;)
{
++r;
if (n->is_structured())
{
r += 1 + count_nodes(member.value());
open.push_back(frame{n->cbegin(), n->cend(), n->is_object()});
}
for (;;)
{
if (open.empty())
{
return r;
}
frame& f = open.back();
if (f.pos == f.end)
{
open.pop_back();
continue;
}
r += f.object ? 1 : 0; // the key
n = &*f.pos;
++f.pos;
break;
}
}
else if (j.is_array())
{
for (const auto& e : j)
{
r += count_nodes(e);
}
}
return r;
}
node* fill_nodes(const BasicJsonType& j, node* out)
void fill_nodes(const BasicJsonType& j, node* out)
{
if (!j.is_structured())
using iterator = typename BasicJsonType::const_iterator;
struct frame
{
*out = json_scalar(j);
return out + 1;
}
node* const self = out++;
*self = plain_node(j.type());
self->len = static_cast<std::uint32_t>(j.size());
if (j.is_object())
iterator pos;
iterator end;
bool object;
node* self; ///< the container in the copy
};
std::vector<frame> open;
const BasicJsonType* n = &j;
for (;;)
{
for (const auto& member : j.items())
if (n->is_structured())
{
check_utf8(member.key().data(), member.key().size());
*out++ = string_node(member.key().data(), member.key().size());
out = fill_nodes(member.value(), out);
node* const self = out++;
*self = plain_node(n->type());
self->len = static_cast<std::uint32_t>(n->size());
open.push_back(frame{n->cbegin(), n->cend(), n->is_object(), self});
}
else
{
*out++ = json_scalar(*n);
}
for (;;)
{
if (open.empty())
{
return;
}
frame& f = open.back();
if (f.pos == f.end)
{
f.self->next = static_cast<std::uint32_t>(out - f.self);
open.pop_back();
continue;
}
if (f.object)
{
const auto& key = f.pos.key();
check_utf8(key.data(), key.size());
*out++ = string_node(key.data(), key.size());
}
n = &*f.pos;
++f.pos;
break;
}
}
else
{
for (const auto& e : j)
{
out = fill_nodes(e, out);
}
}
self->next = static_cast<std::uint32_t>(out - self);
return out;
}
document_data& m_doc;
+38 -14
View File
@@ -63,6 +63,23 @@ inline node* alloc_nodes(document_data& d, std::size_t k)
return r;
}
/// The capacity of the edit arena after it grows by n bytes (`used` of `cap`
/// are taken): doubled, or what is needed plus some room, but never more than
/// the 4 GiB - 1 bytes that the 32-bit offsets of nodes can address. An error
/// if n more bytes do not fit even then.
inline std::size_t text_capacity(std::size_t cap, std::size_t used, std::size_t n)
{
constexpr std::size_t limit = 0xFFFFFFFFu;
if (NLOHMANN_VIEW_UNLIKELY(used > limit || n > limit - used))
{
throw_out_of_range(416, "edits of 4 GiB or more are not supported by json_document");
}
const std::size_t needed = used + n;
const std::size_t wanted = needed + (std::min)(limit - needed, std::size_t{256});
const std::size_t doubled = cap > limit / 2 ? limit : cap * 2;
return (std::max)(doubled, wanted);
}
/// copy n bytes into the edit arena and return their offset; a new buffer
/// leaves the old one alive, so that string views into it remain valid
inline std::uint32_t append_text(document_data& d, const char* s, std::size_t n)
@@ -70,11 +87,7 @@ inline std::uint32_t append_text(document_data& d, const char* s, std::size_t n)
document_data::edit_state& e = edit_state_of(d);
if (NLOHMANN_VIEW_UNLIKELY(e.text_cap - e.text_used < n))
{
const std::size_t cap = (std::max)(e.text_cap * 2, e.text_used + n + 256);
if (cap > 0xFFFFFFFFu)
{
throw_out_of_range(416, "edits of 4 GiB or more are not supported by json_document"); // LCOV_EXCL_LINE (4 GiB)
}
const std::size_t cap = text_capacity(e.text_cap, e.text_used, n);
std::unique_ptr<char[]> fresh(new char[cap]); // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
if (e.text_used != 0)
{
@@ -101,16 +114,12 @@ inline std::size_t moved_capacity(const document_data& d, const node* n) noexcep
return d.edits->moved_cap[n->off];
}
/// let container n take its elements from `seq` (header node first)
inline void set_moved(document_data& d, node* n, node* seq, std::size_t cap)
/// Make room for one more moved container. This is the part of set_moved()
/// that can throw: a caller that changes a node before it calls set_moved()
/// calls this first, so that a failure leaves the node as it was.
inline void reserve_moved(document_data& d)
{
document_data::edit_state& e = edit_state_of(d);
if ((n->flags & node_flags::moved) != 0)
{
e.moved[n->off] = seq;
e.moved_cap[n->off] = cap;
return;
}
if (e.moved.size() >= 0xFFFFFFFFu)
{
throw_out_of_range(416, "more than 4294967295 edited arrays and objects are not supported by json_document"); // LCOV_EXCL_LINE
@@ -121,6 +130,20 @@ inline void set_moved(document_data& d, node* n, node* seq, std::size_t cap)
e.moved.reserve((2 * e.moved.size()) + 16);
e.moved_cap.reserve((2 * e.moved.size()) + 16);
}
}
/// let container n take its elements from `seq` (header node first); cannot
/// throw if n is moved already or reserve_moved() was called
inline void set_moved(document_data& d, node* n, node* seq, std::size_t cap)
{
document_data::edit_state& e = edit_state_of(d);
if ((n->flags & node_flags::moved) != 0)
{
e.moved[n->off] = seq;
e.moved_cap[n->off] = cap;
return;
}
reserve_moved(d);
e.moved.push_back(seq);
e.moved_cap.push_back(cap);
n->off = static_cast<std::uint32_t>(e.moved.size() - 1);
@@ -146,6 +169,7 @@ inline node* block_of(document_data& d, node* n, std::size_t extra)
set_moved(d, n, nh, cap);
return nh;
}
reserve_moved(d); // (so that set_moved() below cannot throw: the links are marked before)
const bool object = n->kind == static_cast<std::uint8_t>(value_t::object);
const std::size_t used = 1 + (static_cast<std::size_t>(n->len) * (object ? 2 : 1));
const std::size_t cap = used + extra;
@@ -161,7 +185,7 @@ inline node* block_of(document_data& d, node* n, std::size_t extra)
{
*o++ = *c++; // the key
}
make_link(*o, document_data::deref(c));
make_link(*o, const_cast<node*>(document_data::deref(c))); // NOLINT(cppcoreguidelines-pro-type-const-cast): the nodes belong to the document
++o;
c = document_data::after(c);
}
+2 -3
View File
@@ -66,9 +66,8 @@ template<typename BasicJsonType>
{
if (f.code == error_code::input_too_large)
{
// LCOV_EXCL_START (4 GiB)
NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr));
// LCOV_EXCL_STOP
// (the limit is detail::view::max_input_size: 4 GiB minus 16 bytes)
NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4294967280 bytes or more is not supported by json_document", nullptr));
}
const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas);
// LCOV_EXCL_START (only if parse() accepts what the view rejects: a bug)
+195 -50
View File
@@ -8,6 +8,7 @@
#pragma once
#include <algorithm> // sort
#include <array> // array
#include <cstddef> // size_t
#include <cstdint> // int64_t, uint8_t, uint16_t, uint32_t, uint64_t
@@ -252,17 +253,25 @@ inline std::vector<std::uint8_t> save_image(const document_data& d)
return image;
}
/// whether a number node matches its token the way the parser records it
/// (after the bounds check)
inline bool check_number(const node& n, const unsigned char* text)
/// the parts of a number token that the checks need
struct number_token
{
const unsigned char* int_start;
std::size_t int_digits;
std::size_t frac_digits;
std::int64_t exponent;
bool negative;
bool is_float; ///< a fraction or an exponent
};
/// whether [s, s + len) is a JSON number (the grammar the parser accepts)
inline bool scan_number_token(const unsigned char* s, std::size_t len, number_token& t)
{
const std::size_t len = number_length(n);
const unsigned char* const s = text + n.off;
const unsigned char* const e = s + len;
const unsigned char* p = s;
const bool negative = *p == '-';
p += negative ? 1 : 0;
const unsigned char* const int_start = p;
t.negative = p != e && *p == '-';
p += t.negative ? 1 : 0;
t.int_start = p;
if (p == e)
{
return false;
@@ -282,9 +291,9 @@ inline bool check_number(const node& n, const unsigned char* text)
{
return false;
}
const auto int_digits = static_cast<std::size_t>(p - int_start);
std::size_t frac_digits = 0;
bool is_float = false;
t.int_digits = static_cast<std::size_t>(p - t.int_start);
t.frac_digits = 0;
t.is_float = false;
if (p != e && *p == '.')
{
const unsigned char* const f0 = ++p;
@@ -296,10 +305,10 @@ inline bool check_number(const node& n, const unsigned char* text)
{
return false;
}
frac_digits = static_cast<std::size_t>(p - f0);
is_float = true;
t.frac_digits = static_cast<std::size_t>(p - f0);
t.is_float = true;
}
std::int64_t exponent = 0;
t.exponent = 0;
if (p != e && (*p | 0x20u) == 'e')
{
++p;
@@ -311,56 +320,171 @@ inline bool check_number(const node& n, const unsigned char* text)
}
while (p != e && is_digit(*p))
{
exponent = exponent < 100000 ? (exponent * 10) + (*p - '0') : exponent;
t.exponent = t.exponent < 100000 ? (t.exponent * 10) + (*p - '0') : t.exponent;
++p;
}
exponent = exp_negative ? -exponent : exponent;
is_float = true;
t.exponent = exp_negative ? -t.exponent : t.exponent;
t.is_float = true;
}
if (p != e)
return p == e;
}
/// the digit layout the parser records for a float token (compaction
/// writes "many" instead; the caller accepts both)
inline std::uint16_t float_layout(const number_token& t)
{
return static_cast<std::uint16_t>((t.int_digits < 255 ? t.int_digits : 255) | ((t.frac_digits < 255 ? t.frac_digits : 255) << 8u));
}
/// whether a float token (well-formed, per scan_number_token) is finite as
/// double; parse() rejects floats that overflow, and as there, only a number
/// whose magnitude could reach 1e308 needs the conversion
inline bool float_token_finite(const unsigned char* s, std::size_t len, const number_token& t)
{
if (static_cast<std::int64_t>(t.int_digits) + t.exponent > 300)
{
node n{};
n.kind = static_cast<std::uint8_t>(value_t::number_float);
n.len = static_cast<std::uint32_t>(len);
const auto v = float_value<double>(reinterpret_cast<const char*>(s), n); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
return v <= (std::numeric_limits<double>::max)() && v >= -(std::numeric_limits<double>::max)();
}
return true;
}
/// whether an integer node matches its token the way the parser records it
/// (after the bounds check; the token has at most 256 characters)
inline bool check_integer(const node& n, const unsigned char* text)
{
const std::size_t len = number_length(n);
const unsigned char* const s = text + n.off;
number_token t{};
if (!scan_number_token(s, len, t))
{
return false;
}
if (n.kind == static_cast<std::uint8_t>(value_t::number_float))
{
// the digit layout the parser records (or "many", as compaction
// writes it), and a finite value
const auto layout = static_cast<std::uint16_t>((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8u));
if (n.extra != layout && n.extra != 0xFFFFu)
{
return false;
}
// parse() rejects floats that overflow; as there, only a number whose
// magnitude could reach 1e308 needs the conversion
if (static_cast<std::int64_t>(int_digits) + exponent > 300)
{
const auto v = float_value<double>(reinterpret_cast<const char*>(s), n); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
return v <= (std::numeric_limits<double>::max)() && v >= -(std::numeric_limits<double>::max)();
}
return true;
}
// integers: the token's value is the stored one; number_integer nodes of
// edits can be non-negative (as basic_json keeps the type of a value)
// the token's value is the stored one; number_integer nodes of edits can
// be non-negative (as basic_json keeps the type of a value)
const bool integer = n.kind == static_cast<std::uint8_t>(value_t::number_integer);
if (is_float || int_digits > 20 || (negative && !integer))
if (t.is_float || t.int_digits > 20 || (t.negative && !integer))
{
return false;
}
// (at most 19 digits cannot overflow; 20 digits are compared with 2^64 - 1)
if (int_digits == 20 && std::memcmp(int_start, "18446744073709551615", 20) > 0)
if (t.int_digits == 20 && std::memcmp(t.int_start, "18446744073709551615", 20) > 0)
{
return false;
}
std::uint64_t m = 0;
for (const unsigned char* d = int_start; d != int_start + int_digits; ++d)
for (const unsigned char* d = t.int_start; d != t.int_start + t.int_digits; ++d)
{
m = (m * 10) + static_cast<std::uint64_t>(*d - '0');
}
if (integer && m > (negative ? std::uint64_t{1} << 63u : (std::uint64_t{1} << 63u) - 1))
if (integer && m > (t.negative ? std::uint64_t{1} << 63u : (std::uint64_t{1} << 63u) - 1))
{
return false;
}
return integer_bits(n) == (negative ? 0 - m : m);
return integer_bits(n) == (t.negative ? 0 - m : m);
}
/// a byte range of the text or of the decoded strings
struct byte_range
{
std::uint32_t off;
std::uint32_t len;
};
/// a float token and the digit layout its node records
struct float_range
{
std::uint32_t off;
std::uint32_t len;
std::uint16_t extra;
};
/// Run check once for each distinct range of ranges (which are within bounds
/// and not empty). Ranges that are not identical must not overlap: save()
/// writes every string and token to a place of its own, and nodes that share
/// a value (copies within a document) share the whole range. This bounds the
/// work by the size of the text, however many nodes point to the same bytes.
template<typename Check>
bool check_distinct_ranges(std::vector<byte_range>& ranges, Check check)
{
std::sort(ranges.begin(), ranges.end(), [](const byte_range & a, const byte_range & b)
{
return a.off != b.off ? a.off < b.off : a.len < b.len;
});
std::size_t end = 0;
for (std::size_t i = 0; i < ranges.size(); ++i)
{
const byte_range r = ranges[i];
if (i != 0 && r.off == ranges[i - 1].off && r.len == ranges[i - 1].len)
{
continue;
}
if (r.off < end || !check(r))
{
return false;
}
end = static_cast<std::size_t>(r.off) + r.len;
}
return true;
}
/// Check the float nodes: each token once (nodes of the same token must
/// record layouts that match it), tokens must not overlap.
inline bool check_float_ranges(std::vector<float_range>& ranges, const unsigned char* text)
{
std::sort(ranges.begin(), ranges.end(), [](const float_range & a, const float_range & b)
{
return a.off != b.off ? a.off < b.off : (a.len != b.len ? a.len < b.len : a.extra < b.extra);
});
std::size_t end = 0;
std::size_t i = 0;
while (i < ranges.size())
{
const float_range r = ranges[i];
if (r.off < end)
{
return false;
}
number_token t{};
const unsigned char* const s = text + r.off;
if (!scan_number_token(s, r.len, t) || !float_token_finite(s, r.len, t))
{
return false;
}
// the digit layout the parser records (or "many", as compaction writes it)
const std::uint16_t layout = float_layout(t);
for (; i < ranges.size() && ranges[i].off == r.off && ranges[i].len == r.len; ++i)
{
if (ranges[i].extra != layout && ranges[i].extra != 0xFFFFu)
{
return false;
}
}
end = static_cast<std::size_t>(r.off) + r.len;
}
return true;
}
/// the content checks of check_image: source strings as the parser leaves
/// them (no quotes, backslashes, or control characters), decoded strings
/// (valid UTF-8), and float tokens
inline bool check_contents(const unsigned char* text, const unsigned char* arena,
std::vector<byte_range>& source_strings, std::vector<byte_range>& decoded_strings,
std::vector<float_range>& floats)
{
return check_distinct_ranges(source_strings, [text](const byte_range & r)
{
const unsigned char* const b = text + r.off;
return scan_string_run(b, b + r.len) == b + r.len;
})
&& check_distinct_ranges(decoded_strings, [arena](const byte_range & r)
{
return valid_utf8_prefix(arena + r.off, r.len) == r.len;
})
&& check_float_ranges(floats, text);
}
/// Check the nodes of a loaded image against its text and decoded strings:
@@ -368,6 +492,12 @@ inline bool check_number(const node& n, const unsigned char* text)
/// objects; keys; bounds; string contents (source strings as the parser
/// leaves them: no quotes, backslashes, or control characters; all strings
/// valid UTF-8); and number tokens.
///
/// The structure and the bounds are checked node by node. The contents of
/// strings and of float tokens are checked afterwards, once for each distinct
/// range (see check_distinct_ranges), so that the full check is linear in the
/// size of the image plus the sorting of the ranges, and not in the number of
/// nodes times the size of the text.
inline bool check_image(const node* nodes, std::size_t count, const unsigned char* text, std::size_t text_size,
const unsigned char* arena, std::size_t arena_size, bool full)
{
@@ -380,6 +510,9 @@ inline bool check_image(const node* nodes, std::size_t count, const unsigned cha
bool expect_key;
};
std::vector<frame> stack;
std::vector<byte_range> source_strings;
std::vector<byte_range> decoded_strings;
std::vector<float_range> floats;
const auto check_string = [&](const node & n) -> bool
{
if ((n.flags & ~node_flags::escaped) != 0 || n.extra != 0)
@@ -387,18 +520,16 @@ inline bool check_image(const node* nodes, std::size_t count, const unsigned cha
return false;
}
const bool decoded = (n.flags & node_flags::escaped) != 0;
const unsigned char* const base = decoded ? arena : text;
const std::size_t limit = decoded ? arena_size : text_size;
if (n.off > limit || n.len > limit - n.off)
{
return false;
}
if (!full)
if (full && n.len != 0)
{
return true;
(decoded ? decoded_strings : source_strings).push_back(byte_range{n.off, n.len});
}
const unsigned char* const b = base + n.off;
return decoded ? valid_utf8_prefix(b, n.len) == n.len : scan_string_run(b, b + n.len) == b + n.len;
return true;
};
// bounds of a number token; the recorded digit layout must lie within it
const auto number_in_bounds = [&](const node & n) -> bool
@@ -440,7 +571,7 @@ inline bool check_image(const node* nodes, std::size_t count, const unsigned cha
}
if (i == count)
{
return stack.empty();
return stack.empty() && (!full || check_contents(text, arena, source_strings, decoded_strings, floats));
}
if (i != 0 && stack.empty())
{
@@ -482,10 +613,23 @@ inline bool check_image(const node* nodes, std::size_t count, const unsigned cha
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
if (n.flags != 0 || !number_in_bounds(n) || (full && !check_number(n, text)))
if (n.flags != 0 || !number_in_bounds(n))
{
return false;
}
if (full)
{
// (the contents of float tokens are checked later; the token of an
// integer has at most 256 characters)
if (n.kind == static_cast<std::uint8_t>(value_t::number_float))
{
floats.push_back(float_range{n.off, n.len, n.extra});
}
else if (!check_integer(n, text))
{
return false;
}
}
break;
case value_t::array:
case value_t::object:
@@ -594,6 +738,7 @@ inline void load_image(document_data& d, const std::uint8_t* image, std::size_t
}
}
build_object_indexes(d);
std::vector<std::uint32_t>().swap(d.large_objects); // (only needed while building)
d.discarded = false;
}
+12 -5
View File
@@ -9,7 +9,7 @@
#pragma once
#include <string> // basic_string, char_traits, string
#include <type_traits> // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference
#include <type_traits> // decay, integral_constant, is_array, is_const, is_integral, is_lvalue_reference, is_pointer, is_same, remove_reference
#include <utility> // forward
#include <nlohmann/json.hpp>
@@ -28,12 +28,12 @@ namespace view
/// how a document takes its input
enum class input_kind
{
move_string, ///< rvalue std::string: owned without a copy
move_string, ///< non-const rvalue std::string: owned without a copy
c_string, ///< const char* (NUL-terminated): borrowed
char_array, ///< char array (e.g. a string literal): borrowed
borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed
copy_range, ///< rvalue contiguous byte container: copied
adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer
copy_range, ///< rvalue contiguous byte container (a const rvalue std::string too): copied
adapter, ///< streams, wide strings, and the rest of what the library's input adapter reads: read into a buffer
};
template<typename InputType>
@@ -52,13 +52,20 @@ struct classify_input
static constexpr input_kind value =
std::is_array<R>::value ? input_kind::char_array
: std::is_pointer<D>::value ? input_kind::c_string
: (is_rvalue && std::is_same<D, std::string>::value) ? input_kind::move_string
: (is_rvalue && !std::is_const<R>::value && std::is_same<D, std::string>::value) ? input_kind::move_string
: (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range
: is_bytes ? input_kind::copy_range
: input_kind::adapter;
// NOLINTEND(readability-avoid-nested-conditional-operator)
};
/// an integer type other than bool: a length passed where a flag is expected
template<typename T>
struct is_integer_not_bool : std::is_integral<T> {};
template<>
struct is_integer_not_bool<bool> : std::false_type {};
/// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel)
template<typename T>
struct is_std_string : std::false_type {};
+28 -6
View File
@@ -11,6 +11,8 @@
#include <cstddef> // size_t
#include <cstdint> // uint16_t, uint32_t, uint64_t
#include <cstring> // memcmp, memcpy
#include <limits> // numeric_limits
#include <type_traits> // integral_constant, is_integral, is_same
#include <nlohmann/json.hpp>
#include <nlohmann/detail/view/document_data.hpp>
@@ -82,8 +84,9 @@ class short_key
std::uint64_t m_b = 0;
};
/// the key node of the first member of an object with the given key, or
/// nullptr; most keys are rejected by their length, from the index alone
/// the key node of the last member of an object with the given key, or
/// nullptr (the last one, as materialize() and parse() keep it); most keys are
/// rejected by their length, from the index alone
template<bool Editable>
const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept
{
@@ -94,6 +97,7 @@ const node* find_member(const document_data& d, const node* object, const char*
}
const node* const end = nav::end(d, object);
const auto* const k = reinterpret_cast<const unsigned char*>(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
const node* last = nullptr;
if (NLOHMANN_VIEW_LIKELY(n <= 16))
{
const short_key probe(k, n);
@@ -101,19 +105,37 @@ const node* find_member(const document_data& d, const node* object, const char*
{
if (m->len == n && probe.matches(reinterpret_cast<const unsigned char*>(d.str(*m)))) // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
{
return m;
last = m;
}
}
return nullptr;
return last;
}
for (const node* m = nav::first(d, object); m != end; m = document_data::after(m + 1))
{
if (m->len == n && std::memcmp(d.str(*m), key, n) == 0)
{
return m;
last = m;
}
}
return nullptr;
return last;
}
/// whether an integer type is accepted as an array index by the view's
/// operator[] and at(): every integer type but bool and size_t, which has its
/// own overload
template<typename T>
struct is_index_type : std::integral_constant < bool,
std::is_integral<T>::value && !std::is_same<T, bool>::value && !std::is_same<T, std::size_t>::value >
{};
/// an integer as an index: negative values, and values that do not fit a
/// size_t, map to the largest size_t (out of range for every array)
template<typename SizeType, typename IntegerType>
SizeType to_index(IntegerType idx) noexcept
{
const IntegerType zero = 0;
const auto result = static_cast<SizeType>(idx);
return (idx < zero || static_cast<IntegerType>(result) != idx) ? (std::numeric_limits<SizeType>::max)() : result;
}
/// the entry of the element of an array at an index below its size (a link
+21 -3
View File
@@ -29,6 +29,11 @@ static_assert(static_cast<std::uint8_t>(value_t::null) == 0 && static_cast<std::
&& static_cast<std::uint8_t>(value_t::number_unsigned) == 6 && static_cast<std::uint8_t>(value_t::number_float) == 7,
"the node format depends on the numbering of value_t");
/// The largest input a document accepts, in bytes. Offsets and node counts are
/// 32 bits wide; the limit keeps 16 bytes (the width of the scanner's steps)
/// below 2^32, so that a position one step past the end of the text fits.
static constexpr std::size_t max_input_size = 0xFFFFFFEFu;
/// node flags
struct node_flags
{
@@ -38,6 +43,7 @@ struct node_flags
static constexpr std::uint8_t is_true = 4; ///< boolean value
static constexpr std::uint8_t moved = 8; ///< array/object: the elements live in a separate sequence (editable documents)
static constexpr std::uint8_t is_new = 16; ///< written by an edit: no source position
static constexpr std::uint8_t linked = 32; ///< an entry of a moved sequence links to this value (editable documents): its extent in the parsed layout no longer matters
};
/// kind of an entry of an edited sequence that stands for a value stored
@@ -52,7 +58,7 @@ struct node
{
std::uint8_t kind; ///< value_t, or kind_link
std::uint8_t flags; ///< node_flags
std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; objects: number of the hash index; otherwise 0
std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; objects: number of the hash index (1-based, 0 = none); otherwise 0
std::uint32_t off; ///< source offset (string content, number token, literal, bracket); arena offset if escaped/edited; number of the element sequence if moved
std::uint32_t len; ///< string: decoded bytes; float: token bytes; array/object: element count
std::uint32_t next; ///< array/object: number of nodes of the subtree (its extent in the enclosing sequence)
@@ -72,24 +78,36 @@ NLOHMANN_VIEW_ALWAYS_INLINE const node* link_target(const node& n) noexcept
return t;
}
inline void make_link(node& n, const node* target) noexcept
/// let the entry n stand for the value at target (and mark the value)
inline void make_link(node& n, node* target) noexcept
{
target->flags = static_cast<std::uint8_t>(target->flags | node_flags::linked);
n = node{};
n.kind = kind_link;
std::memcpy(reinterpret_cast<unsigned char*>(&n) + 8, static_cast<const void*>(&target), sizeof(const node*)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
}
/// the converted value of an integer node (stored in len/next)
/// the converted value of an integer node: len is its low half, next its high
/// half (on little-endian targets the two words are the value in memory)
NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept
{
#if NLOHMANN_VIEW_LITTLE_ENDIAN
std::uint64_t v = 0;
std::memcpy(&v, reinterpret_cast<const unsigned char*>(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
return v;
#else
return static_cast<std::uint64_t>(n.len) | (static_cast<std::uint64_t>(n.next) << 32);
#endif
}
NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept
{
#if NLOHMANN_VIEW_LITTLE_ENDIAN
std::memcpy(reinterpret_cast<unsigned char*>(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
#else
n.len = static_cast<std::uint32_t>(v);
n.next = static_cast<std::uint32_t>(v >> 32);
#endif
}
/// token length of a number node
+34 -5
View File
@@ -22,8 +22,14 @@
// large objects). An object with document_data::index_min_members members or
// more gets an open-addressing table after parsing; its node stores the
// number of the table (1-based) in `extra`. A slot holds the offset of a key
// node from its object node (0: empty). Of duplicate keys, the first is kept,
// as for the linear search.
// node from its object node (0: empty). Of duplicate keys, the last is kept,
// as for the linear search, and as basic_json::parse() does.
//
// The hash is not seeded, so keys chosen to collide could make the build
// quadratic. A key therefore sits at most index_max_displacement slots away
// from its home slot; if a key would sit further away, the table is dropped
// and the object is searched linearly (like a small one). For the same
// reason, a lookup visits at most index_max_displacement + 1 slots.
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -31,6 +37,11 @@ namespace detail
namespace view
{
/// the farthest a key may sit from its home slot (a table with at most half of
/// its slots in use gives random keys a distance of about 50 for millions of
/// members; and every member costs at most this many steps while building)
constexpr std::size_t index_max_displacement = 64;
/// hash of a key: its bytes, eight at a time, in a fixed byte order
inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept
{
@@ -68,27 +79,44 @@ inline void build_object_index(document_data& d, node* obj)
d.index_slots.resize(start + cap, 0);
std::uint32_t* const slots = d.index_slots.data() + start;
const std::size_t mask = cap - 1;
bool degenerate = false;
for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1))
{
const char* const key = d.str(*k);
const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t)
std::size_t i = static_cast<std::size_t>(hash) & mask;
bool duplicate = false;
std::size_t distance = 0;
while (slots[i] != 0)
{
const node* const other = obj + slots[i];
if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0))
{
duplicate = true; // keep the first
duplicate = true; // keep the last: the key's slot now leads to this member
slots[i] = static_cast<std::uint32_t>(k - obj);
break;
}
if (++distance > index_max_displacement)
{
degenerate = true; // too many keys share a home region
break;
}
i = (i + 1) & mask;
}
if (degenerate)
{
break;
}
if (!duplicate)
{
slots[i] = static_cast<std::uint32_t>(k - obj);
}
}
if (degenerate)
{
d.index_slots.resize(start); // no table: the object is searched linearly
return;
}
d.indexes.push_back(document_data::object_index{start, static_cast<std::uint32_t>(mask)});
obj->extra = static_cast<std::uint16_t>(d.indexes.size());
}
@@ -102,7 +130,7 @@ inline void build_object_indexes(document_data& d)
}
}
/// the key node of the first member with this key of an indexed object, or
/// the key node of the last member with this key of an indexed object, or
/// nullptr
inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept
{
@@ -110,7 +138,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c
const std::uint32_t* const slots = d.index_slots.data() + ix.start;
const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t)
std::size_t i = static_cast<std::size_t>(hash) & ix.mask;
for (;;)
for (std::size_t distance = 0; distance <= index_max_displacement; ++distance)
{
const std::uint32_t s = slots[i];
if (s == 0)
@@ -124,6 +152,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c
}
i = (i + 1) & ix.mask;
}
return nullptr; // (no key sits further from its home slot)
}
} // namespace view
+26 -15
View File
@@ -48,7 +48,13 @@ class output_buffer
void finish()
{
m_out.resize(static_cast<std::size_t>(m_pos - m_out.data()));
const auto size = static_cast<std::size_t>(m_pos - m_out.data());
m_out.resize(size);
// do not keep a buffer that was sized for a much larger output
if (m_out.capacity() > 1024 && m_out.capacity() / 2 > size)
{
m_out.shrink_to_fit();
}
}
NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n)
@@ -356,16 +362,6 @@ class view_serializer
}
private:
/*!
@brief the compact output without ensure_ascii (the default dump())
The same walk as dump(), with the write position in a local variable
(stores through char pointers would otherwise force a reload of the
buffer's members after each one), and with strings and number tokens of
the source copied by fixed-size moves of 32 bytes where the source has
that many bytes left, instead of a library call per token. The buffer
keeps 64 bytes of slack for the overshoot.
*/
/// a string that is not a plain string of the source (decoded, or written
/// by an edit), without ensure_ascii: runs without characters to escape
/// are copied
@@ -404,6 +400,16 @@ class view_serializer
return option;
}
/*!
@brief the compact output without ensure_ascii (the default dump())
The same walk as dump(), with the write position in a local variable
(stores through char pointers would otherwise force a reload of the
buffer's members after each one), and with strings and number tokens of
the source copied by fixed-size moves of 32 bytes where the source has
that many bytes left, instead of a library call per token. The buffer
keeps 64 bytes of slack for the overshoot.
*/
template<bool SourceNumbers>
void dump_compact(const node* root)
{
@@ -525,7 +531,7 @@ class view_serializer
room(n->len);
copy(src + n->off, n->len);
}
else if (std::is_same<number_float_t, double>::value)
else if (enabled(std::is_same<number_float_t, double>::value))
{
room(64);
w = write_double_at(w, *n);
@@ -752,9 +758,14 @@ class view_serializer
{
*w = '-';
w += d.negative ? 1 : 0;
// (without leading zeros, all digits of the token count)
const unsigned char lead = first[d.negative ? 1 : 0];
return lead != '0' ? ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast<int>(int_digits + frac_digits), static_cast<int>(d.exponent))
// (without leading zeros, all digits of the token count; the
// check also keeps an image that was only checked for bounds,
// whose token may not be made of digits, from the counted
// overload)
const auto& powers = ::nlohmann::detail::dtoa_impl::powers_of_ten_16();
const unsigned count = int_digits + frac_digits;
return count - 1u < 15u && d.w >= powers[count - 1u] && d.w < powers[count]
? ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast<int>(count), static_cast<int>(d.exponent))
: ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast<int>(d.exponent));
}
return write_double_value_at(w, decimal_to_float<double>(d)); // (without reading the token again)
+2 -1
View File
@@ -35,7 +35,8 @@
#else
#define NLOHMANN_VIEW_NEON 0
#endif
#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2))
// (x86 only: other targets can define __SSE2__ as well, e.g., WebAssembly with -msse2, but have no <cpuid.h>)
#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__x86_64__) || defined(__i386__) || defined(_M_X64) || defined(_M_IX86)) && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2))
#include <emmintrin.h>
#define NLOHMANN_VIEW_SSE2 1
#else
+131 -39
View File
@@ -7,16 +7,18 @@
// SPDX-License-Identifier: MIT
/****************************************************************************\
* Zero-copy, read-only view of a parsed JSON text. *
* Zero-copy view of a parsed JSON text. *
* *
* json_document::parse() builds a flat index of the values of a JSON text *
* (16 bytes per value) instead of a tree of basic_json values. Strings and *
* numbers stay in the source text; only strings with escapes are decoded, *
* into one buffer. json_view is a handle to one value of the document, with *
* the read-only part of the basic_json interface; materialize() turns a *
* subtree into the basic_json value that parse() would produce. *
* subtree into the basic_json value that parse() would produce. An editable *
* document (json_editable_document) also has set(), push_back(), insert(), *
* and erase(): edits never write to the source text, and views stay valid. *
* *
* The source text must outlive a document that borrows it (lvalue byte *
* The source text must outlive a document that borrows it (lvalue byte *
* containers, C strings); rvalue strings, streams, and other inputs are *
* owned by the document. *
\****************************************************************************/
@@ -24,6 +26,7 @@
#ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_
#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_
#include <algorithm> // all_of, min
#include <cstddef> // size_t
#include <cstdint> // uint8_t, uint32_t
#include <cstring> // memcpy, strlen
@@ -181,12 +184,6 @@ class basic_json_view
return type() == value_t::discarded;
}
/// false for discarded views
explicit operator bool() const noexcept
{
return m_node != nullptr;
}
/// the name of the type, as basic_json::type_name()
const char* type_name() const noexcept
{
@@ -246,13 +243,18 @@ class basic_json_view
// element access //
////////////////////
/// the value of the member with this key (the first one, should the key
/// occur more than once); a discarded view if there is none. Throws
/// type_error.305 if this is not an object.
/// the value of the member with this key (the last one, should the key
/// occur more than once); a discarded view if there is none, or if this
/// is a discarded view (so that v["a"]["b"] is safe). Throws type_error.305
/// if this is any other value but an object.
NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view operator[](string_view_t key) const
{
if (NLOHMANN_VIEW_UNLIKELY(!is_object()))
{
if (is_discarded())
{
return basic_json_view();
}
detail::view::throw_type_error(305, "cannot use operator[] with a string argument with ", type_name());
}
return lookup(key);
@@ -269,31 +271,43 @@ class basic_json_view
}
/// the element at this index; a discarded view if the index is out of
/// range. Throws type_error.305 if this is not an array.
/// range, or if this is a discarded view. Throws type_error.305 if this is
/// any other value but an array.
basic_json_view operator[](size_type idx) const
{
if (NLOHMANN_VIEW_UNLIKELY(!is_array()))
{
if (is_discarded())
{
return basic_json_view();
}
detail::view::throw_type_error(305, "cannot use operator[] with a numeric argument with ", type_name());
}
return idx < m_node->len ? basic_json_view(m_doc, navigation::value(detail::view::element_at<Editable>(*m_doc, m_node, idx))) : basic_json_view();
}
/// (an int argument would be ambiguous between size_type and const char*)
basic_json_view operator[](int idx) const
/// any other integer type (int, unsigned, long, std::int64_t, ...; a
/// single overload for size_type alone would be ambiguous for all of them
/// and for const char*); negative values are out of range
template < typename IntegerType, typename std::enable_if < detail::view::is_index_type<IntegerType>::value, int >::type = 0 >
basic_json_view operator[](IntegerType idx) const
{
return operator[](static_cast<size_type>(idx));
return operator[](detail::view::to_index<size_type>(idx));
}
/// the value a JSON pointer refers to; a discarded view if a key is
/// missing or an index is out of range. Other errors throw what const
/// basic_json::operator[] throws.
/// missing or an index is out of range, or if this is a discarded view.
/// Other errors throw what const basic_json::operator[] throws.
basic_json_view operator[](const json_pointer& ptr) const
{
if (NLOHMANN_VIEW_UNLIKELY(is_discarded()))
{
return basic_json_view();
}
return detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::unchecked);
}
/// the value of the member with this key (the first one, should the key
/// the value of the member with this key (the last one, should the key
/// occur more than once). Throws type_error.304 if this is not an object,
/// and out_of_range.403 if there is no such member.
basic_json_view at(string_view_t key) const
@@ -303,7 +317,7 @@ class basic_json_view
detail::view::throw_type_error(304, "cannot use at() with ", type_name());
}
const basic_json_view r = lookup(key);
if (NLOHMANN_VIEW_UNLIKELY(!r))
if (NLOHMANN_VIEW_UNLIKELY(r.is_discarded()))
{
detail::view::throw_out_of_range(403, detail::concat("key '", std::string(key.data(), key.size()), "' not found"));
}
@@ -335,9 +349,12 @@ class basic_json_view
return basic_json_view(m_doc, navigation::value(detail::view::element_at<Editable>(*m_doc, m_node, idx)));
}
basic_json_view at(int idx) const
/// any other integer type, see operator[]; negative values are out of
/// range
template < typename IntegerType, typename std::enable_if < detail::view::is_index_type<IntegerType>::value, int >::type = 0 >
basic_json_view at(IntegerType idx) const
{
return at(static_cast<size_type>(idx));
return at(detail::view::to_index<size_type>(idx));
}
/// the value a JSON pointer refers to; throws what basic_json::at()
@@ -348,7 +365,7 @@ class basic_json_view
}
/// the member with this key converted to T, or the default value if there
/// is no such member (the first one, should the key occur more than
/// is no such member (the last one, should the key occur more than
/// once). Throws type_error.306 if this is not an object.
template < typename T, typename std::enable_if < !std::is_same<typename std::decay<T>::type, const char*>::value, int >::type = 0 >
T value(string_view_t key, const T& default_value) const
@@ -358,7 +375,7 @@ class basic_json_view
detail::view::throw_type_error(306, "cannot use value() with ", type_name());
}
const basic_json_view r = lookup(key);
return r ? r.template get<T>() : default_value;
return r.is_discarded() ? default_value : r.template get<T>();
}
string_t value(string_view_t key, const char* default_value) const
@@ -377,7 +394,7 @@ class basic_json_view
detail::view::throw_type_error(306, "cannot use value() with ", type_name());
}
const basic_json_view r = detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::value);
return r ? r.template get<T>() : default_value;
return r.is_discarded() ? default_value : r.template get<T>();
}
string_t value(const json_pointer& ptr, const char* default_value) const
@@ -413,7 +430,7 @@ class basic_json_view
// lookup //
////////////
/// an iterator to the member with this key (the first one, should the
/// an iterator to the member with this key (the last one, should the
/// key occur more than once), or end(); end() also for non-objects
iterator find(string_view_t key) const
{
@@ -455,7 +472,7 @@ class basic_json_view
/// basic_json::contains())
bool contains(const json_pointer& ptr) const
{
return static_cast<bool>(detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains));
return !detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains).is_discarded();
}
/// 1 if this is an object with a member with this key, else 0 (duplicate
@@ -714,7 +731,7 @@ class basic_json_view
}
/// the number of source bytes of this value (estimated for values with
/// decoded strings)
/// decoded strings); the estimate sizes the output buffer of dump()
std::size_t source_extent() const noexcept
{
if (editable() && m_doc->edits != nullptr)
@@ -722,20 +739,33 @@ class basic_json_view
// positions of moved and new values are not source offsets
return m_node == m_doc->tape ? m_doc->size + m_doc->edits->text_used : 64;
}
const node* const next = document_data::after(m_node);
const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0;
if (!in_source)
const node* const end = m_doc->tape + m_doc->tape_size;
if ((m_node->flags & detail::view::node_flags::storage) != 0)
{
return m_node->len;
}
if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off)
// the value ends where the next node in the source begins; nodes
// with decoded strings (their offset is in the arena) are skipped,
// but only a few of them, to keep the walk short
const node* next = document_data::after(m_node);
for (int skipped = 0; next != end && skipped < 16; ++skipped, ++next)
{
return next->off - m_node->off;
if ((next->flags & detail::view::node_flags::storage) == 0)
{
return next->off >= m_node->off ? next->off - m_node->off : 0;
}
}
return m_doc->size - m_node->off;
if (next == end)
{
return m_doc->size - m_node->off;
}
// the end is unknown: assume a few bytes per node, the output buffer
// grows should the value be larger
const auto nodes = static_cast<std::size_t>(document_data::after(m_node) - m_node);
return (std::min)(m_doc->size - m_node->off, static_cast<std::size_t>(1024) + nodes * 16);
}
/// the value of the first member with this key, or a discarded view
/// the value of the last member with this key, or a discarded view
/// (object required)
NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view lookup(string_view_t key) const noexcept
{
@@ -869,6 +899,13 @@ class basic_json_document
return d;
}
/// parse(ptr, len) does not compile: len would convert to allow_exceptions
/// and ptr be read as a C string (as for the overloads of parse_copy,
/// accept, and read below)
template<typename InputType, typename IntegerType, typename... Flags>
static typename std::enable_if<detail::view::is_integer_not_bool<IntegerType>::value, basic_json_document>::type
parse(InputType&& input, IntegerType value, Flags&&... flags) = delete;
/// parse [first, last)
template<typename IteratorType, typename std::enable_if<
std::is_base_of<std::input_iterator_tag, typename std::iterator_traits<IteratorType>::iterator_category>::value, int>::type = 0>
@@ -896,6 +933,10 @@ class basic_json_document
return d;
}
template<typename InputType, typename IntegerType, typename... Flags>
static typename std::enable_if<detail::view::is_integer_not_bool<IntegerType>::value, basic_json_document>::type
parse_copy(InputType&& input, IntegerType value, Flags&&... flags) = delete;
/// check whether the input is valid JSON (the result of basic_json::accept)
template<typename InputType>
static bool accept(InputType&& input, const bool ignore_comments = false, const bool ignore_trailing_commas = false)
@@ -905,6 +946,10 @@ class basic_json_document
return !d.is_discarded();
}
template<typename InputType, typename IntegerType, typename... Flags>
static typename std::enable_if<detail::view::is_integer_not_bool<IntegerType>::value, bool>::type
accept(InputType&& input, IntegerType value, Flags&&... flags) = delete;
/// parse into this document, reusing its memory
template<typename InputType>
// flawfinder: ignore (a member function, not POSIX read())
@@ -917,12 +962,17 @@ class basic_json_document
std::integral_constant<detail::view::input_kind, detail::view::classify_input<InputType>::value> {});
}
template<typename InputType, typename IntegerType, typename... Flags>
// flawfinder: ignore (a member function, not POSIX read())
typename std::enable_if<detail::view::is_integer_not_bool<IntegerType>::value, void>::type
read(InputType&& input, IntegerType value, Flags&&... flags) = delete;
////////////
// access //
////////////
/// the root value (discarded if parsing failed without exceptions)
view_type root() const noexcept
view_type root() const& noexcept
{
if (!m_data || m_data->discarded)
{
@@ -931,6 +981,9 @@ class basic_json_document
return view_type(m_data.get(), m_data->tape);
}
/// deleted: the view of a temporary document would dangle
view_type root() const&& = delete;
bool is_discarded() const noexcept
{
return !m_data || m_data->discarded;
@@ -990,6 +1043,8 @@ class basic_json_document
// (edits link to the nodes of the index, which then stays in place)
const bool shrink_tape = d.tape != d.inline_tape && d.tape_size != d.tape_cap && d.edits == nullptr;
const bool into_header = d.tape_size <= d.inline_cap;
std::vector<document_data::object_index> indexes(d.indexes.capacity() > d.indexes.size() ? d.indexes : std::vector<document_data::object_index>());
std::vector<std::uint32_t> index_slots(d.index_slots.capacity() > d.index_slots.size() ? d.index_slots : std::vector<std::uint32_t>());
node* fresh = (shrink_tape && !into_header) ? static_cast<node*>(::operator new (d.tape_size * sizeof(node))) : d.inline_tape;
if (shrink_tape)
@@ -1007,6 +1062,14 @@ class basic_json_document
d.base[1] = d.arena.data();
}
}
if (d.indexes.capacity() > d.indexes.size())
{
d.indexes.swap(indexes);
}
if (d.index_slots.capacity() > d.index_slots.size())
{
d.index_slots.swap(index_slots);
}
}
////////////
@@ -1097,7 +1160,9 @@ class basic_json_document
/// set the value at a JSON pointer: its parent must exist; an object
/// member is set (added if missing), an array element assigned, and "-"
/// or the size of the array appends
/// or the size of the array appends. A null parent becomes what
/// basic_json's operator[](json_pointer) makes of it: an array for "-"
/// and for digits (padded with nulls up to the index), an object otherwise.
template<typename V>
view_type set(const json_pointer& ptr, V&& value)
{
@@ -1107,6 +1172,31 @@ class basic_json_document
}
const view_type parent = root().at(ptr.parent_pointer());
const auto& token = ptr.back();
if (parent.is_null())
{
const bool digits = std::all_of(token.begin(), token.end(), [](const char c)
{
return c >= '0' && c <= '9';
});
if (token == "-")
{
return push_back(parent, std::forward<V>(value));
}
if (digits)
{
// (an invalid index is an error before the parent changes)
const std::size_t idx = pointer_index(token);
if (idx >= 0xFFFFFFFFu)
{
detail::view::throw_out_of_range(401, detail::concat("array index ", std::to_string(idx), " is out of range"));
}
for (std::size_t i = 0; i < idx; ++i)
{
push_back(parent, nullptr);
}
return push_back(parent, std::forward<V>(value));
}
}
if (parent.is_array())
{
const std::size_t idx = token == "-" ? parent.size() : pointer_index(token);
@@ -1254,9 +1344,9 @@ class basic_json_document
d.discarded = true;
detail::view::parse_failure failure;
bool ok = false;
if (NLOHMANN_VIEW_UNLIKELY(size >= 0xFFFFFFF0u))
if (NLOHMANN_VIEW_UNLIKELY(size > detail::view::max_input_size))
{
failure.code = detail::view::error_code::input_too_large; // LCOV_EXCL_LINE (4 GiB)
failure.code = detail::view::error_code::input_too_large;
}
else
{
@@ -1268,9 +1358,11 @@ class basic_json_document
d.base[1] = d.arena.data();
d.arena_size = d.arena.size();
detail::view::build_object_indexes(d);
std::vector<std::uint32_t>().swap(d.large_objects); // (only needed while parsing)
d.discarded = false;
return;
}
std::vector<std::uint32_t>().swap(d.large_objects);
if (allow_exceptions)
{
detail::view::throw_parse_failure<BasicJsonType>(failure, src, size, comments, trailing_commas);
+79 -117
View File
@@ -7544,10 +7544,11 @@ namespace detail
{
/*!
@brief the configuration macros that change the library's behavior
@brief the configuration macros that json_view.hpp reads
json.hpp undefines these macros at its end (see macro_unscope.hpp), so code
that builds on the library after it (json_view.hpp) reads them here. Like the
that builds on the library after it (json_view.hpp) reads them here. A macro
is added when the view starts to depend on it. Like the
macros, they are part of the ABI namespace, so they always match the
basic_json they are used with.
*/
@@ -7555,8 +7556,6 @@ struct abi_config
{
/// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input
static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0;
/// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0;
};
} // namespace detail
@@ -8863,8 +8862,9 @@ NLOHMANN_JSON_NAMESPACE_END
#include <cstdint> // uint64_t
#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
#include <intrin0.h> // __umulh, _umul128
#include <cstring> // memcpy
#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) && (!defined(__SIZEOF_INT128__) || (!defined(__GNUC__) && !defined(__clang__)))
#include <intrin0.h> // __umulh, _umul128, _BitScanForward64, _BitScanReverse64
#endif
// #include <nlohmann/detail/macro_scope.hpp>
@@ -8883,6 +8883,10 @@ inline int count_leading_zeros(std::uint64_t x) noexcept
{
#if defined(__GNUC__) || defined(__clang__)
return __builtin_clzll(x);
#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
unsigned long index = 0;
_BitScanReverse64(&index, x);
return 63 - static_cast<int>(index);
#else
int n = 0;
for (int shift = 32; shift != 0; shift >>= 1)
@@ -8902,6 +8906,10 @@ inline int count_trailing_zeros(std::uint64_t x) noexcept
{
#if defined(__GNUC__) || defined(__clang__)
return __builtin_ctzll(x);
#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
unsigned long index = 0;
_BitScanForward64(&index, x);
return static_cast<int>(index);
#else
int n = 0;
for (int shift = 32; shift != 0; shift >>= 1)
@@ -8949,15 +8957,21 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc
#endif
}
/// eight bytes as a little-endian word (compilers fold this into one load on
/// little-endian targets; always inlined, as GCC otherwise calls it in the
/// number loops)
/// eight bytes as a little-endian word (a single load on little-endian
/// targets; always inlined, as GCC otherwise calls it in the number loops)
JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
{
#if defined(_MSC_VER) || defined(__x86_64__) || defined(__i386__) || (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__)
// the byte order already matches (all MSVC targets are little-endian)
std::uint64_t result = 0;
std::memcpy(&result, b, sizeof(result));
return result;
#else
return static_cast<std::uint64_t>(b[0]) | (static_cast<std::uint64_t>(b[1]) << 8u)
| (static_cast<std::uint64_t>(b[2]) << 16u) | (static_cast<std::uint64_t>(b[3]) << 24u)
| (static_cast<std::uint64_t>(b[4]) << 32u) | (static_cast<std::uint64_t>(b[5]) << 40u)
| (static_cast<std::uint64_t>(b[6]) << 48u) | (static_cast<std::uint64_t>(b[7]) << 56u);
#endif
}
/// eight bytes as a little-endian word
@@ -25172,6 +25186,7 @@ NLOHMANN_JSON_NAMESPACE_END
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2009 Florian Loitsch <https://florian.loitsch.com/>
// SPDX-FileCopyrightText: 2025 Victor Zverovich <https://github.com/vitaut/zmij>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
@@ -25247,13 +25262,6 @@ computed from the compressed tables of Zmij beyond it.
namespace zmij
{
/// significand * 10^exponent
struct decimal
{
std::uint64_t significand;
int exponent;
};
/// the compressed powers of ten of Zmij
inline const std::array<std::uint64_t, 28>& pow10_minor() noexcept
{
@@ -25432,18 +25440,6 @@ JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexc
return shortest_decimal{integral, dec_exp, static_cast<unsigned char>(digit), !round_up && !round_down};
}
/// The shortest decimal in the rounding interval of a positive finite double
/// given by its bits, as one number. The significand can end in zeros.
inline decimal to_decimal(std::uint64_t bits) noexcept
{
const shortest_decimal d = to_shortest(bits);
if (d.has_digit)
{
return decimal{(d.integral * 10) + d.digit, d.exponent};
}
return decimal{d.integral, d.exponent + 1};
}
} // namespace zmij
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
@@ -26351,88 +26347,6 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value)
grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus);
}
/*!
@brief the shortest digits of a positive finite float (other than double): Grisu2
*/
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1)
void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value)
{
grisu2(buf, len, decimal_exponent, value);
}
/*!
@brief the shortest digits of a positive finite double: the conversion of
Zmij (see zmij.hpp), which always finds the shortest digits that read back as
the same value (Grisu2 does not for about one double in a thousand), and the
closest of them if there are several
v = buf * 10^decimal_exponent, as for grisu2()
*/
JSON_HEDLEY_NON_NULL(1)
inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value)
{
static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
JSON_ASSERT(std::isfinite(value));
JSON_ASSERT(value > 0);
std::uint64_t bits = 0;
std::memcpy(&bits, &value, sizeof(bits));
zmij::decimal d = zmij::to_decimal(bits);
// without trailing zeros (up to 16): 8, 4, 2, 1 at a time
while (d.significand % 100000000 == 0)
{
d.significand /= 100000000;
d.exponent += 8;
}
if (d.significand % 10000 == 0)
{
d.significand /= 10000;
d.exponent += 4;
}
if (d.significand % 100 == 0)
{
d.significand /= 100;
d.exponent += 2;
}
if (d.significand % 10 == 0)
{
d.significand /= 10;
d.exponent += 1;
}
// at most 17 digits, written from the back two at a time
static constexpr const char* pairs =
"00010203040506070809101112131415161718192021222324252627282930313233343536373839"
"40414243444546474849505152535455565758596061626364656667686970717273747576777879"
"8081828384858687888990919293949596979899";
std::array<char, 20> digits{};
std::size_t n = digits.size();
while (d.significand >= 100)
{
const std::uint64_t two_digits = d.significand % 100; // a variable: GCC calls a cast of the remainder useless where std::uint64_t is std::size_t
const auto i = static_cast<std::size_t>(two_digits) * 2;
d.significand /= 100;
n -= 2;
digits[n] = pairs[i];
digits[n + 1] = pairs[i + 1];
}
if (d.significand >= 10)
{
const auto i = static_cast<std::size_t>(d.significand) * 2;
n -= 2;
digits[n] = pairs[i];
digits[n + 1] = pairs[i + 1];
}
else
{
digits[--n] = static_cast<char>('0' + d.significand);
}
len = static_cast<int>(digits.size() - n);
std::memcpy(buf, digits.data() + n, static_cast<std::size_t>(len));
decimal_exponent = d.exponent;
}
/*!
@brief appends a decimal representation of e to buf
@return a pointer to the element following the exponent.
@@ -26877,11 +26791,30 @@ inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noe
return write_short_decimal(first, digits, count, exp);
}
/// a positive finite float (other than double): Grisu2 and format_buffer()
/*!
@brief whether FloatType is an IEEE 754 binary64 type (a double, or a long double
that has the same format, as with MSVC and on Apple's Arm CPUs)
These are the types the conversion of Zmij (see zmij.hpp) is used for; all
others (binary32, or a format the library does not know) use Grisu2.
*/
template<typename FloatType>
constexpr bool has_binary64_format() noexcept
{
return std::numeric_limits<FloatType>::is_iec559
&& std::numeric_limits<FloatType>::digits == 53
&& std::numeric_limits<FloatType>::max_exponent == 1024
&& sizeof(FloatType) == sizeof(std::uint64_t);
}
template<typename FloatType>
struct is_binary64 : std::integral_constant<bool, has_binary64_format<FloatType>()> {};
/// a positive finite float (other than binary64): Grisu2 and format_buffer()
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value)
char* write_positive_grisu2(char* first, const char* last, FloatType value)
{
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10);
static_cast<void>(last); // (only used in the assertion)
@@ -26892,7 +26825,7 @@ char* write_positive(char* first, const char* last, FloatType value)
// len is the length of the buffer, i.e., the number of decimal digits.
int len = 0;
int decimal_exponent = 0;
shortest_digits(first, len, decimal_exponent, value);
grisu2(first, len, decimal_exponent, value);
JSON_ASSERT(len <= std::numeric_limits<FloatType>::max_digits10);
@@ -26908,15 +26841,16 @@ char* write_positive(char* first, const char* last, FloatType value)
return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp);
}
/// a positive finite double: the shortest digits (Zmij), laid out by
/// a positive finite binary64 number: the shortest digits (Zmij), laid out by
/// write_shortest() (through a local buffer if [first, last) is shorter than
/// the 41 bytes it may write)
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
inline char* write_positive(char* first, const char* last, double value)
char* write_positive_zmij(char* first, const char* last, FloatType value)
{
static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
static_assert(is_binary64<FloatType>::value,
"internal error: the conversion of Zmij needs IEEE 754 binary64 numbers");
std::uint64_t bits = 0;
std::memcpy(&bits, &value, sizeof(bits));
const zmij::shortest_decimal d = zmij::to_shortest(bits);
@@ -26931,6 +26865,34 @@ inline char* write_positive(char* first, const char* last, double value)
return first + len;
}
/// a positive finite binary64 number: Zmij (as a long double has the format of
/// a double here, its bits are those of the double of the same value)
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value, std::true_type /*is_binary64*/)
{
return write_positive_zmij(first, last, value);
}
/// a positive finite float of any other format: Grisu2
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value, std::false_type /*is_binary64*/)
{
return write_positive_grisu2(first, last, value);
}
/// a positive finite float: Zmij for binary64 numbers, Grisu2 otherwise
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value)
{
return write_positive(first, last, value, is_binary64<FloatType> {});
}
} // namespace dtoa_impl
/*!
File diff suppressed because it is too large. Load diff
+4 -1
View File
@@ -6,7 +6,10 @@ Each parser of the library (JSON, BJData, BON8, BSON, CBOR, MessagePack, and UBJ
Additionally, `parse_json_view_fuzzer` (`tests/src/fuzzer-parse_json_view.cpp`) cross-checks `json_document`/`json_view`
(the zero-copy, read-only view declared in `json_view.hpp`) against `basic_json` on the same JSON text: it asserts that
`json_document::accept` agrees with `json::accept`, that an accepted input materializes to the same value `json::parse`
produces, and that a rejected input makes both parsers throw with an identical `what()`. It takes plain JSON text, so it
produces, and that a rejected input makes both parsers throw with an identical `what()`. It checks this for a
`std::string` input (borrowed, with a NUL after the last byte) and for an exact-size `std::vector<std::uint8_t>`
(borrowed, with nothing after the last byte), and for the `ignore_comments` and `ignore_trailing_commas` options, which
are taken from the low bits of the first input byte (the byte stays part of the text). It takes plain JSON text, so it
reuses the `corpus_json` corpus rather than a format of its own.
`json_view_image_fuzzer` (`tests/src/fuzzer-json_view_image.cpp`) tests the images of `json_document` (`save()` and
+89 -34
View File
@@ -9,7 +9,9 @@
/*
This file implements a parser test suitable for fuzz testing. It checks that
json_document (the zero-copy, read-only view of a parsed JSON text declared in
json_view.hpp) agrees with basic_json on every input:
json_view.hpp) agrees with basic_json on every input, for the parse options
selected by the low bits of the first input byte (bit 0: ignore_comments, bit 1:
ignore_trailing_commas; the byte stays part of the text):
- json_document::accept(data) must equal json::accept(data)
- if the input is accepted, json_document::parse(data).root().materialize()
@@ -18,12 +20,20 @@ json_view.hpp) agrees with basic_json on every input:
enabled) must throw a json::parse_error or json::out_of_range whose what()
is identical to the one json::parse(data) throws
This is checked for two kinds of input: a std::string, which the document
borrows and which ends in the NUL the parser uses as sentinel, and an
exact-size byte vector, which has no NUL after its last byte and takes the
parser's bounds-checked path (AddressSanitizer reports any read past the end).
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
drivers.
*/
#include <cassert>
#include <cstddef>
#include <cstdint>
#include <string>
#include <vector>
#include <nlohmann/json.hpp>
#include <nlohmann/json_view.hpp>
@@ -35,65 +45,110 @@ drivers.
using json = nlohmann::json;
using json_document = nlohmann::json_document;
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
namespace
{
// json_document::accept only has a single-argument overload; wrap the raw
// bytes in a (borrowed) std::string so the same bytes can be handed to it
const std::string input(reinterpret_cast<const char*>(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
// what json::parse does with a text: the value, or the message of the exception
struct reference_result
{
bool accepted = false;
json value{};
std::string what{}; // NOLINT(readability-redundant-member-init)
};
const bool accepted_by_json = json::accept(data, data + size);
const bool accepted_by_view = json_document::accept(input);
reference_result parse_reference(const std::uint8_t* data, std::size_t size, bool comments, bool trailing_commas)
{
reference_result r;
r.accepted = json::accept(data, data + size, comments, trailing_commas);
bool json_threw = false;
try
{
r.value = json::parse(data, data + size, nullptr, true, comments, trailing_commas);
}
catch (const json::parse_error& e)
{
r.what = e.what();
json_threw = true;
}
catch (const json::out_of_range& e)
{
r.what = e.what();
json_threw = true;
}
// json::accept and json::parse must agree
assert(json_threw == !r.accepted);
static_cast<void>(json_threw);
return r;
}
// json_document must agree with the reference for this input (a container
// that json_document::parse borrows)
template<typename Input>
void check_input(const Input& input, const reference_result& expected, bool comments, bool trailing_commas)
{
// json_document::accept must agree with json::accept on every input
assert(accepted_by_json == accepted_by_view);
const bool accepted_by_view = json_document::accept(input, comments, trailing_commas);
assert(expected.accepted == accepted_by_view);
static_cast<void>(accepted_by_view);
if (accepted_by_json)
if (expected.accepted)
{
// both parsers must agree on the resulting value
json const j1 = json::parse(data, data + size);
json_document const doc = json_document::parse(input);
json_document const doc = json_document::parse(input, true, comments, trailing_commas);
assert(!doc.is_discarded());
json const j2 = doc.root().materialize();
assert(j1 == j2);
assert(expected.value == j2);
static_cast<void>(j2);
// (without exceptions, the same document)
json_document const quiet = json_document::parse(input, false, comments, trailing_commas);
assert(!quiet.is_discarded());
assert(quiet.node_count() == doc.node_count());
}
else
{
// both parsers must reject the input the same way when exceptions are used
std::string expected_what;
bool json_threw = false;
try
{
static_cast<void>(json::parse(data, data + size));
}
catch (const json::parse_error& e)
{
expected_what = e.what();
json_threw = true;
}
catch (const json::out_of_range& e)
{
expected_what = e.what();
json_threw = true;
}
assert(json_threw);
bool view_threw = false;
try
{
static_cast<void>(json_document::parse(input));
static_cast<void>(json_document::parse(input, true, comments, trailing_commas));
}
catch (const json::parse_error& e)
{
assert(e.what() == expected_what);
assert(e.what() == expected.what);
view_threw = true;
}
catch (const json::out_of_range& e)
{
assert(e.what() == expected_what);
assert(e.what() == expected.what);
view_threw = true;
}
assert(view_threw);
static_cast<void>(view_threw);
// and without exceptions, the document is discarded
json_document const quiet = json_document::parse(input, false, comments, trailing_commas);
assert(quiet.is_discarded());
static_cast<void>(quiet);
}
}
} // namespace
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
// the parse options are taken from the low bits of the first byte
const bool comments = size > 0 && (data[0] & 1U) != 0;
const bool trailing_commas = size > 0 && (data[0] & 2U) != 0;
const reference_result expected = parse_reference(data, size, comments, trailing_commas);
// a std::string: borrowed, with the NUL of std::string as sentinel
const std::string input(reinterpret_cast<const char*>(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
check_input(input, expected, comments, trailing_commas);
// an exact-size byte vector: borrowed, with nothing after its last byte
const std::vector<std::uint8_t> exact(data, data + size);
check_input(exact, expected, comments, trailing_commas);
// return 0 - non-zero return values are reserved for future use
return 0;
+7
View File
@@ -1792,4 +1792,11 @@ TEST_CASE("string scanning kernels")
CHECK(nlohmann::detail::count_trailing_zeros(bit) == k);
CHECK(nlohmann::detail::count_trailing_zeros(bit | (bit << 1u) | 0x8000000000000000u) == k);
}
// eight bytes as a little-endian word, at any alignment
const unsigned char bytes[16] = {0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF};
CHECK(nlohmann::detail::read_eight_bytes(bytes) == 0x0807060504030201u);
CHECK(nlohmann::detail::read_eight_bytes(bytes + 1) == 0x0908070605040302u);
CHECK(nlohmann::detail::read_eight_bytes(bytes + 8) == 0xFF0F0E0D0C0B0A09u);
CHECK(nlohmann::detail::read_eight_bytes(reinterpret_cast<const char*>(bytes) + 3) == 0x0B0A090807060504u);
}
+585 -50
View File
@@ -19,16 +19,19 @@ using nlohmann::ordered_json_view;
#include <algorithm>
#include <array>
#include <cmath>
#include <cstddef>
#include <cstdint>
#include <cstdio>
#include <cstring>
#include <iomanip>
#include <iterator>
#include <limits>
#include <list>
#include <map>
#include <random>
#include <sstream>
#include <string>
#include <type_traits>
#include <unordered_map>
#include <utility>
#include <vector>
@@ -39,6 +42,55 @@ using nlohmann::ordered_json_view;
namespace
{
// the value of a text, through a named document: the views of a temporary
// document would dangle (root() of an rvalue document does not compile)
template<typename Document, typename... Args>
auto materialized(Args&& ... args) -> decltype(std::declval<typename Document::view_type>().materialize())
{
const Document d = Document::parse(std::forward<Args>(args)...);
return d.root().materialize();
}
template<typename Document, typename Input>
auto materialized_copy(Input&& input) -> decltype(std::declval<typename Document::view_type>().materialize())
{
const Document d = Document::parse_copy(std::forward<Input>(input));
return d.root().materialize();
}
// a "byte container" that claims to hold `size` bytes, to reach the limit on
// the size of the input without allocating gigabytes; nothing past the first
// bytes is ever read, because the size is checked before the parse starts
struct oversized_input
{
using value_type = char;
std::size_t claimed;
const char* data() const
{
return "[1]";
}
std::size_t size() const
{
return claimed;
}
};
// detection of calls that must not compile
template<typename... Args>
using parse_call_t = decltype(json_document::parse(std::declval<Args>()...));
template<typename... Args>
using parse_copy_call_t = decltype(json_document::parse_copy(std::declval<Args>()...));
template<typename... Args>
using accept_call_t = decltype(json_document::accept(std::declval<Args>()...));
template<typename... Args>
using read_call_t = decltype(std::declval<json_document&>().read(std::declval<Args>()...));
template<typename Document>
using root_call_t = decltype(std::declval<Document>().root());
template<typename View>
using bool_conversion_t = decltype(static_cast<bool>(std::declval<View>()));
#if !defined(JSON_NOEXCEPTION)
// the exception parse() throws for a text, or "" if it accepts it
std::string parse_exception(const std::string& text, bool comments = false, bool trailing_commas = false)
@@ -152,7 +204,6 @@ TEST_CASE("json_view")
CHECK(v.is_primitive() == j.is_primitive());
CHECK(v.is_structured() == j.is_structured());
CHECK(!v.is_discarded());
CHECK(static_cast<bool>(v));
CHECK(v.size() == j.size());
CHECK(v.empty() == j.empty());
CHECK(v.materialize() == j);
@@ -160,7 +211,6 @@ TEST_CASE("json_view")
const json_view invalid{};
CHECK(invalid.is_discarded());
CHECK(!static_cast<bool>(invalid));
CHECK(invalid.type() == json::value_t::discarded);
CHECK(invalid.size() == 0);
CHECK(invalid.empty());
@@ -176,19 +226,19 @@ TEST_CASE("json_view")
std::string text;
g.value(text, 0);
CAPTURE(text)
CHECK(json_document::parse(text).root().materialize() == json::parse(text));
CHECK(materialized<json_document>(text) == json::parse(text));
// member order as ordered_json::parse keeps it
CHECK(ordered_json_document::parse(text).root().materialize().dump() == ordered_json::parse(text).dump());
CHECK(materialized<ordered_json_document>(text).dump() == ordered_json::parse(text).dump());
}
// duplicate keys: the last value, at the position of the first key
CHECK(json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize() == json::parse(R"({"a":1,"b":2,"a":3})"));
CHECK(ordered_json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize().dump() == R"({"a":3,"b":2})");
CHECK(materialized<json_document>(R"({"a":1,"b":2,"a":3})") == json::parse(R"({"a":1,"b":2,"a":3})"));
CHECK(materialized<ordered_json_document>(R"({"a":1,"b":2,"a":3})").dump() == R"({"a":3,"b":2})");
// very deep nesting (iterative, as parse())
const std::string deep = std::string(100000, '[') + std::string(100000, ']');
CHECK(json_document::parse(deep).root().materialize() == json::parse(deep));
CHECK(materialized<json_document>(deep) == json::parse(deep));
#if JSON_DIAGNOSTICS
// the parents are set, so errors name the path
const json m = json_document::parse(R"({"a":{"b":[1]}})").root().materialize();
const json m = materialized<json_document>(R"({"a":{"b":[1]}})");
CHECK_THROWS_WITH_AS(m.at("a").at("b").at(0).at("x"), "[json.exception.type_error.304] (/a/b/0) cannot use at() with number", json::type_error&);
#endif
}
@@ -261,7 +311,7 @@ TEST_CASE("json_view")
CHECK(json_document::accept(text));
if (accepted)
{
CHECK(float_document::parse(text).root().materialize() == json_float::parse(text));
CHECK(materialized<float_document>(text) == json_float::parse(text));
}
}
float_document f;
@@ -275,7 +325,7 @@ TEST_CASE("json_view")
CHECK(json_document::accept(with_nul) == json::accept(with_nul));
const std::string nul_in_comment("[1, // c\0\n2]", 12);
CHECK(json_document::accept(nul_in_comment, true) == json::accept(nul_in_comment, true));
CHECK(json_document::parse("\xEF\xBB\xBF[1]").root().materialize() == json::parse("\xEF\xBB\xBF[1]"));
CHECK(materialized<json_document>("\xEF\xBB\xBF[1]") == json::parse("\xEF\xBB\xBF[1]"));
#if !defined(JSON_NOEXCEPTION)
CHECK(view_exception("\xEF\xBB") == parse_exception("\xEF\xBB"));
#endif
@@ -291,18 +341,18 @@ TEST_CASE("json_view")
CHECK(!borrowed.owns_source());
CHECK(borrowed.source().data() == text.data());
CHECK(borrowed.root().materialize() == expected);
CHECK(json_document::parse(text.c_str()).root().materialize() == expected);
CHECK(json_document::parse(R"([1, "two", {"three": 3.5}])").root().materialize() == expected);
CHECK(json_document::parse(text.data(), text.data() + text.size()).root().materialize() == expected);
CHECK(materialized<json_document>(text.c_str()) == expected);
CHECK(materialized<json_document>(R"([1, "two", {"three": 3.5}])") == expected);
CHECK(materialized<json_document>(text.data(), text.data() + text.size()) == expected);
const std::vector<char> chars(text.begin(), text.end());
CHECK(!json_document::parse(chars).owns_source());
CHECK(json_document::parse(chars).root().materialize() == expected);
CHECK(materialized<json_document>(chars) == expected);
const std::vector<std::uint8_t> bytes(text.begin(), text.end());
CHECK(json_document::parse(bytes).root().materialize() == expected);
CHECK(materialized<json_document>(bytes) == expected);
#ifdef JSON_HAS_CPP_17
const std::string_view sv = text;
CHECK(!json_document::parse(sv).owns_source());
CHECK(json_document::parse(sv).root().materialize() == expected);
CHECK(materialized<json_document>(sv) == expected);
#endif
// owned
@@ -311,15 +361,28 @@ TEST_CASE("json_view")
CHECK(from_rvalue.owns_source());
CHECK(from_rvalue.root().materialize() == expected);
CHECK(json_document::parse(std::vector<char>(text.begin(), text.end())).owns_source());
// a const rvalue cannot be moved from, and is not borrowed (it may be a
// temporary): it is copied, as is a const rvalue of any container
const std::string const_text = text;
const json_document from_const_rvalue = json_document::parse(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg)
CHECK(from_const_rvalue.owns_source());
CHECK(from_const_rvalue.source().data() != const_text.data());
CHECK(from_const_rvalue.root().materialize() == expected);
json_document read_const_rvalue;
read_const_rvalue.read(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg)
CHECK(read_const_rvalue.owns_source());
CHECK(read_const_rvalue.root().materialize() == expected);
const std::vector<char> const_chars(text.begin(), text.end());
CHECK(json_document::parse(std::move(const_chars)).owns_source()); // NOLINT(performance-move-const-arg,hicpp-move-const-arg)
CHECK(json_document::parse_copy(text).owns_source());
CHECK(json_document::parse_copy(text).root().materialize() == expected);
CHECK(materialized_copy<json_document>(text) == expected);
std::istringstream stream(text);
const json_document from_stream = json_document::parse(stream);
CHECK(from_stream.owns_source());
CHECK(from_stream.root().materialize() == expected);
const std::list<char> list(text.begin(), text.end());
CHECK(json_document::parse(list.begin(), list.end()).owns_source());
CHECK(json_document::parse(list.begin(), list.end()).root().materialize() == expected);
CHECK(materialized<json_document>(list.begin(), list.end()) == expected);
// iterator pairs: pointers are borrowed, and so are contiguous library
// iterators where the input adapter detects them (C++20)
@@ -330,14 +393,85 @@ TEST_CASE("json_view")
CHECK((from_iterators.source().data() == chars.data()) == contiguous);
CHECK(from_iterators.root().materialize() == expected);
const std::string padded = "x" + text + "x";
CHECK(json_document::parse(padded.begin() + 1, padded.end() - 1).root().materialize() == expected);
CHECK(materialized<json_document>(padded.begin() + 1, padded.end() - 1) == expected);
CHECK(json_document::parse(chars.cbegin(), chars.cbegin(), false).is_discarded());
const std::wstring wide = L"[\"\u00e4\u20ac\", 1]";
CHECK(json_document::parse(wide).root().materialize() == json::parse(wide));
CHECK(materialized<json_document>(wide) == json::parse(wide));
CHECK(json_document::parse(static_cast<const char*>(nullptr), false).is_discarded());
CHECK(json_document::parse("", false).is_discarded());
}
SECTION("integer arguments do not compile")
{
using nlohmann::detail::is_detected;
// a length is not a flag: parse(ptr, len) would convert len to
// allow_exceptions and read ptr as a C string, which need not end
static_assert(is_detected<parse_call_t, const char*, bool>::value, "parse(ptr, bool) is valid");
static_assert(is_detected<parse_call_t, const char*, bool, bool, bool>::value, "parse(ptr, bool, bool, bool) is valid");
static_assert(is_detected<parse_call_t, const char*, const char*>::value, "parse(first, last) is valid");
static_assert(is_detected<parse_call_t, const char*, const char*, bool>::value, "parse(first, last, bool) is valid");
static_assert(!is_detected<parse_call_t, const char*, std::size_t>::value, "parse(ptr, len) must not compile");
static_assert(!is_detected<parse_call_t, const char*, int>::value, "parse(ptr, int) must not compile");
static_assert(!is_detected<parse_call_t, const char*, char>::value, "parse(ptr, char) must not compile");
static_assert(!is_detected<parse_call_t, const char*, std::size_t, bool>::value, "parse(ptr, len, bool) must not compile");
static_assert(!is_detected<parse_call_t, const std::string&, std::size_t>::value, "parse(string, len) must not compile");
static_assert(!is_detected<parse_call_t, const std::vector<char>&, std::size_t>::value, "parse(vector, len) must not compile");
static_assert(is_detected<parse_copy_call_t, const char*, bool>::value, "parse_copy(ptr, bool) is valid");
static_assert(!is_detected<parse_copy_call_t, const char*, std::size_t>::value, "parse_copy(ptr, len) must not compile");
static_assert(is_detected<accept_call_t, const char*, bool>::value, "accept(ptr, bool) is valid");
static_assert(!is_detected<accept_call_t, const char*, std::size_t>::value, "accept(ptr, len) must not compile");
static_assert(is_detected<read_call_t, const char*, bool>::value, "read(ptr, bool) is valid");
static_assert(!is_detected<read_call_t, const char*, std::size_t>::value, "read(ptr, len) must not compile");
// json_view has no conversion to bool: unlike basic_json's, it would
// mean "exists", not "is not null"; use is_discarded()
static_assert(!is_detected<bool_conversion_t, json_view>::value, "json_view must not convert to bool");
// the valid calls still work
const char* const text = "[1]";
CHECK(materialized<json_document>(text, true) == json::parse(text));
CHECK(json_document::accept(text, true, true));
}
SECTION("root of a temporary document does not compile")
{
using nlohmann::detail::is_detected;
// the view would dangle: auto v = json_document::parse(text).root();
static_assert(is_detected<root_call_t, json_document&>::value, "root() of an lvalue is valid");
static_assert(is_detected<root_call_t, const json_document&>::value, "root() of a const lvalue is valid");
static_assert(!is_detected<root_call_t, json_document>::value, "root() of an rvalue must not compile");
static_assert(!is_detected < root_call_t, json_document && >::value, "root() of an rvalue must not compile");
static_assert(!is_detected < root_call_t, const json_document && >::value, "root() of a const rvalue must not compile");
static_assert(!is_detected<root_call_t, ordered_json_document>::value, "root() of an rvalue must not compile");
// a named document is fine, also after a move
json_document d = json_document::parse("[1]");
CHECK(d.root().size() == 1);
const json_document moved = std::move(d);
CHECK(moved.root().size() == 1);
}
SECTION("input size limit")
{
// 32-bit offsets: the limit is 4 GiB minus 16 bytes (a margin below 2^32),
// which is what the exception message and the documentation say
const std::size_t limit = nlohmann::detail::view::max_input_size;
CHECK(limit == std::size_t{4294967279u});
const oversized_input input{limit + 1};
CHECK(!json_document::accept(input));
CHECK(json_document::parse(input, false).is_discarded());
#if !defined(JSON_NOEXCEPTION)
json_document d;
CHECK_THROWS_WITH_AS(d = json_document::parse(input), "[json.exception.out_of_range.416] input of 4294967280 bytes or more is not supported by json_document", json::out_of_range&);
#endif
}
SECTION("document lifetime and reuse")
{
json_document d;
@@ -443,7 +577,7 @@ std::string exception_of(F f)
// compares a view with the ordered_json value materialize() gives for it:
// types, sizes, elements and members (by index, key, and iteration), in
// document order; duplicate keys are found as their first occurrence
// document order; duplicate keys are found as their last occurrence
void check_access(const ordered_json_view& v, const ordered_json& j)
{
REQUIRE(v.type() == j.type());
@@ -460,7 +594,7 @@ void check_access(const ordered_json_view& v, const ordered_json& j)
++i;
}
CHECK(i == v.size());
CHECK(!v[v.size()]);
CHECK(v[v.size()].is_discarded());
std::size_t index = 0;
for (const auto& item : v.items())
{
@@ -484,11 +618,23 @@ void check_access(const ordered_json_view& v, const ordered_json& j)
const std::string key(it.key().data(), it.key().size());
CHECK(v.contains(key));
CHECK(v.count(key) == 1);
if (std::find(keys.begin(), keys.end(), key) != keys.end())
if (std::find(keys.begin(), keys.end(), key) == keys.end())
{
continue; // a duplicate: lookups find the first one
keys.push_back(key);
}
// lookups find the last member with the key, which is this one if
// there is no later one
auto next = it;
++next;
bool is_last = true;
for (; next != v.end(); ++next)
{
is_last = is_last && next.key() != it.key();
}
if (!is_last)
{
continue;
}
keys.push_back(key);
CHECK(v.find(key) == it);
CHECK(v[key].materialize() == it->materialize());
CHECK(v.at(key).materialize() == it.value().materialize());
@@ -514,7 +660,7 @@ void check_access(const ordered_json_view& v, const ordered_json& j)
CHECK(v.back().materialize() == j.back());
}
}
CHECK(!v["not a key in the generated documents"]);
CHECK(v["not a key in the generated documents"].is_discarded());
CHECK(v.find("not a key in the generated documents") == v.end());
}
else
@@ -579,15 +725,35 @@ TEST_CASE("json_view element access and iteration")
#endif
}
SECTION("duplicate keys: lookups find the first member, iteration all")
SECTION("duplicate keys: lookups find the last member, iteration all")
{
const json_document d = json_document::parse(R"({"a":1,"b":2,"a":3})");
const json_view v = d.root();
CHECK(v.size() == 3);
CHECK(v["a"].materialize() == 1);
CHECK(v.at("a").materialize() == 1);
CHECK(v.find("a") == v.begin());
CHECK(v["a"].materialize() == 3);
CHECK(v.at("a").materialize() == 3);
CHECK(v.find("a") == std::next(v.begin(), 2));
CHECK(v.find("a").value().materialize() == 3);
CHECK(v.find("b") == std::next(v.begin()));
CHECK(v.count("a") == 1);
CHECK(v.contains("a"));
CHECK(v.value("a", 0) == 3);
CHECK(v["a"].materialize() == v.materialize()["a"]); // as materialize()
// keys of every length class (the 16-byte short compare and memcmp)
for (const std::size_t n :
{
0u, 1u, 3u, 7u, 8u, 15u, 16u, 17u, 40u
})
{
const std::string key(n, 'k');
const json_document dk = json_document::parse("{\"" + key + "\":1,\"" + key + "x\":2,\"" + key + "\":3,\"" + key + "\":4}");
CAPTURE(n)
CHECK(dk.root()[key].materialize() == 4);
CHECK(dk.root().at(key).materialize() == 4);
CHECK(dk.root().find(key) == std::next(dk.root().begin(), 3));
CHECK(dk.root().value(key, 0) == 4);
CHECK(dk.root()[key + "x"].materialize() == 2);
}
std::string order;
for (auto it = v.begin(); it != v.end(); ++it)
{
@@ -638,14 +804,131 @@ TEST_CASE("json_view element access and iteration")
// where basic_json has undefined behavior, the view answers safely
const json_document d = json_document::parse(R"({"a":[]})");
CHECK(!d.root()["b"]);
CHECK(!d.root()["a"][0]);
CHECK(d.root()["b"].is_discarded());
CHECK(d.root()["a"][0].is_discarded());
CHECK_THROWS_WITH_AS(d.root()["a"].front(), "[json.exception.invalid_iterator.214] cannot get value", json::invalid_iterator&);
CHECK_THROWS_WITH_AS(d.root()["a"].back(), "[json.exception.invalid_iterator.214] cannot get value", json::invalid_iterator&);
const json_view invalid{};
CHECK(invalid.begin() == invalid.end());
CHECK(std::string(invalid.type_name()) == "discarded");
CHECK_THROWS_WITH_AS(invalid["a"], "[json.exception.type_error.305] cannot use operator[] with a string argument with discarded", json::type_error&);
}
SECTION("chained access is safe: operator[] of a discarded view is discarded")
{
const json_document d = json_document::parse(R"({"a":{"b":[10,20]},"s":"str"})");
const json_view v = d.root();
// missing keys and indexes
CHECK(v["x"].is_discarded());
CHECK(v["x"]["y"].is_discarded());
CHECK(v["x"]["y"]["z"].is_discarded());
CHECK(v["x"][0].is_discarded());
CHECK(v["x"][0u][1L].is_discarded());
CHECK(v["a"]["b"][2].is_discarded());
CHECK(v["a"]["b"][2]["c"].is_discarded());
CHECK(v["a"]["b"][2][json_view::json_pointer("/c")].is_discarded());
CHECK(v["x"][json_view::json_pointer("/a/b")].is_discarded());
CHECK(v["x"][json_view::json_pointer("")].is_discarded());
CHECK(v["x"]["y"].is_discarded());
CHECK(v["x"][std::string("y")].is_discarded());
// a resolvable path still resolves
CHECK(v["a"]["b"][1].materialize() == 20);
CHECK(v[json_view::json_pointer("/a/b/1")].materialize() == 20);
// the discarded view of an unresolved pointer is discarded too
CHECK(v[json_view::json_pointer("/x/y")]["z"].is_discarded());
CHECK(v[json_view::json_pointer("/a/b/5")][0].is_discarded());
#if !defined(JSON_NOEXCEPTION)
// type errors on values that are not discarded stay
CHECK_THROWS_WITH_AS(v[0], "[json.exception.type_error.305] cannot use operator[] with a numeric argument with object", json::type_error&);
CHECK_THROWS_WITH_AS(v["a"]["b"]["c"], "[json.exception.type_error.305] cannot use operator[] with a string argument with array", json::type_error&);
CHECK_THROWS_WITH_AS(v["s"]["c"], "[json.exception.type_error.305] cannot use operator[] with a string argument with string", json::type_error&);
CHECK_THROWS_WITH_AS(v["s"][0], "[json.exception.type_error.305] cannot use operator[] with a numeric argument with string", json::type_error&);
CHECK_THROWS_AS(v["a"]["b"][0]["c"], json::type_error&);
CHECK_THROWS_AS(v["s"][json_view::json_pointer("/x")], json::out_of_range&);
// at() keeps throwing on a discarded view
const json_view invalid{};
CHECK(invalid["a"].is_discarded());
CHECK(invalid[0].is_discarded());
CHECK(invalid[json_view::json_pointer("/a")].is_discarded());
CHECK_THROWS_WITH_AS(invalid.at("a"), "[json.exception.type_error.304] cannot use at() with discarded", json::type_error&);
CHECK_THROWS_WITH_AS(invalid.at(0), "[json.exception.type_error.304] cannot use at() with discarded", json::type_error&);
CHECK_THROWS_AS(v.at("x").at("y"), json::out_of_range&);
CHECK_THROWS_AS(v["x"].at("y"), json::type_error&);
CHECK_THROWS_AS(v.at(json_view::json_pointer("/x/y")), json::out_of_range&);
CHECK_THROWS_AS(v["x"].at(json_view::json_pointer("/y")), json::out_of_range&);
#endif
}
SECTION("integer types as array indexes")
{
const json_document d = json_document::parse("[10,20,30]");
const json_view v = d.root();
const json j = v.materialize();
// (compile-time: no overload is ambiguous)
CHECK(v[0].materialize() == 10);
CHECK(v[1].materialize() == 20);
CHECK(v[0u].materialize() == 10);
CHECK(v[1u].materialize() == 20);
CHECK(v[1L].materialize() == 20);
CHECK(v[2UL].materialize() == 30);
CHECK(v[1LL].materialize() == 20);
CHECK(v[2ULL].materialize() == 30);
CHECK(v[static_cast<short>(1)].materialize() == 20);
CHECK(v[static_cast<unsigned short>(2)].materialize() == 30);
CHECK(v[static_cast<signed char>(1)].materialize() == 20);
CHECK(v[static_cast<unsigned char>(2)].materialize() == 30);
CHECK(v[std::int8_t(1)].materialize() == 20);
CHECK(v[std::int16_t(2)].materialize() == 30);
CHECK(v[std::int32_t(1)].materialize() == 20);
CHECK(v[std::int64_t(2)].materialize() == 30);
CHECK(v[std::uint32_t(0)].materialize() == 10);
CHECK(v[std::uint64_t(1)].materialize() == 20);
CHECK(v[std::size_t(2)].materialize() == 30);
CHECK(v[std::ptrdiff_t(1)].materialize() == 20);
CHECK(j[0u] == 10); // as basic_json
CHECK(v.at(0).materialize() == 10);
CHECK(v.at(1u).materialize() == 20);
CHECK(v.at(1L).materialize() == 20);
CHECK(v.at(2LL).materialize() == 30);
CHECK(v.at(2ULL).materialize() == 30);
CHECK(v.at(static_cast<short>(1)).materialize() == 20);
CHECK(v.at(static_cast<unsigned short>(2)).materialize() == 30);
CHECK(v.at(std::int32_t(0)).materialize() == 10);
CHECK(v.at(std::uint32_t(0)).materialize() == 10);
CHECK(v.at(std::int64_t(0)).materialize() == 10);
CHECK(v.at(std::uint64_t(1)).materialize() == 20);
CHECK(v.at(std::size_t(2)).materialize() == 30);
CHECK(j.at(std::uint32_t(0)) == 10); // as basic_json
// out of range, including negative values (no wrap-around)
CHECK(v[3].is_discarded());
CHECK(v[3u].is_discarded());
CHECK(v[3L].is_discarded());
CHECK(v[-1].is_discarded());
CHECK(v[-1L].is_discarded());
CHECK(v[-1LL].is_discarded());
CHECK(v[static_cast<short>(-1)].is_discarded());
CHECK(v[std::int64_t(-3)].is_discarded());
CHECK(v[(std::numeric_limits<std::int64_t>::min)()].is_discarded());
CHECK(v[(std::numeric_limits<std::uint64_t>::max)()].is_discarded());
CHECK(v[(std::numeric_limits<std::size_t>::max)()].is_discarded());
CHECK(v[std::numeric_limits<int>::max()].is_discarded());
#if !defined(JSON_NOEXCEPTION)
CHECK_THROWS_WITH_AS(v.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&);
CHECK_THROWS_WITH_AS(v.at(3u), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&);
CHECK_THROWS_WITH_AS(v.at(std::int64_t(3)), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&);
CHECK_THROWS_AS(v.at(-1), json::out_of_range&);
CHECK_THROWS_AS(v.at(-1L), json::out_of_range&);
CHECK_THROWS_AS(v.at(std::int64_t(-1)), json::out_of_range&);
CHECK_THROWS_AS(v.at((std::numeric_limits<std::int64_t>::min)()), json::out_of_range&);
CHECK_THROWS_AS(v.at((std::numeric_limits<std::uint64_t>::max)()), json::out_of_range&);
// not an array
const json_document o = json_document::parse("{}");
CHECK_THROWS_AS(o.root()[0u], json::type_error&);
CHECK_THROWS_AS(o.root()[1L], json::type_error&);
CHECK_THROWS_AS(o.root().at(std::uint32_t(0)), json::type_error&);
CHECK_THROWS_AS(o.root().at(std::int64_t(0)), json::type_error&);
#endif
}
SECTION("iterators")
@@ -919,10 +1202,12 @@ TEST_CASE("json_view values")
CAPTURE(token)
const std::string text = "[" + token + "]";
const double b = json::parse(text)[0].get<double>();
CHECK(bits(json_document::parse(text).root()[0].get<double>()) == bits(b));
const json_document dd = json_document::parse(text);
CHECK(bits(dd.root()[0].get<double>()) == bits(b));
if (std::abs(b) < 1e38)
{
CHECK(bits(nlohmann::basic_json_document<json_float>::parse(text).root()[0].get<float>()) == bits(json_float::parse(text)[0].get<float>()));
const nlohmann::basic_json_document<json_float> df = nlohmann::basic_json_document<json_float>::parse(text);
CHECK(bits(df.root()[0].get<float>()) == bits(json_float::parse(text)[0].get<float>()));
}
}
}
@@ -979,7 +1264,8 @@ TEST_CASE("json_view values")
CHECK(count == 3);
// a duplicate key: the last value, as parse()
CHECK((json_document::parse(R"({"a":1,"a":2})").root().get<std::map<std::string, int>>() == std::map<std::string, int> {{"a", 2}}));
const json_document dup = json_document::parse(R"({"a":1,"a":2})");
CHECK((dup.root().get<std::map<std::string, int>>() == std::map<std::string, int> {{"a", 2}}));
const json_view invalid{};
CHECK_THROWS_WITH_AS(invalid.get<int>(), "[json.exception.type_error.302] type must be number, but is discarded", json::type_error&);
@@ -1070,7 +1356,7 @@ TEST_CASE("json_view JSON pointers")
else if (at_error.find("out_of_range.401") != std::string::npos || at_error.find("out_of_range.403") != std::string::npos) // NOLINT(abseil-string-find-str-contains)
{
// undefined behavior for const basic_json::operator[]
CHECK(!v[p]);
CHECK(v[p].is_discarded());
}
else
{
@@ -1198,7 +1484,8 @@ TEST_CASE("json_view dump")
many_tokens += (i != 0 ? "," : "") + token;
}
many_tokens += ']';
CHECK(json_document::parse(many_tokens).root().dump() == json::parse(many_tokens).dump());
const json_document many_doc = json_document::parse(many_tokens);
CHECK(many_doc.root().dump() == json::parse(many_tokens).dump());
}
// random doubles, written as parse() and dump() would
@@ -1215,10 +1502,12 @@ TEST_CASE("json_view dump")
}
}
many += ']';
CHECK(json_document::parse(many).root().dump() == json::parse(many).dump());
const json_document many_document = json_document::parse(many);
CHECK(many_document.root().dump() == json::parse(many).dump());
using json_float = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
CHECK(nlohmann::basic_json_document<json_float>::parse("[0.1, 1.5e10, 3.4028235e38]").root().dump() == json_float::parse("[0.1, 1.5e10, 3.4028235e38]").dump());
const nlohmann::basic_json_document<json_float> float_document = nlohmann::basic_json_document<json_float>::parse("[0.1, 1.5e10, 3.4028235e38]");
CHECK(float_document.root().dump() == json_float::parse("[0.1, 1.5e10, 3.4028235e38]").dump());
}
SECTION("members in document order, all of them")
@@ -1231,7 +1520,42 @@ TEST_CASE("json_view dump")
SECTION("deep nesting")
{
const std::string deep = std::string(100000, '[') + std::string(100000, ']');
CHECK(json_document::parse(deep).root().dump() == deep);
const json_document deep_document = json_document::parse(deep);
CHECK(deep_document.root().dump() == deep);
}
SECTION("the output buffer of a small value is small")
{
// an escaped key after the value: its node lies in the arena, so the
// source extent of the value cannot be read from the next node
const std::string big(100000, 'a');
const std::string text = R"({"small":1,"list":[1,2,3],"k\n":")" + big + R"("})";
const json_document d = json_document::parse(text);
const auto small = d.root()["small"].dump();
CHECK(small == "1");
CHECK(small.capacity() < 4096);
const auto list = d.root()["list"].dump();
CHECK(list == "[1,2,3]");
CHECK(list.capacity() < 4096);
CHECK(d.root()["list"].dump(2).capacity() < 4096);
// the whole document and the large value are unaffected
CHECK(d.root().dump() == ordered_json::parse(text).dump());
CHECK(d.root()["k\n"].dump() == "\"" + big + "\"");
}
SECTION("output that outgrows the estimate")
{
// ensure_ascii writes six bytes for each two-byte character
std::string chars;
for (int i = 0; i < 5000; ++i)
{
chars += "\xC3\xA9";
}
const json_document d = json_document::parse("{\"a\":\"" + chars + R"(","k\n":1})");
const json expected = json::parse("\"" + chars + "\"");
CHECK(d.root()["a"].dump(-1, ' ', true) == expected.dump(-1, ' ', true));
CHECK(d.root()["a"].dump(-1, ' ', true).size() == 2 + 5000 * 6);
}
SECTION("streams and discarded views")
@@ -1293,7 +1617,9 @@ TEST_CASE("json_view comparison")
{
const auto same = [](const char* x, const char* y)
{
return json_document::parse(x).root() == json_document::parse(y).root();
const json_document dx = json_document::parse(x);
const json_document dy = json_document::parse(y);
return dx.root() == dy.root();
};
CHECK(same("1", "1.0"));
CHECK(same("[1, -1, 2.5]", "[1.0, -1.0, 25e-1]"));
@@ -1308,15 +1634,20 @@ TEST_CASE("json_view comparison")
CHECK(same("\"\\u00e9\"", "\"\xc3\xa9\""));
CHECK(!same("null", "false"));
CHECK(!same("[]", "{}"));
CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2, "a": 3})").root() == ordered_json_document::parse(R"({"a": 3, "b": 2})").root());
CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2})").root() != ordered_json_document::parse(R"({"b": 2, "a": 1})").root());
const ordered_json_document dup = ordered_json_document::parse(R"({"a": 1, "b": 2, "a": 3})");
const ordered_json_document last = ordered_json_document::parse(R"({"a": 3, "b": 2})");
CHECK(dup.root() == last.root());
const ordered_json_document ab = ordered_json_document::parse(R"({"a": 1, "b": 2})");
const ordered_json_document ba = ordered_json_document::parse(R"({"b": 2, "a": 1})");
CHECK(ab.root() != ba.root());
// discarded values compare as basic_json's do
const json discarded(json::value_t::discarded);
CHECK((json_view() == json_view()) == (discarded == discarded)); // NOLINT(readability-container-size-empty): operator== is tested
CHECK((json_view() == discarded) == (discarded == discarded));
CHECK(!(json_view() == json_document::parse("null").root())); // NOLINT(readability-container-size-empty)
CHECK(!(json_document::parse("null").root() == discarded));
const json_document null_document = json_document::parse("null");
CHECK(!(json_view() == null_document.root())); // NOLINT(readability-container-size-empty)
CHECK(!(null_document.root() == discarded));
}
SECTION("deep nesting")
@@ -1327,7 +1658,8 @@ TEST_CASE("json_view comparison")
CHECK(a.root() == b.root());
CHECK(a.root() == json::parse(deep));
const std::string other = std::string(100000, '[') + "1" + std::string(100000, ']');
CHECK(a.root() != json_document::parse(other).root());
const json_document c = json_document::parse(other);
CHECK(a.root() != c.root());
}
}
@@ -1352,20 +1684,223 @@ TEST_CASE("json_view large objects")
for (std::size_t i = 0; i < members; ++i)
{
const std::string key = std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\n" : "");
CHECK(v[key].get<std::size_t>() == i);
CHECK(v.contains(key));
CHECK(v.find(key).key() == key);
CHECK(v.at(key).get<std::size_t>() == i);
CHECK(!v.contains(key + "x"));
if (key == "k1")
{
continue; // repeated below: the last member wins
}
CHECK(v[key].get<std::size_t>() == i);
CHECK(v.at(key).get<std::size_t>() == i);
}
CHECK(v[""].get_string() == "empty key");
CHECK(v["k1"].get<int>() == 1); // the first of duplicate keys, as for small objects
CHECK(v["k1"].get_string() == "a duplicate of an earlier key"); // the last of duplicate keys, as for small objects
CHECK(v.at("k1").get_string() == "a duplicate of an earlier key");
CHECK(!v.contains("missing"));
CHECK_THROWS_WITH_AS(v.at("missing"), "[json.exception.out_of_range.403] key 'missing' not found", json::out_of_range&);
CHECK(v == j);
CHECK(v.materialize() == j);
}
SECTION("duplicate keys: the last member wins, with and without a table")
{
// an object of `total` members: the keys "k0".."k<n-1>" in order, then
// three keys repeated twice more (one copy in the middle, one at the
// end), and two keys repeated once; the value of a member is its
// position, so that the last member of a key can be told apart
struct member
{
std::string key;
std::size_t position;
};
const auto make_members = [](std::size_t total)
{
std::vector<std::string> keys;
for (std::size_t i = 0; i + 8 < total; ++i)
{
keys.push_back("k" + std::to_string(i));
}
const std::size_t n = keys.size();
const std::array<std::size_t, 3> triple = {{3, 17, n - 1}};
const std::array<std::size_t, 2> twice = {{5, n / 2}};
std::vector<std::string> ordered = keys;
for (const std::size_t i : triple)
{
ordered.insert(ordered.begin() + static_cast<std::ptrdiff_t>(ordered.size() / 2), keys[i]);
}
for (const std::size_t i : twice)
{
ordered.insert(ordered.begin() + static_cast<std::ptrdiff_t>(ordered.size() / 3), keys[i]);
}
for (const std::size_t i : triple)
{
ordered.push_back(keys[i]);
}
std::vector<member> result;
for (std::size_t i = 0; i < ordered.size(); ++i)
{
result.push_back({ordered[i], i});
}
return result;
};
// 100 and 127 members: no table; 128 members and more: a table
for (const std::size_t total :
{
100u, 127u, 128u, 200u, 5000u
})
{
CAPTURE(total)
const std::vector<member> members = make_members(total);
REQUIRE(members.size() >= total);
REQUIRE(members.size() >= 100);
std::string text = "{";
std::map<std::string, std::size_t> last;
for (const member& m : members)
{
text += (text.size() > 1 ? ",\"" : "\"") + m.key + "\":" + std::to_string(m.position);
last[m.key] = m.position;
}
text += '}';
REQUIRE(last.size() < members.size());
const json_document d = json_document::parse(text);
const json_view v = d.root();
const json j = json::parse(text);
CHECK(v.size() == members.size()); // every occurrence is visited
for (const auto& entry : last)
{
CAPTURE(entry.first)
const std::size_t expected = entry.second;
CHECK(v[entry.first].get<std::size_t>() == expected);
CHECK(v.at(entry.first).get<std::size_t>() == expected);
CHECK(v.find(entry.first).value().get<std::size_t>() == expected);
CHECK(v.value(entry.first, std::size_t{0}) == expected);
CHECK(v.contains(entry.first));
CHECK(v.count(entry.first) == 1);
const std::string pointer = "/" + entry.first;
CHECK(v[json::json_pointer(pointer)].get<std::size_t>() == expected);
CHECK(v.at(json::json_pointer(pointer)).get<std::size_t>() == expected);
CHECK(v.value(json::json_pointer(pointer), std::size_t{0}) == expected);
CHECK(v.contains(json::json_pointer(pointer)));
CHECK(j[entry.first].get<std::size_t>() == expected); // as materialize() and parse() keep it
}
CHECK(!v.contains("k"));
CHECK(v["missing"].is_discarded());
CHECK(v.materialize() == j);
CHECK(v == j);
}
}
SECTION("colliding keys")
{
// the hash is not seeded, so keys that all land in one place must not
// make the table build quadratic: such an object gets no table and is
// searched linearly
constexpr std::size_t members = 300;
constexpr std::size_t slots = 1024; // the table size for 300 members: the next power of two >= 600
std::vector<std::string> colliding;
std::vector<std::string> spread;
for (std::uint64_t counter = 0; colliding.size() < members || spread.size() < members; ++counter)
{
std::string key(8, 'a');
for (std::uint64_t x = counter, i = 0; i < 8; ++i, x /= 26)
{
key[i] = static_cast<char>('a' + (x % 26));
}
const bool lands_in_slot_zero = (nlohmann::detail::view::key_hash(key.data(), key.size()) & (slots - 1)) == 0;
if (lands_in_slot_zero && colliding.size() < members)
{
colliding.push_back(key);
}
else if (!lands_in_slot_zero && spread.size() < members)
{
spread.push_back(key);
}
}
const auto make_text = [](const std::vector<std::string>& keys)
{
std::string text = "{";
for (std::size_t i = 0; i < keys.size(); ++i)
{
text += (i != 0 ? ",\"" : "\"") + keys[i] + "\":" + std::to_string(i % 10);
}
return text + "}";
};
const std::string colliding_text = make_text(colliding);
const std::string spread_text = make_text(spread);
json_document with_collisions = json_document::parse(colliding_text);
json_document without_collisions = json_document::parse(spread_text);
for (const auto* pair :
{
&colliding, &spread
})
{
const json_view v = (pair == &colliding ? with_collisions : without_collisions).root();
for (std::size_t i = 0; i < members; ++i)
{
CAPTURE(i)
CHECK(v[(*pair)[i]].get<std::size_t>() == i % 10);
CHECK(v.at((*pair)[i]).get<std::size_t>() == i % 10);
CHECK(v.find((*pair)[i]).key() == (*pair)[i]);
CHECK(!v.contains((*pair)[i] + "x"));
}
CHECK(!v.contains("missing"));
}
CHECK(with_collisions.root() == json::parse(colliding_text));
// only the object with the spread keys got a table (both texts have the
// same length, so the tables are the only difference)
with_collisions.shrink_to_fit();
without_collisions.shrink_to_fit();
CHECK(without_collisions.memory_usage() >= with_collisions.memory_usage() + (slots * sizeof(std::uint32_t)));
}
SECTION("shrink_to_fit releases the tables' spare capacity")
{
const auto make_text = [](int objects, int members)
{
std::string text = "[";
for (int object = 0; object < objects; ++object)
{
text += object != 0 ? ",{" : "{";
for (int i = 0; i < members + object; ++i)
{
text += (i != 0 ? ",\"" : "\"") + std::to_string(i) + "\":" + std::to_string(i);
}
text += '}';
}
return text + "]";
};
const std::string small_text = make_text(5, 150);
const std::string big_text = make_text(40, 400);
// reading a big text, and then a small one, leaves the spare capacity
// of the big one: shrink_to_fit() brings the document to the size of
// one parsed from the small text alone
json_document d = json_document::parse(big_text);
const std::size_t big = d.memory_usage();
d.read(small_text);
CHECK(d.memory_usage() >= big);
d.shrink_to_fit();
json_document fresh = json_document::parse(small_text);
fresh.shrink_to_fit();
CHECK(d.memory_usage() == fresh.memory_usage());
CHECK(d.memory_usage() < big / 2);
CHECK(d.root() == json::parse(small_text));
for (int object = 0; object < 5; ++object)
{
const json_view v = d.root()[static_cast<std::size_t>(object)];
for (int i = 0; i < 150 + object; ++i)
{
CHECK(v[std::to_string(i)].get<int>() == i);
}
}
}
SECTION("nested, reused, and in arrays")
{
std::string inner = "{";
+88
View File
@@ -20,8 +20,10 @@ using nlohmann::json;
#include <array>
#include <cstdint>
#include <fstream>
#include <limits>
#include <map>
#include <memory>
#include <new>
#include <random>
#include <sstream>
#include <string>
@@ -506,3 +508,89 @@ TEST_CASE("json_view builder: strings across vector blocks")
}
}
}
TEST_CASE("json_view node integer bits")
{
using nlohmann::detail::view::integer_bits;
using nlohmann::detail::view::set_integer_bits;
// an integer lives in len (low half) and next (high half), on any byte
// order; a big-endian target must not store the native word over both
SECTION("set_integer_bits and integer_bits")
{
node n = {};
for (const std::uint64_t v :
{
std::uint64_t{0}, std::uint64_t{1}, std::uint64_t{0xFFFFFFFFu}, std::uint64_t{0x100000000u},
std::uint64_t{0x0000000200000003u}, std::uint64_t{0x0123456789ABCDEFu}, std::uint64_t{0xFFFFFFFFFFFFFFFEu}
})
{
CAPTURE(v)
n.kind = 0x5A;
n.flags = 0xA5;
n.extra = 0x1234;
n.off = 0x89ABCDEFu;
set_integer_bits(n, v);
CHECK(integer_bits(n) == v);
CHECK(n.len == static_cast<std::uint32_t>(v));
CHECK(n.next == static_cast<std::uint32_t>(v >> 32));
// the other fields are untouched
CHECK(n.kind == 0x5A);
CHECK(n.flags == 0xA5);
CHECK(n.extra == 0x1234);
CHECK(n.off == 0x89ABCDEFu);
}
}
SECTION("parsed integers")
{
struct integer_case
{
const char* text;
std::uint64_t bits;
};
for (const integer_case c :
{
integer_case{"[8589934595]", 0x0000000200000003u}, integer_case{"[4294967296]", 0x100000000u}, integer_case{"[4294967295]", 0xFFFFFFFFu},
integer_case{"[-2]", 0xFFFFFFFFFFFFFFFEu}, integer_case{"[-4294967297]", 0xFFFFFFFEFFFFFFFFu}, integer_case{"[18446744073709551615]", 0xFFFFFFFFFFFFFFFFu},
integer_case{"[7]", 7u}
})
{
CAPTURE(c.text)
const built b = build(c.text, false, false, true);
REQUIRE(b.ok);
const node& n = b.data->tape[1];
CHECK(integer_bits(n) == c.bits);
CHECK(n.len == static_cast<std::uint32_t>(c.bits));
CHECK(n.next == static_cast<std::uint32_t>(c.bits >> 32));
}
}
}
TEST_CASE("json_view node array size limit")
{
// a node array larger than the address space is refused, not wrapped to a
// small allocation (the size computation overflows on 32-bit targets, and
// for absurd counts everywhere)
std::unique_ptr<document_data, document_data::deleter> d(document_data::create(0));
d->reserve(8);
REQUIRE(d->tape_cap >= 8);
d->tape_size = 2;
const std::size_t cap = d->tape_cap;
node* const tape = d->tape;
#if !defined(JSON_NOEXCEPTION)
const std::size_t too_many = document_data::max_nodes() + 1;
CHECK_THROWS_AS(d->reserve(too_many), std::bad_alloc&);
CHECK_THROWS_AS(d->reserve((std::numeric_limits<std::size_t>::max)()), std::bad_alloc&);
// the array is unchanged
CHECK(d->tape == tape);
CHECK(d->tape_cap == cap);
CHECK(d->tape_size == 2);
#endif
// the largest count that fits is not refused by the check (nothing is
// allocated for a count that is already there)
d->reserve(cap);
CHECK(d->tape == tape);
}
+501 -4
View File
@@ -181,6 +181,104 @@ void compare(const ordered_json_editable_view& v, const ordered_json& j)
}
}
// the text of j, in which some objects repeat a key of theirs: before their
// members (the real member is then the last), or after (the repeat is)
std::string text_with_duplicates(const ordered_json& j)
{
if (j.is_object())
{
std::vector<std::string> keys;
for (const auto& kv : j.items())
{
keys.push_back(kv.key());
}
const auto repeated = [&keys]()
{
return ordered_json(keys[static_cast<std::size_t>(r(static_cast<int>(keys.size())))]).dump() + ":" + random_value(2).dump();
};
std::string text = "{";
if (!keys.empty() && r(4) == 0)
{
text += repeated() + ",";
}
bool first = true;
for (const auto& kv : j.items())
{
text += (first ? "" : ",") + ordered_json(kv.key()).dump() + ":" + text_with_duplicates(kv.value());
first = false;
}
for (int i = keys.empty() ? 0 : r(3); i > 0; --i)
{
text += "," + repeated();
}
return text + "}";
}
if (j.is_array())
{
std::string text = "[";
for (std::size_t i = 0; i < j.size(); ++i)
{
text += (i != 0 ? "," : "") + text_with_duplicates(j[i]);
}
return text + "]";
}
return j.dump();
}
// every lookup of the edited view finds what j holds, although the view may
// have several members for a key (j has the last value, at the position of the
// first member: what parse() and materialize() make of it)
void check_lookups(const ordered_json_editable_view& v, const ordered_json& j)
{
REQUIRE(v.type() == j.type());
if (j.is_object())
{
for (const auto& kv : j.items())
{
const std::string& key = kv.key();
const ptr_t ptr = ptr_t() / key;
CAPTURE(key)
const ordered_json_editable_view m = v[key];
REQUIRE(!m.is_discarded());
CHECK(m.materialize() == kv.value());
CHECK(v.at(key).materialize() == kv.value());
CHECK(v[ptr].materialize() == kv.value());
CHECK(v.at(ptr).materialize() == kv.value());
CHECK(v.contains(key));
CHECK(v.contains(ptr));
CHECK(v.count(key) == 1);
const auto it = v.find(key);
REQUIRE(it != v.end());
CHECK((*it).materialize() == kv.value());
CHECK(v.value(ptr, ordered_json(nullptr)) == kv.value());
if (kv.value().is_string())
{
CHECK(v.value(key, std::string("-")) == kv.value().get<std::string>());
}
check_lookups(m, kv.value());
}
CHECK(v["missing#key"].is_discarded());
CHECK(!v.contains("missing#key"));
CHECK(v.find("missing#key") == v.end());
}
else if (j.is_array())
{
for (std::size_t i = 0; i < j.size(); ++i)
{
check_lookups(v[i], j[i]);
check_lookups(v.at(i), j[i]);
}
}
}
// (for documents whose objects repeat keys: dump(), size(), and iteration
// list all members, so only the lookups and materialize() are compared)
void check_duplicates(const ordered_json_editable_document& d, const ordered_json& j)
{
CHECK(d.root().materialize() == j);
check_lookups(d.root(), j);
}
void check_all(const ordered_json_editable_document& d, const ordered_json& j, bool deep)
{
const std::string text = d.root().dump();
@@ -204,14 +302,16 @@ TEST_CASE("json_view edits: differential")
// random edits are applied to an ordered_json_editable_document and to the
// ordered_json parse() produces; after every edit both must serialize,
// materialize, and read back the same
for (int n = 0; n < 150; ++n)
for (int n = 0; n < 250; ++n)
{
// the last 100 documents repeat keys in some of their objects
const bool duplicates = n >= 150;
ordered_json j = random_value(0);
if (r(4) == 0)
{
j = ordered_json::object({{"a", random_value(1)}, {"b", random_value(1)}});
}
const std::string text = j.dump(r(2) == 0 ? -1 : 2);
const std::string text = duplicates ? text_with_duplicates(j) : j.dump(r(2) == 0 ? -1 : 2);
CAPTURE(text)
ordered_json_editable_document d = ordered_json_editable_document::parse(text);
j = ordered_json::parse(text);
@@ -340,7 +440,14 @@ TEST_CASE("json_view edits: differential")
}
CAPTURE(p.to_string())
CAPTURE(op)
check_all(d, j, e % 8 == 7 || e == edits - 1);
if (duplicates)
{
check_duplicates(d, j);
}
else
{
check_all(d, j, e % 8 == 7 || e == edits - 1);
}
}
}
}
@@ -517,7 +624,7 @@ TEST_CASE("json_view edits: views and values")
SECTION("duplicate keys")
{
json_editable_document d = json_editable_document::parse(R"({"a": 1, "b": 2, "a": 3})");
d.set(d.root(), "a", 4); // the first member is assigned, the others dropped
d.set(d.root(), "a", 4); // the last member is assigned (at the position of the first), the others dropped
CHECK(d.root().dump() == R"({"a":4,"b":2})");
d = json_editable_document::parse(R"({"a": 1, "b": 2, "a": 3})");
CHECK(d.erase(d.root(), "a") == 2);
@@ -577,3 +684,393 @@ TEST_CASE("json_view edits: views and values")
CHECK(d.root().dump() == "[true,false]");
}
}
TEST_CASE("json_view edits: deeply nested values")
{
// copying a value into a document must not recurse per nesting level
const std::size_t depth = 100000;
const std::string brackets = std::string(depth, '[') + std::string(depth, ']');
std::string braces;
for (std::size_t i = 0; i < depth; ++i)
{
braces += "{\"a\":";
}
braces += '1';
braces += std::string(depth, '}');
SECTION("a view of a read-only document")
{
const json_document source = json_document::parse(brackets);
json_editable_document d = json_editable_document::parse("[]");
d.push_back(d.root(), source.root());
CHECK(d.root().dump() == "[" + brackets + "]");
}
SECTION("a view of an editable document")
{
const json_editable_document source = json_editable_document::parse(braces);
json_editable_document d = json_editable_document::parse("{}");
d.set(d.root(), "deep", source.root());
CHECK(d.root().dump() == "{\"deep\":" + braces + "}");
}
SECTION("a view of an edited document (values behind links)")
{
const json_document source = json_document::parse(brackets);
json_editable_document edited = json_editable_document::parse("[[]]");
edited.push_back(edited.root()[0], source.root());
edited.push_back(edited.root(), source.root());
json_editable_document d = json_editable_document::parse("null");
d.set(d.root(), edited.root());
CHECK(d.root().dump() == "[[" + brackets + "]," + brackets + "]");
}
SECTION("a basic_json value")
{
json deep = json::array();
json* inner = &deep;
for (std::size_t i = 1; i < depth; ++i)
{
inner->push_back(json::array());
inner = &inner->back();
}
json_editable_document d = json_editable_document::parse("[]");
d.push_back(d.root(), deep);
CHECK(d.root().dump() == "[" + brackets + "]");
}
SECTION("a basic_json value with objects")
{
json deep = 1;
for (std::size_t i = 0; i < depth; ++i)
{
json outer = json::object();
outer["a"] = std::move(deep);
deep = std::move(outer);
}
json_editable_document d = json_editable_document::parse("{}");
d.set(d.root(), "deep", deep);
CHECK(d.root().dump() == "{\"deep\":" + braces + "}");
}
}
#if !defined(JSON_NOEXCEPTION)
TEST_CASE("json_view edits: pointers below a null value")
{
// a null value on the way becomes what basic_json makes of it: an array
// for "-" and for digits, an object otherwise
struct test_case
{
const char* document;
const char* pointer;
};
const std::array<test_case, 16> cases =
{
{
{R"({"a":null})", "/a/0"},
{R"({"a":null})", "/a/-"},
{R"({"a":null})", "/a/3"},
{R"({"a":null})", "/a/x"},
{R"({"a":null})", "/a/+1"},
{R"({"a":null})", "/a/01"},
{R"({"a":null})", "/a/"},
{R"({"a":{"b":null}})", "/a/b/1"},
{R"({"a":{"b":null}})", "/a/b/-"},
{R"({"a":[null]})", "/a/0/0"},
{R"({"a":[null,null]})", "/a/1/k"},
{R"([null])", "/0"},
{"null", "/0"},
{"null", "/-"},
{"null", "/k"},
{"null", ""},
}
};
for (const test_case& c : cases)
{
CAPTURE(c.document)
CAPTURE(c.pointer)
json expected = json::parse(c.document);
const std::string error = exception_of_call([&]
{
expected[json::json_pointer(c.pointer)] = 1;
});
json_editable_document d = json_editable_document::parse(c.document);
if (error.empty())
{
d.set(json::json_pointer(c.pointer), 1);
CHECK(d.root().dump() == expected.dump());
CHECK(d.root().materialize() == expected);
}
else
{
// the same error, and the document is not changed
CHECK(exception_of_call([&] { d.set(json::json_pointer(c.pointer), 1); }) == error);
CHECK(d.root().dump() == json::parse(c.document).dump());
}
}
}
TEST_CASE("json_view edits: strings of other documents are checked")
{
// A document borrows the text it was parsed from, and sees later changes
// of the text: a way to get ill-formed UTF-8 into a view. Copying it into
// an editable document is an error, as for any other string.
std::string text = R"({"key":"abc","list":["abc"]})";
const json_document source = json_document::parse(text);
const auto message_of = [](const std::string & bad)
{
return exception_of_call([&]
{
const std::string dumped = json(bad).dump();
static_cast<void>(dumped);
});
};
json_editable_document d = json_editable_document::parse("[1]");
d.push_back(d.root(), source.root());
CHECK(d.root().dump() == R"([1,{"key":"abc","list":["abc"]}])");
text[text.find("abc") + 1] = '\xC3'; // "a\xC3c"
text[text.rfind("abc") + 1] = '\xC3';
const std::string bad_value = message_of(std::string("a\xC3" "c"));
CHECK(!bad_value.empty());
CHECK(exception_of_call([&] { d.push_back(d.root(), source.root()["key"]); }) == bad_value);
CHECK(exception_of_call([&] { d.set(d.root()[0], source.root()["list"][0]); }) == bad_value);
CHECK(exception_of_call([&] { d.set(d.root(), 0, source.root()["list"]); }) == bad_value);
CHECK(exception_of_call([&] { d.set(d.root(), 0, source.root()); }) == bad_value);
text[text.find("key") + 1] = '\xC3'; // a key is checked as well
const std::string bad_key = message_of(std::string("k\xC3" "y"));
CHECK(exception_of_call([&] { d.set(d.root(), 0, source.root()); }) == bad_key);
CHECK(exception_of_call([&] { d.push_back(d.root(), source.root()); }) == bad_key);
// nothing of the failed edits is visible
CHECK(d.root().dump() == R"([1,{"key":"abc","list":["abc"]}])");
}
TEST_CASE("json_view edits: the size of a text arena")
{
using nlohmann::detail::view::text_capacity;
constexpr std::size_t limit = 0xFFFFFFFFu;
// grows by doubling, or to what is needed (plus some room)
CHECK(text_capacity(0, 0, 10) == 266);
CHECK(text_capacity(1000, 990, 20) == 2000);
CHECK(text_capacity(100, 100, 5000) == 5356);
// an arena beyond 2 GiB: doubling is clamped to 4 GiB - 1
CHECK(text_capacity(0x90000000u, 0x8FFFFFFFu, 2) == limit);
CHECK(text_capacity(limit, limit - 10, 10) == limit);
// exactly what fits is accepted, without room to spare
CHECK(text_capacity(100, 90, limit - 90) == limit);
CHECK(text_capacity(limit - 100, limit - 100, 100) == limit);
// what does not fit is an error
CHECK_THROWS_WITH_AS(text_capacity(100, 90, limit - 89), "[json.exception.out_of_range.416] edits of 4 GiB or more are not supported by json_document", json::out_of_range&);
CHECK_THROWS_WITH_AS(text_capacity(limit, limit, 1), "[json.exception.out_of_range.416] edits of 4 GiB or more are not supported by json_document", json::out_of_range&);
}
#endif
TEST_CASE("json_view edits: replacing arrays and objects by scalars")
{
// the first assignment switches the parent to links; the later ones do
// not need to look for the parent again
std::string text = "[";
json expected = json::array();
for (int i = 0; i < 300; ++i)
{
text += (i != 0 ? ",[" : "[") + std::to_string(i) + ",{\"k\":" + std::to_string(i) + "}]";
expected.push_back(json::array({i, json{{"k", i}}}));
}
text += ']';
json_editable_document d = json_editable_document::parse(text);
CHECK(d.root().dump() == expected.dump());
for (int i = 0; i < 300; i += 2)
{
d.set(d.root()[static_cast<std::size_t>(i)], i);
expected[static_cast<std::size_t>(i)] = i;
}
CHECK(d.root().dump() == expected.dump());
for (int i = 1; i < 300; i += 2) // (elements that are still in the parsed layout of their parent)
{
d.set(d.root()[static_cast<std::size_t>(i)][1], "x"); // replaces an object
expected[static_cast<std::size_t>(i)][1] = "x";
}
CHECK(d.root().dump() == expected.dump());
// new values, and values in new values
d.push_back(d.root(), json::parse(R"([[1,2],{"a":[3]}])"));
expected.push_back(json::parse(R"([[1,2],{"a":[3]}])"));
d.set(d.root()[300][0], 7);
expected[300][0] = 7;
d.insert(d.root(), 0, json::array({1, 2}));
expected.insert(expected.begin(), json::array({1, 2}));
d.set(d.root()[0], nullptr);
expected[0] = nullptr;
d.set(d.root()[301], 5);
expected[301] = 5;
CHECK(d.root().dump() == expected.dump());
CHECK(d.root().materialize() == expected);
}
namespace
{
// reads of "a" in an object that repeats it: the last member wins, however the
// object is stored
template<typename View>
void check_last_wins(const View& o, int expected)
{
CAPTURE(expected)
CHECK(o["a"].template get<int>() == expected);
CHECK(o.at("a").template get<int>() == expected);
CHECK(o.find("a")->template get<int>() == expected);
CHECK(o.value("a", -1) == expected);
CHECK(o[json::json_pointer("/a")].template get<int>() == expected);
CHECK(o.at(json::json_pointer("/a")).template get<int>() == expected);
CHECK(o.value(json::json_pointer("/a"), -1) == expected);
CHECK(o.contains("a"));
CHECK(o.contains(json::json_pointer("/a")));
CHECK(o.count("a") == 1);
}
} // namespace
TEST_CASE("json_view edits: duplicate keys")
{
SECTION("an object in its parsed layout, then moved by edits")
{
json_editable_document d = json_editable_document::parse(R"({"a": 1, "b": 2, "a": 3})");
check_last_wins(d.root(), 3);
CHECK(d.root().size() == 3);
CHECK(d.root().dump() == R"({"a":1,"b":2,"a":3})");
// an edit of a value does not move the object
d.set(d.root()["b"], 5);
check_last_wins(d.root(), 3);
d.set(d.root()["a"], 4); // the member that reads find
check_last_wins(d.root(), 4);
CHECK(d.root().dump() == R"({"a":1,"b":5,"a":4})");
// an appended member moves it
d.set(d.root(), "c", true);
check_last_wins(d.root(), 4);
CHECK(d.root().size() == 4);
CHECK(d.root().dump() == R"({"a":1,"b":5,"a":4,"c":true})");
d.set(d.root()["a"], 6);
check_last_wins(d.root(), 6);
CHECK(d.root().dump() == R"({"a":1,"b":5,"a":6,"c":true})");
d.set(json::json_pointer("/a"), 7);
check_last_wins(d.root(), 7);
}
SECTION("set assigns the member that reads find")
{
// the key keeps the position of its first occurrence (as in materialize()), the later members are dropped
ordered_json_editable_document d = ordered_json_editable_document::parse(R"({"a": 1, "b": 2, "a": 3, "c": 4, "a": 5})");
const ordered_json_editable_view held = d.root()["a"];
CHECK(held.get<int>() == 5);
const ordered_json_editable_view assigned = d.set(d.root(), "a", "x");
CHECK(held.get<std::string>() == "x");
CHECK(assigned.get<std::string>() == "x");
CHECK(d.root().dump() == R"({"a":"x","b":2,"c":4})");
CHECK(d.root().materialize() == ordered_json::parse(R"({"a": "x", "b": 2, "c": 4})"));
CHECK(d.root().size() == 3);
// as for the object that parse() makes of the text
ordered_json j = ordered_json::parse(R"({"a": 1, "b": 2, "a": 3, "c": 4, "a": 5})");
j["a"] = "x";
CHECK(d.root().materialize().dump() == j.dump());
// through a pointer, below a duplicate
d = ordered_json_editable_document::parse(R"({"a": {"x": 1}, "b": 2, "a": {"x": 3}})");
d.set(ptr_t("/a/x"), 4);
CHECK(d.root().dump() == R"({"a":{"x":1},"b":2,"a":{"x":4}})");
d.set(ptr_t("/a"), 0);
CHECK(d.root().dump() == R"({"a":0,"b":2})");
}
SECTION("erase removes every member")
{
json_editable_document d = json_editable_document::parse(R"({"a": 1, "b": 2, "a": 3})");
d.set(d.root(), "c", 4);
check_last_wins(d.root(), 3);
CHECK(d.erase(d.root(), "a") == 2);
CHECK(d.root().dump() == R"({"b":2,"c":4})");
CHECK(!d.root().contains("a"));
CHECK(d.root()["a"].is_discarded());
CHECK(d.erase(d.root(), "a") == 0);
d = json_editable_document::parse(R"({"a": 1, "b": 2, "a": 3})");
CHECK(d.erase(json::json_pointer("/a")) == 2);
CHECK(d.root().dump() == R"({"b":2})");
}
SECTION("a large object with an index")
{
std::string text = R"({"a":0,"k7":"first")";
for (int i = 0; i < 200; ++i)
{
text += ",\"k" + std::to_string(i) + "\":" + std::to_string(i);
}
text += R"(,"a":1,"k7":"last","a":2})";
json_editable_document d = json_editable_document::parse(text);
check_last_wins(d.root(), 2);
CHECK(d.root()["k7"].get_string() == "last");
CHECK(d.root()["k199"].get<int>() == 199);
// a value assigned in place: the index stays in use
d.set(d.root()["a"], 3);
check_last_wins(d.root(), 3);
d.set(d.root()["k7"], "changed");
CHECK(d.root()["k7"].get_string() == "changed");
CHECK(d.root()["k199"].get<int>() == 199);
// an appended member moves the members: the lookup scans them
d.set(d.root(), "new", 1);
check_last_wins(d.root(), 3);
CHECK(d.root()["k7"].get_string() == "changed");
CHECK(d.root()["k7"].get_string() == d.root().at("k7").get_string());
CHECK(d.root()["k199"].get<int>() == 199);
CHECK(d.root()["new"].get<int>() == 1);
// a value replaced by a container, in an object that was not moved
d = json_editable_document::parse(text);
d.set(d.root()["a"], json{{"x", 1}});
CHECK(d.root()["a"]["x"].get<int>() == 1);
CHECK(d.root()["k7"].get_string() == "last");
CHECK(d.erase(d.root(), "k7") == 3); // (the key occurs three times)
CHECK(d.root()["k7"].is_discarded());
CHECK(d.root()["a"]["x"].get<int>() == 1);
}
SECTION("objects in moved arrays and in new values")
{
json_editable_document d = json_editable_document::parse(R"([{"a": 1, "a": 2}, {"a": 3, "b": 0, "a": 4}])");
check_last_wins(d.root()[0], 2);
check_last_wins(d.root()[1], 4);
d.insert(d.root(), 0, json::parse(R"({"a": 0})"));
d.push_back(d.root(), json::parse(R"({"a": 5, "a": 6})")); // (a basic_json value has one member)
check_last_wins(d.root()[1], 2);
check_last_wins(d.root()[2], 4);
CHECK(d.root()[3]["a"].get<int>() == 6);
d.set(d.root()[1], "a", 8);
check_last_wins(d.root()[1], 8);
CHECK(d.root()[1].dump() == R"({"a":8})");
d.erase(d.root(), 0);
check_last_wins(d.root()[1], 4);
CHECK(d.root().dump() == R"([{"a":8},{"a":3,"b":0,"a":4},{"a":6}])");
}
SECTION("values of other documents")
{
const json_document source = json_document::parse(R"({"list": [{"a": 1, "a": 2}], "o": {"b": {"a": 3, "a": 4}, "a": 5, "a": 6}})");
json_editable_document d = json_editable_document::parse("{}");
d.set(d.root(), "copy", source.root()["list"]);
d.set(d.root(), "o", source.root()["o"]);
// (the copies keep the repeated members)
CHECK(d.root().dump() == R"({"copy":[{"a":1,"a":2}],"o":{"b":{"a":3,"a":4},"a":5,"a":6}})");
check_last_wins(d.root()["copy"][0], 2);
check_last_wins(d.root()["o"]["b"], 4);
check_last_wins(d.root()["o"], 6);
d.set(d.root()["o"]["b"], "c", 0);
check_last_wins(d.root()["o"]["b"], 4);
d.set(d.root()["o"], "d", 0);
check_last_wins(d.root()["o"], 6);
CHECK(d.root()["o"]["b"]["a"].get<int>() == 4);
CHECK(d.root()["o"]["d"].get<int>() == 0);
}
}
+450 -4
View File
@@ -18,6 +18,7 @@ using nlohmann::ordered_json_editable_document;
using image_check = json_document::image_check;
using nlohmann::detail::view::node;
#include <algorithm>
#include <array>
#include <cstdint>
#include <cstring>
@@ -61,6 +62,14 @@ std::string read_file(const std::string& name)
return ss.str();
}
// the dump of a loaded image, through a named document (root() of a temporary
// document does not compile, and its views would dangle)
std::string loaded_dump(const std::vector<std::uint8_t>& image, image_check check = image_check::full)
{
const json_document d = json_document::load(image, check);
return d.root().dump();
}
// the offsets of the parts of an image
constexpr std::size_t header_size = 64;
@@ -191,7 +200,8 @@ TEST_CASE("json_view images: round trips")
const json_document d = json_document::parse(text);
check_round_trip(d);
// what a loaded document reads is what parse() produces
CHECK(json_document::load(d.save()).root().materialize() == json::parse(text));
const json_document l = json_document::load(d.save());
CHECK(l.root().materialize() == json::parse(text));
}
}
@@ -231,7 +241,7 @@ TEST_CASE("json_view images: round trips")
{
CHECK(l.root()["k" + std::to_string(i)] == d.root()["k" + std::to_string(i)]);
}
CHECK(l.root()["k7"].get<int>() == 7); // the first of duplicate keys
CHECK(l.root()["k7"] == "a duplicate"); // the last of duplicate keys
CHECK(l.root()["inner"]["m199"].get<int>() == -199);
CHECK(!l.root().contains("k1000"));
// the index is not part of the image
@@ -240,6 +250,155 @@ TEST_CASE("json_view images: round trips")
// the nodes of objects in the image do not carry the number of an index
CHECK(node_at(image, 0).extra == 0);
}
SECTION("duplicate keys of a large object: lookups return the last member")
{
std::string text = "{";
for (int i = 0; i < 200; ++i)
{
text += (i != 0 ? ",\"k" : "\"k") + std::to_string(i) + "\":" + std::to_string(i);
}
// three members for k5 (the third is the last), and duplicates of k0 and k199
text += R"(,"k5":"two","k0":null,"k199":[],"k5":"three"})";
const json_document d = json_document::parse(text);
REQUIRE(d.root().size() == 204);
CHECK(d.root()["k5"] == "three");
const std::vector<std::uint8_t> image = d.save();
for (const image_check check :
{
image_check::full, image_check::bounds, image_check::none
})
{
const json_document l = json_document::load(image, check);
CHECK(l.root().size() == 204);
CHECK(l.root()["k5"] == "three");
CHECK(l.root().at("k5") == "three");
CHECK(l.root().find("k5").value() == "three");
CHECK(l.root()["k0"].is_null());
CHECK(l.root()["k199"] == json::array());
CHECK(l.root()["k100"] == 100);
CHECK(l.root().materialize() == d.root().materialize());
}
}
SECTION("objects with colliding keys")
{
// the hash is not seeded: keys can be found that land in one slot of
// a table (see json_view: "colliding keys")
const auto keys_for = [](std::size_t slots, std::size_t colliding_count, std::size_t spread_count)
{
std::vector<std::string> colliding;
std::vector<std::string> spread;
for (std::uint64_t counter = 0; colliding.size() < colliding_count || spread.size() < spread_count; ++counter)
{
std::string key(8, 'a');
for (std::uint64_t x = counter, i = 0; i < 8; ++i, x /= 26)
{
key[i] = static_cast<char>('a' + (x % 26));
}
const bool lands_in_slot_zero = (nlohmann::detail::view::key_hash(key.data(), key.size()) & (slots - 1)) == 0;
if (lands_in_slot_zero && colliding.size() < colliding_count)
{
colliding.push_back(key);
}
else if (!lands_in_slot_zero && spread.size() < spread_count)
{
spread.push_back(key);
}
}
colliding.insert(colliding.end(), spread.begin(), spread.end());
return colliding;
};
const auto make_text = [](const std::vector<std::string>& keys)
{
std::string text = "{";
for (std::size_t i = 0; i < keys.size(); ++i)
{
text += (i != 0 ? ",\"" : "\"") + keys[i] + "\":" + std::to_string(i);
}
return text + "}";
};
// 300 keys in one slot (the table is 1024 slots): the table would
// exceed the probe limit, so the object has none and is searched
// linearly; 40 of 200 keys in one slot (512 slots): a table with a
// long chain
const struct
{
std::size_t slots;
std::size_t colliding;
std::size_t spread;
} cases[] = {{1024, 300, 0}, {512, 40, 160}};
for (const auto& c : cases)
{
CAPTURE(c.slots)
std::vector<std::string> keys = keys_for(c.slots, c.colliding, c.spread);
const std::size_t members = keys.size();
// a duplicate of a colliding key, of the first and of a spread key
const std::vector<std::string> duplicated = {keys[c.colliding - 1], keys[0], keys[members - 1]};
std::string text = make_text(keys);
text.pop_back();
for (const std::string& key : duplicated)
{
text += ",\"" + key + "\":\"last\"";
}
text += "}";
const json_document d = json_document::parse(text);
const std::vector<std::uint8_t> image = d.save();
for (const image_check check :
{
image_check::full, image_check::bounds, image_check::none
})
{
const json_document l = json_document::load(image, check);
REQUIRE(l.root().size() == members + duplicated.size());
for (std::size_t i = 0; i < members; ++i)
{
CAPTURE(i)
const bool duplicate = std::find(duplicated.begin(), duplicated.end(), keys[i]) != duplicated.end();
if (duplicate)
{
CHECK(l.root()[keys[i]] == "last");
}
else
{
CHECK(l.root()[keys[i]] == i);
}
CHECK(l.root().at(keys[i]) == l.root()[keys[i]]);
CHECK(l.root().find(keys[i]).key() == keys[i]);
CHECK(!l.root().contains(keys[i] + "x"));
}
CHECK(!l.root().contains("missing"));
CHECK(l.root().materialize() == json::parse(text));
}
}
}
SECTION("loading releases the list of large objects")
{
std::string text = "[";
for (int object = 0; object < 400; ++object)
{
text += object != 0 ? ",{" : "{";
for (int i = 0; i < 128; ++i)
{
text += (i != 0 ? ",\"" : "\"") + std::to_string(i) + "\":" + std::to_string(i);
}
text += '}';
}
text += "]";
json_document parsed = json_document::parse(text);
const std::vector<std::uint8_t> image = parsed.save();
json_document loaded = json_document::load(image);
CHECK(loaded.root()[399]["127"] == 127);
// Parsing and loading build the same tables, and keep nothing else:
// not the positions of the objects to index (2 KiB here). The slack
// covers the nodes in the header of the document, which are sized
// differently.
parsed.shrink_to_fit();
loaded.shrink_to_fit();
CHECK(loaded.memory_usage() <= parsed.memory_usage() + 512);
}
}
TEST_CASE("json_view images: edited documents")
@@ -305,6 +464,153 @@ TEST_CASE("json_view images: edited documents")
CHECK(l.root()["list"].size() == 4);
}
SECTION("the targets of links do not reach the image")
{
// Edits that make an entry of a moved sequence link to a value mark the
// value as linked (node_flags::linked). An image has no links and no
// such flag: the full and the bounds check reject it.
const std::string nested = R"({"a": [1, 2, {"x": [3]}, "s", 1.5, true, null, [4, 5]], "b": {"c": 1, "d": [1, 2, 3], "e": {"f": "g"}}, "h": "str"})";
const auto check_image = [](const json_editable_document & d)
{
const std::vector<std::uint8_t> image = d.save();
for (std::size_t i = 0; i < node_count(image); ++i)
{
CAPTURE(i)
CHECK((node_at(image, i).flags & nlohmann::detail::view::node_flags::linked) == 0);
CHECK(node_at(image, i).kind != nlohmann::detail::view::kind_link);
}
for (const image_check check :
{
image_check::full, image_check::bounds
})
{
CHECK(load_result(image, check).empty());
}
check_round_trip(d);
// a loaded document is edited and saved again
json_editable_document e = json_editable_document::load(image);
CHECK(e.root().dump() == d.root().dump());
e.set(e.root(), "after", 1);
CHECK(loaded_dump(e.save()) == e.root().dump());
};
SECTION("a container replaced by a scalar after an earlier edit")
{
json_editable_document d = json_editable_document::parse(nested);
d.set(d.root()["a"][0], 9);
d.set(d.root()["a"][2], 5);
check_image(d);
d.set(d.root()["a"][2], "now a string");
check_image(d);
d.set(d.root()["a"][7], 0.25);
check_image(d);
d.set(d.root()["b"]["e"], true);
d.set(d.root()["b"]["d"], 7);
check_image(d);
}
SECTION("insert and push_back into parsed arrays")
{
json_editable_document d = json_editable_document::parse(nested);
d.push_back(d.root()["a"], 6);
check_image(d);
d.insert(d.root()["a"], 0, "first");
check_image(d);
d.insert(d.root()["b"]["d"], 1, json::array({1, 2}));
d.push_back(d.root()["b"]["d"], json::object({{"k", nullptr}}));
check_image(d);
// links to values that are replaced afterwards
d.set(d.root()["a"][3], json::object());
d.set(d.root()["a"][4], false);
d.set(d.root()["a"][5], "replaced");
d.set(d.root()["a"][9], 1e300);
check_image(d);
}
SECTION("members set in parsed objects")
{
json_editable_document d = json_editable_document::parse(nested);
d.set(d.root(), "new", 1);
d.set(d.root()["b"], "c", "changed");
d.set(d.root()["b"], "e", json::array({1, {{"z", 2}}}));
d.set(d.root()["b"], "more", d.root()["a"]);
check_image(d);
d.set(d.root()["a"][2], 3); // a link target in a copied subtree
d.erase(d.root()["b"], "c");
d.set(d.root(), "h", json::object());
check_image(d);
}
SECTION("erase")
{
json_editable_document d = json_editable_document::parse(nested);
d.push_back(d.root()["a"], 6);
d.erase(d.root()["a"], 0);
d.erase(d.root()["a"], 1);
d.set(d.root()["a"][0], 1);
d.erase(d.root()["b"], "d");
check_image(d);
}
}
#if !defined(JSON_NOEXCEPTION)
SECTION("invalid UTF-8 of a loaded image is not copied into an editable document")
{
// loading with the bounds check (or none) does not look at the encoding
// of strings: dump() of such a view throws, and an editable document
// must not take the string over
const auto patched = [](const std::string & json_text)
{
std::vector<std::uint8_t> image = json_document::parse(json_text).save();
bool found = false;
for (std::size_t i = 0; i + 1 < image.size(); ++i)
{
if (image[i] == 'Q' && image[i + 1] == 'Z')
{
image[i] = 0xC3;
image[i + 1] = 0x28; // an invalid sequence
found = true;
}
}
REQUIRE(found);
return image;
};
for (const image_check check :
{
image_check::bounds, image_check::none
})
{
// as a value
{
const json_document loaded = json_document::load(patched(R"(["abQZ"])"), check);
json_editable_document e = json_editable_document::parse("[]");
CHECK_THROWS_WITH_AS(e.push_back(e.root(), loaded.root()[0]), "[json.exception.type_error.316] invalid UTF-8 byte at index 3: 0x28", json::type_error&);
CHECK_THROWS_WITH_AS(e.insert(e.root(), 0, loaded.root()[0]), "[json.exception.type_error.316] invalid UTF-8 byte at index 3: 0x28", json::type_error&);
CHECK_THROWS_WITH_AS(e.set(e.root(), loaded.root()[0]), "[json.exception.type_error.316] invalid UTF-8 byte at index 3: 0x28", json::type_error&);
CHECK_THROWS_WITH_AS(e.push_back(e.root(), loaded.root()), "[json.exception.type_error.316] invalid UTF-8 byte at index 3: 0x28", json::type_error&);
// nothing changed
CHECK(e.root().dump() == "[]");
CHECK(e.root().size() == 0);
CHECK(loaded_dump(e.save()) == "[]");
// the valid part of the same document can be copied
e.push_back(e.root(), loaded.root().size());
CHECK(e.root().dump() == "[1]");
}
// as a key, and in a nested value
{
const json_document loaded = json_document::load(patched(R"([{"abQZ": 1}, [["x", "QZ"]]])"), check);
json_editable_document e = json_editable_document::parse(R"({"keep": [1]})");
CHECK_THROWS_WITH_AS(e.set(e.root(), "k", loaded.root()[0]), "[json.exception.type_error.316] invalid UTF-8 byte at index 3: 0x28", json::type_error&);
CHECK_THROWS_WITH_AS(e.set(e.root(), "k", loaded.root()[1]), "[json.exception.type_error.316] invalid UTF-8 byte at index 1: 0x28", json::type_error&);
CHECK_THROWS_WITH_AS(e.push_back(e.root()["keep"], loaded.root()[1]), "[json.exception.type_error.316] invalid UTF-8 byte at index 1: 0x28", json::type_error&);
CHECK(e.root().dump() == R"({"keep":[1]})");
CHECK(loaded_dump(e.save()) == R"({"keep":[1]})");
}
}
}
#endif
SECTION("the root replaced")
{
json_editable_document d = json_editable_document::parse(text);
@@ -530,6 +836,15 @@ TEST_CASE("json_view images: check")
{
n.extra = 1; // a hash index
}), true);
// the flag of the targets of links (editable documents) is not part of an image
for (std::size_t i = 0; i < node_count(image); ++i)
{
CAPTURE(i)
rejected(corrupted(image, i, [](node & n)
{
n.flags = static_cast<std::uint8_t>(n.flags | nlohmann::detail::view::node_flags::linked);
}), true);
}
}
SECTION("bounds")
@@ -625,7 +940,7 @@ TEST_CASE("json_view images: check")
});
CHECK(load_result(as_array, image_check::full).empty());
const std::string expected_dump = R"({"s":"x\"y","i":-12,"u":7,"f":1.5e+300,"b":true,"n":null,"a":["t",[]]})";
CHECK(json_document::load(as_array).root().dump() == expected_dump);
CHECK(loaded_dump(as_array) == expected_dump);
}
SECTION("strings")
@@ -680,7 +995,8 @@ TEST_CASE("json_view images: check")
n.extra = 0;
});
CHECK(load_result(positive, image_check::full).empty());
CHECK(json_document::load(positive).root()["u"].is_number_integer());
const json_document pos = json_document::load(positive);
CHECK(pos.root()["u"].is_number_integer());
rejected(corrupted(image, 8, [](node & n)
{
n.kind = 6; // a float token as integer
@@ -712,6 +1028,136 @@ TEST_CASE("json_view images: check")
}
}
SECTION("float tokens with fewer digits than the layout records")
{
// The bytes of the token are not digits (they read as zeros or as
// other values), so the value has fewer (or more) digits than the
// layout says: dump() must still write a number.
for (const auto& source : std::vector<std::pair<std::string, std::string>>
{
{"123456789012345678.5", std::string("@") + std::string(16, '0') + "1.1"}, // 19 digits
{"1234.5", "@001.1"}, // 5 digits
{"1234.5", "9??.??"} // more than 5 digits
})
{
CAPTURE(source.second)
const std::vector<std::uint8_t> img = json_document::parse("[" + source.first + "]").save();
const node n = node_at(img, 1);
REQUIRE(source.second.size() == n.len);
std::vector<std::uint8_t> b = img;
std::memcpy(b.data() + text_at(img) + n.off, source.second.data(), n.len);
const json_document d = json_document::load(b, image_check::bounds);
const std::string dumped = d.root().dump();
CAPTURE(dumped)
CHECK(json::parse(dumped)[0].is_number());
}
}
SECTION("nodes that share a range")
{
// [big string, then n strings made to point to the big string]: every
// node shares the one range (the check reads it once)
const std::size_t n = 200;
std::string long_text = "[\"" + std::string(5000, 'a') + "\"";
for (std::size_t k = 0; k < n; ++k)
{
long_text += ",\"x\"";
}
long_text += "]";
const std::vector<std::uint8_t> img = json_document::parse(long_text).save();
const node big = node_at(img, 1);
std::vector<std::uint8_t> b = img;
for (std::size_t k = 0; k < n; ++k)
{
set_node(b, 2 + k, big);
}
CHECK(load_result(b, image_check::full).empty());
const json_document d = json_document::load(b);
CHECK(d.root().size() == n + 1);
CHECK(d.root()[n].get<std::string>() == std::string(5000, 'a'));
CHECK(d.root().dump() == loaded_dump(b, image_check::none));
// the same for decoded strings and float tokens, with a node that
// records another digit layout than its token
const std::vector<std::uint8_t> img2 = json_document::parse(R"(["a\"b", "a\"b", 1.25, 1.25, 1.25e3])").save();
CHECK(load_result(img2, image_check::full).empty());
std::vector<std::uint8_t> same = img2;
set_node(same, 2, node_at(img2, 1));
set_node(same, 4, node_at(img2, 3));
CHECK(load_result(same, image_check::full).empty());
CHECK(loaded_dump(same) == R"(["a\"b","a\"b",1.25,1.25,1250.0])");
std::vector<std::uint8_t> layout = same;
node f4 = node_at(layout, 4);
f4.extra = 0x0100u; // the layout of "1.", and not that of "1.25"
set_node(layout, 4, f4);
rejected(layout, false);
f4.extra = 0xFFFFu; // "many" digits: fine, as compaction writes it
set_node(layout, 4, f4);
CHECK(load_result(layout, image_check::full).empty());
}
SECTION("ranges that overlap")
{
// save() writes every string and every token to a place of its own
// (nodes that share a value share the whole range): ranges that
// overlap without being identical are a damaged image, though each
// range is a valid string or token
// nodes: 0 [ 1 "abcdef" 2 "ghijkl" 3 1.2525 4 9.9 5 "a\"bcd" (decoded) 6 "e\"fgh" (decoded)
const std::vector<std::uint8_t> img = json_document::parse(R"(["abcdef", "ghijkl", 1.2525, 9.9, "a\"bcd", "e\"fgh"])").save();
REQUIRE(load_result(img, image_check::full).empty());
// node i with the range (off of node of + shift, length), and extra
const auto aliased = [&](std::size_t i, std::size_t of, std::uint32_t shift, std::uint32_t length, std::uint16_t extra)
{
std::vector<std::uint8_t> b = img;
node n = node_at(b, of);
n.off += shift;
n.len = length;
n.extra = extra;
set_node(b, i, n);
return b;
};
// source strings
rejected(aliased(2, 1, 0, 4, 0), false); // "abcd": the start of another string
rejected(aliased(2, 1, 1, 4, 0), false); // "bcde": inside
rejected(aliased(2, 1, 2, 4, 0), false); // "cdef": the end
CHECK(load_result(aliased(2, 1, 0, 6, 0), image_check::full).empty()); // the whole range: identical
// decoded strings: inside, and partially overlapping (the arena holds a"bcde"fgh)
rejected(aliased(6, 5, 1, 3, 0), false);
rejected(aliased(6, 5, 3, 5, 0), false);
CHECK(load_result(aliased(6, 5, 0, 5, 0), image_check::full).empty());
// float tokens: "1.25" and "525" (an integer token as a float) inside "1.2525"
rejected(aliased(4, 3, 0, 4, 0x0201u), false);
rejected(aliased(4, 3, 3, 3, 0x0003u), false);
CHECK(load_result(aliased(4, 3, 0, 6, 0x0401u), image_check::full).empty());
// a string and a float token may use the same bytes (each is checked by its own kind)
std::vector<std::uint8_t> shared = img;
node as_string = node_at(img, 2);
as_string.off = node_at(img, 3).off;
as_string.len = 6;
set_node(shared, 2, as_string);
CHECK(load_result(shared, image_check::full).empty());
// empty strings inside others are not ranges
CHECK(load_result(aliased(2, 1, 2, 0, 0), image_check::full).empty());
}
SECTION("copies within an edited document")
{
// copies within a document share the value: identical ranges
json_editable_document d = json_editable_document::parse(R"({"s": "ab", "e": "x\"y", "f": 1.5, "g": 1.5e300, "a": [1.5, "ab"]})");
d.set(d.root(), "c1", d.root()["a"]);
d.set(d.root(), "c2", d.root()["a"]);
d.set(d.root(), "c3", d.root());
d.push_back(d.root()["a"], d.root()["e"]);
d.push_back(d.root()["a"], d.root()["g"]);
d.push_back(d.root()["a"], d.root()["g"]);
d.set(d.root()["s"], "changed");
d.set(d.root()["f"], 7.25);
const std::vector<std::uint8_t> saved = d.save();
CHECK(load_result(saved, image_check::full).empty());
CHECK(loaded_dump(saved) == d.root().dump());
check_round_trip(d);
}
SECTION("integer ranges")
{
// tokens of many digits, which the parser stores as floats
+154 -5
View File
@@ -15,6 +15,7 @@
#include <nlohmann/json.hpp>
using nlohmann::detail::dtoa_impl::reinterpret_bits;
#include <algorithm>
#include <array>
#include <cmath>
#include <cstdint>
@@ -666,13 +667,24 @@ void check_shortest(double v)
const std::string text(buf.data(), end);
CAPTURE(text)
CHECK(parse_double(text) == v);
// the layout is that of format_buffer() for the same digits
// the layout is that of format_buffer() for the digits of Zmij
const auto sd = nlohmann::detail::zmij::to_shortest(reinterpret_bits<std::uint64_t>(v));
const std::uint64_t significand = sd.has_digit ? (sd.integral * 10) + sd.digit : sd.integral;
int exponent = sd.has_digit ? sd.exponent : sd.exponent + 1;
std::string significand_digits = std::to_string(significand);
while (significand_digits.size() > 1 && significand_digits.back() == '0')
{
significand_digits.pop_back();
++exponent;
}
std::array<char, 64> reference{};
int len = 0;
int exponent = 0;
nlohmann::detail::dtoa_impl::shortest_digits(reference.data(), len, exponent, v);
const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), len, exponent, -4, 15);
std::copy(significand_digits.begin(), significand_digits.end(), reference.begin());
const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), static_cast<int>(significand_digits.size()), exponent, -4, 15);
CHECK(text == std::string(reference.data(), static_cast<std::size_t>(reference_end - reference.data())));
// and write_positive() is what to_chars() calls
std::array<char, 64> positive{};
const char* const positive_end = nlohmann::detail::dtoa_impl::write_positive(positive.data(), positive.data() + positive.size(), v);
CHECK(text == std::string(positive.data(), static_cast<std::size_t>(positive_end - positive.data())));
const auto de = digits_and_exponent(text);
const std::string& digits = de.first;
if (digits.size() > 1)
@@ -785,3 +797,140 @@ TEST_CASE("shortest digits of doubles")
}
}
}
TEST_CASE("choice of the conversion")
{
using nlohmann::detail::dtoa_impl::is_binary64;
SECTION("by the format of the type")
{
// Zmij needs binary64 numbers; everything else uses Grisu2
static_assert(!is_binary64<float>::value, "float is not binary64");
static_assert(is_binary64<double>::value == (std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53),
"double is binary64 where it is IEEE 754 with 53 digits");
static_assert(!is_binary64<int>::value, "integers are not binary64");
static_assert(is_binary64<long double>::value == (std::numeric_limits<long double>::is_iec559 && std::numeric_limits<long double>::digits == 53 && sizeof(long double) == 8),
"long double is binary64 where it has the format of a double");
CHECK(!is_binary64<float>::value);
CHECK(is_binary64<double>::value);
}
SECTION("float: Grisu2, double: Zmij")
{
// 5.3165205877497296e+16 is one of the doubles for which Grisu2 does not find the shortest digits
constexpr double value = 5.3165205877497296e+16;
std::array<char, 64> buf{};
const char* const last = buf.data() + buf.size();
char* end = nlohmann::detail::dtoa_impl::write_positive(buf.data(), last, value);
CHECK(std::string(buf.data(), end) == "5.31652058774973e+16");
end = nlohmann::detail::dtoa_impl::write_positive_grisu2(buf.data(), last, value);
CHECK(std::string(buf.data(), end) == "5.3165205877497296e+16");
constexpr float f = 1.1754944e-38f;
end = nlohmann::detail::dtoa_impl::write_positive(buf.data(), last, f);
const std::string dispatched(buf.data(), end);
end = nlohmann::detail::dtoa_impl::write_positive_grisu2(buf.data(), last, f);
CHECK(dispatched == std::string(buf.data(), end));
}
SECTION("long double with the format of a double: Zmij")
{
// (on platforms where long double is wider, Grisu2 does not apply either: the snprintf fallback does)
if (std::numeric_limits<long double>::digits == 53 && std::numeric_limits<long double>::is_iec559)
{
using long_double_json = nlohmann::json::with_float_t<long double>;
for (const double d :
{
5.3165205877497296e+16, 1.0, 0.1, 123456.789, 2.2250738585072014e-308, 1.7976931348623157e+308, -5.3165205877497296e+16
})
{
CAPTURE(d)
CHECK(long_double_json(static_cast<long double>(d)).dump() == nlohmann::json(d).dump());
}
CHECK(long_double_json(5.3165205877497296e+16L).dump() == "5.31652058774973e+16");
}
}
}
TEST_CASE("short decimals")
{
// write_short_decimal() writes digits * 10^exp for the digits of a double
// that need no conversion (at most 15, the first not 0): as to_chars()
// writes the (positive) double that has these digits
const auto written = [](std::uint64_t digits, int exp)
{
std::array<char, 64> buf{}; // (up to 41 bytes are written)
char* const end = nlohmann::detail::dtoa_impl::write_short_decimal(buf.data(), digits, exp);
return std::string(buf.data(), end);
};
const auto written_counted = [](std::uint64_t digits, int count, int exp)
{
std::array<char, 64> buf{};
char* const end = nlohmann::detail::dtoa_impl::write_short_decimal(buf.data(), digits, count, exp);
return std::string(buf.data(), end);
};
const auto expected = [](std::uint64_t digits, int exp)
{
const double value = std::strtod((std::to_string(digits) + "e" + std::to_string(exp)).c_str(), nullptr);
std::array<char, 64> buf{};
char* const end = nlohmann::detail::to_chars(buf.data(), buf.data() + 32, value);
return std::string(buf.data(), end);
};
SECTION("powers of ten")
{
const auto& powers = nlohmann::detail::dtoa_impl::powers_of_ten_16();
std::uint64_t power = 1;
for (const std::uint64_t p : powers)
{
CHECK(p == power);
power *= 10;
}
}
SECTION("examples")
{
CHECK(written(1, 0) == "1.0");
CHECK(written(15, -1) == "1.5");
CHECK(written(125, -2) == "1.25");
CHECK(written(1, 22) == "1e+22");
CHECK(written(123456789012345, -2) == "1234567890123.45");
CHECK(written(999999999999999, -15) == "0.999999999999999");
CHECK(written(5, -324) == "5e-324");
CHECK(written_counted(1, 1, 0) == "1.0");
CHECK(written_counted(125, 3, -2) == "1.25");
CHECK(written_counted(100, 3, -2) == "1.0");
CHECK(written_counted(999999999999999, 15, -15) == "0.999999999999999");
}
SECTION("random digits, exponents and trailing zeros")
{
std::mt19937_64 rng(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed)
for (int i = 0; i < 100000; ++i)
{
// 1 to 15 digits, the first not 0, and up to 14 of them trailing zeros
std::uint64_t count = 1 + (rng() % 15u);
std::uint64_t power = 1;
for (std::uint64_t k = 1; k < count; ++k)
{
power *= 10;
}
std::uint64_t digits = power + (rng() % (9 * power));
const std::uint64_t zeros = (rng() % 3u == 0) ? (rng() % count) : 0;
for (std::uint64_t k = 0; k < zeros; ++k)
{
digits = (digits / 10) * 10;
}
// (a value between 1e-300 and 1e300)
const int exp = static_cast<int>(rng() % 560u) - 300 - static_cast<int>(count);
CAPTURE(digits)
CAPTURE(count)
CAPTURE(exp)
const std::string want = expected(digits, exp);
CHECK(written(digits, exp) == want);
CHECK(written_counted(digits, static_cast<int>(count), exp) == want);
}
}
}