mirror of
https://github.com/nlohmann/json.git
synced 2026-09-29 19:20:30 +00:00
Compare commits
6
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a04aa3d095 | ||
|
|
cbdc502fbf | ||
|
|
300ab011b0 | ||
|
|
8d9088ab05 | ||
|
|
5a936a2ef3 | ||
|
|
5876a22712 |
+1
-1
@@ -51,7 +51,7 @@ labels:
|
||||
- "include/nlohmann/detail/view/.*"
|
||||
- "single_include/nlohmann/json_view\\.hpp"
|
||||
- "tests/src/unit-json_view.*"
|
||||
- "tests/src/fuzzer-parse_json_view\\.cpp"
|
||||
- "tests/src/fuzzer-(parse_json_view|json_view_image)\\.cpp"
|
||||
- "tests/benchmarks/json_view/.*"
|
||||
- "tools/amalgamate/config_json_view\\.json"
|
||||
- "docs/mkdocs/docs/features/json_view\\.md"
|
||||
|
||||
@@ -26,6 +26,7 @@ cc_library(
|
||||
"include/nlohmann/detail/conversions/from_json.hpp",
|
||||
"include/nlohmann/detail/conversions/to_chars.hpp",
|
||||
"include/nlohmann/detail/conversions/to_json.hpp",
|
||||
"include/nlohmann/detail/conversions/zmij.hpp",
|
||||
"include/nlohmann/detail/exceptions.hpp",
|
||||
"include/nlohmann/detail/hash.hpp",
|
||||
"include/nlohmann/detail/input/binary_reader.hpp",
|
||||
@@ -72,6 +73,7 @@ cc_library(
|
||||
"include/nlohmann/detail/view/edit.hpp",
|
||||
"include/nlohmann/detail/view/edit_storage.hpp",
|
||||
"include/nlohmann/detail/view/errors.hpp",
|
||||
"include/nlohmann/detail/view/image.hpp",
|
||||
"include/nlohmann/detail/view/input.hpp",
|
||||
"include/nlohmann/detail/view/iterator.hpp",
|
||||
"include/nlohmann/detail/view/lookup.hpp",
|
||||
|
||||
@@ -1401,6 +1401,7 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I
|
||||
|
||||
- The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2008-2009 [Björn Hoehrmann](https://bjoern.hoehrmann.de/) <bjoern@hoehrmann.de>
|
||||
- The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/)
|
||||
- The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut)
|
||||
- The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/).
|
||||
- The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0).
|
||||
- The class contains an adapted version of the Eisel-Lemire algorithm and its table of powers of five from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors
|
||||
|
||||
@@ -134,6 +134,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::accept',
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::erase', 'Method', 'api/basic_json_document/erase/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::insert', 'Method', 'api/basic_json_document/insert/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::is_discarded', 'Method', 'api/basic_json_document/is_discarded/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::load', 'Function', 'api/basic_json_document/load/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::memory_usage', 'Method', 'api/basic_json_document/memory_usage/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::node_count', 'Method', 'api/basic_json_document/node_count/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::owns_source', 'Method', 'api/basic_json_document/owns_source/index.html');
|
||||
@@ -142,6 +143,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::parse_co
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::push_back', 'Method', 'api/basic_json_document/push_back/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::read', 'Method', 'api/basic_json_document/read/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::root', 'Method', 'api/basic_json_document/root/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::save', 'Method', 'api/basic_json_document/save/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::set', 'Method', 'api/basic_json_document/set/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::shrink_to_fit', 'Method', 'api/basic_json_document/shrink_to_fit/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::source', 'Method', 'api/basic_json_document/source/index.html');
|
||||
|
||||
@@ -60,6 +60,9 @@ Linear.
|
||||
|
||||
## Notes
|
||||
|
||||
Floating-point numbers are written with the fewest digits that read back as the same value (for `#!cpp double`; see
|
||||
[number handling](../../features/types/number_handling.md#number-serialization)).
|
||||
|
||||
Binary values are serialized as an object containing two keys:
|
||||
|
||||
- "bytes": an array of bytes as integers
|
||||
@@ -96,3 +99,5 @@ Binary values are serialized as an object containing two keys:
|
||||
- Indentation character `indent_char`, option `ensure_ascii` and exceptions added in version 3.0.0.
|
||||
- Error handlers added in version 3.4.0.
|
||||
- Serialization of binary values added in version 3.8.0.
|
||||
- Doubles are written with the shortest digits (Żmij instead of Grisu2) since version 3.13.0; about 0.1% of doubles are
|
||||
written differently, most of them with fewer digits.
|
||||
|
||||
@@ -52,10 +52,16 @@ bookkeeping edits need, and calling any of them on one fails to compile (`#!cpp
|
||||
## Member functions
|
||||
|
||||
- [(constructor)](basic_json_document.md)
|
||||
|
||||
### Parsing
|
||||
|
||||
- [**parse**](parse.md) (_static_) - deserialize from a compatible input, borrowing or owning it as appropriate
|
||||
- [**parse_copy**](parse_copy.md) (_static_) - deserialize a copy of a compatible input
|
||||
- [**accept**](accept.md) (_static_) - check whether the input is valid JSON
|
||||
- [**read**](read.md) - (re-)parse into this document, reusing its memory
|
||||
|
||||
### Access
|
||||
|
||||
- [**root**](root.md) - the view of the root value
|
||||
- [**is_discarded**](is_discarded.md) - return whether the last parse failed
|
||||
- [**source**](source.md) - the parsed text
|
||||
@@ -63,6 +69,14 @@ bookkeeping edits need, and calling any of them on one fails to compile (`#!cpp
|
||||
- [**node_count**](node_count.md) - the number of index entries (values plus object keys)
|
||||
- [**memory_usage**](memory_usage.md) - the number of bytes held by the document
|
||||
- [**shrink_to_fit**](shrink_to_fit.md) - release unused index capacity
|
||||
|
||||
### Images
|
||||
|
||||
- [**save**](save.md) - the document as an image that `load()` reads without parsing
|
||||
- [**load**](load.md) (_static_) - read an image written by `save()`
|
||||
|
||||
### Edits
|
||||
|
||||
- [**set**](set.md) - replace a value, or set an object member, an array element, or the value a JSON pointer refers
|
||||
to (`#!cpp Editable` documents only)
|
||||
- [**push_back**](push_back.md) - append to an array (`#!cpp Editable` documents only)
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
# <small>nlohmann::basic_json_document::</small>load
|
||||
|
||||
```cpp
|
||||
// (1)
|
||||
static basic_json_document load(const std::uint8_t* image, std::size_t size,
|
||||
const image_check check = image_check::full);
|
||||
|
||||
// (2)
|
||||
static basic_json_document load(const std::vector<std::uint8_t>& image,
|
||||
const image_check check = image_check::full);
|
||||
|
||||
// (3)
|
||||
static basic_json_document load(std::vector<std::uint8_t>&& image,
|
||||
const image_check check = image_check::full);
|
||||
```
|
||||
|
||||
1. Reads an image [`save()`](save.md) wrote, from a pointer and a byte count. The image is **borrowed**: `image`
|
||||
must stay alive and unchanged for as long as the returned document, and any view taken from it, is used.
|
||||
2. Reads an image from a `#!cpp std::vector`. Also **borrowed** -- equivalent to overload 1 called with
|
||||
`#!cpp image.data()` and `#!cpp image.size()`.
|
||||
3. Reads an image, keeping the vector instead of copying it: `image` is moved into the document (no copy), which
|
||||
then owns it for as long as it needs the text and the decoded strings. [`owns_source()`](owns_source.md) is
|
||||
`#!cpp true` afterward.
|
||||
|
||||
In every overload, the node index is copied into storage the document itself owns -- so that it is properly aligned,
|
||||
and, for an [editable](index.md#edits) document, can be edited -- while the text and the decoded strings stay in
|
||||
`image`. The hash indexes [large objects](../../features/json_view.md) use for lookup are rebuilt, exactly as after
|
||||
parsing.
|
||||
|
||||
## Parameters
|
||||
|
||||
`image` (in)
|
||||
: the image [`save()`](save.md) wrote (overloads 1 and 2), or one to take ownership of (overload 3)
|
||||
|
||||
`size` (in)
|
||||
: the number of bytes at `image` (overload 1)
|
||||
|
||||
`check` (in)
|
||||
: how thoroughly to validate `image` before trusting it; see [`image_check`](#image_check) below (optional,
|
||||
`#!cpp image_check::full` by default)
|
||||
|
||||
## Return value
|
||||
|
||||
The document read from the image.
|
||||
|
||||
## Exception safety
|
||||
|
||||
Overloads 1 and 2 give the strong guarantee: `image` is only read, never written, so a thrown exception leaves the
|
||||
caller's buffer untouched.
|
||||
|
||||
Overload 3 moves `image` into the document *before* validating it, so that a good image is kept without a copy. If
|
||||
loading then fails, the partially built document -- and the vector now inside it -- is discarded along with the
|
||||
exception, and `image` itself is left **empty**, not restored to what was passed in. Move a copy in instead, or
|
||||
validate with overload 2 first, if the original vector must survive a failed load.
|
||||
|
||||
## Exceptions
|
||||
|
||||
On a big-endian target, throws [`type_error.320`](../../home/exceptions.md#jsonexceptiontype_error320) -- the same
|
||||
exception [`save()`](save.md#exceptions) throws there, since the image format is little-endian only.
|
||||
|
||||
Otherwise throws [`parse_error.116`](../../home/exceptions.md#jsonexceptionparse_error116) if `image` is not one
|
||||
`save()` could have written, or fails the requested `check`:
|
||||
|
||||
| message | when |
|
||||
|------------------------|------------------------------------------------------------------------------------------------------|
|
||||
| `too short` | `image` is `#!cpp nullptr`, or `size` is smaller than the 64-byte header |
|
||||
| `unknown format` | the header's magic bytes or version do not match, or a reserved header field is not zero |
|
||||
| `sizes out of range` | the node count, text size, or decoded-string size the header describes does not fit `size`, or the `#!cpp '\0'` after the text or after the decoded strings is missing |
|
||||
| `the check failed` | `check` is not `#!cpp image_check::none`, and the image fails it -- see [`image_check`](#image_check) |
|
||||
|
||||
!!! failure "Example messages"
|
||||
|
||||
```
|
||||
[json.exception.parse_error.116] parse error: invalid json_document image: too short
|
||||
```
|
||||
```
|
||||
[json.exception.parse_error.116] parse error: invalid json_document image: unknown format
|
||||
```
|
||||
```
|
||||
[json.exception.parse_error.116] parse error: invalid json_document image: sizes out of range
|
||||
```
|
||||
```
|
||||
[json.exception.parse_error.116] parse error: invalid json_document image: the check failed
|
||||
```
|
||||
|
||||
## Complexity
|
||||
|
||||
Linear in the number of nodes, which are always copied into the document. With `#!cpp check == image_check::full`,
|
||||
additionally linear in the combined length of the text and the decoded strings; `#!cpp image_check::bounds` and
|
||||
`#!cpp image_check::none` do not read them.
|
||||
|
||||
## `image_check`
|
||||
|
||||
```cpp
|
||||
using image_check = detail::view::image_check;
|
||||
|
||||
enum class image_check
|
||||
{
|
||||
full,
|
||||
bounds,
|
||||
none
|
||||
};
|
||||
```
|
||||
|
||||
How thoroughly `load()` validates `image` before trusting it.
|
||||
|
||||
| value | checks | guarantees |
|
||||
|----------|--------------------------------------------------------------------------------------------------------------|------------|
|
||||
| `full` | everything the parser itself guarantees: structure and bounds; that every string is valid UTF-8 (and, for a string still in the source text, that it contains no quote, backslash, or control character); and that every number token is well-formed and matches the value stored for it | reading and serializing a checked image is safe and always produces valid JSON, exactly as for a parsed document |
|
||||
| `bounds` | structure and bounds only -- that every offset and count in the node index stays inside the image | reading and serializing stay memory-safe, but a crafted image can hold strings that are not valid UTF-8 or that serialize to invalid JSON ([`dump()`](../basic_json_view/dump.md) writes them unchanged or throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316)), and numbers whose values differ from their text |
|
||||
| `none` | nothing | images from a trusted source only -- reading a damaged image is undefined behavior |
|
||||
|
||||
`full` is the default and the right choice for an image from anything you do not fully control -- a file, a cache
|
||||
shared with other processes, a peer on the network. `bounds` skips scanning the text and the decoded strings, so it
|
||||
fits a cache your own process just wrote and reads straight back, where damage would mean a bug or a hardware fault
|
||||
rather than adversarial input; it still cannot crash or read out of bounds. `none` skips validation entirely and
|
||||
should only be used for an image you trust as much as your own memory.
|
||||
|
||||
## Notes
|
||||
|
||||
**Lifetime.** Overloads 1 and 2 borrow `image`: it must stay alive and byte-for-byte unchanged for as long as the
|
||||
returned document, and any [view](../basic_json_view/index.md) taken from it, is used -- exactly like a document
|
||||
[`parse()`](parse.md) borrowed its input for. Overload 3 avoids this by keeping the vector itself; see
|
||||
[`owns_source`](owns_source.md).
|
||||
|
||||
!!! warning "Experimental"
|
||||
|
||||
The image format is versioned but not yet stable, and may change in an incompatible way before it is declared
|
||||
stable; `load()` already rejects an image written by a different format version with `parse_error.116`
|
||||
("unknown format"). Use images to cache a document within one build of the library, or to hand one to another
|
||||
process running the *same* build on the *same* (little-endian) machine -- not as a long-term storage format.
|
||||
|
||||
**What `image_check::bounds` does not guarantee.** A bounds-checked image can never make `load()`,
|
||||
[`root()`](root.md), element access, or [`materialize()`](../basic_json_view/materialize.md) read outside the image,
|
||||
so those stay safe on a damaged one. It does *not* guarantee that the image describes valid JSON: a string
|
||||
that a `full` check would have rejected can make [`dump()`](../basic_json_view/dump.md) write invalid UTF-8 or invalid
|
||||
JSON, or throw `type_error.316`, and a number can read back with a value that does not match how it is spelled. Reserve `bounds` for images you already trust to be well-formed, and use it
|
||||
only to skip the extra scan.
|
||||
|
||||
## Examples
|
||||
|
||||
??? example "Caching a document, ownership, and a rejected image"
|
||||
|
||||
The example below saves a parsed document as an image, checks that `load()` reproduces the original
|
||||
[`dump()`](../basic_json_view/dump.md) without parsing, and shows the difference between
|
||||
`load(std::move(image))` (owned) and `load(image)` (borrowed). It then damages one byte of the image and shows
|
||||
`image_check::full` rejecting it with `parse_error.116`, while `image_check::bounds` -- meant for a cache the
|
||||
process already trusts -- still reads it without going out of bounds.
|
||||
|
||||
```cpp
|
||||
--8<-- "examples/basic_json_document__load.cpp"
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```json
|
||||
--8<-- "examples/basic_json_document__load.output"
|
||||
```
|
||||
|
||||
## See also
|
||||
|
||||
- [save](save.md) - write the document as an image
|
||||
- [owns_source](owns_source.md) - return whether the document holds its own copy of the text
|
||||
- [parse](parse.md) - deserialize from JSON text instead of an image
|
||||
- [Images](../../features/json_view.md#images) - why and when to use images
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.13.0.
|
||||
@@ -132,6 +132,7 @@ integer type becomes a floating-point value.
|
||||
- [accept](accept.md) - check whether the input is valid JSON
|
||||
- [read](read.md) - (re-)parse into this document, reusing its memory
|
||||
- [owns_source](owns_source.md) - return whether the document holds its own copy of the text
|
||||
- [load](load.md) - read a document from an image instead of parsing JSON text
|
||||
- [`BasicJsonType::parse`](../basic_json/parse.md) - the corresponding function of `basic_json`
|
||||
|
||||
## Version history
|
||||
|
||||
@@ -70,6 +70,7 @@ own on the next, since ownership is decided freshly each time.
|
||||
|
||||
- [parse](parse.md) - deserialize from a compatible input
|
||||
- [root](root.md) - the view of the root value
|
||||
- [load](load.md) - read a document from an image instead of parsing JSON text
|
||||
|
||||
## Version history
|
||||
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
# <small>nlohmann::basic_json_document::</small>save
|
||||
|
||||
```cpp
|
||||
std::vector<std::uint8_t> save() const;
|
||||
```
|
||||
|
||||
Writes the document as an *image*: a byte buffer that [`load`](load.md) reads back without parsing. The image holds
|
||||
the node index, the source text (plus, for an edited document, the number tokens edits wrote), and the decoded
|
||||
strings (plus the strings edits wrote) -- everything [`root()`](root.md) needs, with nothing left to parse.
|
||||
|
||||
An edited document is written in its *current* state, with its values in document order, the way the library's own
|
||||
parser would have produced them for that JSON text: a member [`set`](set.md) added goes at the end, an
|
||||
[`erase`](erase.md)d member leaves no trace, and a float that is not finite (NaN or positive/negative infinity)
|
||||
becomes null, the same substitution [`dump()`](../basic_json_view/dump.md) makes. The same document always saves to
|
||||
the same bytes -- also across `BasicJsonType` and `#!cpp Editable`, since the image reflects document order and
|
||||
values only, not which specialization produced them.
|
||||
|
||||
## Return value
|
||||
|
||||
The image, as a `#!cpp std::vector<std::uint8_t>`. Pass it, or a pointer to its data together with its size, to
|
||||
[`load`](load.md) to read the document back.
|
||||
|
||||
## Exception safety
|
||||
|
||||
Strong guarantee: `save()` does not modify `#!cpp *this` (it is `#!cpp const`), so if it throws, the document is left
|
||||
exactly as it was, and the partially built image is discarded with the exception.
|
||||
|
||||
## Exceptions
|
||||
|
||||
Throws [`type_error.320`](../../home/exceptions.md#jsonexceptiontype_error320) if the document is
|
||||
[discarded](is_discarded.md) -- a default-constructed document, or one a failed [`parse()`](parse.md)/
|
||||
[`read()`](read.md) with `allow_exceptions == false` left discarded.
|
||||
|
||||
On a big-endian target, throws `type_error.320` with a different message instead: the image format is little-endian
|
||||
only (see [Notes](#notes)).
|
||||
|
||||
Throws [`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the node count, the text, or
|
||||
the decoded strings of the image would individually reach 4 GiB -- the same 32-bit offsets
|
||||
[`parse()`](parse.md#exceptions) and, for edits, [`set`](set.md)/[`push_back`](push_back.md) are already limited to.
|
||||
|
||||
!!! failure "Example messages"
|
||||
|
||||
```
|
||||
[json.exception.type_error.320] cannot save a discarded json_document
|
||||
```
|
||||
```
|
||||
[json.exception.type_error.320] json_document images need a little-endian target
|
||||
```
|
||||
```
|
||||
[json.exception.out_of_range.416] images of 4 GiB or more are not supported by json_document
|
||||
```
|
||||
|
||||
## Complexity
|
||||
|
||||
Linear in the size of the document: the number of nodes, plus the length of the text and the decoded strings that end
|
||||
up in the image.
|
||||
|
||||
## Notes
|
||||
|
||||
**Format.** The image begins with a 64-byte header (the magic bytes `#!cpp "NJVI"`, a version number, the node count,
|
||||
and the sizes of the text and the decoded strings, all little-endian), followed by the nodes
|
||||
([16 bytes each](../../home/architecture.md#node-index-of-json-views)), the text and a `#!cpp '\0'`, and the decoded
|
||||
strings and a `#!cpp '\0'`. [`load`](load.md) checks the header, and the sizes it describes, before reading anything
|
||||
else -- see [`load`'s Exceptions](load.md#exceptions).
|
||||
|
||||
!!! warning "Experimental"
|
||||
|
||||
The image format is versioned but not yet stable: it may change in an incompatible way before it is declared
|
||||
stable. Use images to cache a document within one build of the library, or to hand one to another process running
|
||||
the *same* build on the *same* (little-endian) machine -- not as a long-term storage format. Keep the original
|
||||
JSON text if you need to read a saved document back with a future library version.
|
||||
|
||||
**Little-endian only.** The image is written as raw little-endian bytes, with no byte-swapping. `save()` (and
|
||||
[`load`](load.md)) throw `type_error.320` on a big-endian target rather than silently produce bytes a big-endian
|
||||
reader could not interpret correctly.
|
||||
|
||||
## Examples
|
||||
|
||||
??? example "Caching a parsed document as an image"
|
||||
|
||||
The example below saves a parsed configuration as an image -- the way a service might cache one to answer later
|
||||
requests without parsing the text again -- and confirms that loading it back gives exactly the same result as
|
||||
parsing did, and that saving is deterministic.
|
||||
|
||||
```cpp
|
||||
--8<-- "examples/basic_json_document__save.cpp"
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```json
|
||||
--8<-- "examples/basic_json_document__save.output"
|
||||
```
|
||||
|
||||
## See also
|
||||
|
||||
- [load](load.md) - read an image written by `save()`
|
||||
- [owns_source](owns_source.md) - return whether the document holds its own copy of the text
|
||||
- [`basic_json_view::dump`](../basic_json_view/dump.md) - serialize the document to JSON text instead of an image
|
||||
- [Images](../../features/json_view.md#images) - why and when to use images
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.13.0.
|
||||
@@ -0,0 +1,52 @@
|
||||
#include <cstdint>
|
||||
#include <iostream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
using json = nlohmann::json;
|
||||
using json_document = nlohmann::json_document;
|
||||
using image_check = json_document::image_check;
|
||||
|
||||
int main()
|
||||
{
|
||||
std::cout << std::boolalpha;
|
||||
|
||||
// the image of a parsed document -- as if read back from a cache file or
|
||||
// received from another process running the same build of the library
|
||||
const std::string text = R"({"name": "cache", "note": "caf\u00e9", "replicas": ["db2", "db3"]})";
|
||||
const json_document parsed = json_document::parse(text);
|
||||
const std::vector<std::uint8_t> image = parsed.save();
|
||||
|
||||
// (1)/(2) load() needs no parsing, yet dumps exactly what parsing did
|
||||
const json_document borrowed = json_document::load(image);
|
||||
std::cout << (borrowed.root().dump() == parsed.root().dump()) << '\n';
|
||||
std::cout << borrowed.owns_source() << '\n'; // borrowed: still points into `image`
|
||||
|
||||
// (3) load(std::move(image)) keeps the vector instead of copying it
|
||||
std::vector<std::uint8_t> to_move = image;
|
||||
const json_document owned = json_document::load(std::move(to_move));
|
||||
std::cout << owned.owns_source() << '\n';
|
||||
|
||||
// a damaged image -- the last byte of the decoded string "note" holds
|
||||
// (an escape sequence, so it was unescaped into the document's own
|
||||
// buffer), flipped, as storage or transport corruption might do
|
||||
std::vector<std::uint8_t> damaged = image;
|
||||
damaged[damaged.size() - 2] = 0xFF;
|
||||
|
||||
// image_check::full inspects strings and numbers, so it catches the damage
|
||||
try
|
||||
{
|
||||
static_cast<void>(json_document::load(damaged, image_check::full));
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
std::cout << e.id << '\n';
|
||||
}
|
||||
|
||||
// image_check::bounds only checks structure and bounds, so a cache the
|
||||
// process already trusts loads without the extra scan -- reading a value
|
||||
// the damage did not touch is still safe
|
||||
const json_document trusted = json_document::load(damaged, image_check::bounds);
|
||||
std::cout << trusted.root()["name"].get<std::string>() << '\n';
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
true
|
||||
false
|
||||
true
|
||||
116
|
||||
cache
|
||||
@@ -0,0 +1,30 @@
|
||||
#include <cstdint>
|
||||
#include <iostream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
using json_document = nlohmann::json_document;
|
||||
|
||||
int main()
|
||||
{
|
||||
std::cout << std::boolalpha;
|
||||
|
||||
// a configuration a service parses once and then caches as an image, so
|
||||
// that later requests can load() it instead of parsing the text again
|
||||
const std::string text = R"({"name": "cache", "host": "db1", "port": 6379, "replicas": ["db2", "db3"]})";
|
||||
const json_document config = json_document::parse(text);
|
||||
|
||||
// save() turns the parsed document into a byte buffer: a 64-byte header,
|
||||
// the node index, the source text, and the decoded strings
|
||||
const std::vector<std::uint8_t> image = config.save();
|
||||
std::cout << image.size() << '\n';
|
||||
|
||||
// the same document always saves to the same bytes
|
||||
std::cout << (image == json_document::parse(text).save()) << '\n';
|
||||
|
||||
// loading the image back needs no parsing, yet dumps exactly what
|
||||
// parsing the text produced
|
||||
const json_document reloaded = json_document::load(image);
|
||||
std::cout << (reloaded.root().dump() == config.root().dump()) << '\n';
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
316
|
||||
true
|
||||
true
|
||||
@@ -13,7 +13,8 @@ C++ types, and finally serialize it again.
|
||||
[SAX interface](parsing/sax_interface.md), and [error handling](parsing/parse_exceptions.md).
|
||||
- [Zero-copy JSON views](json_view.md) — read a JSON text through a flat index instead of building a `json` tree;
|
||||
strings and numbers stay in the input and are only decoded when needed.
|
||||
[Editable documents](json_view.md#editing-a-document) can also be modified.
|
||||
[Editable documents](json_view.md#editing-a-document) can also be modified, and [images](json_view.md#images) load a
|
||||
parsed document again without parsing it.
|
||||
- [Comments](comments.md) and [trailing commas](trailing_commas.md) — opt-in relaxations of the JSON grammar.
|
||||
|
||||
## Accessing and modifying values
|
||||
|
||||
@@ -250,6 +250,57 @@ edits. See [`basic_json_document`'s Edits](../api/basic_json_document/index.md#e
|
||||
throws (the *basic* guarantee, not the strong one `dump()` and the read-only functions provide). How edits are kept in
|
||||
the index is described in the [architecture overview](../home/architecture.md#node-index-of-json-views).
|
||||
|
||||
## Images
|
||||
|
||||
[`save()`](../api/basic_json_document/save.md) writes a document as an *image*: a byte buffer that
|
||||
[`load()`](../api/basic_json_document/load.md) reads back into a document without parsing -- no lexing, no building
|
||||
the node index, nothing but copying the nodes and pointing the text and the decoded strings at the image. Where
|
||||
[`parse_copy()`](../api/basic_json_document/parse_copy.md) still has to scan the whole input,
|
||||
[`load()`](../api/basic_json_document/load.md) turns that scan into a copy of the node index alone.
|
||||
|
||||
**Why.** A document that is parsed once and then read many times -- a configuration loaded at startup, a template
|
||||
rendered on every request, a large reference dataset a worker process needs in memory -- pays for parsing once but
|
||||
can amortize [`save()`](../api/basic_json_document/save.md)'s cost across every later load. That makes images useful
|
||||
for a cache: save a document the first time it is parsed (to a file, a shared-memory segment, an in-process cache),
|
||||
and [`load()`](../api/basic_json_document/load.md) it on every later use instead of parsing the source text again.
|
||||
They are just as useful for handing a parsed document to another process (or a forked worker) running the same build
|
||||
of the library, since [`load()`](../api/basic_json_document/load.md) turns the transfer into a copy of the node index
|
||||
plus pointers into the received bytes, not a re-parse.
|
||||
|
||||
**Choosing a check.** [`load()`](../api/basic_json_document/load.md) takes an
|
||||
[`image_check`](../api/basic_json_document/load.md#image_check) that trades validation against speed:
|
||||
`image_check::full` (the default) checks everything the parser itself guarantees, so a checked image is exactly as
|
||||
safe to read and serialize as a freshly parsed document -- the right choice whenever the image did not come straight
|
||||
from this process's own [`save()`](../api/basic_json_document/save.md), such as a file or a network peer.
|
||||
`image_check::bounds` only checks structure and bounds -- cheaper, since it skips scanning the text and the decoded
|
||||
strings -- and fits a cache the process trusts, one it wrote and reads back itself. `image_check::none` skips
|
||||
validation entirely, for an image trusted as much as the process's own memory. See
|
||||
[`load()`'s Notes](../api/basic_json_document/load.md#notes) for exactly what each level does and does not guarantee.
|
||||
|
||||
??? example "Example: cache a parsed configuration as an image"
|
||||
|
||||
```cpp
|
||||
--8<-- "examples/basic_json_document__save.cpp"
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```json
|
||||
--8<-- "examples/basic_json_document__save.output"
|
||||
```
|
||||
|
||||
!!! warning "Experimental"
|
||||
|
||||
The image format is versioned but not yet stable, and may change in an incompatible way before it is declared
|
||||
stable. It is little-endian only, and tied to the library build that wrote it -- use it to cache a document or to
|
||||
hand one to another process running the *same* build, not as a long-term storage format; keep the original JSON
|
||||
text if a saved document needs to be readable by a future library version.
|
||||
|
||||
The idea of a document you can read without parsing comes from zero-copy formats such as
|
||||
[FlatBuffers](https://github.com/google/flatbuffers) and [YaFF](https://github.com/yandex/yaff); the check
|
||||
[`load()`](../api/basic_json_document/load.md) runs follows the idea of FlatBuffers' Verifier. No code is taken from
|
||||
either.
|
||||
|
||||
## Choosing between `json`, `ordered_json`, the SAX interface, and `json_view`
|
||||
|
||||
| | [`json`](../api/json.md) / [`ordered_json`](../api/ordered_json.md) | [SAX interface](parsing/sax_interface.md) | [`json_document`](../api/json_document.md) / [`json_view`](../api/json_view.md) | [`json_editable_document`](../api/json_editable_document.md) / [`json_editable_view`](../api/json_editable_view.md) |
|
||||
@@ -258,6 +309,7 @@ the index is described in the [architecture overview](../home/architecture.md#no
|
||||
| **Mutability** | freely mutable | not applicable (a one-shot event stream) | read-only | [`set`](../api/basic_json_document/set.md)/[`push_back`](../api/basic_json_document/push_back.md)/[`insert`](../api/basic_json_document/insert.md)/[`erase`](../api/basic_json_document/erase.md) edit in place; the source text is never rewritten |
|
||||
| **What you get** | a full tree you can read, write, and keep as long as you like | a sequence of callbacks; whatever your handler builds from them | a flat index plus, on demand, [`materialize()`](../api/basic_json_view/materialize.md)d `json`/`ordered_json` values for the parts you actually use | the same, plus [`dump()`](../api/basic_json_view/dump.md) of an edited document that keeps the member order and, with [`number_format::source`](../api/basic_json_view/number_format.md), the spelling of every untouched number |
|
||||
| **Typical use** | general-purpose JSON handling: config, request/response bodies you build or modify, anything you hold onto | validating or projecting a text into your own data structure without ever holding the whole thing as JSON | large or high-volume input where you only need part of it, or need it repeatedly, and can keep the source text (or a copy) alive for as long as the document lives | a document you read, patch a few fields of, and write back -- a configuration file, for instance -- where the rest of it should come back exactly as it was |
|
||||
| **Caching/reload** | not applicable -- re-parse, or roll your own serialization | not applicable | [`save()`](../api/basic_json_document/save.md)/[`load()`](../api/basic_json_document/load.md): cache the parsed index as an image and reload it without parsing | same, saving the document's current -- possibly edited -- state |
|
||||
|
||||
## Version history
|
||||
|
||||
|
||||
@@ -118,9 +118,10 @@ That is, `-0` is stored as a signed integer, but the serialization does not repr
|
||||
### Number serialization
|
||||
|
||||
- Integer numbers are serialized as is; that is, no scientific notation is used.
|
||||
- Floating-point numbers are serialized as specified by the `#!c %g` printf modifier with
|
||||
[`std::numeric_limits<double>::max_digits10`](https://en.cppreference.com/w/cpp/types/numeric_limits/max_digits10)
|
||||
significant digits. The rationale is to use the shortest representation while still allowing round-tripping.
|
||||
- Floating-point numbers are serialized with the fewest digits that read back as the same value (the closest such
|
||||
digits if there are several), in the layout of the `#!c %g` printf modifier: `#!c 1.5`, `#!c 100.0`, `#!c 1e+100`.
|
||||
Doubles are converted with the algorithm of [Żmij](https://github.com/vitaut/zmij), floats with Grisu2, which
|
||||
can write more digits than necessary.
|
||||
|
||||
!!! hint "Notes regarding precision of floating-point numbers"
|
||||
|
||||
|
||||
@@ -540,9 +540,10 @@ therefore silently changes parse results rather than raising an error. See
|
||||
specifiers, for which the library likewise provides only `#!cpp double` and `#!cpp long double` overloads
|
||||
(`#!cpp float` is promoted to `#!cpp double`).
|
||||
|
||||
If `#!cpp std::numeric_limits<NumberFloatType>` describes an IEEE 754 binary32 or binary64 number, `dump` uses the
|
||||
Grisu2 algorithm, which produces the shortest representation that round-trips. Otherwise the `snprintf` fallback with
|
||||
`max_digits10` digits is used.
|
||||
If `#!cpp std::numeric_limits<NumberFloatType>` describes an IEEE 754 binary64 number, `dump` uses the algorithm of
|
||||
Żmij, which produces the shortest representation that round-trips. For IEEE 754 binary32 numbers, it uses Grisu2,
|
||||
which produces a short representation that round-trips. Otherwise the `snprintf` fallback with `max_digits10` digits is
|
||||
used.
|
||||
|
||||
### Required for the binary formats
|
||||
|
||||
@@ -554,7 +555,7 @@ binary32 or binary64 field and have no encoding for `#!cpp long double`.
|
||||
|
||||
| Type | Support |
|
||||
|--------------------------|-----------------------------------------------------------------------------------------------------------------------|
|
||||
| `#!cpp double` (default) | full; short round-trip output through Grisu2 |
|
||||
| `#!cpp double` (default) | full; shortest round-trip output through Żmij |
|
||||
| `#!cpp float` | full; short round-trip output through Grisu2 |
|
||||
| `#!cpp long double` | `dump` and `parse` only; the binary format writers do not compile, as they only handle IEEE 754 binary32 and binary64 |
|
||||
| any other type | not usable |
|
||||
|
||||
@@ -259,6 +259,14 @@ edited:
|
||||
- Views of read-only documents compile without any of this: how views walk the index is a template parameter
|
||||
(`navigation<Editable>`).
|
||||
|
||||
Images ([`save`](../api/basic_json_document/save.md) and [`load`](../api/basic_json_document/load.md),
|
||||
[`detail/view/image.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/image.hpp)) store
|
||||
the nodes as they are: a 64-byte header (the magic bytes `NJVI`, a format version, the sizes, and reserved bytes that
|
||||
must be zero), the nodes, the text, and the decoded strings. An edited document is first written in document order, as
|
||||
the parser would have written it (without links), and the numbers of the hash indexes are cleared, since `load`
|
||||
rebuilds the indexes. So a change of the node layout is a change of the image format: it must raise `image_version`,
|
||||
and `load` then rejects images of other versions (`parse_error.116`) instead of misreading them.
|
||||
|
||||
## Input adapters
|
||||
|
||||
Input is read via **input adapters** that abstract a source. Every input adapter provides this interface:
|
||||
|
||||
@@ -388,6 +388,23 @@ A UBJSON high-precision number could not be parsed.
|
||||
[json.exception.parse_error.115] parse error at byte 5: syntax error while parsing UBJSON high-precision number: invalid number text: 1A
|
||||
```
|
||||
|
||||
### json.exception.parse_error.116
|
||||
|
||||
[`basic_json_document::load()`](../api/basic_json_document/load.md) rejected an
|
||||
[image](../features/json_view.md#images): either the bytes are not one [`save()`](../api/basic_json_document/save.md)
|
||||
could have written (too short, an unknown magic number or format version, or sizes that do not fit the buffer), or
|
||||
they are, but fail the requested [`image_check`](../api/basic_json_document/load.md#image_check).
|
||||
|
||||
!!! failure "Example message"
|
||||
|
||||
```
|
||||
[json.exception.parse_error.116] parse error: invalid json_document image: the check failed
|
||||
```
|
||||
|
||||
!!! note
|
||||
|
||||
This exception was added in version 3.13.0, together with [images](../features/json_view.md#images).
|
||||
|
||||
## Iterator errors
|
||||
|
||||
This exception is thrown if iterators passed to a library function do not match
|
||||
@@ -800,6 +817,26 @@ from JSON text.
|
||||
|
||||
This exception was added in version 3.13.0, together with editable [`json_document`s](../features/json_view.md).
|
||||
|
||||
### json.exception.type_error.320
|
||||
|
||||
[`basic_json_document::save()`](../api/basic_json_document/save.md) cannot write an
|
||||
[image](../features/json_view.md#images) of a [discarded](../api/basic_json_document/is_discarded.md) document.
|
||||
[`save()`](../api/basic_json_document/save.md) and [`load()`](../api/basic_json_document/load.md) also throw this
|
||||
exception on a big-endian target, since the image format is little-endian only.
|
||||
|
||||
!!! failure "Example messages"
|
||||
|
||||
```
|
||||
[json.exception.type_error.320] cannot save a discarded json_document
|
||||
```
|
||||
```
|
||||
[json.exception.type_error.320] json_document images need a little-endian target
|
||||
```
|
||||
|
||||
!!! note
|
||||
|
||||
This exception was added in version 3.13.0, together with [images](../features/json_view.md#images).
|
||||
|
||||
## Out of range
|
||||
|
||||
This exception is thrown in case a library function is called on an input parameter that exceeds the expected range, for instance, in the case of array indices or nonexisting object keys.
|
||||
@@ -1027,7 +1064,9 @@ MessagePack's ext type and BSON's binary subtype are each stored in a single byt
|
||||
so they do not support an input of 4 GiB or more. The same 32-bit limit applies to an **editable** document's own
|
||||
storage: [`set`](../api/basic_json_document/set.md) and [`push_back`](../api/basic_json_document/push_back.md) throw
|
||||
this exception once the strings and number tokens written by edits reach 4 GiB in total, or once more than
|
||||
4294967295 arrays/objects have had an element set or appended to them.
|
||||
4294967295 arrays/objects have had an element set or appended to them. The same limit applies to an
|
||||
[image](../features/json_view.md#images): [`save()`](../api/basic_json_document/save.md) throws it if the node
|
||||
count, the text, or the decoded strings it would write would individually reach 4 GiB.
|
||||
|
||||
!!! failure "Example messages"
|
||||
|
||||
@@ -1037,6 +1076,9 @@ this exception once the strings and number tokens written by edits reach 4 GiB i
|
||||
```
|
||||
[json.exception.out_of_range.416] edits of 4 GiB or more are not supported by json_document
|
||||
```
|
||||
```
|
||||
[json.exception.out_of_range.416] images of 4 GiB or more are not supported by json_document
|
||||
```
|
||||
|
||||
!!! note
|
||||
|
||||
|
||||
@@ -18,6 +18,8 @@ The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed und
|
||||
|
||||
The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/)
|
||||
|
||||
The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut)
|
||||
|
||||
The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/).
|
||||
|
||||
The class contains an adapted version of the Eisel-Lemire algorithm and its table of powers of five from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors
|
||||
|
||||
@@ -236,6 +236,7 @@ nav:
|
||||
- 'erase': api/basic_json_document/erase.md
|
||||
- 'insert': api/basic_json_document/insert.md
|
||||
- 'is_discarded': api/basic_json_document/is_discarded.md
|
||||
- 'load': api/basic_json_document/load.md
|
||||
- 'memory_usage': api/basic_json_document/memory_usage.md
|
||||
- 'node_count': api/basic_json_document/node_count.md
|
||||
- 'owns_source': api/basic_json_document/owns_source.md
|
||||
@@ -244,6 +245,7 @@ nav:
|
||||
- 'push_back': api/basic_json_document/push_back.md
|
||||
- 'read': api/basic_json_document/read.md
|
||||
- 'root': api/basic_json_document/root.md
|
||||
- 'save': api/basic_json_document/save.md
|
||||
- 'set': api/basic_json_document/set.md
|
||||
- 'shrink_to_fit': api/basic_json_document/shrink_to_fit.md
|
||||
- 'source': api/basic_json_document/source.md
|
||||
|
||||
@@ -11,11 +11,17 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cmath> // signbit, isfinite
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // intN_t, uintN_t
|
||||
#include <cstring> // memcpy, memmove
|
||||
#include <limits> // numeric_limits
|
||||
#include <type_traits> // conditional
|
||||
|
||||
#ifdef _MSC_VER
|
||||
#include <cstdlib> // _byteswap_uint64
|
||||
#endif
|
||||
|
||||
#include <nlohmann/detail/conversions/zmij.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
@@ -918,6 +924,87 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value)
|
||||
grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the shortest digits of a positive finite float (other than double): Grisu2
|
||||
*/
|
||||
template<typename FloatType>
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value)
|
||||
{
|
||||
grisu2(buf, len, decimal_exponent, value);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the shortest digits of a positive finite double: the conversion of
|
||||
Zmij (see zmij.hpp), which always finds the shortest digits that read back as
|
||||
the same value (Grisu2 does not for about one double in a thousand), and the
|
||||
closest of them if there are several
|
||||
|
||||
v = buf * 10^decimal_exponent, as for grisu2()
|
||||
*/
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value)
|
||||
{
|
||||
static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
|
||||
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
|
||||
JSON_ASSERT(std::isfinite(value));
|
||||
JSON_ASSERT(value > 0);
|
||||
|
||||
std::uint64_t bits = 0;
|
||||
std::memcpy(&bits, &value, sizeof(bits));
|
||||
zmij::decimal d = zmij::to_decimal(bits);
|
||||
// without trailing zeros (up to 16): 8, 4, 2, 1 at a time
|
||||
while (d.significand % 100000000 == 0)
|
||||
{
|
||||
d.significand /= 100000000;
|
||||
d.exponent += 8;
|
||||
}
|
||||
if (d.significand % 10000 == 0)
|
||||
{
|
||||
d.significand /= 10000;
|
||||
d.exponent += 4;
|
||||
}
|
||||
if (d.significand % 100 == 0)
|
||||
{
|
||||
d.significand /= 100;
|
||||
d.exponent += 2;
|
||||
}
|
||||
if (d.significand % 10 == 0)
|
||||
{
|
||||
d.significand /= 10;
|
||||
d.exponent += 1;
|
||||
}
|
||||
// at most 17 digits, written from the back two at a time
|
||||
static constexpr const char* pairs =
|
||||
"00010203040506070809101112131415161718192021222324252627282930313233343536373839"
|
||||
"40414243444546474849505152535455565758596061626364656667686970717273747576777879"
|
||||
"8081828384858687888990919293949596979899";
|
||||
std::array<char, 20> digits{};
|
||||
std::size_t n = digits.size();
|
||||
while (d.significand >= 100)
|
||||
{
|
||||
const auto i = static_cast<std::size_t>(d.significand % 100) * 2;
|
||||
d.significand /= 100;
|
||||
n -= 2;
|
||||
digits[n] = pairs[i];
|
||||
digits[n + 1] = pairs[i + 1];
|
||||
}
|
||||
if (d.significand >= 10)
|
||||
{
|
||||
const auto i = static_cast<std::size_t>(d.significand) * 2;
|
||||
n -= 2;
|
||||
digits[n] = pairs[i];
|
||||
digits[n + 1] = pairs[i + 1];
|
||||
}
|
||||
else
|
||||
{
|
||||
digits[--n] = static_cast<char>('0' + d.significand);
|
||||
}
|
||||
len = static_cast<int>(digits.size() - n);
|
||||
std::memcpy(buf, digits.data() + n, static_cast<std::size_t>(len));
|
||||
decimal_exponent = d.exponent;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief appends a decimal representation of e to buf
|
||||
@return a pointer to the element following the exponent.
|
||||
@@ -1047,6 +1134,177 @@ inline char* format_buffer(char* buf, int len, int decimal_exponent,
|
||||
return append_exponent(buf, n - 1);
|
||||
}
|
||||
|
||||
/// eight decimal digits (a value below 10^8) as bytes 0..9, the first digit
|
||||
/// in the most significant byte: three steps that divide all lanes at once
|
||||
/// by a multiplication (the conversion of Xiang JunBo, as in Zmij)
|
||||
inline std::uint64_t eight_digit_bytes(std::uint64_t abcdefgh) noexcept
|
||||
{
|
||||
const std::uint64_t abcd_efgh = abcdefgh + (((std::uint64_t{1} << 32u) - 10000u) * ((abcdefgh * (((std::uint64_t{1} << 40u) / 10000u) + 1u)) >> 40u));
|
||||
const std::uint64_t ab_cd_ef_gh = abcd_efgh + (((std::uint64_t{1} << 16u) - 100u) * (((abcd_efgh * (((std::uint64_t{1} << 19u) / 100u) + 1u)) >> 19u) & 0x7F0000007Fu));
|
||||
return ab_cd_ef_gh + (((std::uint64_t{1} << 8u) - 10u) * (((ab_cd_ef_gh * (((std::uint64_t{1} << 10u) / 10u) + 1u)) >> 10u) & 0x000F000F000F000Fu));
|
||||
}
|
||||
|
||||
/// store the bytes of v, the most significant one first (one byte swap and
|
||||
/// one store where the byte order is known: compilers do not reliably merge
|
||||
/// the byte stores once this is inlined)
|
||||
inline void store_msb_first(char* p, std::uint64_t v) noexcept
|
||||
{
|
||||
#if defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
|
||||
v = __builtin_bswap64(v);
|
||||
std::memcpy(p, &v, sizeof(v));
|
||||
#elif defined(__BYTE_ORDER__) && defined(__ORDER_BIG_ENDIAN__) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__
|
||||
std::memcpy(p, &v, sizeof(v));
|
||||
#elif defined(_MSC_VER) // (little-endian on all its targets)
|
||||
v = _byteswap_uint64(v);
|
||||
std::memcpy(p, &v, sizeof(v));
|
||||
#else
|
||||
for (unsigned i = 0; i < 8; ++i)
|
||||
{
|
||||
p[i] = static_cast<char>(v >> (56u - (8u * i)));
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief digits * 10^exp for a double, in the layout of format_buffer()
|
||||
|
||||
The layout is that of format_buffer() with min_exp -4 and max_exp 15 (the
|
||||
digits10 of double). The digits are converted eight at a time and placed
|
||||
with fixed-size moves instead of per-digit loops and moves of the buffer.
|
||||
|
||||
@param[in] digits the digits (not 0, at most 17 digits; trailing zeros allowed)
|
||||
@param[in] exp the decimal exponent of the last digit
|
||||
@return a pointer past the text; up to 41 bytes at @a first are written
|
||||
(some beyond the returned end)
|
||||
*/
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
inline char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept
|
||||
{
|
||||
JSON_ASSERT(digits != 0 && digits < 100000000000000000u);
|
||||
const std::uint64_t upper = digits / 100000000u;
|
||||
const std::uint64_t b0 = upper / 100000000u; // (one digit: it is its own byte)
|
||||
const std::uint64_t b1 = eight_digit_bytes(upper % 100000000u);
|
||||
const std::uint64_t b2 = eight_digit_bytes(digits % 100000000u);
|
||||
// leading and trailing zero digits: zero bytes, counted without division
|
||||
int leading = 16;
|
||||
int zeros = 16;
|
||||
if (b0 != 0)
|
||||
{
|
||||
leading = count_leading_zeros(b0) / 8;
|
||||
}
|
||||
else if (b1 != 0)
|
||||
{
|
||||
leading = 8 + (count_leading_zeros(b1) / 8);
|
||||
}
|
||||
else
|
||||
{
|
||||
leading += count_leading_zeros(b2) / 8;
|
||||
}
|
||||
if (b2 != 0)
|
||||
{
|
||||
zeros = count_trailing_zeros(b2) / 8;
|
||||
}
|
||||
else if (b1 != 0)
|
||||
{
|
||||
zeros = 8 + (count_trailing_zeros(b1) / 8);
|
||||
}
|
||||
// (else: 16, b0 is the one digit that is not 0)
|
||||
// the digits as text at text + leading, then '0's, so that fixed-size
|
||||
// moves need not check how many digits there are
|
||||
std::array<char, 64> text; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
|
||||
store_msb_first(text.data(), b0 + 0x3030303030303030u);
|
||||
store_msb_first(text.data() + 8, b1 + 0x3030303030303030u);
|
||||
store_msb_first(text.data() + 16, b2 + 0x3030303030303030u);
|
||||
std::memset(text.data() + 24, '0', 40);
|
||||
const int k = 24 - leading - zeros; // significant digits
|
||||
const int n = k + exp + zeros; // position of the decimal point after the first digit
|
||||
const char* const s0 = text.data() + leading;
|
||||
|
||||
if (-4 < n && n <= 15)
|
||||
{
|
||||
// "0.[000]digits" (n <= 0) is the digits after 1 - n leading '0's
|
||||
// with the point after the first; "digits[000].0" (n >= k) and
|
||||
// "dig.its" put the point after n characters
|
||||
const int pad = n <= 0 ? 1 - n : 0;
|
||||
const char* const s = s0 - pad;
|
||||
const int len = k + pad;
|
||||
const int point = n + pad;
|
||||
std::memcpy(first, s, 16);
|
||||
std::memcpy(first + point + 1, s + point, 24);
|
||||
first[point] = '.';
|
||||
return first + (point >= len ? point + 2 : len + 1);
|
||||
}
|
||||
|
||||
// d.igitse+XX, with at least two exponent digits (as append_exponent())
|
||||
std::memcpy(first, s0, 16);
|
||||
std::memcpy(first + 2, s0 + 1, 16);
|
||||
first[1] = '.';
|
||||
char* const end = first + (k == 1 ? 1 : k + 1);
|
||||
const int e = n - 1;
|
||||
const auto ea = static_cast<unsigned>(e < 0 ? -e : e);
|
||||
const bool three = ea >= 100;
|
||||
end[0] = 'e';
|
||||
end[1] = e < 0 ? '-' : '+';
|
||||
end[2] = static_cast<char>('0' + (three ? ea / 100 : (ea / 10) % 10));
|
||||
end[3] = static_cast<char>('0' + (three ? (ea / 10) % 10 : ea % 10));
|
||||
end[4] = static_cast<char>('0' + (ea % 10));
|
||||
return end + (three ? 5 : 4);
|
||||
}
|
||||
|
||||
/// a positive finite float (other than double): Grisu2 and format_buffer()
|
||||
template<typename FloatType>
|
||||
JSON_HEDLEY_NON_NULL(1, 2)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
char* write_positive(char* first, const char* last, FloatType value)
|
||||
{
|
||||
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10);
|
||||
|
||||
// Compute v = buffer * 10^decimal_exponent.
|
||||
// The decimal digits are stored in the buffer, which needs to be interpreted
|
||||
// as an unsigned decimal integer.
|
||||
// len is the length of the buffer, i.e., the number of decimal digits.
|
||||
int len = 0;
|
||||
int decimal_exponent = 0;
|
||||
shortest_digits(first, len, decimal_exponent, value);
|
||||
|
||||
JSON_ASSERT(len <= std::numeric_limits<FloatType>::max_digits10);
|
||||
|
||||
// Format the buffer like printf("%.*g", prec, value)
|
||||
constexpr int kMinExp = -4;
|
||||
// Use digits10 here to increase compatibility with version 2.
|
||||
constexpr int kMaxExp = std::numeric_limits<FloatType>::digits10;
|
||||
|
||||
JSON_ASSERT(last - first >= kMaxExp + 2);
|
||||
JSON_ASSERT(last - first >= 2 + (-kMinExp - 1) + std::numeric_limits<FloatType>::max_digits10);
|
||||
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10 + 6);
|
||||
|
||||
return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp);
|
||||
}
|
||||
|
||||
/// a positive finite double: the shortest digits (Zmij), laid out by
|
||||
/// write_decimal() (through a local buffer if [first, last) is shorter than
|
||||
/// the 41 bytes it may write)
|
||||
JSON_HEDLEY_NON_NULL(1, 2)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
inline char* write_positive(char* first, const char* last, double value)
|
||||
{
|
||||
static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
|
||||
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
|
||||
std::uint64_t bits = 0;
|
||||
std::memcpy(&bits, &value, sizeof(bits));
|
||||
const zmij::decimal d = zmij::to_decimal(bits);
|
||||
if (JSON_HEDLEY_LIKELY(last - first >= 41))
|
||||
{
|
||||
return write_decimal(first, d.significand, d.exponent);
|
||||
}
|
||||
std::array<char, 64> buf; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
|
||||
const auto len = static_cast<std::size_t>(write_decimal(buf.data(), d.significand, d.exponent) - buf.data());
|
||||
JSON_ASSERT(static_cast<std::size_t>(last - first) >= len);
|
||||
std::memcpy(first, buf.data(), len);
|
||||
return first + len;
|
||||
}
|
||||
|
||||
} // namespace dtoa_impl
|
||||
|
||||
/*!
|
||||
@@ -1064,7 +1322,6 @@ JSON_HEDLEY_NON_NULL(1, 2)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
char* to_chars(char* first, const char* last, FloatType value)
|
||||
{
|
||||
static_cast<void>(last); // maybe unused - fix warning
|
||||
JSON_ASSERT(std::isfinite(value));
|
||||
|
||||
// Use signbit(value) instead of (value < 0) since signbit works for -0.
|
||||
@@ -1090,28 +1347,7 @@ char* to_chars(char* first, const char* last, FloatType value)
|
||||
JSON_HEDLEY_DIAGNOSTIC_POP
|
||||
#endif
|
||||
|
||||
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10);
|
||||
|
||||
// Compute v = buffer * 10^decimal_exponent.
|
||||
// The decimal digits are stored in the buffer, which needs to be interpreted
|
||||
// as an unsigned decimal integer.
|
||||
// len is the length of the buffer, i.e., the number of decimal digits.
|
||||
int len = 0;
|
||||
int decimal_exponent = 0;
|
||||
dtoa_impl::grisu2(first, len, decimal_exponent, value);
|
||||
|
||||
JSON_ASSERT(len <= std::numeric_limits<FloatType>::max_digits10);
|
||||
|
||||
// Format the buffer like printf("%.*g", prec, value)
|
||||
constexpr int kMinExp = -4;
|
||||
// Use digits10 here to increase compatibility with version 2.
|
||||
constexpr int kMaxExp = std::numeric_limits<FloatType>::digits10;
|
||||
|
||||
JSON_ASSERT(last - first >= kMaxExp + 2);
|
||||
JSON_ASSERT(last - first >= 2 + (-kMinExp - 1) + std::numeric_limits<FloatType>::max_digits10);
|
||||
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10 + 6);
|
||||
|
||||
return dtoa_impl::format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp);
|
||||
return dtoa_impl::write_positive(first, last, value);
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
|
||||
@@ -0,0 +1,218 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2025 Victor Zverovich <https://github.com/vitaut/zmij>
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t, uint64_t
|
||||
|
||||
#include <nlohmann/detail/abi_macros.hpp>
|
||||
#include <nlohmann/detail/bit_ops.hpp>
|
||||
#include <nlohmann/detail/input/pow5_table.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
/*!
|
||||
@brief the shortest decimal representation of a double
|
||||
|
||||
A C++11 port of the conversion of Zmij by Victor Zverovich
|
||||
(https://github.com/vitaut/zmij, MIT license): the shortest decimal in the
|
||||
rounding interval of a double, the closest one if there are several. Zmij
|
||||
credits Xiang JunBo (producing the shorter candidate without a division) and
|
||||
Dougall Johnson (the compressed powers of ten). The powers of ten are taken
|
||||
from the table for number parsing (pow5_table.hpp) where it holds them, and
|
||||
computed from the compressed tables of Zmij beyond it.
|
||||
*/
|
||||
namespace zmij
|
||||
{
|
||||
|
||||
/// significand * 10^exponent
|
||||
struct decimal
|
||||
{
|
||||
std::uint64_t significand;
|
||||
int exponent;
|
||||
};
|
||||
|
||||
/// the compressed powers of ten of Zmij
|
||||
inline const std::array<std::uint64_t, 28>& pow10_minor() noexcept
|
||||
{
|
||||
static const std::array<std::uint64_t, 28> table =
|
||||
{
|
||||
{
|
||||
0x8000000000000000u, 0xa000000000000000u, 0xc800000000000000u, 0xfa00000000000000u, 0x9c40000000000000u,
|
||||
0xc350000000000000u, 0xf424000000000000u, 0x9896800000000000u, 0xbebc200000000000u, 0xee6b280000000000u,
|
||||
0x9502f90000000000u, 0xba43b74000000000u, 0xe8d4a51000000000u, 0x9184e72a00000000u, 0xb5e620f480000000u,
|
||||
0xe35fa931a0000000u, 0x8e1bc9bf04000000u, 0xb1a2bc2ec5000000u, 0xde0b6b3a76400000u, 0x8ac7230489e80000u,
|
||||
0xad78ebc5ac620000u, 0xd8d726b7177a8000u, 0x878678326eac9000u, 0xa968163f0a57b400u, 0xd3c21bcecceda100u,
|
||||
0x84595161401484a0u, 0xa56fa5b99019a5c8u, 0xcecb8f27f4200f3au
|
||||
}
|
||||
};
|
||||
return table;
|
||||
}
|
||||
|
||||
/// (high, low) pairs
|
||||
inline const std::array<std::uint64_t, 50>& pow10_major() noexcept
|
||||
{
|
||||
static const std::array<std::uint64_t, 50> table =
|
||||
{
|
||||
{
|
||||
0xaddcb9e83c6b1793u, 0xdf4abe242a1bbf3eu, 0xaf8e5410288e1b6fu, 0x07ecf0ae5ee44ddau, 0xb1442798f49ffb4au, 0x99cd11cfdf41779du,
|
||||
0xb2fe3f0b8599ef07u, 0x861fa7e6dcb4aa15u, 0xb4bca50b065abe63u, 0x0fed077a756b53aau, 0xb67f6455292cbf08u, 0x1a3bc84c17b1d543u,
|
||||
0xb84687c269ef3bfbu, 0x3d5d514f40eea742u, 0xba121a4650e4ddebu, 0x92f34d62616ce413u, 0xbbe226efb628afeau, 0x890489f70a55368cu,
|
||||
0xbdb6b8e905cb600fu, 0x5400e987bbc1c921u, 0xbf8fdb78849a5f96u, 0xde98520472bdd034u, 0xc16d9a0095928a27u, 0x75b7053c0f178294u,
|
||||
0xc350000000000000u, 0x0000000000000000u, 0xc5371912364ce305u, 0x6c28000000000000u, 0xc722f0ef9d80aad6u, 0x424d3ad2b7b97ef6u,
|
||||
0xc913936dd571c84cu, 0x03bc3a19cd1e38eau, 0xcb090c8001ab551cu, 0x5cadf5bfd3072cc6u, 0xcd036837130890a1u, 0x36dba887c37a8c10u,
|
||||
0xcf02b2c21207ef2eu, 0x94f967e45e03f4bcu, 0xd106f86e69d785c7u, 0xe13336d701beba52u, 0xd31045a8341ca07cu, 0x1ede48111209a051u,
|
||||
0xd51ea6fa85785631u, 0x552a74227f3ea566u, 0xd732290fbacaf133u, 0xa97c177947ad4096u, 0xd94ad8b1c7380874u, 0x18375281ae7822bdu,
|
||||
0xdb68c2ca82ed2a05u, 0xa67398db9f6820e1u
|
||||
}
|
||||
};
|
||||
return table;
|
||||
}
|
||||
|
||||
/// one bit per power: whether the computed value is one unit too large
|
||||
inline const std::array<std::uint32_t, 21>& pow10_fixups() noexcept
|
||||
{
|
||||
static const std::array<std::uint32_t, 21> table =
|
||||
{
|
||||
{
|
||||
0x8d8fc810u, 0x06100293u, 0x19000000u, 0x00100000u, 0x00000908u, 0x00000000u, 0x04e00300u, 0x3807e0b2u, 0x3d83d793u, 0x0006f5ccu,
|
||||
0x00000000u, 0xffff0000u, 0x8076337du, 0x4ff45ba0u, 0x09405033u, 0x034376d9u, 0x09000000u, 0x4e100501u, 0x076d14dcu, 0xf964f45eu,
|
||||
0x0000003du
|
||||
}
|
||||
};
|
||||
return table;
|
||||
}
|
||||
|
||||
/// the 128-bit significand of 10^k, rounded down, for k in [-307, 341]
|
||||
/// (compute_pow10 of Zmij)
|
||||
inline uint128_parts compute_pow10(int k) noexcept
|
||||
{
|
||||
const auto i = static_cast<unsigned>(k + 307);
|
||||
const std::uint64_t m = pow10_minor()[(i + 24) % 28];
|
||||
const std::size_t j = 2 * static_cast<std::size_t>((i + 24) / 28);
|
||||
const std::uint64_t h_hi = pow10_major()[j];
|
||||
const std::uint64_t h_lo = pow10_major()[j + 1];
|
||||
const std::uint64_t h1 = full_multiplication(h_lo, m).high;
|
||||
const std::uint64_t c0 = h_lo * m;
|
||||
const std::uint64_t c1 = h1 + (h_hi * m);
|
||||
const std::uint64_t c2 = (c1 < h1 ? 1u : 0u) + full_multiplication(h_hi, m).high;
|
||||
uint128_parts r{};
|
||||
if ((c2 >> 63u) != 0)
|
||||
{
|
||||
r.high = c2;
|
||||
r.low = c1;
|
||||
}
|
||||
else
|
||||
{
|
||||
r.high = (c2 << 1u) | (c1 >> 63u);
|
||||
r.low = (c1 << 1u) | (c0 >> 63u);
|
||||
}
|
||||
r.low -= (pow10_fixups()[i >> 5u] >> (i & 31u)) & 1u;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// The 128-bit significand of 10^k, rounded down, for k in [-342, 341].
|
||||
/// Up to 10^308, the table for number parsing holds the same significands
|
||||
/// (those of 5^k), except for k in [-27, -1], where it holds them one unit
|
||||
/// larger (as the Eisel-Lemire algorithm needs them).
|
||||
inline uint128_parts pow10(int k) noexcept
|
||||
{
|
||||
if (k > pow5_128_largest_power)
|
||||
{
|
||||
return compute_pow10(k); // (only for the smallest doubles)
|
||||
}
|
||||
const auto i = 2 * static_cast<std::size_t>(k - pow5_128_smallest_power);
|
||||
uint128_parts r{pow5_128()[i + 1], pow5_128()[i]};
|
||||
const std::uint64_t adjust = static_cast<unsigned>(k + 27) < 27u ? 1u : 0u;
|
||||
r.high -= r.low < adjust ? 1u : 0u;
|
||||
r.low -= adjust;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// (x_hi * 2^64 + x_lo) * y >> 64, as 128 bits
|
||||
inline uint128_parts umul192_hi128(std::uint64_t x_hi, std::uint64_t x_lo, std::uint64_t y) noexcept
|
||||
{
|
||||
const uint128_parts p = full_multiplication(x_hi, y);
|
||||
uint128_parts r{};
|
||||
r.low = p.low + full_multiplication(x_lo, y).high;
|
||||
r.high = p.high + (r.low < p.low ? 1u : 0u);
|
||||
return r;
|
||||
}
|
||||
|
||||
/// (x * y + c) >> 64
|
||||
inline std::uint64_t umul128_add_hi64(std::uint64_t x, std::uint64_t y, std::uint64_t c) noexcept
|
||||
{
|
||||
const uint128_parts p = full_multiplication(x, y);
|
||||
return p.high + (p.low + c < p.low ? 1u : 0u);
|
||||
}
|
||||
|
||||
/// The shortest decimal in the rounding interval of a positive finite double
|
||||
/// given by its bits, the closest one if there are several (to_decimal of
|
||||
/// Zmij). The significand can end in zeros.
|
||||
inline decimal to_decimal(std::uint64_t bits) noexcept
|
||||
{
|
||||
constexpr int extra_shift = 9;
|
||||
const auto raw_exp = static_cast<int>((bits >> 52u) & 0x7FFu);
|
||||
std::uint64_t bin_sig = bits & ((std::uint64_t{1} << 52u) - 1);
|
||||
// a power of two has a narrower interval below (except the smallest normal)
|
||||
const bool regular = bin_sig != 0 || raw_exp <= 1;
|
||||
const int bin_exp = (raw_exp == 0 ? 1 : raw_exp) - 1075;
|
||||
if (raw_exp != 0)
|
||||
{
|
||||
bin_sig |= std::uint64_t{1} << 52u;
|
||||
}
|
||||
// floor(log10(2^bin_exp)), or floor(log10(3/4 * 2^bin_exp)) for the irregular case
|
||||
const int dec_exp = ((bin_exp * 315653) - (regular ? 0 : 131072)) >> 20;
|
||||
// scaled by 10^(-dec_exp - 1): the integral part is the shorter candidate
|
||||
const int shift = bin_exp + ((-(dec_exp + 1) * 217707) >> 16) + 1 + extra_shift;
|
||||
const uint128_parts p10 = pow10(-dec_exp - 1);
|
||||
const uint128_parts p = umul192_hi128(p10.high, p10.low, bin_sig << static_cast<unsigned>(shift));
|
||||
std::uint64_t integral = p.high >> static_cast<unsigned>(extra_shift);
|
||||
const std::uint64_t fractional = (p.high << static_cast<unsigned>(64 - extra_shift)) | (p.low >> static_cast<unsigned>(extra_shift));
|
||||
std::uint64_t digit = 0;
|
||||
bool round_up = false;
|
||||
bool round_down = false;
|
||||
if (JSON_HEDLEY_LIKELY(regular))
|
||||
{
|
||||
const std::uint64_t half_ulp = (p10.high >> static_cast<unsigned>(extra_shift + 1 - shift)) + (1 - (bin_sig & 1u));
|
||||
round_up = fractional + half_ulp < fractional;
|
||||
round_down = half_ulp > fractional;
|
||||
// the last digit of the longer candidate, rounded to nearest
|
||||
digit = umul128_add_hi64(fractional, 10, (std::uint64_t{1} << 63u) + 6);
|
||||
if (fractional == (std::uint64_t{1} << 62u))
|
||||
{
|
||||
digit = 2; // 2.5 rounds to 2
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
const std::uint64_t half_ulp = p10.high >> static_cast<unsigned>(extra_shift + 1 - shift);
|
||||
round_up = half_ulp > ~std::uint64_t{0} - fractional;
|
||||
round_down = (half_ulp >> 1u) > fractional;
|
||||
digit = umul128_add_hi64(fractional, 10, (std::uint64_t{1} << 63u) - 1);
|
||||
const std::uint64_t lowest = umul128_add_hi64(fractional - (half_ulp >> 1u), 10, ~std::uint64_t{0});
|
||||
digit = digit < lowest ? lowest : digit;
|
||||
}
|
||||
integral += round_up ? 1u : 0u;
|
||||
if (!round_up && !round_down)
|
||||
{
|
||||
// the shorter candidate is outside the rounding interval: one digit more
|
||||
return decimal{(integral * 10) + digit, dec_exp};
|
||||
}
|
||||
return decimal{integral, dec_exp + 1};
|
||||
}
|
||||
|
||||
} // namespace zmij
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -10,7 +10,7 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstdint> // uint8_t, uint32_t
|
||||
#include <cstring> // memcpy
|
||||
#include <functional> // less
|
||||
#include <map> // map
|
||||
@@ -41,7 +41,9 @@ struct document_data
|
||||
node* inline_tape = nullptr; ///< node array allocated together with this header
|
||||
std::size_t inline_cap = 0;
|
||||
std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init)
|
||||
std::size_t arena_size = 0; ///< bytes of decoded strings at base[1] (the arena, or those of a loaded image)
|
||||
std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init)
|
||||
std::vector<std::uint8_t> owned_image{}; ///< a loaded image the document owns (the text and the decoded strings point into it) // NOLINT(readability-redundant-member-init)
|
||||
|
||||
// hash indexes of large objects (see object_index.hpp)
|
||||
static constexpr std::uint32_t index_min_members = 128;
|
||||
|
||||
@@ -0,0 +1,602 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint8_t, uint16_t, uint32_t, uint64_t
|
||||
#include <cstring> // memcmp, memcpy
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/document_data.hpp>
|
||||
#include <nlohmann/detail/view/errors.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/node.hpp>
|
||||
#include <nlohmann/detail/view/number.hpp>
|
||||
#include <nlohmann/detail/view/object_index.hpp>
|
||||
#include <nlohmann/detail/view/scan.hpp>
|
||||
|
||||
// Images: a document stored so that loading it needs no parsing.
|
||||
//
|
||||
// Layout (little-endian): a 64-byte header, the nodes, the text (the source,
|
||||
// followed by the number tokens written by edits), a NUL, the decoded strings
|
||||
// (followed by the strings written by edits), a NUL. The idea is that of
|
||||
// zero-copy formats such as FlatBuffers (https://github.com/google/flatbuffers)
|
||||
// and YaFF (https://github.com/yandex/yaff); no code is taken from them.
|
||||
// check_image follows the idea of FlatBuffers' Verifier (bounds and
|
||||
// structure) and also checks what the parser guarantees about strings and
|
||||
// numbers, so that reading and serializing a checked image is safe and yields
|
||||
// valid JSON.
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
/// how load() checks an image
|
||||
enum class image_check
|
||||
{
|
||||
/// everything the parser guarantees: structure and bounds, strings (valid
|
||||
/// UTF-8; source strings without quotes, backslashes, and control
|
||||
/// characters), and numbers (well-formed, matching the stored values)
|
||||
full,
|
||||
/// structure and bounds only: reading and serializing are safe, but a
|
||||
/// crafted image can yield invalid UTF-8, strings that serialize to
|
||||
/// invalid JSON, or numbers that differ from their text
|
||||
bounds,
|
||||
/// none: for images from a trusted source only (a damaged image is
|
||||
/// undefined behavior)
|
||||
none,
|
||||
};
|
||||
|
||||
struct image_header
|
||||
{
|
||||
std::array<char, 4> magic; ///< "NJVI"
|
||||
std::uint32_t version; ///< 1
|
||||
std::uint64_t node_count;
|
||||
std::uint64_t text_size;
|
||||
std::uint64_t arena_size;
|
||||
std::array<std::uint64_t, 4> reserved; ///< zero (for later versions)
|
||||
};
|
||||
static_assert(sizeof(image_header) == 64, "the image header must be 64 bytes");
|
||||
|
||||
constexpr std::uint32_t image_version = 1;
|
||||
|
||||
/// the largest node count and text or string size of an image (as for parsed
|
||||
/// documents, offsets and counts must fit 32 bits)
|
||||
constexpr std::uint64_t image_limit = 0xFFFFFFF0u;
|
||||
|
||||
/// Copy the current structure of an edited document into nodes in document
|
||||
/// order, as the parser would have written them. Text written by edits is
|
||||
/// appended to text_tail (number tokens) and arena_tail (strings); floats that
|
||||
/// are not finite become null, as dump() writes them.
|
||||
inline void compact_nodes(const document_data& d, std::size_t arena_size, std::vector<node>& out, std::string& text_tail, std::string& arena_tail)
|
||||
{
|
||||
struct frame
|
||||
{
|
||||
const node* cur;
|
||||
const node* end;
|
||||
std::size_t index; ///< the container's node in out
|
||||
std::uint32_t count;
|
||||
bool object;
|
||||
};
|
||||
std::vector<frame> stack;
|
||||
const auto string_node = [&](const node & s)
|
||||
{
|
||||
node r = s;
|
||||
r.extra = 0;
|
||||
r.flags = static_cast<std::uint8_t>(s.flags & node_flags::storage);
|
||||
if (r.flags == node_flags::edited)
|
||||
{
|
||||
r.off = static_cast<std::uint32_t>(arena_size + arena_tail.size());
|
||||
arena_tail.append(d.str(s), s.len);
|
||||
r.flags = node_flags::escaped;
|
||||
}
|
||||
return r;
|
||||
};
|
||||
const auto emit = [&](const node * v)
|
||||
{
|
||||
node r = *v;
|
||||
switch (static_cast<value_t>(v->kind))
|
||||
{
|
||||
case value_t::object:
|
||||
case value_t::array:
|
||||
r.flags = 0;
|
||||
r.extra = 0;
|
||||
r.off = (v->flags & (node_flags::moved | node_flags::is_new)) != 0 ? 0 : v->off;
|
||||
r.len = 0; // counted below
|
||||
r.next = 0; // set when the container is complete
|
||||
stack.push_back(frame{d.first_child_edited(v), d.child_end_edited(v), out.size(), 0, v->kind == static_cast<std::uint8_t>(value_t::object)});
|
||||
break;
|
||||
case value_t::string:
|
||||
r = string_node(*v);
|
||||
break;
|
||||
case value_t::number_integer:
|
||||
case value_t::number_unsigned:
|
||||
if ((v->flags & node_flags::storage) == node_flags::edited)
|
||||
{
|
||||
r.off = static_cast<std::uint32_t>(d.size + text_tail.size());
|
||||
text_tail.append(d.str(*v), number_length(*v));
|
||||
}
|
||||
r.flags = 0;
|
||||
break;
|
||||
case value_t::number_float:
|
||||
if ((v->flags & node_flags::storage) == node_flags::edited)
|
||||
{
|
||||
const char* const t = d.str(*v);
|
||||
if (t[0] == 'n' || t[0] == 'i' || (v->len > 1 && t[1] == 'i'))
|
||||
{
|
||||
r = node{}; // nan and infinity: null, as dump() writes them
|
||||
r.kind = static_cast<std::uint8_t>(value_t::null);
|
||||
break;
|
||||
}
|
||||
r.off = static_cast<std::uint32_t>(d.size + text_tail.size());
|
||||
text_tail.append(t, v->len);
|
||||
r.extra = 0xFFFFu; // the digit layout is not recorded
|
||||
}
|
||||
r.flags = 0;
|
||||
break;
|
||||
case value_t::boolean:
|
||||
r.flags = static_cast<std::uint8_t>(v->flags & node_flags::is_true);
|
||||
break;
|
||||
case value_t::null:
|
||||
case value_t::binary:
|
||||
case value_t::discarded:
|
||||
default:
|
||||
r.flags = 0;
|
||||
break;
|
||||
}
|
||||
out.push_back(r);
|
||||
};
|
||||
emit(d.tape);
|
||||
while (!stack.empty())
|
||||
{
|
||||
frame& top = stack.back();
|
||||
if (top.cur == top.end)
|
||||
{
|
||||
node& c = out[top.index];
|
||||
c.len = top.count;
|
||||
c.next = static_cast<std::uint32_t>(out.size() - top.index);
|
||||
stack.pop_back();
|
||||
continue;
|
||||
}
|
||||
++top.count;
|
||||
const node* v = nullptr;
|
||||
if (top.object)
|
||||
{
|
||||
out.push_back(string_node(*top.cur));
|
||||
v = document_data::deref(top.cur + 1);
|
||||
top.cur = document_data::after(top.cur + 1);
|
||||
}
|
||||
else
|
||||
{
|
||||
v = document_data::deref(top.cur);
|
||||
top.cur = document_data::after(top.cur);
|
||||
}
|
||||
emit(v); // may grow the stack (top is not used afterwards)
|
||||
}
|
||||
}
|
||||
|
||||
/// the document as an image
|
||||
inline std::vector<std::uint8_t> save_image(const document_data& d)
|
||||
{
|
||||
#if !NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
throw_type_error(320, "json_document images need a little-endian target"); // LCOV_EXCL_LINE
|
||||
#endif
|
||||
const std::size_t arena_size = d.arena_size;
|
||||
const node* nodes = d.tape;
|
||||
std::size_t count = d.tape_size;
|
||||
std::vector<node> compacted;
|
||||
std::string text_tail;
|
||||
std::string arena_tail;
|
||||
if (d.edits)
|
||||
{
|
||||
compact_nodes(d, arena_size, compacted, text_tail, arena_tail);
|
||||
nodes = compacted.data();
|
||||
count = compacted.size();
|
||||
}
|
||||
const std::size_t text_size = d.size + text_tail.size();
|
||||
const std::size_t total_arena = arena_size + arena_tail.size();
|
||||
if (NLOHMANN_VIEW_UNLIKELY(text_size >= image_limit || total_arena >= image_limit || count >= image_limit))
|
||||
{
|
||||
// LCOV_EXCL_START (4 GiB)
|
||||
throw_out_of_range(416, "images of 4 GiB or more are not supported by json_document");
|
||||
// LCOV_EXCL_STOP
|
||||
}
|
||||
image_header h{};
|
||||
h.magic = {{'N', 'J', 'V', 'I'}};
|
||||
h.version = image_version;
|
||||
h.node_count = count;
|
||||
h.text_size = text_size;
|
||||
h.arena_size = total_arena;
|
||||
std::vector<std::uint8_t> image(sizeof(h) + (count * sizeof(node)) + text_size + 1 + total_arena + 1);
|
||||
std::uint8_t* o = image.data();
|
||||
std::memcpy(o, &h, sizeof(h));
|
||||
o += sizeof(h);
|
||||
std::memcpy(o, nodes, count * sizeof(node));
|
||||
// the hash indexes are rebuilt by load()
|
||||
for (std::size_t i = 0; i < count; ++i)
|
||||
{
|
||||
if (nodes[i].kind == static_cast<std::uint8_t>(value_t::object) && nodes[i].extra != 0)
|
||||
{
|
||||
node n = nodes[i];
|
||||
n.extra = 0;
|
||||
std::memcpy(o + (i * sizeof(node)), &n, sizeof(node));
|
||||
}
|
||||
}
|
||||
o += count * sizeof(node);
|
||||
const auto append = [&o](const char* s, std::size_t n)
|
||||
{
|
||||
if (n != 0)
|
||||
{
|
||||
std::memcpy(o, s, n);
|
||||
o += n;
|
||||
}
|
||||
};
|
||||
append(d.src, d.size);
|
||||
append(text_tail.data(), text_tail.size());
|
||||
*o++ = 0;
|
||||
append(d.base[1], arena_size);
|
||||
append(arena_tail.data(), arena_tail.size());
|
||||
*o = 0;
|
||||
return image;
|
||||
}
|
||||
|
||||
/// whether a number node matches its token the way the parser records it
|
||||
/// (after the bounds check)
|
||||
inline bool check_number(const node& n, const unsigned char* text)
|
||||
{
|
||||
const std::size_t len = number_length(n);
|
||||
const unsigned char* const s = text + n.off;
|
||||
const unsigned char* const e = s + len;
|
||||
const unsigned char* p = s;
|
||||
const bool negative = *p == '-';
|
||||
p += negative ? 1 : 0;
|
||||
const unsigned char* const int_start = p;
|
||||
if (p == e)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if (*p == '0')
|
||||
{
|
||||
++p;
|
||||
}
|
||||
else if (*p >= '1' && *p <= '9')
|
||||
{
|
||||
while (p != e && is_digit(*p))
|
||||
{
|
||||
++p;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
return false;
|
||||
}
|
||||
const auto int_digits = static_cast<std::size_t>(p - int_start);
|
||||
std::size_t frac_digits = 0;
|
||||
bool is_float = false;
|
||||
if (p != e && *p == '.')
|
||||
{
|
||||
const unsigned char* const f0 = ++p;
|
||||
while (p != e && is_digit(*p))
|
||||
{
|
||||
++p;
|
||||
}
|
||||
if (p == f0)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
frac_digits = static_cast<std::size_t>(p - f0);
|
||||
is_float = true;
|
||||
}
|
||||
std::int64_t exponent = 0;
|
||||
if (p != e && (*p | 0x20u) == 'e')
|
||||
{
|
||||
++p;
|
||||
const bool exp_negative = p != e && *p == '-';
|
||||
p += (p != e && (*p == '+' || *p == '-')) ? 1 : 0;
|
||||
if (p == e || !is_digit(*p))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
while (p != e && is_digit(*p))
|
||||
{
|
||||
exponent = exponent < 100000 ? (exponent * 10) + (*p - '0') : exponent;
|
||||
++p;
|
||||
}
|
||||
exponent = exp_negative ? -exponent : exponent;
|
||||
is_float = true;
|
||||
}
|
||||
if (p != e)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if (n.kind == static_cast<std::uint8_t>(value_t::number_float))
|
||||
{
|
||||
// the digit layout the parser records (or "many", as compaction
|
||||
// writes it), and a finite value
|
||||
const auto layout = static_cast<std::uint16_t>((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8u));
|
||||
if (n.extra != layout && n.extra != 0xFFFFu)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
// parse() rejects floats that overflow; as there, only a number whose
|
||||
// magnitude could reach 1e308 needs the conversion
|
||||
if (static_cast<std::int64_t>(int_digits) + exponent > 300)
|
||||
{
|
||||
const auto v = float_value<double>(reinterpret_cast<const char*>(s), n); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
return v <= (std::numeric_limits<double>::max)() && v >= -(std::numeric_limits<double>::max)();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
// integers: the token's value is the stored one; number_integer nodes of
|
||||
// edits can be non-negative (as basic_json keeps the type of a value)
|
||||
const bool integer = n.kind == static_cast<std::uint8_t>(value_t::number_integer);
|
||||
if (is_float || int_digits > 20 || (negative && !integer))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
// (at most 19 digits cannot overflow; 20 digits are compared with 2^64 - 1)
|
||||
if (int_digits == 20 && std::memcmp(int_start, "18446744073709551615", 20) > 0)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
std::uint64_t m = 0;
|
||||
for (const unsigned char* d = int_start; d != int_start + int_digits; ++d)
|
||||
{
|
||||
m = (m * 10) + static_cast<std::uint64_t>(*d - '0');
|
||||
}
|
||||
if (integer && m > (negative ? std::uint64_t{1} << 63u : (std::uint64_t{1} << 63u) - 1))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
return integer_bits(n) == (negative ? 0 - m : m);
|
||||
}
|
||||
|
||||
/// Check the nodes of a loaded image against its text and decoded strings:
|
||||
/// kinds, flags, and `extra`; extents and element counts of arrays and
|
||||
/// objects; keys; bounds; string contents (source strings as the parser
|
||||
/// leaves them: no quotes, backslashes, or control characters; all strings
|
||||
/// valid UTF-8); and number tokens.
|
||||
inline bool check_image(const node* nodes, std::size_t count, const unsigned char* text, std::size_t text_size,
|
||||
const unsigned char* arena, std::size_t arena_size, bool full)
|
||||
{
|
||||
struct frame
|
||||
{
|
||||
std::size_t end;
|
||||
std::uint32_t len;
|
||||
std::uint32_t seen;
|
||||
bool object;
|
||||
bool expect_key;
|
||||
};
|
||||
std::vector<frame> stack;
|
||||
const auto check_string = [&](const node & n) -> bool
|
||||
{
|
||||
if ((n.flags & ~node_flags::escaped) != 0 || n.extra != 0)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
const bool decoded = (n.flags & node_flags::escaped) != 0;
|
||||
const unsigned char* const base = decoded ? arena : text;
|
||||
const std::size_t limit = decoded ? arena_size : text_size;
|
||||
if (n.off > limit || n.len > limit - n.off)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if (!full)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
const unsigned char* const b = base + n.off;
|
||||
return decoded ? valid_utf8_prefix(b, n.len) == n.len : scan_string_run(b, b + n.len) == b + n.len;
|
||||
};
|
||||
// bounds of a number token; the recorded digit layout must lie within it
|
||||
const auto number_in_bounds = [&](const node & n) -> bool
|
||||
{
|
||||
const std::size_t len = number_length(n);
|
||||
if (len == 0 || n.off > text_size || len > text_size - n.off)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if (n.kind != static_cast<std::uint8_t>(value_t::number_float))
|
||||
{
|
||||
return (n.extra >> 8u) == 0;
|
||||
}
|
||||
// float_value() reads the sign, the integer digits, and the point and
|
||||
// fraction digits the layout records (a layout of more than 19 digits
|
||||
// means the general conversion, which stays within the token)
|
||||
const std::size_t int_digits = n.extra & 0xFFu;
|
||||
const std::size_t frac_digits = n.extra >> 8u;
|
||||
const std::size_t need = (text[n.off] == '-' ? 1u : 0u) + int_digits + (frac_digits != 0 ? frac_digits + 1 : 0);
|
||||
return int_digits + frac_digits > 19 || need <= len;
|
||||
};
|
||||
std::size_t i = 0;
|
||||
for (;;)
|
||||
{
|
||||
// close finished arrays and objects
|
||||
while (!stack.empty() && i == stack.back().end)
|
||||
{
|
||||
const frame f = stack.back();
|
||||
if (f.seen != f.len || (f.object && !f.expect_key))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
stack.pop_back();
|
||||
if (!stack.empty())
|
||||
{
|
||||
++stack.back().seen;
|
||||
stack.back().expect_key = true;
|
||||
}
|
||||
}
|
||||
if (i == count)
|
||||
{
|
||||
return stack.empty();
|
||||
}
|
||||
if (i != 0 && stack.empty())
|
||||
{
|
||||
return false; // nodes after the root
|
||||
}
|
||||
const node& n = nodes[i];
|
||||
if (!stack.empty() && stack.back().object && stack.back().expect_key)
|
||||
{
|
||||
if (n.kind != static_cast<std::uint8_t>(value_t::string) || !check_string(n))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
stack.back().expect_key = false;
|
||||
++i;
|
||||
continue;
|
||||
}
|
||||
bool complete = true;
|
||||
switch (static_cast<value_t>(n.kind))
|
||||
{
|
||||
case value_t::null:
|
||||
// (the offset of a literal is read to size the output of dump())
|
||||
if (n.flags != 0 || n.extra != 0 || n.off > text_size)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
case value_t::boolean:
|
||||
if ((n.flags & ~node_flags::is_true) != 0 || n.extra != 0 || n.off > text_size)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
case value_t::string:
|
||||
if (!check_string(n))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
case value_t::number_integer:
|
||||
case value_t::number_unsigned:
|
||||
case value_t::number_float:
|
||||
if (n.flags != 0 || !number_in_bounds(n) || (full && !check_number(n, text)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
case value_t::array:
|
||||
case value_t::object:
|
||||
{
|
||||
const std::size_t limit = stack.empty() ? count : stack.back().end;
|
||||
if (n.flags != 0 || n.extra != 0 || n.next == 0 || n.next > limit - i || n.off > text_size)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
stack.push_back(frame{i + n.next, n.len, 0, n.kind == static_cast<std::uint8_t>(value_t::object), true});
|
||||
complete = false;
|
||||
break;
|
||||
}
|
||||
case value_t::binary:
|
||||
case value_t::discarded:
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
++i;
|
||||
if (complete && !stack.empty())
|
||||
{
|
||||
++stack.back().seen;
|
||||
stack.back().expect_key = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_invalid_image(const char* what)
|
||||
{
|
||||
throw_parse_error(116, concat("invalid json_document image: ", what));
|
||||
}
|
||||
|
||||
/// Read an image into d. The text and the decoded strings stay in the image;
|
||||
/// the nodes are copied (so that they are aligned, and edits can change them).
|
||||
inline void load_image(document_data& d, const std::uint8_t* image, std::size_t size, image_check check)
|
||||
{
|
||||
#if !NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
throw_type_error(320, "json_document images need a little-endian target"); // LCOV_EXCL_LINE
|
||||
#endif
|
||||
if (image == nullptr || size < sizeof(image_header))
|
||||
{
|
||||
throw_invalid_image("too short");
|
||||
}
|
||||
image_header h{};
|
||||
std::memcpy(&h, image, sizeof(h));
|
||||
// (the reserved fields are for later versions)
|
||||
if (std::memcmp(h.magic.data(), "NJVI", 4) != 0 || h.version != image_version
|
||||
|| (h.reserved[0] | h.reserved[1] | h.reserved[2] | h.reserved[3]) != 0)
|
||||
{
|
||||
throw_invalid_image("unknown format");
|
||||
}
|
||||
const std::size_t room = size - sizeof(h);
|
||||
if (h.node_count == 0 || h.node_count > room / sizeof(node) || h.node_count >= image_limit || h.text_size >= image_limit || h.arena_size >= image_limit)
|
||||
{
|
||||
throw_invalid_image("sizes out of range");
|
||||
}
|
||||
const auto count = static_cast<std::size_t>(h.node_count);
|
||||
const auto text_size = static_cast<std::size_t>(h.text_size);
|
||||
const auto arena_size = static_cast<std::size_t>(h.arena_size);
|
||||
const std::size_t text_at = sizeof(h) + (count * sizeof(node));
|
||||
// the text, a NUL, the decoded strings, a NUL, and nothing after them
|
||||
if (size - text_at < 2 || text_size > size - text_at - 2 || arena_size != size - text_at - text_size - 2
|
||||
|| image[text_at + text_size] != 0 || image[size - 1] != 0)
|
||||
{
|
||||
throw_invalid_image("sizes out of range");
|
||||
}
|
||||
|
||||
d.discarded = true;
|
||||
d.edits.reset();
|
||||
d.base[2] = nullptr;
|
||||
d.owned.clear();
|
||||
if (d.owned_image.empty() || image != d.owned_image.data())
|
||||
{
|
||||
d.owned_image.clear();
|
||||
}
|
||||
d.arena.clear();
|
||||
d.indexes.clear();
|
||||
d.index_slots.clear();
|
||||
d.large_objects.clear();
|
||||
d.tape_size = 0;
|
||||
d.reserve(count);
|
||||
std::memcpy(d.tape, image + sizeof(h), count * sizeof(node));
|
||||
d.tape_size = count;
|
||||
const std::uint8_t* const text = image + text_at;
|
||||
const std::uint8_t* const arena = text + text_size + 1;
|
||||
d.src = reinterpret_cast<const char*>(text); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
d.size = text_size;
|
||||
d.base[0] = d.src;
|
||||
d.base[1] = reinterpret_cast<const char*>(arena); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
d.arena_size = arena_size;
|
||||
if (check != image_check::none && !check_image(d.tape, count, text, text_size, arena, arena_size, check == image_check::full))
|
||||
{
|
||||
throw_invalid_image("the check failed");
|
||||
}
|
||||
// the hash indexes of large objects, as after parsing
|
||||
for (std::size_t i = 0; i < count; ++i)
|
||||
{
|
||||
node& n = d.tape[i];
|
||||
if (n.kind == static_cast<std::uint8_t>(value_t::object))
|
||||
{
|
||||
n.extra = 0;
|
||||
if (n.len >= document_data::index_min_members)
|
||||
{
|
||||
d.large_objects.push_back(static_cast<std::uint32_t>(i));
|
||||
}
|
||||
}
|
||||
}
|
||||
build_object_indexes(d);
|
||||
d.discarded = false;
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -29,43 +29,76 @@ namespace detail
|
||||
namespace view
|
||||
{
|
||||
|
||||
/*!
|
||||
@brief locate the decimal point and the end of the mantissa of a float token
|
||||
|
||||
Also checks that the token is a JSON number. Tokens of the parser and of edits
|
||||
always are; an image loaded with image_check::bounds can hold any bytes, which
|
||||
must not reach the conversion (it expects a well-formed token).
|
||||
*/
|
||||
inline bool float_token_layout(const char* first, const char* last, std::size_t& dot, std::size_t& mantissa_end) noexcept
|
||||
{
|
||||
const auto digit = [last](const char* q)
|
||||
{
|
||||
return q != last && is_digit(static_cast<unsigned char>(*q));
|
||||
};
|
||||
const char* p = first;
|
||||
p += (p != last && *p == '-') ? 1 : 0;
|
||||
if (!digit(p) || (*p == '0' && digit(p + 1)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
while (digit(p))
|
||||
{
|
||||
++p;
|
||||
}
|
||||
dot = std::string::npos;
|
||||
if (p != last && *p == '.')
|
||||
{
|
||||
dot = static_cast<std::size_t>(p - first);
|
||||
if (!digit(++p))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
while (digit(p))
|
||||
{
|
||||
++p;
|
||||
}
|
||||
}
|
||||
mantissa_end = static_cast<std::size_t>(p - first);
|
||||
if (p != last && (*p == 'e' || *p == 'E'))
|
||||
{
|
||||
++p;
|
||||
p += (p != last && (*p == '+' || *p == '-')) ? 1 : 0;
|
||||
if (!digit(p))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
while (digit(p))
|
||||
{
|
||||
++p;
|
||||
}
|
||||
}
|
||||
return p == last;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the value of the float token of a node, as parse() converts it
|
||||
|
||||
Uses the lexer's conversion (detail::convert_float_fast, then the locale-aware
|
||||
strtod fallback), so that the values are bit-identical to parse(). The digit
|
||||
layout recorded while parsing locates the decimal point and the exponent
|
||||
without scanning the token.
|
||||
strtod fallback), so that the values are bit-identical to parse(). A token
|
||||
that is not a JSON number (only in a damaged image loaded with
|
||||
image_check::bounds) yields 0.
|
||||
*/
|
||||
template<typename FloatType>
|
||||
NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n)
|
||||
{
|
||||
const char* const last = first + n.len;
|
||||
const std::size_t neg = first[0] == '-' ? 1 : 0;
|
||||
const std::size_t int_digits = n.extra & 0xFFu;
|
||||
const std::size_t frac_digits = n.extra >> 8u;
|
||||
std::size_t dot = std::string::npos;
|
||||
std::size_t mantissa_end = n.len;
|
||||
if (int_digits != 255 && frac_digits != 255)
|
||||
std::size_t dot = 0;
|
||||
std::size_t mantissa_end = 0;
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!float_token_layout(first, last, dot, mantissa_end)))
|
||||
{
|
||||
dot = frac_digits != 0 ? neg + int_digits : std::string::npos;
|
||||
mantissa_end = neg + int_digits + (frac_digits != 0 ? 1 + frac_digits : 0);
|
||||
}
|
||||
else
|
||||
{
|
||||
// more digits than the layout records: locate them
|
||||
for (std::size_t i = 0; i < n.len; ++i)
|
||||
{
|
||||
if (first[i] == '.')
|
||||
{
|
||||
dot = i;
|
||||
}
|
||||
else if (first[i] == 'e' || first[i] == 'E')
|
||||
{
|
||||
mantissa_end = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return FloatType{};
|
||||
}
|
||||
FloatType v{};
|
||||
if (!convert_float_fast(first, last, dot, mantissa_end, v))
|
||||
@@ -104,16 +137,20 @@ NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const u
|
||||
}
|
||||
if (p != e)
|
||||
{
|
||||
// [eE][+-]digits; huge exponents saturate (the parser rejected overflow)
|
||||
// [eE][+-]digits; huge exponents saturate (the parser rejected
|
||||
// overflow). The token is not read beyond e, and the digits are taken
|
||||
// as unsigned, so that a token that is not well-formed (a damaged
|
||||
// image loaded with image_check::bounds) yields a wrong value, but no
|
||||
// overflow.
|
||||
++p;
|
||||
const bool exp_negative = *p == '-';
|
||||
p += (*p == '-' || *p == '+') ? 1 : 0;
|
||||
const bool exp_negative = p != e && *p == '-';
|
||||
p += (p != e && (*p == '-' || *p == '+')) ? 1 : 0;
|
||||
std::int64_t exp_value = 0;
|
||||
for (; p != e; ++p)
|
||||
{
|
||||
if (exp_value < 0x10000000)
|
||||
{
|
||||
exp_value = (exp_value * 10) + (*p - '0');
|
||||
exp_value = (exp_value * 10) + static_cast<unsigned char>(*p - '0');
|
||||
}
|
||||
}
|
||||
q += exp_negative ? -exp_value : exp_value;
|
||||
|
||||
@@ -75,6 +75,23 @@ class output_buffer
|
||||
m_pos += n;
|
||||
}
|
||||
|
||||
/// the write position and the end of the writable space, for a writer
|
||||
/// that keeps the position in a local variable (set_cursor() hands it back)
|
||||
char* cursor() const noexcept
|
||||
{
|
||||
return m_pos;
|
||||
}
|
||||
|
||||
char* limit() const noexcept
|
||||
{
|
||||
return m_end;
|
||||
}
|
||||
|
||||
void set_cursor(char* p) noexcept
|
||||
{
|
||||
m_pos = p;
|
||||
}
|
||||
|
||||
private:
|
||||
static StringType& sized(StringType& out, std::size_t estimate)
|
||||
{
|
||||
@@ -95,6 +112,105 @@ class output_buffer
|
||||
char* m_end;
|
||||
};
|
||||
|
||||
/// The length of the run at s that dump() writes unchanged without
|
||||
/// ensure_ascii: all bytes but quotes, backslashes, and control characters.
|
||||
/// Unlike detail::string_bulk_run(), non-ASCII bytes are not validated: the
|
||||
/// strings of a document are valid UTF-8 (a damaged image loaded with
|
||||
/// image_check::bounds can have others, which are then written unchanged).
|
||||
inline std::size_t plain_output_run(const unsigned char* s, std::size_t n) noexcept
|
||||
{
|
||||
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
const std::uint64_t v = read_eight_bytes(s + i);
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"'
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\'
|
||||
const std::uint64_t stop = (((q - ones) & ~q) | ((b - ones) & ~b) | ((v - 0x2020202020202020ull) & ~v)) & high;
|
||||
if (stop != 0)
|
||||
{
|
||||
// the lowest flagged byte is the first stop: borrows only flag bytes above a true one
|
||||
return i + (static_cast<std::size_t>(count_trailing_zeros(stop)) / 8);
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
{
|
||||
if (s[i] == '"' || s[i] == '\\' || s[i] < 0x20)
|
||||
{
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
/// A stack that starts in a buffer of the caller (a local array) and moves to
|
||||
/// the heap (a vector of the caller) only when that is full, so that dumps of
|
||||
/// shallow documents need no allocation. The top is a pointer, as in
|
||||
/// std::vector. The address of the stack never escapes (the growth gets the
|
||||
/// vector and returns the new storage), so its pointers stay in registers.
|
||||
template<typename T>
|
||||
class small_stack
|
||||
{
|
||||
public:
|
||||
small_stack(T* buffer, std::size_t capacity, std::vector<T>& heap) noexcept
|
||||
: m_begin(buffer), m_top(buffer), m_end(buffer + capacity), m_heap(&heap)
|
||||
{}
|
||||
small_stack(const small_stack&) = delete;
|
||||
small_stack(small_stack&&) = delete;
|
||||
small_stack& operator=(const small_stack&) = delete;
|
||||
small_stack& operator=(small_stack&&) = delete;
|
||||
~small_stack() = default;
|
||||
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE void push_back(const T& x)
|
||||
{
|
||||
if (NLOHMANN_VIEW_UNLIKELY(m_top == m_end))
|
||||
{
|
||||
const std::size_t used = size();
|
||||
const std::size_t capacity = 2 * static_cast<std::size_t>(m_end - m_begin);
|
||||
m_begin = grow(*m_heap, m_begin, used, capacity);
|
||||
m_top = m_begin + used;
|
||||
m_end = m_begin + capacity;
|
||||
}
|
||||
*m_top++ = x;
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE T& back() noexcept
|
||||
{
|
||||
return m_top[-1];
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE void pop_back() noexcept
|
||||
{
|
||||
--m_top;
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE bool empty() const noexcept
|
||||
{
|
||||
return m_top == m_begin;
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE std::size_t size() const noexcept
|
||||
{
|
||||
return static_cast<std::size_t>(m_top - m_begin);
|
||||
}
|
||||
|
||||
private:
|
||||
/// the used entries moved to heap storage of the given capacity
|
||||
NLOHMANN_VIEW_NOINLINE static T* grow(std::vector<T>& heap, const T* begin, std::size_t used, std::size_t capacity)
|
||||
{
|
||||
std::vector<T> bigger(capacity);
|
||||
std::copy(begin, begin + used, bigger.begin());
|
||||
heap.swap(bigger);
|
||||
return heap.data();
|
||||
}
|
||||
|
||||
T* m_begin;
|
||||
T* m_top;
|
||||
T* m_end;
|
||||
std::vector<T>* m_heap;
|
||||
};
|
||||
|
||||
/// how the view's dump() writes a value
|
||||
struct dump_style
|
||||
{
|
||||
@@ -129,6 +245,18 @@ class view_serializer
|
||||
|
||||
void dump(const node* root)
|
||||
{
|
||||
if (!m_style.pretty && !m_style.ensure_ascii)
|
||||
{
|
||||
if (m_style.source_numbers)
|
||||
{
|
||||
dump_compact<true>(root);
|
||||
}
|
||||
else
|
||||
{
|
||||
dump_compact<false>(root);
|
||||
}
|
||||
return;
|
||||
}
|
||||
struct frame
|
||||
{
|
||||
const node* pos; ///< next element, or key of the next member
|
||||
@@ -136,7 +264,9 @@ class view_serializer
|
||||
bool object;
|
||||
bool first; ///< nothing written yet
|
||||
};
|
||||
std::vector<frame> stack;
|
||||
std::array<frame, 32> buffer; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
|
||||
std::vector<frame> heap;
|
||||
small_stack<frame> stack(buffer.data(), buffer.size(), heap);
|
||||
const node* n = root;
|
||||
for (;;)
|
||||
{
|
||||
@@ -207,6 +337,284 @@ class view_serializer
|
||||
}
|
||||
|
||||
private:
|
||||
/*!
|
||||
@brief the compact output without ensure_ascii (the default dump())
|
||||
|
||||
The same walk as dump(), with the write position in a local variable
|
||||
(stores through char pointers would otherwise force a reload of the
|
||||
buffer's members after each one), and with strings and number tokens of
|
||||
the source copied by fixed-size moves of 32 bytes where the source has
|
||||
that many bytes left, instead of a library call per token. The buffer
|
||||
keeps 64 bytes of slack for the overshoot.
|
||||
*/
|
||||
/// a string that is not a plain string of the source (decoded, or written
|
||||
/// by an edit), without ensure_ascii: runs without characters to escape
|
||||
/// are copied
|
||||
NLOHMANN_VIEW_NOINLINE void write_decoded(const node& n)
|
||||
{
|
||||
const auto* const s = reinterpret_cast<const unsigned char*>(m_doc.str(n)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
m_out.put('"');
|
||||
for (std::size_t i = 0; i < n.len;)
|
||||
{
|
||||
const std::size_t run = plain_output_run(s + i, n.len - i);
|
||||
if (run != 0)
|
||||
{
|
||||
m_out.put(reinterpret_cast<const char*>(s + i), run); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
i += run;
|
||||
continue;
|
||||
}
|
||||
write_codepoint<false>(s[i], s + i, 1); // a quote, a backslash, or a control character
|
||||
++i;
|
||||
}
|
||||
m_out.put('"');
|
||||
}
|
||||
|
||||
/// the copies of dump_compact() that are not fixed-size moves (long
|
||||
/// strings, or near the end of the source); out of line, so that the
|
||||
/// compiler does not merge the fixed-size moves into this call
|
||||
NLOHMANN_VIEW_NOINLINE static void copy_long(char* to, const char* from, std::size_t n) noexcept
|
||||
{
|
||||
std::memcpy(to, from, n);
|
||||
}
|
||||
|
||||
template<bool SourceNumbers>
|
||||
void dump_compact(const node* root)
|
||||
{
|
||||
struct frame
|
||||
{
|
||||
const node* pos; ///< (editable documents) next element, or key of the next member
|
||||
const node* end;
|
||||
bool object;
|
||||
};
|
||||
std::array<frame, 32> buffer; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
|
||||
std::vector<frame> heap;
|
||||
small_stack<frame> stack(buffer.data(), buffer.size(), heap);
|
||||
const char* const src = m_doc.src;
|
||||
const char* const src_end = src + m_doc.size;
|
||||
char* w = m_out.cursor();
|
||||
char* lim = m_out.limit();
|
||||
// room for n bytes and the slack
|
||||
const auto room = [&](std::size_t n)
|
||||
{
|
||||
if (NLOHMANN_VIEW_UNLIKELY(static_cast<std::size_t>(lim - w) < n + 64))
|
||||
{
|
||||
m_out.set_cursor(w);
|
||||
m_out.reserve(n + 64);
|
||||
w = m_out.cursor();
|
||||
lim = m_out.limit();
|
||||
}
|
||||
};
|
||||
// copy n bytes of the source (after room(n))
|
||||
const auto copy = [&](const char* from, std::size_t n)
|
||||
{
|
||||
if (n <= 32 && src_end - from >= 32)
|
||||
{
|
||||
std::memcpy(w, from, 32);
|
||||
}
|
||||
else if (n <= 256 && src_end - from >= static_cast<std::ptrdiff_t>(n) + 32)
|
||||
{
|
||||
for (std::size_t i = 0; i < n; i += 32)
|
||||
{
|
||||
std::memcpy(w + i, from + i, 32);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
copy_long(w, from, n);
|
||||
}
|
||||
w += n;
|
||||
};
|
||||
// a literal of n bytes (after room(n))
|
||||
const auto literal = [&](const char* text, std::size_t n)
|
||||
{
|
||||
std::memcpy(w, text, n);
|
||||
w += n;
|
||||
};
|
||||
// a string that is not a plain string of the source (out of line, so
|
||||
// that the cursor stays in a register here)
|
||||
const auto escaped = [&](const node & n)
|
||||
{
|
||||
m_out.set_cursor(w);
|
||||
write_decoded(n);
|
||||
w = m_out.cursor();
|
||||
lim = m_out.limit();
|
||||
};
|
||||
|
||||
// Read-only documents: the elements of a container follow it in the
|
||||
// node array, so the walk goes through the array in order, and a
|
||||
// frame only needs the end of its container. Editable documents: the
|
||||
// elements of a moved container live elsewhere, so a frame keeps the
|
||||
// position of the next element (see navigation).
|
||||
// The innermost open container is kept in registers (cur; end ==
|
||||
// nullptr: none), the stack holds the ones around it.
|
||||
frame cur{nullptr, nullptr, false};
|
||||
const node* n = root;
|
||||
for (;;)
|
||||
{
|
||||
// write the value at n (read-only documents: and advance n)
|
||||
bool opened = false;
|
||||
switch (static_cast<value_t>(n->kind))
|
||||
{
|
||||
case value_t::string:
|
||||
if ((n->flags & node_flags::storage) == 0)
|
||||
{
|
||||
room(n->len + 2);
|
||||
*w++ = '"';
|
||||
copy(src + n->off, n->len);
|
||||
*w++ = '"';
|
||||
}
|
||||
else
|
||||
{
|
||||
escaped(*n);
|
||||
}
|
||||
break;
|
||||
case value_t::number_integer:
|
||||
case value_t::number_unsigned:
|
||||
{
|
||||
const std::uint32_t len = number_length(*n);
|
||||
room(len);
|
||||
if (Editable && (n->flags & node_flags::storage) != 0)
|
||||
{
|
||||
copy_long(w, m_doc.str(*n), len); // a canonical token written by an edit
|
||||
w += len;
|
||||
break;
|
||||
}
|
||||
const char* const token = src + n->off;
|
||||
if (!SourceNumbers && NLOHMANN_VIEW_UNLIKELY(len == 2 && token[0] == '-' && token[1] == '0'))
|
||||
{
|
||||
*w++ = '0'; // parse() reads -0 as the integer 0
|
||||
}
|
||||
else
|
||||
{
|
||||
copy(token, len);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case value_t::number_float:
|
||||
if (SourceNumbers && (n->flags & node_flags::storage) != node_flags::edited)
|
||||
{
|
||||
room(n->len);
|
||||
copy(src + n->off, n->len);
|
||||
}
|
||||
else
|
||||
{
|
||||
m_out.set_cursor(w);
|
||||
write_float(float_value<number_float_t>(m_doc, *n));
|
||||
w = m_out.cursor();
|
||||
lim = m_out.limit();
|
||||
}
|
||||
break;
|
||||
case value_t::boolean:
|
||||
room(8);
|
||||
if ((n->flags & node_flags::is_true) != 0)
|
||||
{
|
||||
literal("true", 4);
|
||||
}
|
||||
else
|
||||
{
|
||||
literal("false", 5);
|
||||
}
|
||||
break;
|
||||
case value_t::object:
|
||||
case value_t::array:
|
||||
{
|
||||
const bool object = n->kind == static_cast<std::uint8_t>(value_t::object);
|
||||
room(8);
|
||||
if (n->len == 0)
|
||||
{
|
||||
literal(object ? "{}" : "[]", 2);
|
||||
}
|
||||
else
|
||||
{
|
||||
*w++ = object ? '{' : '[';
|
||||
stack.push_back(cur);
|
||||
if (Editable)
|
||||
{
|
||||
cur = frame{nav::first(m_doc, n), nav::end(m_doc, n), object};
|
||||
}
|
||||
else
|
||||
{
|
||||
cur = frame{nullptr, n + n->next, object};
|
||||
}
|
||||
opened = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case value_t::null:
|
||||
room(8);
|
||||
literal("null", 4);
|
||||
break;
|
||||
case value_t::binary: // LCOV_EXCL_LINE (not in a document)
|
||||
case value_t::discarded: // LCOV_EXCL_LINE
|
||||
default: // LCOV_EXCL_LINE
|
||||
break; // LCOV_EXCL_LINE
|
||||
}
|
||||
if (!Editable)
|
||||
{
|
||||
++n; // the next node: the first element of an opened container, or the node after a scalar
|
||||
}
|
||||
|
||||
// go to the next value: close finished containers, then separate
|
||||
// (a container just opened has an element)
|
||||
if (!opened)
|
||||
{
|
||||
for (;;)
|
||||
{
|
||||
if (cur.end == nullptr)
|
||||
{
|
||||
m_out.set_cursor(w);
|
||||
m_out.finish();
|
||||
return;
|
||||
}
|
||||
if ((Editable ? cur.pos : n) != cur.end)
|
||||
{
|
||||
break;
|
||||
}
|
||||
room(1);
|
||||
*w++ = cur.object ? '}' : ']';
|
||||
cur = stack.back();
|
||||
stack.pop_back();
|
||||
}
|
||||
room(1);
|
||||
*w++ = ',';
|
||||
}
|
||||
const node* const at = Editable ? cur.pos : n;
|
||||
if (cur.object)
|
||||
{
|
||||
const node& key = *at;
|
||||
if ((key.flags & node_flags::storage) == 0)
|
||||
{
|
||||
room(key.len + 3);
|
||||
*w++ = '"';
|
||||
copy(src + key.off, key.len);
|
||||
w[0] = '"';
|
||||
w[1] = ':';
|
||||
w += 2;
|
||||
}
|
||||
else
|
||||
{
|
||||
escaped(key);
|
||||
room(1);
|
||||
*w++ = ':';
|
||||
}
|
||||
if (Editable)
|
||||
{
|
||||
n = nav::value(at + 1);
|
||||
cur.pos = document_data::after(at + 1);
|
||||
}
|
||||
else
|
||||
{
|
||||
++n;
|
||||
}
|
||||
}
|
||||
else if (Editable)
|
||||
{
|
||||
n = nav::value(at);
|
||||
cur.pos = document_data::after(at);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void newline(std::size_t level)
|
||||
{
|
||||
if (m_style.pretty)
|
||||
@@ -317,7 +725,9 @@ class view_serializer
|
||||
m_out.put('"');
|
||||
}
|
||||
|
||||
/// as serializer::dump_escaped() for valid UTF-8 (the view has no other)
|
||||
/// as serializer::dump_escaped(); strings of a document are valid UTF-8,
|
||||
/// except in a damaged image loaded with image_check::bounds, for which
|
||||
/// this throws what basic_json::dump() throws for the string
|
||||
template<bool EnsureAscii>
|
||||
void write_escaped(const unsigned char* s, std::size_t n)
|
||||
{
|
||||
@@ -341,12 +751,13 @@ class view_serializer
|
||||
}
|
||||
std::uint32_t codepoint = s[i];
|
||||
std::size_t len = 1;
|
||||
if (codepoint >= 0xC0)
|
||||
if (codepoint >= 0x80)
|
||||
{
|
||||
len = 2;
|
||||
if (codepoint >= 0xE0)
|
||||
len = validate_one_utf8(s + i, n - i);
|
||||
if (NLOHMANN_VIEW_UNLIKELY(len == 0))
|
||||
{
|
||||
len = codepoint >= 0xF0 ? 4 : 3;
|
||||
invalid_utf8(s, n);
|
||||
return;
|
||||
}
|
||||
codepoint &= 0xFFu >> (len + 1);
|
||||
for (std::size_t k = 1; k < len; ++k)
|
||||
@@ -359,6 +770,13 @@ class view_serializer
|
||||
}
|
||||
}
|
||||
|
||||
/// throw what basic_json::dump() throws for a string that is not valid UTF-8
|
||||
NLOHMANN_VIEW_NOINLINE static void invalid_utf8(const unsigned char* s, std::size_t n)
|
||||
{
|
||||
const string_t dumped = BasicJsonType(string_t(reinterpret_cast<const char*>(s), n)).dump(); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
static_cast<void>(dumped);
|
||||
}
|
||||
|
||||
template<bool EnsureAscii>
|
||||
void write_codepoint(std::uint32_t codepoint, const unsigned char* bytes, std::size_t len)
|
||||
{
|
||||
|
||||
@@ -25,7 +25,7 @@
|
||||
#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstdint> // uint8_t, uint32_t
|
||||
#include <cstring> // memcpy, strlen
|
||||
#include <iterator> // distance, input_iterator_tag, iterator_traits
|
||||
#include <map> // map
|
||||
@@ -53,6 +53,7 @@
|
||||
#include <nlohmann/detail/view/edit.hpp>
|
||||
#include <nlohmann/detail/view/edit_storage.hpp>
|
||||
#include <nlohmann/detail/view/errors.hpp>
|
||||
#include <nlohmann/detail/view/image.hpp>
|
||||
#include <nlohmann/detail/view/input.hpp>
|
||||
#include <nlohmann/detail/view/iterator.hpp>
|
||||
#include <nlohmann/detail/view/lookup.hpp>
|
||||
@@ -591,8 +592,10 @@ class basic_json_view
|
||||
style.indent_char = indent_char;
|
||||
style.ensure_ascii = ensure_ascii;
|
||||
style.source_numbers = numbers == number_format::source;
|
||||
// the compact text is about as long as the source text of the value
|
||||
const std::size_t estimate = source_extent() + (style.pretty ? source_extent() / 2 : 0) + 64;
|
||||
// the compact text is about as long as the source text of the value;
|
||||
// the compact writer keeps 64 bytes of slack, so that it does not grow
|
||||
// the buffer just before the end
|
||||
const std::size_t estimate = source_extent() + (style.pretty ? source_extent() / 2 : 0) + 160;
|
||||
detail::view::view_serializer<BasicJsonType, Editable>(*m_doc, out, estimate, style).dump(m_node);
|
||||
return out;
|
||||
}
|
||||
@@ -933,7 +936,7 @@ class basic_json_document
|
||||
/// whether the document holds its own copy of the text
|
||||
bool owns_source() const noexcept
|
||||
{
|
||||
return m_data && !m_data->owned.empty() && m_data->src == m_data->owned.data();
|
||||
return m_data && ((!m_data->owned.empty() && m_data->src == m_data->owned.data()) || !m_data->owned_image.empty());
|
||||
}
|
||||
|
||||
/// number of index nodes (values plus object keys)
|
||||
@@ -942,7 +945,7 @@ class basic_json_document
|
||||
return m_data ? m_data->tape_size : 0;
|
||||
}
|
||||
|
||||
/// bytes held by the document (index, decoded strings, owned text)
|
||||
/// bytes held by the document (index, decoded strings, owned text or image)
|
||||
std::size_t memory_usage() const noexcept
|
||||
{
|
||||
if (!m_data)
|
||||
@@ -951,7 +954,7 @@ class basic_json_document
|
||||
}
|
||||
return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node))
|
||||
+ (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0)
|
||||
+ m_data->arena.capacity() + m_data->owned.capacity()
|
||||
+ m_data->arena.capacity() + m_data->owned.capacity() + m_data->owned_image.capacity()
|
||||
+ (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t))
|
||||
+ (m_data->large_objects.capacity() * sizeof(std::uint32_t))
|
||||
+ (m_data->edits != nullptr ? m_data->edits->bytes : 0);
|
||||
@@ -971,8 +974,10 @@ class basic_json_document
|
||||
|
||||
// allocate everything first, so that an exception leaves the document
|
||||
// unchanged
|
||||
// (the decoded strings of a loaded image stay in the image)
|
||||
const bool arena_in_use = d.base[1] == d.arena.data();
|
||||
const bool shrink_arena = d.arena.capacity() > d.arena.size();
|
||||
std::string arena(shrink_arena ? d.arena : std::string());
|
||||
std::string arena(shrink_arena && arena_in_use ? d.arena : std::string());
|
||||
// (edits link to the nodes of the index, which then stays in place)
|
||||
const bool shrink_tape = d.tape != d.inline_tape && d.tape_size != d.tape_cap && d.edits == nullptr;
|
||||
const bool into_header = d.tape_size <= d.inline_cap;
|
||||
@@ -988,10 +993,62 @@ class basic_json_document
|
||||
if (shrink_arena)
|
||||
{
|
||||
d.arena.swap(arena);
|
||||
d.base[1] = d.arena.data();
|
||||
if (arena_in_use)
|
||||
{
|
||||
d.base[1] = d.arena.data();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
////////////
|
||||
// images //
|
||||
////////////
|
||||
|
||||
/// how load() checks an image (full, bounds, or none)
|
||||
using image_check = detail::view::image_check;
|
||||
|
||||
/// The document as an image that load() reads without parsing: the node
|
||||
/// index, the text, and the decoded strings. An edited document is
|
||||
/// written in its current state (floats that are not finite become null,
|
||||
/// as in dump()).
|
||||
std::vector<std::uint8_t> save() const
|
||||
{
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!m_data || m_data->discarded))
|
||||
{
|
||||
detail::view::throw_type_error(320, "cannot save a discarded json_document");
|
||||
}
|
||||
return detail::view::save_image(*m_data);
|
||||
}
|
||||
|
||||
/// Read an image written by save(). The image is borrowed: it must stay
|
||||
/// alive and unchanged while the document is used.
|
||||
NLOHMANN_VIEW_NODISCARD
|
||||
static basic_json_document load(const std::uint8_t* image, std::size_t size, const image_check check = image_check::full)
|
||||
{
|
||||
basic_json_document d;
|
||||
d.ensure_data(nullptr, 0);
|
||||
detail::view::load_image(*d.m_data, image, size, check);
|
||||
return d;
|
||||
}
|
||||
|
||||
/// read an image (borrowed)
|
||||
NLOHMANN_VIEW_NODISCARD
|
||||
static basic_json_document load(const std::vector<std::uint8_t>& image, const image_check check = image_check::full)
|
||||
{
|
||||
return load(image.data(), image.size(), check);
|
||||
}
|
||||
|
||||
/// read an image and keep it (no copy)
|
||||
NLOHMANN_VIEW_NODISCARD
|
||||
static basic_json_document load(std::vector<std::uint8_t>&& image, const image_check check = image_check::full)
|
||||
{
|
||||
basic_json_document d;
|
||||
d.ensure_data(nullptr, 0);
|
||||
d.m_data->owned_image = std::move(image);
|
||||
detail::view::load_image(*d.m_data, d.m_data->owned_image.data(), d.m_data->owned_image.size(), check);
|
||||
return d;
|
||||
}
|
||||
|
||||
///////////
|
||||
// edits //
|
||||
///////////
|
||||
@@ -1175,6 +1232,7 @@ class basic_json_document
|
||||
{
|
||||
d.owned.clear();
|
||||
}
|
||||
d.owned_image.clear();
|
||||
d.src = src;
|
||||
d.size = size;
|
||||
d.tape_size = 0;
|
||||
@@ -1199,6 +1257,7 @@ class basic_json_document
|
||||
{
|
||||
d.base[0] = d.src;
|
||||
d.base[1] = d.arena.data();
|
||||
d.arena_size = d.arena.size();
|
||||
detail::view::build_object_indexes(d);
|
||||
d.discarded = false;
|
||||
return;
|
||||
|
||||
@@ -23918,11 +23918,240 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
#include <array> // array
|
||||
#include <cmath> // signbit, isfinite
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // intN_t, uintN_t
|
||||
#include <cstring> // memcpy, memmove
|
||||
#include <limits> // numeric_limits
|
||||
#include <type_traits> // conditional
|
||||
|
||||
#ifdef _MSC_VER
|
||||
#include <cstdlib> // _byteswap_uint64
|
||||
#endif
|
||||
|
||||
// #include <nlohmann/detail/conversions/zmij.hpp>
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2025 Victor Zverovich <https://github.com/vitaut/zmij>
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t, uint64_t
|
||||
|
||||
// #include <nlohmann/detail/abi_macros.hpp>
|
||||
|
||||
// #include <nlohmann/detail/bit_ops.hpp>
|
||||
|
||||
// #include <nlohmann/detail/input/pow5_table.hpp>
|
||||
|
||||
// #include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
/*!
|
||||
@brief the shortest decimal representation of a double
|
||||
|
||||
A C++11 port of the conversion of Zmij by Victor Zverovich
|
||||
(https://github.com/vitaut/zmij, MIT license): the shortest decimal in the
|
||||
rounding interval of a double, the closest one if there are several. Zmij
|
||||
credits Xiang JunBo (producing the shorter candidate without a division) and
|
||||
Dougall Johnson (the compressed powers of ten). The powers of ten are taken
|
||||
from the table for number parsing (pow5_table.hpp) where it holds them, and
|
||||
computed from the compressed tables of Zmij beyond it.
|
||||
*/
|
||||
namespace zmij
|
||||
{
|
||||
|
||||
/// significand * 10^exponent
|
||||
struct decimal
|
||||
{
|
||||
std::uint64_t significand;
|
||||
int exponent;
|
||||
};
|
||||
|
||||
/// the compressed powers of ten of Zmij
|
||||
inline const std::array<std::uint64_t, 28>& pow10_minor() noexcept
|
||||
{
|
||||
static const std::array<std::uint64_t, 28> table =
|
||||
{
|
||||
{
|
||||
0x8000000000000000u, 0xa000000000000000u, 0xc800000000000000u, 0xfa00000000000000u, 0x9c40000000000000u,
|
||||
0xc350000000000000u, 0xf424000000000000u, 0x9896800000000000u, 0xbebc200000000000u, 0xee6b280000000000u,
|
||||
0x9502f90000000000u, 0xba43b74000000000u, 0xe8d4a51000000000u, 0x9184e72a00000000u, 0xb5e620f480000000u,
|
||||
0xe35fa931a0000000u, 0x8e1bc9bf04000000u, 0xb1a2bc2ec5000000u, 0xde0b6b3a76400000u, 0x8ac7230489e80000u,
|
||||
0xad78ebc5ac620000u, 0xd8d726b7177a8000u, 0x878678326eac9000u, 0xa968163f0a57b400u, 0xd3c21bcecceda100u,
|
||||
0x84595161401484a0u, 0xa56fa5b99019a5c8u, 0xcecb8f27f4200f3au
|
||||
}
|
||||
};
|
||||
return table;
|
||||
}
|
||||
|
||||
/// (high, low) pairs
|
||||
inline const std::array<std::uint64_t, 50>& pow10_major() noexcept
|
||||
{
|
||||
static const std::array<std::uint64_t, 50> table =
|
||||
{
|
||||
{
|
||||
0xaddcb9e83c6b1793u, 0xdf4abe242a1bbf3eu, 0xaf8e5410288e1b6fu, 0x07ecf0ae5ee44ddau, 0xb1442798f49ffb4au, 0x99cd11cfdf41779du,
|
||||
0xb2fe3f0b8599ef07u, 0x861fa7e6dcb4aa15u, 0xb4bca50b065abe63u, 0x0fed077a756b53aau, 0xb67f6455292cbf08u, 0x1a3bc84c17b1d543u,
|
||||
0xb84687c269ef3bfbu, 0x3d5d514f40eea742u, 0xba121a4650e4ddebu, 0x92f34d62616ce413u, 0xbbe226efb628afeau, 0x890489f70a55368cu,
|
||||
0xbdb6b8e905cb600fu, 0x5400e987bbc1c921u, 0xbf8fdb78849a5f96u, 0xde98520472bdd034u, 0xc16d9a0095928a27u, 0x75b7053c0f178294u,
|
||||
0xc350000000000000u, 0x0000000000000000u, 0xc5371912364ce305u, 0x6c28000000000000u, 0xc722f0ef9d80aad6u, 0x424d3ad2b7b97ef6u,
|
||||
0xc913936dd571c84cu, 0x03bc3a19cd1e38eau, 0xcb090c8001ab551cu, 0x5cadf5bfd3072cc6u, 0xcd036837130890a1u, 0x36dba887c37a8c10u,
|
||||
0xcf02b2c21207ef2eu, 0x94f967e45e03f4bcu, 0xd106f86e69d785c7u, 0xe13336d701beba52u, 0xd31045a8341ca07cu, 0x1ede48111209a051u,
|
||||
0xd51ea6fa85785631u, 0x552a74227f3ea566u, 0xd732290fbacaf133u, 0xa97c177947ad4096u, 0xd94ad8b1c7380874u, 0x18375281ae7822bdu,
|
||||
0xdb68c2ca82ed2a05u, 0xa67398db9f6820e1u
|
||||
}
|
||||
};
|
||||
return table;
|
||||
}
|
||||
|
||||
/// one bit per power: whether the computed value is one unit too large
|
||||
inline const std::array<std::uint32_t, 21>& pow10_fixups() noexcept
|
||||
{
|
||||
static const std::array<std::uint32_t, 21> table =
|
||||
{
|
||||
{
|
||||
0x8d8fc810u, 0x06100293u, 0x19000000u, 0x00100000u, 0x00000908u, 0x00000000u, 0x04e00300u, 0x3807e0b2u, 0x3d83d793u, 0x0006f5ccu,
|
||||
0x00000000u, 0xffff0000u, 0x8076337du, 0x4ff45ba0u, 0x09405033u, 0x034376d9u, 0x09000000u, 0x4e100501u, 0x076d14dcu, 0xf964f45eu,
|
||||
0x0000003du
|
||||
}
|
||||
};
|
||||
return table;
|
||||
}
|
||||
|
||||
/// the 128-bit significand of 10^k, rounded down, for k in [-307, 341]
|
||||
/// (compute_pow10 of Zmij)
|
||||
inline uint128_parts compute_pow10(int k) noexcept
|
||||
{
|
||||
const auto i = static_cast<unsigned>(k + 307);
|
||||
const std::uint64_t m = pow10_minor()[(i + 24) % 28];
|
||||
const std::size_t j = 2 * static_cast<std::size_t>((i + 24) / 28);
|
||||
const std::uint64_t h_hi = pow10_major()[j];
|
||||
const std::uint64_t h_lo = pow10_major()[j + 1];
|
||||
const std::uint64_t h1 = full_multiplication(h_lo, m).high;
|
||||
const std::uint64_t c0 = h_lo * m;
|
||||
const std::uint64_t c1 = h1 + (h_hi * m);
|
||||
const std::uint64_t c2 = (c1 < h1 ? 1u : 0u) + full_multiplication(h_hi, m).high;
|
||||
uint128_parts r{};
|
||||
if ((c2 >> 63u) != 0)
|
||||
{
|
||||
r.high = c2;
|
||||
r.low = c1;
|
||||
}
|
||||
else
|
||||
{
|
||||
r.high = (c2 << 1u) | (c1 >> 63u);
|
||||
r.low = (c1 << 1u) | (c0 >> 63u);
|
||||
}
|
||||
r.low -= (pow10_fixups()[i >> 5u] >> (i & 31u)) & 1u;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// The 128-bit significand of 10^k, rounded down, for k in [-342, 341].
|
||||
/// Up to 10^308, the table for number parsing holds the same significands
|
||||
/// (those of 5^k), except for k in [-27, -1], where it holds them one unit
|
||||
/// larger (as the Eisel-Lemire algorithm needs them).
|
||||
inline uint128_parts pow10(int k) noexcept
|
||||
{
|
||||
if (k > pow5_128_largest_power)
|
||||
{
|
||||
return compute_pow10(k); // (only for the smallest doubles)
|
||||
}
|
||||
const auto i = 2 * static_cast<std::size_t>(k - pow5_128_smallest_power);
|
||||
uint128_parts r{pow5_128()[i + 1], pow5_128()[i]};
|
||||
const std::uint64_t adjust = static_cast<unsigned>(k + 27) < 27u ? 1u : 0u;
|
||||
r.high -= r.low < adjust ? 1u : 0u;
|
||||
r.low -= adjust;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// (x_hi * 2^64 + x_lo) * y >> 64, as 128 bits
|
||||
inline uint128_parts umul192_hi128(std::uint64_t x_hi, std::uint64_t x_lo, std::uint64_t y) noexcept
|
||||
{
|
||||
const uint128_parts p = full_multiplication(x_hi, y);
|
||||
uint128_parts r{};
|
||||
r.low = p.low + full_multiplication(x_lo, y).high;
|
||||
r.high = p.high + (r.low < p.low ? 1u : 0u);
|
||||
return r;
|
||||
}
|
||||
|
||||
/// (x * y + c) >> 64
|
||||
inline std::uint64_t umul128_add_hi64(std::uint64_t x, std::uint64_t y, std::uint64_t c) noexcept
|
||||
{
|
||||
const uint128_parts p = full_multiplication(x, y);
|
||||
return p.high + (p.low + c < p.low ? 1u : 0u);
|
||||
}
|
||||
|
||||
/// The shortest decimal in the rounding interval of a positive finite double
|
||||
/// given by its bits, the closest one if there are several (to_decimal of
|
||||
/// Zmij). The significand can end in zeros.
|
||||
inline decimal to_decimal(std::uint64_t bits) noexcept
|
||||
{
|
||||
constexpr int extra_shift = 9;
|
||||
const auto raw_exp = static_cast<int>((bits >> 52u) & 0x7FFu);
|
||||
std::uint64_t bin_sig = bits & ((std::uint64_t{1} << 52u) - 1);
|
||||
// a power of two has a narrower interval below (except the smallest normal)
|
||||
const bool regular = bin_sig != 0 || raw_exp <= 1;
|
||||
const int bin_exp = (raw_exp == 0 ? 1 : raw_exp) - 1075;
|
||||
if (raw_exp != 0)
|
||||
{
|
||||
bin_sig |= std::uint64_t{1} << 52u;
|
||||
}
|
||||
// floor(log10(2^bin_exp)), or floor(log10(3/4 * 2^bin_exp)) for the irregular case
|
||||
const int dec_exp = ((bin_exp * 315653) - (regular ? 0 : 131072)) >> 20;
|
||||
// scaled by 10^(-dec_exp - 1): the integral part is the shorter candidate
|
||||
const int shift = bin_exp + ((-(dec_exp + 1) * 217707) >> 16) + 1 + extra_shift;
|
||||
const uint128_parts p10 = pow10(-dec_exp - 1);
|
||||
const uint128_parts p = umul192_hi128(p10.high, p10.low, bin_sig << static_cast<unsigned>(shift));
|
||||
std::uint64_t integral = p.high >> static_cast<unsigned>(extra_shift);
|
||||
const std::uint64_t fractional = (p.high << static_cast<unsigned>(64 - extra_shift)) | (p.low >> static_cast<unsigned>(extra_shift));
|
||||
std::uint64_t digit = 0;
|
||||
bool round_up = false;
|
||||
bool round_down = false;
|
||||
if (JSON_HEDLEY_LIKELY(regular))
|
||||
{
|
||||
const std::uint64_t half_ulp = (p10.high >> static_cast<unsigned>(extra_shift + 1 - shift)) + (1 - (bin_sig & 1u));
|
||||
round_up = fractional + half_ulp < fractional;
|
||||
round_down = half_ulp > fractional;
|
||||
// the last digit of the longer candidate, rounded to nearest
|
||||
digit = umul128_add_hi64(fractional, 10, (std::uint64_t{1} << 63u) + 6);
|
||||
if (fractional == (std::uint64_t{1} << 62u))
|
||||
{
|
||||
digit = 2; // 2.5 rounds to 2
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
const std::uint64_t half_ulp = p10.high >> static_cast<unsigned>(extra_shift + 1 - shift);
|
||||
round_up = half_ulp > ~std::uint64_t{0} - fractional;
|
||||
round_down = (half_ulp >> 1u) > fractional;
|
||||
digit = umul128_add_hi64(fractional, 10, (std::uint64_t{1} << 63u) - 1);
|
||||
const std::uint64_t lowest = umul128_add_hi64(fractional - (half_ulp >> 1u), 10, ~std::uint64_t{0});
|
||||
digit = digit < lowest ? lowest : digit;
|
||||
}
|
||||
integral += round_up ? 1u : 0u;
|
||||
if (!round_up && !round_down)
|
||||
{
|
||||
// the shorter candidate is outside the rounding interval: one digit more
|
||||
return decimal{(integral * 10) + digit, dec_exp};
|
||||
}
|
||||
return decimal{integral, dec_exp + 1};
|
||||
}
|
||||
|
||||
} // namespace zmij
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// #include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
|
||||
@@ -24826,6 +25055,87 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value)
|
||||
grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the shortest digits of a positive finite float (other than double): Grisu2
|
||||
*/
|
||||
template<typename FloatType>
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value)
|
||||
{
|
||||
grisu2(buf, len, decimal_exponent, value);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the shortest digits of a positive finite double: the conversion of
|
||||
Zmij (see zmij.hpp), which always finds the shortest digits that read back as
|
||||
the same value (Grisu2 does not for about one double in a thousand), and the
|
||||
closest of them if there are several
|
||||
|
||||
v = buf * 10^decimal_exponent, as for grisu2()
|
||||
*/
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value)
|
||||
{
|
||||
static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
|
||||
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
|
||||
JSON_ASSERT(std::isfinite(value));
|
||||
JSON_ASSERT(value > 0);
|
||||
|
||||
std::uint64_t bits = 0;
|
||||
std::memcpy(&bits, &value, sizeof(bits));
|
||||
zmij::decimal d = zmij::to_decimal(bits);
|
||||
// without trailing zeros (up to 16): 8, 4, 2, 1 at a time
|
||||
while (d.significand % 100000000 == 0)
|
||||
{
|
||||
d.significand /= 100000000;
|
||||
d.exponent += 8;
|
||||
}
|
||||
if (d.significand % 10000 == 0)
|
||||
{
|
||||
d.significand /= 10000;
|
||||
d.exponent += 4;
|
||||
}
|
||||
if (d.significand % 100 == 0)
|
||||
{
|
||||
d.significand /= 100;
|
||||
d.exponent += 2;
|
||||
}
|
||||
if (d.significand % 10 == 0)
|
||||
{
|
||||
d.significand /= 10;
|
||||
d.exponent += 1;
|
||||
}
|
||||
// at most 17 digits, written from the back two at a time
|
||||
static constexpr const char* pairs =
|
||||
"00010203040506070809101112131415161718192021222324252627282930313233343536373839"
|
||||
"40414243444546474849505152535455565758596061626364656667686970717273747576777879"
|
||||
"8081828384858687888990919293949596979899";
|
||||
std::array<char, 20> digits{};
|
||||
std::size_t n = digits.size();
|
||||
while (d.significand >= 100)
|
||||
{
|
||||
const auto i = static_cast<std::size_t>(d.significand % 100) * 2;
|
||||
d.significand /= 100;
|
||||
n -= 2;
|
||||
digits[n] = pairs[i];
|
||||
digits[n + 1] = pairs[i + 1];
|
||||
}
|
||||
if (d.significand >= 10)
|
||||
{
|
||||
const auto i = static_cast<std::size_t>(d.significand) * 2;
|
||||
n -= 2;
|
||||
digits[n] = pairs[i];
|
||||
digits[n + 1] = pairs[i + 1];
|
||||
}
|
||||
else
|
||||
{
|
||||
digits[--n] = static_cast<char>('0' + d.significand);
|
||||
}
|
||||
len = static_cast<int>(digits.size() - n);
|
||||
std::memcpy(buf, digits.data() + n, static_cast<std::size_t>(len));
|
||||
decimal_exponent = d.exponent;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief appends a decimal representation of e to buf
|
||||
@return a pointer to the element following the exponent.
|
||||
@@ -24955,6 +25265,177 @@ inline char* format_buffer(char* buf, int len, int decimal_exponent,
|
||||
return append_exponent(buf, n - 1);
|
||||
}
|
||||
|
||||
/// eight decimal digits (a value below 10^8) as bytes 0..9, the first digit
|
||||
/// in the most significant byte: three steps that divide all lanes at once
|
||||
/// by a multiplication (the conversion of Xiang JunBo, as in Zmij)
|
||||
inline std::uint64_t eight_digit_bytes(std::uint64_t abcdefgh) noexcept
|
||||
{
|
||||
const std::uint64_t abcd_efgh = abcdefgh + (((std::uint64_t{1} << 32u) - 10000u) * ((abcdefgh * (((std::uint64_t{1} << 40u) / 10000u) + 1u)) >> 40u));
|
||||
const std::uint64_t ab_cd_ef_gh = abcd_efgh + (((std::uint64_t{1} << 16u) - 100u) * (((abcd_efgh * (((std::uint64_t{1} << 19u) / 100u) + 1u)) >> 19u) & 0x7F0000007Fu));
|
||||
return ab_cd_ef_gh + (((std::uint64_t{1} << 8u) - 10u) * (((ab_cd_ef_gh * (((std::uint64_t{1} << 10u) / 10u) + 1u)) >> 10u) & 0x000F000F000F000Fu));
|
||||
}
|
||||
|
||||
/// store the bytes of v, the most significant one first (one byte swap and
|
||||
/// one store where the byte order is known: compilers do not reliably merge
|
||||
/// the byte stores once this is inlined)
|
||||
inline void store_msb_first(char* p, std::uint64_t v) noexcept
|
||||
{
|
||||
#if defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
|
||||
v = __builtin_bswap64(v);
|
||||
std::memcpy(p, &v, sizeof(v));
|
||||
#elif defined(__BYTE_ORDER__) && defined(__ORDER_BIG_ENDIAN__) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__
|
||||
std::memcpy(p, &v, sizeof(v));
|
||||
#elif defined(_MSC_VER) // (little-endian on all its targets)
|
||||
v = _byteswap_uint64(v);
|
||||
std::memcpy(p, &v, sizeof(v));
|
||||
#else
|
||||
for (unsigned i = 0; i < 8; ++i)
|
||||
{
|
||||
p[i] = static_cast<char>(v >> (56u - (8u * i)));
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief digits * 10^exp for a double, in the layout of format_buffer()
|
||||
|
||||
The layout is that of format_buffer() with min_exp -4 and max_exp 15 (the
|
||||
digits10 of double). The digits are converted eight at a time and placed
|
||||
with fixed-size moves instead of per-digit loops and moves of the buffer.
|
||||
|
||||
@param[in] digits the digits (not 0, at most 17 digits; trailing zeros allowed)
|
||||
@param[in] exp the decimal exponent of the last digit
|
||||
@return a pointer past the text; up to 41 bytes at @a first are written
|
||||
(some beyond the returned end)
|
||||
*/
|
||||
JSON_HEDLEY_NON_NULL(1)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
inline char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept
|
||||
{
|
||||
JSON_ASSERT(digits != 0 && digits < 100000000000000000u);
|
||||
const std::uint64_t upper = digits / 100000000u;
|
||||
const std::uint64_t b0 = upper / 100000000u; // (one digit: it is its own byte)
|
||||
const std::uint64_t b1 = eight_digit_bytes(upper % 100000000u);
|
||||
const std::uint64_t b2 = eight_digit_bytes(digits % 100000000u);
|
||||
// leading and trailing zero digits: zero bytes, counted without division
|
||||
int leading = 16;
|
||||
int zeros = 16;
|
||||
if (b0 != 0)
|
||||
{
|
||||
leading = count_leading_zeros(b0) / 8;
|
||||
}
|
||||
else if (b1 != 0)
|
||||
{
|
||||
leading = 8 + (count_leading_zeros(b1) / 8);
|
||||
}
|
||||
else
|
||||
{
|
||||
leading += count_leading_zeros(b2) / 8;
|
||||
}
|
||||
if (b2 != 0)
|
||||
{
|
||||
zeros = count_trailing_zeros(b2) / 8;
|
||||
}
|
||||
else if (b1 != 0)
|
||||
{
|
||||
zeros = 8 + (count_trailing_zeros(b1) / 8);
|
||||
}
|
||||
// (else: 16, b0 is the one digit that is not 0)
|
||||
// the digits as text at text + leading, then '0's, so that fixed-size
|
||||
// moves need not check how many digits there are
|
||||
std::array<char, 64> text; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
|
||||
store_msb_first(text.data(), b0 + 0x3030303030303030u);
|
||||
store_msb_first(text.data() + 8, b1 + 0x3030303030303030u);
|
||||
store_msb_first(text.data() + 16, b2 + 0x3030303030303030u);
|
||||
std::memset(text.data() + 24, '0', 40);
|
||||
const int k = 24 - leading - zeros; // significant digits
|
||||
const int n = k + exp + zeros; // position of the decimal point after the first digit
|
||||
const char* const s0 = text.data() + leading;
|
||||
|
||||
if (-4 < n && n <= 15)
|
||||
{
|
||||
// "0.[000]digits" (n <= 0) is the digits after 1 - n leading '0's
|
||||
// with the point after the first; "digits[000].0" (n >= k) and
|
||||
// "dig.its" put the point after n characters
|
||||
const int pad = n <= 0 ? 1 - n : 0;
|
||||
const char* const s = s0 - pad;
|
||||
const int len = k + pad;
|
||||
const int point = n + pad;
|
||||
std::memcpy(first, s, 16);
|
||||
std::memcpy(first + point + 1, s + point, 24);
|
||||
first[point] = '.';
|
||||
return first + (point >= len ? point + 2 : len + 1);
|
||||
}
|
||||
|
||||
// d.igitse+XX, with at least two exponent digits (as append_exponent())
|
||||
std::memcpy(first, s0, 16);
|
||||
std::memcpy(first + 2, s0 + 1, 16);
|
||||
first[1] = '.';
|
||||
char* const end = first + (k == 1 ? 1 : k + 1);
|
||||
const int e = n - 1;
|
||||
const auto ea = static_cast<unsigned>(e < 0 ? -e : e);
|
||||
const bool three = ea >= 100;
|
||||
end[0] = 'e';
|
||||
end[1] = e < 0 ? '-' : '+';
|
||||
end[2] = static_cast<char>('0' + (three ? ea / 100 : (ea / 10) % 10));
|
||||
end[3] = static_cast<char>('0' + (three ? (ea / 10) % 10 : ea % 10));
|
||||
end[4] = static_cast<char>('0' + (ea % 10));
|
||||
return end + (three ? 5 : 4);
|
||||
}
|
||||
|
||||
/// a positive finite float (other than double): Grisu2 and format_buffer()
|
||||
template<typename FloatType>
|
||||
JSON_HEDLEY_NON_NULL(1, 2)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
char* write_positive(char* first, const char* last, FloatType value)
|
||||
{
|
||||
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10);
|
||||
|
||||
// Compute v = buffer * 10^decimal_exponent.
|
||||
// The decimal digits are stored in the buffer, which needs to be interpreted
|
||||
// as an unsigned decimal integer.
|
||||
// len is the length of the buffer, i.e., the number of decimal digits.
|
||||
int len = 0;
|
||||
int decimal_exponent = 0;
|
||||
shortest_digits(first, len, decimal_exponent, value);
|
||||
|
||||
JSON_ASSERT(len <= std::numeric_limits<FloatType>::max_digits10);
|
||||
|
||||
// Format the buffer like printf("%.*g", prec, value)
|
||||
constexpr int kMinExp = -4;
|
||||
// Use digits10 here to increase compatibility with version 2.
|
||||
constexpr int kMaxExp = std::numeric_limits<FloatType>::digits10;
|
||||
|
||||
JSON_ASSERT(last - first >= kMaxExp + 2);
|
||||
JSON_ASSERT(last - first >= 2 + (-kMinExp - 1) + std::numeric_limits<FloatType>::max_digits10);
|
||||
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10 + 6);
|
||||
|
||||
return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp);
|
||||
}
|
||||
|
||||
/// a positive finite double: the shortest digits (Zmij), laid out by
|
||||
/// write_decimal() (through a local buffer if [first, last) is shorter than
|
||||
/// the 41 bytes it may write)
|
||||
JSON_HEDLEY_NON_NULL(1, 2)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
inline char* write_positive(char* first, const char* last, double value)
|
||||
{
|
||||
static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
|
||||
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
|
||||
std::uint64_t bits = 0;
|
||||
std::memcpy(&bits, &value, sizeof(bits));
|
||||
const zmij::decimal d = zmij::to_decimal(bits);
|
||||
if (JSON_HEDLEY_LIKELY(last - first >= 41))
|
||||
{
|
||||
return write_decimal(first, d.significand, d.exponent);
|
||||
}
|
||||
std::array<char, 64> buf; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read
|
||||
const auto len = static_cast<std::size_t>(write_decimal(buf.data(), d.significand, d.exponent) - buf.data());
|
||||
JSON_ASSERT(static_cast<std::size_t>(last - first) >= len);
|
||||
std::memcpy(first, buf.data(), len);
|
||||
return first + len;
|
||||
}
|
||||
|
||||
} // namespace dtoa_impl
|
||||
|
||||
/*!
|
||||
@@ -24972,7 +25453,6 @@ JSON_HEDLEY_NON_NULL(1, 2)
|
||||
JSON_HEDLEY_RETURNS_NON_NULL
|
||||
char* to_chars(char* first, const char* last, FloatType value)
|
||||
{
|
||||
static_cast<void>(last); // maybe unused - fix warning
|
||||
JSON_ASSERT(std::isfinite(value));
|
||||
|
||||
// Use signbit(value) instead of (value < 0) since signbit works for -0.
|
||||
@@ -24998,28 +25478,7 @@ char* to_chars(char* first, const char* last, FloatType value)
|
||||
JSON_HEDLEY_DIAGNOSTIC_POP
|
||||
#endif
|
||||
|
||||
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10);
|
||||
|
||||
// Compute v = buffer * 10^decimal_exponent.
|
||||
// The decimal digits are stored in the buffer, which needs to be interpreted
|
||||
// as an unsigned decimal integer.
|
||||
// len is the length of the buffer, i.e., the number of decimal digits.
|
||||
int len = 0;
|
||||
int decimal_exponent = 0;
|
||||
dtoa_impl::grisu2(first, len, decimal_exponent, value);
|
||||
|
||||
JSON_ASSERT(len <= std::numeric_limits<FloatType>::max_digits10);
|
||||
|
||||
// Format the buffer like printf("%.*g", prec, value)
|
||||
constexpr int kMinExp = -4;
|
||||
// Use digits10 here to increase compatibility with version 2.
|
||||
constexpr int kMaxExp = std::numeric_limits<FloatType>::digits10;
|
||||
|
||||
JSON_ASSERT(last - first >= kMaxExp + 2);
|
||||
JSON_ASSERT(last - first >= 2 + (-kMinExp - 1) + std::numeric_limits<FloatType>::max_digits10);
|
||||
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10 + 6);
|
||||
|
||||
return dtoa_impl::format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp);
|
||||
return dtoa_impl::write_positive(first, last, value);
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
|
||||
+1336
-210
File diff suppressed because it is too large
Load Diff
+4
-1
@@ -10,7 +10,7 @@ CXXFLAGS += -std=c++11
|
||||
CPPFLAGS += -I ../single_include
|
||||
|
||||
FUZZER_ENGINE = src/fuzzer-driver_afl.cpp
|
||||
FUZZERS = parse_afl_fuzzer parse_bson_fuzzer parse_cbor_fuzzer parse_msgpack_fuzzer parse_ubjson_fuzzer parse_bjdata_fuzzer parse_bon8_fuzzer parse_json_view_fuzzer
|
||||
FUZZERS = parse_afl_fuzzer parse_bson_fuzzer parse_cbor_fuzzer parse_msgpack_fuzzer parse_ubjson_fuzzer parse_bjdata_fuzzer parse_bon8_fuzzer parse_json_view_fuzzer json_view_image_fuzzer
|
||||
fuzzers: $(FUZZERS)
|
||||
|
||||
parse_afl_fuzzer:
|
||||
@@ -19,6 +19,9 @@ parse_afl_fuzzer:
|
||||
parse_json_view_fuzzer:
|
||||
$(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-parse_json_view.cpp -o $@
|
||||
|
||||
json_view_image_fuzzer:
|
||||
$(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-json_view_image.cpp -o $@
|
||||
|
||||
parse_bson_fuzzer:
|
||||
$(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-parse_bson.cpp -o $@
|
||||
|
||||
|
||||
@@ -10,6 +10,12 @@ produces, and that a rejected input makes both parsers throw with an identical `
|
||||
reuses the `corpus_json` corpus (or, for the `make fuzz_testing_json_view` target below, `tests/data/json_tests`) rather
|
||||
than a format of its own.
|
||||
|
||||
`json_view_image_fuzzer` (`tests/src/fuzzer-json_view_image.cpp`) tests the images of `json_document` (`save()` and
|
||||
`load()`). It uses each input twice: as an image, which `load()` must either reject with `parse_error.116` or read
|
||||
safely (with `image_check::full`, the document must also serialize to the JSON it reads as), and as a JSON text, whose
|
||||
image must load and serialize to the same text. A corpus of images can be made from JSON files with a small program
|
||||
that calls `json_document::parse(text).save()`; plain JSON files work as well.
|
||||
|
||||
## Corpus creation
|
||||
|
||||
For most effective fuzzing, a [corpus](https://llvm.org/docs/LibFuzzer.html#corpus) should be provided. A corpus is a
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
/*
|
||||
This file implements a test of json_document images suitable for fuzz
|
||||
testing. The input is used twice:
|
||||
|
||||
- as an image: json_document::load() with image_check::full must either throw
|
||||
a parse_error or yield a document that serializes to the JSON text it reads
|
||||
as; with image_check::bounds, reading and serializing must be safe (checked
|
||||
by the sanitizers), and serializing may only throw type_error.316
|
||||
- as a JSON text: if json_document::parse() accepts it, the image of the
|
||||
document must load (with every check) and serialize to the same text
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
|
||||
#include <cassert>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
// the checks below are assertions; NDEBUG would compile them away
|
||||
#ifdef NDEBUG
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
using json = nlohmann::json;
|
||||
using json_document = nlohmann::json_document;
|
||||
using image_check = json_document::image_check;
|
||||
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// the input as an image
|
||||
for (const image_check check : {image_check::full, image_check::bounds})
|
||||
{
|
||||
json_document d;
|
||||
try
|
||||
{
|
||||
d = json_document::load(data, size, check);
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
assert(e.id == 116);
|
||||
continue;
|
||||
}
|
||||
std::string dumped;
|
||||
try
|
||||
{
|
||||
dumped = d.root().dump();
|
||||
}
|
||||
catch (const json::type_error& e)
|
||||
{
|
||||
// invalid UTF-8 can only pass the bounds check
|
||||
assert(check == image_check::bounds && e.id == 316);
|
||||
continue;
|
||||
}
|
||||
const json j = d.root().materialize();
|
||||
if (check == image_check::full)
|
||||
{
|
||||
assert(json::parse(dumped) == j);
|
||||
// an image of the loaded document is the input
|
||||
assert(d.save() == std::vector<std::uint8_t>(data, data + size));
|
||||
}
|
||||
}
|
||||
|
||||
// the input as a JSON text
|
||||
const std::string text(reinterpret_cast<const char*>(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
const json_document parsed = json_document::parse(text, false);
|
||||
if (!parsed.is_discarded())
|
||||
{
|
||||
const std::vector<std::uint8_t> image = parsed.save();
|
||||
for (const image_check check : {image_check::full, image_check::bounds, image_check::none})
|
||||
{
|
||||
const json_document loaded = json_document::load(image, check);
|
||||
assert(loaded.root().dump() == parsed.root().dump());
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -33,7 +33,7 @@ TEST_CASE("Binary Formats" * doctest::skip())
|
||||
const auto ubjson_2_size = json::to_ubjson(j, true).size();
|
||||
const auto ubjson_3_size = json::to_ubjson(j, true, true).size();
|
||||
|
||||
CHECK(json_size == 2090303);
|
||||
CHECK(json_size == 2090234);
|
||||
CHECK(bjdata_1_size == 1112030);
|
||||
CHECK(bjdata_2_size == 1224148);
|
||||
CHECK(bjdata_3_size == 1224148);
|
||||
@@ -46,16 +46,16 @@ TEST_CASE("Binary Formats" * doctest::skip())
|
||||
CHECK(ubjson_3_size == 1169069);
|
||||
|
||||
CHECK((100.0 * double(json_size) / double(json_size)) == Approx(100.0));
|
||||
CHECK((100.0 * double(bjdata_1_size) / double(json_size)) == Approx(53.199));
|
||||
CHECK((100.0 * double(bjdata_2_size) / double(json_size)) == Approx(58.563));
|
||||
CHECK((100.0 * double(bjdata_3_size) / double(json_size)) == Approx(58.563));
|
||||
CHECK((100.0 * double(bon8_size) / double(json_size)) == Approx(50.509));
|
||||
CHECK((100.0 * double(bson_size) / double(json_size)) == Approx(85.849));
|
||||
CHECK((100.0 * double(cbor_size) / double(json_size)) == Approx(50.497));
|
||||
CHECK((100.0 * double(msgpack_size) / double(json_size)) == Approx(50.526));
|
||||
CHECK((100.0 * double(ubjson_1_size) / double(json_size)) == Approx(53.199));
|
||||
CHECK((100.0 * double(ubjson_2_size) / double(json_size)) == Approx(58.563));
|
||||
CHECK((100.0 * double(ubjson_3_size) / double(json_size)) == Approx(55.928));
|
||||
CHECK((100.0 * double(bjdata_1_size) / double(json_size)) == Approx(53.201));
|
||||
CHECK((100.0 * double(bjdata_2_size) / double(json_size)) == Approx(58.565));
|
||||
CHECK((100.0 * double(bjdata_3_size) / double(json_size)) == Approx(58.565));
|
||||
CHECK((100.0 * double(bon8_size) / double(json_size)) == Approx(50.511));
|
||||
CHECK((100.0 * double(bson_size) / double(json_size)) == Approx(85.853));
|
||||
CHECK((100.0 * double(cbor_size) / double(json_size)) == Approx(50.499));
|
||||
CHECK((100.0 * double(msgpack_size) / double(json_size)) == Approx(50.528));
|
||||
CHECK((100.0 * double(ubjson_1_size) / double(json_size)) == Approx(53.201));
|
||||
CHECK((100.0 * double(ubjson_2_size) / double(json_size)) == Approx(58.565));
|
||||
CHECK((100.0 * double(ubjson_3_size) / double(json_size)) == Approx(55.930));
|
||||
}
|
||||
|
||||
SECTION("twitter.json")
|
||||
|
||||
@@ -1089,6 +1089,9 @@ TEST_CASE("json_view dump")
|
||||
CHECK(d.root().dump() == json::parse(text).dump());
|
||||
CHECK(d.root().dump() == "[1.5,100.0,0,-0.0,1.2345678901234568e+29,18446744073709551615,-9223372036854775808,0.1,1e-07,5e-324]");
|
||||
CHECK(d.root().dump(-1, ' ', false, json_view::number_format::source) == "[1.50,1E2,-0,-0.0,123456789012345678901234567890,18446744073709551615,-9223372036854775808,0.1,1e-7,5e-324]");
|
||||
// also indented, and with ensure_ascii
|
||||
CHECK(d.root().dump(0, ' ', false, json_view::number_format::source) == "[\n1.50,\n1E2,\n-0,\n-0.0,\n123456789012345678901234567890,\n18446744073709551615,\n-9223372036854775808,\n0.1,\n1e-7,\n5e-324\n]");
|
||||
CHECK(d.root().dump(-1, ' ', true, json_view::number_format::source) == "[1.50,1E2,-0,-0.0,123456789012345678901234567890,18446744073709551615,-9223372036854775808,0.1,1e-7,5e-324]");
|
||||
|
||||
// random doubles, written as parse() and dump() would
|
||||
std::mt19937_64 rng(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed)
|
||||
|
||||
@@ -0,0 +1,791 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
#include <nlohmann/json_view.hpp>
|
||||
using nlohmann::json;
|
||||
using nlohmann::ordered_json;
|
||||
using nlohmann::json_document;
|
||||
using nlohmann::json_editable_document;
|
||||
using nlohmann::ordered_json_document;
|
||||
using nlohmann::ordered_json_editable_document;
|
||||
using image_check = json_document::image_check;
|
||||
using nlohmann::detail::view::node;
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
#include <limits>
|
||||
#include <random>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include <test_data.hpp>
|
||||
|
||||
#if !(defined(__BYTE_ORDER__) && defined(__ORDER_BIG_ENDIAN__) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
|
||||
|
||||
namespace
|
||||
{
|
||||
std::string exception_of(const std::function<void()>& f)
|
||||
{
|
||||
try
|
||||
{
|
||||
f();
|
||||
}
|
||||
catch (const json::exception& e)
|
||||
{
|
||||
return e.what();
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
const char* const check_failed = "[json.exception.parse_error.116] parse error: invalid json_document image: the check failed";
|
||||
|
||||
std::string read_file(const std::string& name)
|
||||
{
|
||||
std::ifstream f(std::string(TEST_DATA_DIRECTORY) + name, std::ios::binary);
|
||||
std::stringstream ss;
|
||||
ss << f.rdbuf();
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
// the offsets of the parts of an image
|
||||
constexpr std::size_t header_size = 64;
|
||||
|
||||
std::uint64_t header_field(const std::vector<std::uint8_t>& image, std::size_t offset)
|
||||
{
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, image.data() + offset, sizeof(v));
|
||||
return v;
|
||||
}
|
||||
|
||||
void set_header_field(std::vector<std::uint8_t>& image, std::size_t offset, std::uint64_t v)
|
||||
{
|
||||
std::memcpy(image.data() + offset, &v, sizeof(v));
|
||||
}
|
||||
|
||||
std::size_t node_count(const std::vector<std::uint8_t>& image)
|
||||
{
|
||||
return static_cast<std::size_t>(header_field(image, 8));
|
||||
}
|
||||
|
||||
std::size_t text_at(const std::vector<std::uint8_t>& image)
|
||||
{
|
||||
return header_size + (node_count(image) * sizeof(node));
|
||||
}
|
||||
|
||||
node node_at(const std::vector<std::uint8_t>& image, std::size_t i)
|
||||
{
|
||||
node n{};
|
||||
std::memcpy(&n, image.data() + header_size + (i * sizeof(node)), sizeof(node));
|
||||
return n;
|
||||
}
|
||||
|
||||
void set_node(std::vector<std::uint8_t>& image, std::size_t i, const node& n)
|
||||
{
|
||||
std::memcpy(image.data() + header_size + (i * sizeof(node)), &n, sizeof(node));
|
||||
}
|
||||
|
||||
/// the result of loading an image with a check: "" or the exception message
|
||||
std::string load_result(const std::vector<std::uint8_t>& image, image_check check)
|
||||
{
|
||||
return exception_of([&]
|
||||
{
|
||||
const json_document d = json_document::load(image, check);
|
||||
static_cast<void>(d);
|
||||
});
|
||||
}
|
||||
|
||||
/// a copy of the image with node i changed by f
|
||||
template<typename F>
|
||||
std::vector<std::uint8_t> corrupted(const std::vector<std::uint8_t>& image, std::size_t i, F f)
|
||||
{
|
||||
std::vector<std::uint8_t> b = image;
|
||||
node n = node_at(b, i);
|
||||
f(n);
|
||||
set_node(b, i, n);
|
||||
return b;
|
||||
}
|
||||
|
||||
/// a document and the documents loaded from its image must be equal
|
||||
template<typename Document>
|
||||
void check_round_trip(const Document& d)
|
||||
{
|
||||
const std::vector<std::uint8_t> image = d.save();
|
||||
for (const image_check check :
|
||||
{
|
||||
image_check::full, image_check::bounds, image_check::none
|
||||
})
|
||||
{
|
||||
const json_document l = json_document::load(image, check);
|
||||
CHECK(l.root().dump() == d.root().dump());
|
||||
CHECK(l.root().dump(2) == d.root().dump(2));
|
||||
CHECK(l.root().materialize() == json(d.root().materialize()));
|
||||
// an image of a loaded document is the same image
|
||||
CHECK(l.save() == image);
|
||||
}
|
||||
// an editable document can be loaded, too
|
||||
const ordered_json_editable_document e = ordered_json_editable_document::load(image);
|
||||
CHECK(e.root().dump() == d.root().dump());
|
||||
}
|
||||
|
||||
std::uint32_t rng()
|
||||
{
|
||||
static std::mt19937 generator(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed): reproducible
|
||||
return generator();
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("json_view images: round trips")
|
||||
{
|
||||
SECTION("small documents")
|
||||
{
|
||||
for (const char* text :
|
||||
{
|
||||
"null", "true", "false", "0", "-0", "42", "-42", "18446744073709551615", "-9223372036854775808",
|
||||
"123456789012345678901234567890", "1.5", "-1.25e-300", "1E308", "0.1000000000000000000000000001",
|
||||
"\"\"", "\"text\"", R"("esc\"aped\n\u00e9\ud83d\ude00")", "\"\xc3\xa9\xe3\x81\x82\"",
|
||||
"[]", "{}", "[[]]", "[{}]", "{\"\":{}}",
|
||||
R"({"a": [1, 2.5, "x\ty", true, null, {"b": []}], "c": {"d": -3, "eA": "f"}})",
|
||||
R"({"k": 1, "k": 2, "l": [], "k": 3})",
|
||||
" [1 , 2 ] "
|
||||
})
|
||||
{
|
||||
CAPTURE(text);
|
||||
check_round_trip(json_document::parse(text));
|
||||
check_round_trip(ordered_json_document::parse(text));
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("files")
|
||||
{
|
||||
for (const char* name :
|
||||
{
|
||||
"/json_testsuite/sample.json", "/nativejson-benchmark/canada.json", "/nativejson-benchmark/citm_catalog.json",
|
||||
"/nativejson-benchmark/twitter.json", "/json_tests/pass1.json", "/json_tests/pass2.json", "/json_tests/pass3.json"
|
||||
})
|
||||
{
|
||||
CAPTURE(name);
|
||||
const std::string text = read_file(name);
|
||||
const json_document d = json_document::parse(text);
|
||||
check_round_trip(d);
|
||||
// what a loaded document reads is what parse() produces
|
||||
CHECK(json_document::load(d.save()).root().materialize() == json::parse(text));
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("images are deterministic")
|
||||
{
|
||||
const std::string text = R"({"b": [1, 2, {"c": "\u00e9"}], "a": 1.5})";
|
||||
const json_document d = json_document::parse(text);
|
||||
CHECK(d.save() == json_document::parse(text).save());
|
||||
CHECK(d.save() == json_editable_document::parse(text).save());
|
||||
CHECK(d.save() == ordered_json_document::parse(text).save());
|
||||
const json_document copy = json_document::parse_copy(text);
|
||||
CHECK(copy.save() == d.save());
|
||||
}
|
||||
|
||||
SECTION("large objects get their hash index again")
|
||||
{
|
||||
std::string text = "{";
|
||||
for (int i = 0; i < 1000; ++i)
|
||||
{
|
||||
text += (i != 0 ? ",\"k" : "\"k") + std::to_string(i) + "\":" + std::to_string(i);
|
||||
}
|
||||
text += R"(,"k7":"a duplicate","inner":{)";
|
||||
for (int i = 0; i < 200; ++i)
|
||||
{
|
||||
text += (i != 0 ? ",\"m" : "\"m") + std::to_string(i) + "\":" + std::to_string(-i);
|
||||
}
|
||||
text += "}}";
|
||||
const json_document d = json_document::parse(text);
|
||||
const std::vector<std::uint8_t> image = d.save();
|
||||
for (const image_check check :
|
||||
{
|
||||
image_check::full, image_check::none
|
||||
})
|
||||
{
|
||||
const json_document l = json_document::load(image, check);
|
||||
for (int i = 0; i < 1000; ++i)
|
||||
{
|
||||
CHECK(l.root()["k" + std::to_string(i)] == d.root()["k" + std::to_string(i)]);
|
||||
}
|
||||
CHECK(l.root()["k7"].get<int>() == 7); // the first of duplicate keys
|
||||
CHECK(l.root()["inner"]["m199"].get<int>() == -199);
|
||||
CHECK(!l.root().contains("k1000"));
|
||||
// the index is not part of the image
|
||||
CHECK(l.save() == image);
|
||||
}
|
||||
// the nodes of objects in the image do not carry the number of an index
|
||||
CHECK(node_at(image, 0).extra == 0);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view images: edited documents")
|
||||
{
|
||||
const std::string text = R"({"name": "x", "n": 1, "f": 2.5, "list": [1, 2, 3], "obj": {"a": "\u00e9", "b": [true]}, "s": "a\"b"})";
|
||||
|
||||
SECTION("every kind of edit")
|
||||
{
|
||||
ordered_json_editable_document d = ordered_json_editable_document::parse(text);
|
||||
d.set(d.root()["name"], "a new \"name\""); // string in the edit arena
|
||||
d.set(d.root()["n"], -17); // negative integer
|
||||
d.set(d.root(), "p", 5); // non-negative number_integer
|
||||
d.set(d.root(), "u", 18446744073709551615u); // unsigned
|
||||
d.set(d.root()["f"], 0.1); // float token
|
||||
d.set(d.root(), "nan", std::numeric_limits<double>::quiet_NaN());
|
||||
d.set(d.root(), "inf", -std::numeric_limits<double>::infinity());
|
||||
d.push_back(d.root()["list"], "pushed"); // moved array
|
||||
d.insert(d.root()["list"], 0, ordered_json::object({{"new", {1, 2}}}));
|
||||
d.erase(d.root()["list"], 2);
|
||||
d.erase(d.root(), "s");
|
||||
d.set(d.root()["obj"], "c", ordered_json::array({1, "two", 3.5, nullptr, false})); // new object member with a new array
|
||||
d.set(d.root(), "copy", d.root()["obj"]); // a copy of a subtree
|
||||
d.set(d.root(), "key \xc3\xa9", true); // a key in the edit arena
|
||||
|
||||
const std::vector<std::uint8_t> image = d.save();
|
||||
const ordered_json expected = ordered_json::parse(d.root().dump());
|
||||
for (const image_check check :
|
||||
{
|
||||
image_check::full, image_check::bounds, image_check::none
|
||||
})
|
||||
{
|
||||
const ordered_json_document l = ordered_json_document::load(image, check);
|
||||
CHECK(l.root().dump() == d.root().dump());
|
||||
CHECK(l.root().materialize() == expected);
|
||||
CHECK(l.root()["nan"].is_null());
|
||||
CHECK(l.root()["inf"].is_null());
|
||||
CHECK(l.root()["p"].is_number_integer());
|
||||
CHECK(l.root()["p"].get<int>() == 5);
|
||||
CHECK(l.root()["f"].get<double>() == 0.1);
|
||||
CHECK(l.root()["u"].get<std::uint64_t>() == 18446744073709551615u);
|
||||
}
|
||||
// the node index is in document order again: an image of the loaded
|
||||
// document is the same image
|
||||
CHECK(ordered_json_document::load(image).save() == image);
|
||||
// number tokens of edits follow the source; the text is the source's
|
||||
// prefix
|
||||
const ordered_json_document l = ordered_json_document::load(image);
|
||||
REQUIRE(l.source().size() > text.size());
|
||||
CHECK(std::string(l.source().data(), text.size()) == text);
|
||||
}
|
||||
|
||||
SECTION("a loaded document can be edited and saved again")
|
||||
{
|
||||
const std::vector<std::uint8_t> first = json_editable_document::parse(text).save();
|
||||
json_editable_document d = json_editable_document::load(first);
|
||||
d.set(d.root()["obj"]["a"], "changed");
|
||||
d.push_back(d.root()["list"], 4);
|
||||
d.set(d.root(), "z", json::array({json::object()}));
|
||||
const std::vector<std::uint8_t> second = d.save();
|
||||
const json_document l = json_document::load(second);
|
||||
CHECK(l.root().dump() == d.root().dump());
|
||||
CHECK(l.root()["obj"]["a"] == "changed");
|
||||
CHECK(l.root()["list"].size() == 4);
|
||||
}
|
||||
|
||||
SECTION("the root replaced")
|
||||
{
|
||||
json_editable_document d = json_editable_document::parse(text);
|
||||
d.set(d.root(), json::array({1, "x"}));
|
||||
check_round_trip(d);
|
||||
d.set(d.root(), 3.5);
|
||||
check_round_trip(d);
|
||||
d.set(d.root(), "text");
|
||||
check_round_trip(d);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view images: ownership")
|
||||
{
|
||||
const std::string text = R"({"a": "esc\u00e9aped", "b": [1, 2]})";
|
||||
const std::vector<std::uint8_t> image = json_document::parse(text).save();
|
||||
|
||||
SECTION("borrowed")
|
||||
{
|
||||
const json_document d = json_document::load(image);
|
||||
CHECK(!d.owns_source());
|
||||
CHECK(d.root()["a"] == "esc\xc3\xa9" "aped");
|
||||
const json_document p = json_document::load(image.data(), image.size());
|
||||
CHECK(!p.owns_source());
|
||||
CHECK(p.root() == d.root());
|
||||
// the text is the image's
|
||||
CHECK(d.source().data() == reinterpret_cast<const char*>(image.data() + text_at(image)));
|
||||
}
|
||||
|
||||
SECTION("owned")
|
||||
{
|
||||
std::vector<std::uint8_t> copy = image;
|
||||
const std::uint8_t* const data = copy.data();
|
||||
json_document d = json_document::load(std::move(copy));
|
||||
CHECK(d.owns_source());
|
||||
CHECK(d.source().data() == reinterpret_cast<const char*>(data + text_at(image)));
|
||||
CHECK(d.memory_usage() >= image.size());
|
||||
CHECK(d.root()["b"][1] == 2);
|
||||
// read() replaces the image
|
||||
d.read(std::string("[1]"));
|
||||
CHECK(d.owns_source());
|
||||
CHECK(d.root().dump() == "[1]");
|
||||
const std::string borrowed = "[2]";
|
||||
d.read(borrowed);
|
||||
CHECK(!d.owns_source());
|
||||
}
|
||||
|
||||
SECTION("shrink_to_fit keeps the decoded strings of the image")
|
||||
{
|
||||
json_document d = json_document::parse(R"(["\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9"])");
|
||||
d.shrink_to_fit();
|
||||
const std::vector<std::uint8_t> img = d.save();
|
||||
json_document l = json_document::load(img);
|
||||
l.shrink_to_fit();
|
||||
CHECK(l.root().dump() == d.root().dump());
|
||||
CHECK(l.save() == img);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view images: errors")
|
||||
{
|
||||
SECTION("a literal as the root: dump() after loading")
|
||||
{
|
||||
for (const char* text :
|
||||
{
|
||||
"null", "true", "false"
|
||||
})
|
||||
{
|
||||
std::vector<std::uint8_t> image = json_document::parse(text).save();
|
||||
node n = node_at(image, 0);
|
||||
n.off = static_cast<std::uint32_t>(image.size());
|
||||
set_node(image, 0, n);
|
||||
CHECK(load_result(image, image_check::full) == check_failed);
|
||||
CHECK(load_result(image, image_check::bounds) == check_failed);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("saving a discarded document")
|
||||
{
|
||||
const json_document empty;
|
||||
CHECK(exception_of([&] { static_cast<void>(empty.save()); }) == "[json.exception.type_error.320] cannot save a discarded json_document");
|
||||
const json_document failed = json_document::parse("[1,", false);
|
||||
CHECK(exception_of([&] { static_cast<void>(failed.save()); }) == "[json.exception.type_error.320] cannot save a discarded json_document");
|
||||
}
|
||||
|
||||
const std::vector<std::uint8_t> image = json_document::parse(R"({"a": [1, "\u00e9"]})").save();
|
||||
const std::string prefix = "[json.exception.parse_error.116] parse error: invalid json_document image: ";
|
||||
|
||||
SECTION("header and sizes")
|
||||
{
|
||||
CHECK(exception_of([]
|
||||
{
|
||||
const json_document d = json_document::load(nullptr, 0);
|
||||
static_cast<void>(d);
|
||||
}) == prefix + "too short");
|
||||
CHECK(exception_of([&]
|
||||
{
|
||||
const json_document d = json_document::load(image.data(), 63);
|
||||
static_cast<void>(d);
|
||||
}) == prefix + "too short");
|
||||
|
||||
std::vector<std::uint8_t> bad = image;
|
||||
bad[0] = 'X';
|
||||
CHECK(load_result(bad, image_check::full) == prefix + "unknown format");
|
||||
bad = image;
|
||||
bad[4] = 2; // version
|
||||
CHECK(load_result(bad, image_check::full) == prefix + "unknown format");
|
||||
for (std::size_t reserved = 32; reserved < 64; reserved += 8)
|
||||
{
|
||||
bad = image;
|
||||
bad[reserved + 3] = 1;
|
||||
CHECK(load_result(bad, image_check::none) == prefix + "unknown format");
|
||||
}
|
||||
|
||||
const auto sizes = [&](std::size_t offset, std::uint64_t v)
|
||||
{
|
||||
std::vector<std::uint8_t> b = image;
|
||||
set_header_field(b, offset, v);
|
||||
return load_result(b, image_check::none);
|
||||
};
|
||||
CHECK(sizes(8, 0) == prefix + "sizes out of range"); // no nodes
|
||||
CHECK(sizes(8, 1000) == prefix + "sizes out of range"); // more nodes than bytes
|
||||
CHECK(sizes(8, 0xFFFFFFF0u) == prefix + "sizes out of range");
|
||||
CHECK(sizes(16, 0xFFFFFFF0u) == prefix + "sizes out of range"); // text size
|
||||
CHECK(sizes(16, header_field(image, 16) + 1) == prefix + "sizes out of range");
|
||||
CHECK(sizes(16, image.size()) == prefix + "sizes out of range");
|
||||
CHECK(sizes(24, 0xFFFFFFF0u) == prefix + "sizes out of range"); // decoded string size
|
||||
CHECK(sizes(24, header_field(image, 24) - 1) == prefix + "sizes out of range");
|
||||
|
||||
// the NULs after the text and the decoded strings
|
||||
bad = image;
|
||||
bad[text_at(image) + header_field(image, 16)] = 'x';
|
||||
CHECK(load_result(bad, image_check::none) == prefix + "sizes out of range");
|
||||
bad = image;
|
||||
bad.back() = 'x';
|
||||
CHECK(load_result(bad, image_check::none) == prefix + "sizes out of range");
|
||||
// nothing after the image
|
||||
bad = image;
|
||||
bad.push_back(0);
|
||||
CHECK(load_result(bad, image_check::none) == prefix + "sizes out of range");
|
||||
// nodes, but not even room for the NULs
|
||||
bad.assign(image.begin(), image.begin() + static_cast<std::ptrdiff_t>(text_at(image)));
|
||||
CHECK(load_result(bad, image_check::none) == prefix + "sizes out of range");
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view images: check")
|
||||
{
|
||||
// nodes: 0 { 1 "s" 2 "x\"y" (escaped) 3 "i" 4 -12 5 "u" 6 7 7 "f" 8 1.5e300 9 "b" 10 true 11 "n" 12 null
|
||||
// 13 "a" 14 [ 15 "t" 16 {} ]
|
||||
const std::string text = R"({"s":"x\"y","i":-12,"u":7,"f":1.5e300,"b":true,"n":null,"a":["t",{}]})";
|
||||
const std::vector<std::uint8_t> image = json_document::parse(text).save();
|
||||
REQUIRE(load_result(image, image_check::full).empty());
|
||||
REQUIRE(node_count(image) == 17);
|
||||
|
||||
// bounds: rejected by both checks; content: only by the full one
|
||||
const auto rejected = [&](const std::vector<std::uint8_t>& b, bool bounds)
|
||||
{
|
||||
CHECK(load_result(b, image_check::full) == check_failed);
|
||||
CHECK(load_result(b, image_check::bounds) == (bounds ? check_failed : ""));
|
||||
};
|
||||
|
||||
SECTION("kinds")
|
||||
{
|
||||
const std::array<std::uint8_t, 4> kinds = {{8, 9, 10, 200}}; // binary, discarded, link, unknown
|
||||
for (const std::uint8_t kind : kinds)
|
||||
{
|
||||
rejected(corrupted(image, 12, [&](node & n)
|
||||
{
|
||||
n.kind = kind;
|
||||
}), true);
|
||||
}
|
||||
// a key that is not a string
|
||||
rejected(corrupted(image, 1, [](node & n)
|
||||
{
|
||||
n.kind = 0;
|
||||
n.len = 0;
|
||||
n.off = 0;
|
||||
}), true);
|
||||
}
|
||||
|
||||
SECTION("flags and extra")
|
||||
{
|
||||
rejected(corrupted(image, 12, [](node & n)
|
||||
{
|
||||
n.flags = 4;
|
||||
}), true);
|
||||
rejected(corrupted(image, 12, [](node & n)
|
||||
{
|
||||
n.extra = 1;
|
||||
}), true);
|
||||
rejected(corrupted(image, 10, [](node & n)
|
||||
{
|
||||
n.flags = 5;
|
||||
}), true);
|
||||
rejected(corrupted(image, 10, [](node & n)
|
||||
{
|
||||
n.extra = 1;
|
||||
}), true);
|
||||
rejected(corrupted(image, 1, [](node & n)
|
||||
{
|
||||
n.flags = 2; // a string in the edit arena
|
||||
}), true);
|
||||
rejected(corrupted(image, 1, [](node & n)
|
||||
{
|
||||
n.extra = 3;
|
||||
}), true);
|
||||
rejected(corrupted(image, 4, [](node & n)
|
||||
{
|
||||
n.flags = 2;
|
||||
}), true);
|
||||
rejected(corrupted(image, 4, [](node & n)
|
||||
{
|
||||
n.extra = static_cast<std::uint16_t>(n.extra | 0x100u); // an integer with fraction digits
|
||||
}), true);
|
||||
rejected(corrupted(image, 0, [](node & n)
|
||||
{
|
||||
n.flags = 8; // moved
|
||||
}), true);
|
||||
rejected(corrupted(image, 0, [](node & n)
|
||||
{
|
||||
n.extra = 1; // a hash index
|
||||
}), true);
|
||||
}
|
||||
|
||||
SECTION("bounds")
|
||||
{
|
||||
const std::size_t text_size = header_field(image, 16);
|
||||
const std::size_t arena_size = header_field(image, 24);
|
||||
rejected(corrupted(image, 1, [&](node & n)
|
||||
{
|
||||
n.off = static_cast<std::uint32_t>(text_size + 1);
|
||||
}), true);
|
||||
rejected(corrupted(image, 1, [&](node & n)
|
||||
{
|
||||
n.len = static_cast<std::uint32_t>(text_size);
|
||||
}), true);
|
||||
rejected(corrupted(image, 2, [&](node & n)
|
||||
{
|
||||
n.len = static_cast<std::uint32_t>(arena_size + 1);
|
||||
}), true);
|
||||
rejected(corrupted(image, 6, [&](node & n)
|
||||
{
|
||||
n.off = static_cast<std::uint32_t>(text_size);
|
||||
}), true);
|
||||
rejected(corrupted(image, 6, [&](node & n)
|
||||
{
|
||||
n.off = static_cast<std::uint32_t>(text_size + 5);
|
||||
}), true);
|
||||
rejected(corrupted(image, 6, [](node & n)
|
||||
{
|
||||
n.extra = 0; // no digits
|
||||
}), true);
|
||||
rejected(corrupted(image, 8, [&](node & n)
|
||||
{
|
||||
n.len = static_cast<std::uint32_t>(text_size);
|
||||
}), true);
|
||||
rejected(corrupted(image, 8, [](node & n)
|
||||
{
|
||||
n.len = 2; // shorter than the recorded digits
|
||||
}), true);
|
||||
rejected(corrupted(image, 14, [&](node & n)
|
||||
{
|
||||
n.off = static_cast<std::uint32_t>(text_size + 1);
|
||||
}), true);
|
||||
// literals: their offset sizes the output of dump()
|
||||
rejected(corrupted(image, 10, [&](node & n)
|
||||
{
|
||||
n.off = static_cast<std::uint32_t>(text_size + 1);
|
||||
}), true);
|
||||
rejected(corrupted(image, 12, [&](node & n)
|
||||
{
|
||||
n.off = 0xFFFFFFFFu;
|
||||
}), true);
|
||||
}
|
||||
|
||||
SECTION("structure")
|
||||
{
|
||||
rejected(corrupted(image, 0, [](node & n)
|
||||
{
|
||||
n.next = 0;
|
||||
}), true);
|
||||
rejected(corrupted(image, 0, [](node & n)
|
||||
{
|
||||
n.next = 18; // beyond the image
|
||||
}), true);
|
||||
rejected(corrupted(image, 14, [](node & n)
|
||||
{
|
||||
n.next = 4; // beyond the enclosing object
|
||||
}), true);
|
||||
rejected(corrupted(image, 0, [](node & n)
|
||||
{
|
||||
n.len = 6; // member count
|
||||
}), true);
|
||||
rejected(corrupted(image, 14, [](node & n)
|
||||
{
|
||||
n.len = 3; // element count
|
||||
}), true);
|
||||
rejected(corrupted(image, 0, [](node & n)
|
||||
{
|
||||
n.next = 14; // the object ends after the key "a"
|
||||
n.len = 7;
|
||||
}), true);
|
||||
rejected(corrupted(image, 0, [](node & n)
|
||||
{
|
||||
n.next = 13; // nodes after the root
|
||||
n.len = 6;
|
||||
}), true);
|
||||
rejected(corrupted(image, 0, [](node & n)
|
||||
{
|
||||
n.kind = 2; // an array: the "keys" are values, and the counts do not match
|
||||
}), true);
|
||||
const std::vector<std::uint8_t> as_array = corrupted(image, 16, [](node & n)
|
||||
{
|
||||
n.kind = 2; // {} as []: fine
|
||||
});
|
||||
CHECK(load_result(as_array, image_check::full).empty());
|
||||
CHECK(json_document::load(as_array).root().dump() == R"({"s":"x\"y","i":-12,"u":7,"f":1.5e+300,"b":true,"n":null,"a":["t",[]]})");
|
||||
}
|
||||
|
||||
SECTION("strings")
|
||||
{
|
||||
// a quote in a source string (the full check only)
|
||||
std::vector<std::uint8_t> b = image;
|
||||
const std::size_t t = text_at(image);
|
||||
const node t15 = node_at(image, 15);
|
||||
b[t + t15.off] = '"';
|
||||
rejected(b, false);
|
||||
// a control character
|
||||
b[t + t15.off] = '\n';
|
||||
rejected(b, false);
|
||||
// invalid UTF-8 in a decoded string
|
||||
b = image;
|
||||
const node s2 = node_at(image, 2);
|
||||
b[t + header_field(image, 16) + 1 + s2.off] = 0xFF;
|
||||
rejected(b, false);
|
||||
}
|
||||
|
||||
SECTION("numbers")
|
||||
{
|
||||
const std::size_t t = text_at(image);
|
||||
const node i4 = node_at(image, 4);
|
||||
const node u6 = node_at(image, 6);
|
||||
const node f8 = node_at(image, 8);
|
||||
const auto at_token = [&](const node & n, std::size_t k, std::uint8_t c)
|
||||
{
|
||||
std::vector<std::uint8_t> b = image;
|
||||
b[t + n.off + k] = c;
|
||||
return b;
|
||||
};
|
||||
rejected(at_token(i4, 1, 'x'), false); // -x2
|
||||
rejected(at_token(i4, 1, '0'), false); // -02
|
||||
rejected(at_token(i4, 0, '1'), false); // 112 != -12
|
||||
rejected(at_token(f8, 1, 'x'), false); // 1x5e300
|
||||
rejected(at_token(f8, 2, 'e'), false); // 1.ee300
|
||||
rejected(at_token(f8, 4, 'x'), false); // 1.5ex00
|
||||
rejected(at_token(f8, 3, '0'), false); // 1.50300: another layout
|
||||
rejected(at_token(f8, 4, '9'), false); // 1.5e900: overflow
|
||||
rejected(at_token(f8, 0, 'x'), false);
|
||||
rejected(at_token(u6, 0, '8'), false); // 8 != 7
|
||||
rejected(corrupted(image, 4, [](node & n)
|
||||
{
|
||||
n.kind = 6; // "-12" as unsigned: a sign
|
||||
n.extra = 3;
|
||||
}), false);
|
||||
// a non-negative number_integer (as edits write it): fine
|
||||
const std::vector<std::uint8_t> positive = corrupted(image, 6, [](node & n)
|
||||
{
|
||||
n.kind = 5;
|
||||
n.extra = 0;
|
||||
});
|
||||
CHECK(load_result(positive, image_check::full).empty());
|
||||
CHECK(json_document::load(positive).root()["u"].is_number_integer());
|
||||
rejected(corrupted(image, 8, [](node & n)
|
||||
{
|
||||
n.kind = 6; // a float token as integer
|
||||
n.extra = 7;
|
||||
}), false);
|
||||
}
|
||||
|
||||
SECTION("float tokens of an image checked for bounds only")
|
||||
{
|
||||
// A float node whose layout records "many" digits is converted from
|
||||
// its token alone; a token that is not a JSON number reads as 0.
|
||||
const std::vector<std::uint8_t> img = json_document::parse("[1.5e300,2]").save();
|
||||
const std::size_t t = text_at(img);
|
||||
for (const char* token :
|
||||
{
|
||||
"x.5e300", "01.5e30", "1.xe300", "1.5ex00", "1.5e+x0", "1.5e30x", "-.5e300", "1.5E300"
|
||||
})
|
||||
{
|
||||
CAPTURE(token);
|
||||
std::vector<std::uint8_t> b = img;
|
||||
node n = node_at(b, 1);
|
||||
n.extra = 0xFFFFu;
|
||||
set_node(b, 1, n);
|
||||
std::memcpy(b.data() + t + n.off, token, n.len);
|
||||
const json_document d = json_document::load(b, image_check::bounds);
|
||||
const auto v = d.root()[0].get<double>();
|
||||
CHECK(v == (std::string(token) == "1.5E300" ? 1.5e300 : 0.0));
|
||||
CHECK(load_result(b, image_check::full) == (std::string(token) == "1.5E300" ? "" : check_failed));
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("integer ranges")
|
||||
{
|
||||
// tokens of many digits, which the parser stores as floats
|
||||
const std::string big = R"([123456789012345678901234, 99999999999999999999, 9223372036854775808])";
|
||||
const std::vector<std::uint8_t> img = json_document::parse(big).save();
|
||||
const auto as_integer = [&](std::size_t i, std::uint8_t kind, std::uint16_t extra)
|
||||
{
|
||||
std::vector<std::uint8_t> b = img;
|
||||
node n = node_at(b, i);
|
||||
n.kind = kind;
|
||||
n.extra = extra;
|
||||
set_node(b, i, n);
|
||||
return load_result(b, image_check::full);
|
||||
};
|
||||
CHECK(as_integer(1, 6, 24) == check_failed); // more than 20 digits
|
||||
CHECK(as_integer(2, 6, 20) == check_failed); // more than 2^64 - 1
|
||||
CHECK(as_integer(3, 5, 18) == check_failed); // more than 2^63 - 1 as number_integer
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view images: damaged images")
|
||||
{
|
||||
// A damaged image must be rejected, or read safely; with the full check,
|
||||
// it also serializes to the JSON it reads as.
|
||||
const std::vector<std::string> texts =
|
||||
{
|
||||
R"({"a": [1, -2, 3.25, "x\u00e9y", true, null], "b": {"c": "\"q\"", "d": 1e10}, "e": ""})",
|
||||
R"([[[[]]], {"k": {"k": {"k": 12345678901234567890}}}, "\ud83d\ude00", -0.0, 0])",
|
||||
};
|
||||
for (const std::string& text : texts)
|
||||
{
|
||||
const std::vector<std::uint8_t> image = json_document::parse(text).save();
|
||||
for (int round = 0; round < 3000; ++round)
|
||||
{
|
||||
std::vector<std::uint8_t> b = image;
|
||||
const std::uint32_t flips = 1 + (rng() % 3);
|
||||
for (std::uint32_t k = 0; k < flips; ++k)
|
||||
{
|
||||
// mostly the nodes, where the damage matters most
|
||||
const std::size_t at = rng() % 4 != 0 ? header_size + (rng() % (b.size() - header_size)) : rng() % b.size();
|
||||
b[at] = static_cast<std::uint8_t>(rng() % 3 == 0 ? rng() : b[at] ^ (1u << (rng() % 8)));
|
||||
}
|
||||
for (const image_check check :
|
||||
{
|
||||
image_check::full, image_check::bounds
|
||||
})
|
||||
{
|
||||
json_document d;
|
||||
try
|
||||
{
|
||||
d = json_document::load(b, check);
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
CHECK(e.id == 116);
|
||||
continue;
|
||||
}
|
||||
std::string dumped;
|
||||
std::string dumped_ascii;
|
||||
try
|
||||
{
|
||||
dumped = d.root().dump();
|
||||
dumped_ascii = d.root().dump(-1, ' ', true);
|
||||
}
|
||||
catch (const json::type_error& e)
|
||||
{
|
||||
// invalid UTF-8 (the bounds check only)
|
||||
CHECK(check == image_check::bounds);
|
||||
CHECK(e.id == 316);
|
||||
continue;
|
||||
}
|
||||
const json j = d.root().materialize();
|
||||
if (check == image_check::full)
|
||||
{
|
||||
CHECK(json::parse(dumped) == j);
|
||||
CHECK(json::parse(dumped_ascii) == j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
TEST_CASE("json_view images: big-endian targets")
|
||||
{
|
||||
const json_document d = json_document::parse("[1]");
|
||||
CHECK_THROWS_WITH_AS(d.save(), "[json.exception.type_error.320] json_document images need a little-endian target", json::type_error&);
|
||||
}
|
||||
|
||||
#endif
|
||||
+255
-1
@@ -15,6 +15,20 @@
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::detail::dtoa_impl::reinterpret_bits;
|
||||
|
||||
#include <array>
|
||||
#include <cmath>
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <limits>
|
||||
#include <random>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
#if defined(JSON_HAS_CPP_17)
|
||||
#include <charconv>
|
||||
#endif
|
||||
|
||||
namespace
|
||||
{
|
||||
float make_float(uint32_t sign_bit, uint32_t biased_exponent, uint32_t significand)
|
||||
@@ -450,7 +464,7 @@ TEST_CASE("formatting")
|
||||
check_double( 1.2345e+18, "1.2345e+18" ); // 1.2345e+18 1.2345e+18 1.2345e18
|
||||
check_double( 1.2345e+19, "1.2345e+19" ); // 1.2345e+19 1.2345e+19 1.2345e19
|
||||
check_double( 1.2345e+20, "1.2345e+20" ); // 1.2345e+20 1.2345e+20 1.2345e20
|
||||
check_double( 1.2345e+21, "1.2344999999999999e+21" ); // 1.2345e+21 1.2344999999999999e+21 1.2345e21
|
||||
check_double( 1.2345e+21, "1.2345e+21" ); // 1.2345e+21 1.2344999999999999e+21 1.2345e21
|
||||
check_double( 1.2345e+22, "1.2345e+22" ); // 1.2345e+22 1.2345e+22 1.2345e22
|
||||
}
|
||||
|
||||
@@ -514,3 +528,243 @@ TEST_CASE("formatting")
|
||||
check_integer(1000000000000000000LL, "1000000000000000000");
|
||||
}
|
||||
}
|
||||
|
||||
namespace
|
||||
{
|
||||
// a small unsigned big integer (32-bit limbs, least significant first), to
|
||||
// recompute the powers of ten of the shortest double conversion
|
||||
using big = std::vector<std::uint32_t>;
|
||||
|
||||
void big_mul_small(big& x, std::uint32_t m)
|
||||
{
|
||||
std::uint64_t carry = 0;
|
||||
for (auto& limb : x)
|
||||
{
|
||||
const std::uint64_t v = (static_cast<std::uint64_t>(limb) * m) + carry;
|
||||
limb = static_cast<std::uint32_t>(v);
|
||||
carry = v >> 32u;
|
||||
}
|
||||
if (carry != 0)
|
||||
{
|
||||
x.push_back(static_cast<std::uint32_t>(carry));
|
||||
}
|
||||
}
|
||||
|
||||
void big_div_small(big& x, std::uint32_t d)
|
||||
{
|
||||
std::uint64_t rest = 0;
|
||||
for (std::size_t i = x.size(); i-- > 0;)
|
||||
{
|
||||
const std::uint64_t v = (rest << 32u) | x[i];
|
||||
x[i] = static_cast<std::uint32_t>(v / d);
|
||||
rest = v % d;
|
||||
}
|
||||
while (!x.empty() && x.back() == 0)
|
||||
{
|
||||
x.pop_back();
|
||||
}
|
||||
}
|
||||
|
||||
std::size_t big_bit_length(const big& x)
|
||||
{
|
||||
std::size_t n = 32 * x.size();
|
||||
for (std::uint32_t top = x.back(); (top & 0x80000000u) == 0; top <<= 1u)
|
||||
{
|
||||
--n;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
bool big_bit(const big& x, std::size_t i)
|
||||
{
|
||||
return ((x[i / 32] >> (i % 32)) & 1u) != 0;
|
||||
}
|
||||
|
||||
/// the 128 most significant bits of x (floor), shifted left if x has fewer bits
|
||||
std::pair<std::uint64_t, std::uint64_t> big_top128(const big& x)
|
||||
{
|
||||
const std::size_t n = big_bit_length(x);
|
||||
std::uint64_t high = 0;
|
||||
std::uint64_t low = 0;
|
||||
for (std::size_t k = 0; k < 128; ++k)
|
||||
{
|
||||
const bool bit = k < n && big_bit(x, n - 1 - k);
|
||||
if (k < 64)
|
||||
{
|
||||
high = (high << 1u) | (bit ? 1u : 0u);
|
||||
}
|
||||
else
|
||||
{
|
||||
low = (low << 1u) | (bit ? 1u : 0u);
|
||||
}
|
||||
}
|
||||
return {high, low};
|
||||
}
|
||||
|
||||
/// the digits (without trailing zeros) and the decimal exponent of a
|
||||
/// representation "[-]d[.ddd][e[+-]x]"
|
||||
std::pair<std::string, int> digits_and_exponent(const std::string& s)
|
||||
{
|
||||
std::string digits;
|
||||
int point = -1;
|
||||
int exponent = 0;
|
||||
for (std::size_t i = 0; i < s.size(); ++i)
|
||||
{
|
||||
const char c = s[i];
|
||||
if (c >= '0' && c <= '9')
|
||||
{
|
||||
digits += c;
|
||||
}
|
||||
else if (c == '.')
|
||||
{
|
||||
point = static_cast<int>(digits.size());
|
||||
}
|
||||
else if (c == 'e' || c == 'E')
|
||||
{
|
||||
exponent = std::stoi(s.substr(i + 1));
|
||||
break;
|
||||
}
|
||||
}
|
||||
int e = exponent + (point < 0 ? static_cast<int>(digits.size()) : point) - static_cast<int>(digits.size());
|
||||
const std::size_t first = digits.find_first_not_of('0');
|
||||
digits = first == std::string::npos ? "0" : digits.substr(first);
|
||||
while (digits.size() > 1 && digits.back() == '0')
|
||||
{
|
||||
digits.pop_back();
|
||||
++e;
|
||||
}
|
||||
return {digits, e};
|
||||
}
|
||||
|
||||
/// whether the decimal digits * 10^e reads back as v
|
||||
bool reads_back(const std::string& digits, int e, double v)
|
||||
{
|
||||
const std::string text = digits + "e" + std::to_string(e);
|
||||
return std::strtod(text.c_str(), nullptr) == v;
|
||||
}
|
||||
|
||||
/// Check the representation of a positive finite double: it reads back as
|
||||
/// the same value, and no representation with fewer digits does.
|
||||
void check_shortest(double v)
|
||||
{
|
||||
std::array<char, 33> buf{};
|
||||
char* end = nlohmann::detail::to_chars(buf.data(), buf.data() + 32, v);
|
||||
const std::string text(buf.data(), end);
|
||||
CAPTURE(text);
|
||||
CHECK(std::strtod(text.c_str(), nullptr) == v);
|
||||
// the layout is that of format_buffer() for the same digits
|
||||
std::array<char, 64> reference{};
|
||||
int len = 0;
|
||||
int exponent = 0;
|
||||
nlohmann::detail::dtoa_impl::shortest_digits(reference.data(), len, exponent, v);
|
||||
const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), len, exponent, -4, 15);
|
||||
CHECK(text == std::string(reference.data(), static_cast<std::size_t>(reference_end - reference.data())));
|
||||
const auto de = digits_and_exponent(text);
|
||||
const std::string& digits = de.first;
|
||||
if (digits.size() > 1)
|
||||
{
|
||||
// the decimals of one digit fewer next to the value
|
||||
std::array<char, 64> shorter{};
|
||||
const int n = std::snprintf(shorter.data(), shorter.size(), "%.*e", static_cast<int>(digits.size()) - 2, v); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
const auto near = digits_and_exponent(std::string(shorter.data(), static_cast<std::size_t>(n)));
|
||||
// as an integer with digits.size() - 1 digits
|
||||
std::string m = near.first;
|
||||
int e = near.second;
|
||||
while (m.size() < digits.size() - 1)
|
||||
{
|
||||
m += '0';
|
||||
--e;
|
||||
}
|
||||
const std::uint64_t mid = std::stoull(m);
|
||||
for (const std::uint64_t candidate :
|
||||
{
|
||||
mid - 1, mid, mid + 1
|
||||
})
|
||||
{
|
||||
CAPTURE(candidate);
|
||||
CHECK(!reads_back(std::to_string(candidate), e, v));
|
||||
}
|
||||
}
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars)
|
||||
// the closest of the shortest representations, as std::to_chars finds it
|
||||
std::array<char, 64> std_text{};
|
||||
const auto r = std::to_chars(std_text.data(), std_text.data() + std_text.size(), v, std::chars_format::scientific);
|
||||
CHECK(digits_and_exponent(std::string(std_text.data(), r.ptr)) == de);
|
||||
#endif
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("shortest digits of doubles")
|
||||
{
|
||||
SECTION("powers of ten")
|
||||
{
|
||||
// the 128-bit significands of 10^k, rounded down, recomputed
|
||||
for (int k = -342; k <= 341; ++k)
|
||||
{
|
||||
CAPTURE(k);
|
||||
big x{1};
|
||||
if (k >= 0)
|
||||
{
|
||||
for (int i = 0; i < k; ++i)
|
||||
{
|
||||
big_mul_small(x, 10);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// floor(2^b / 10^-k) for a b that leaves more than 128 bits
|
||||
const int b = 128 + 64 + (4 * -k);
|
||||
x.assign(static_cast<std::size_t>(b / 32) + 1, 0);
|
||||
x.back() = 1u << (b % 32);
|
||||
for (int i = 0; i < -k; ++i)
|
||||
{
|
||||
big_div_small(x, 10);
|
||||
}
|
||||
}
|
||||
const auto expected = big_top128(x);
|
||||
const auto actual = nlohmann::detail::zmij::pow10(k);
|
||||
CHECK(actual.high == expected.first);
|
||||
CHECK(actual.low == expected.second);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("boundary values")
|
||||
{
|
||||
for (const double v :
|
||||
{
|
||||
std::numeric_limits<double>::min(), std::numeric_limits<double>::max(), std::numeric_limits<double>::denorm_min(),
|
||||
std::nextafter(std::numeric_limits<double>::min(), 0.0), 1.0, 2.0, 0.1, 0.3, 1e21, 1e22, 1e23, 5e-324, 9007199254740993.0,
|
||||
1.2345e+21, 2.2250738585072014e-308, 1.7976931348623157e308, 4.9406564584124654e-324, 123456789012345680.0
|
||||
})
|
||||
{
|
||||
check_shortest(v);
|
||||
}
|
||||
// all powers of two (their rounding interval is narrower below)
|
||||
for (int e = -1074; e <= 1023; ++e)
|
||||
{
|
||||
check_shortest(std::ldexp(1.0, e));
|
||||
}
|
||||
// powers of ten and their neighbors
|
||||
for (int e = -323; e <= 308; ++e)
|
||||
{
|
||||
const double p = std::strtod(("1e" + std::to_string(e)).c_str(), nullptr);
|
||||
check_shortest(p);
|
||||
check_shortest(std::nextafter(p, 0.0));
|
||||
check_shortest(std::nextafter(p, std::numeric_limits<double>::infinity()));
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("random doubles")
|
||||
{
|
||||
std::mt19937_64 rng(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed): reproducible
|
||||
for (int i = 0; i < 100000; ++i)
|
||||
{
|
||||
const std::uint64_t bits = rng() & 0x7FFFFFFFFFFFFFFFu;
|
||||
const auto v = reinterpret_bits<double>(bits);
|
||||
if (std::isfinite(v) && v != 0)
|
||||
{
|
||||
check_shortest(v);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user