mirror of
https://github.com/nlohmann/json.git
synced 2026-07-21 01:44:54 +00:00
Compare commits
@@ -38,14 +38,14 @@ jobs:
|
||||
|
||||
# Initializes the CodeQL tools for scanning.
|
||||
- name: Initialize CodeQL
|
||||
uses: github/codeql-action/init@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
uses: github/codeql-action/init@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
with:
|
||||
languages: c-cpp
|
||||
|
||||
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
||||
# If this step fails, then you should remove it and run the build manually (see below)
|
||||
- name: Autobuild
|
||||
uses: github/codeql-action/autobuild@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
uses: github/codeql-action/autobuild@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
|
||||
- name: Perform CodeQL Analysis
|
||||
uses: github/codeql-action/analyze@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
uses: github/codeql-action/analyze@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
|
||||
@@ -43,6 +43,6 @@ jobs:
|
||||
output: 'flawfinder_results.sarif'
|
||||
|
||||
- name: Upload analysis results to GitHub Security tab
|
||||
uses: github/codeql-action/upload-sarif@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
uses: github/codeql-action/upload-sarif@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
with:
|
||||
sarif_file: ${{github.workspace}}/flawfinder_results.sarif
|
||||
|
||||
@@ -76,6 +76,6 @@ jobs:
|
||||
|
||||
# Upload the results to GitHub's code scanning dashboard.
|
||||
- name: "Upload to code-scanning"
|
||||
uses: github/codeql-action/upload-sarif@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
uses: github/codeql-action/upload-sarif@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
with:
|
||||
sarif_file: results.sarif
|
||||
|
||||
@@ -61,7 +61,7 @@ jobs:
|
||||
|
||||
# Upload SARIF file generated in previous step
|
||||
- name: Upload SARIF file
|
||||
uses: github/codeql-action/upload-sarif@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
uses: github/codeql-action/upload-sarif@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
with:
|
||||
sarif_file: semgrep.sarif
|
||||
if: always()
|
||||
|
||||
@@ -20,7 +20,7 @@ jobs:
|
||||
with:
|
||||
egress-policy: audit
|
||||
|
||||
- uses: actions/stale@eb5cf3af3ac0a1aa4c9c45633dd1ae542a27a899 # v10.3.0
|
||||
- uses: actions/stale@1e223db275d687790206a7acac4d1a11bd6fe629 # v10.4.0
|
||||
with:
|
||||
stale-issue-label: 'state: stale'
|
||||
stale-pr-label: 'state: stale'
|
||||
|
||||
@@ -8,8 +8,8 @@ static bool accept(InputType&& i,
|
||||
const bool ignore_trailing_commas = false);
|
||||
|
||||
// (2)
|
||||
template<typename IteratorType>
|
||||
static bool accept(IteratorType first, IteratorType last,
|
||||
template<typename IteratorType, typename SentinelType = IteratorType>
|
||||
static bool accept(IteratorType first, SentinelType last,
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas = false);
|
||||
```
|
||||
@@ -17,10 +17,11 @@ static bool accept(IteratorType first, IteratorType last,
|
||||
Checks whether the input is valid JSON.
|
||||
|
||||
1. Reads from a compatible input.
|
||||
2. Reads from a pair of character iterators
|
||||
2. Reads from a pair of character iterators, or an iterator and a sentinel of a different type (C++20 ranges support)
|
||||
|
||||
The value_type of the iterator must be an integral type with a size of 1, 2, or 4 bytes, which will be interpreted
|
||||
respectively as UTF-8, UTF-16, and UTF-32.
|
||||
respectively as UTF-8, UTF-16, and UTF-32. If `SentinelType` differs from `IteratorType`, it must be comparable to
|
||||
the iterator type with `operator!=`.
|
||||
|
||||
Unlike the [`parse()`](parse.md) function, this function neither throws an exception in case of invalid JSON input
|
||||
(i.e., a parse error) nor creates diagnostic information.
|
||||
@@ -44,6 +45,12 @@ Unlike the [`parse()`](parse.md) function, this function neither throws an excep
|
||||
- a pair of `std::string::iterator` or `std::vector<std::uint8_t>::iterator`
|
||||
- a pair of pointers such as `ptr` and `ptr + len`
|
||||
|
||||
`SentinelType`
|
||||
: defaults to `IteratorType`; may be a different type comparable to `IteratorType` via `operator!=`, for instance.
|
||||
|
||||
- a custom sentinel type for C++20 ranges
|
||||
- `std::default_sentinel_t`, when `IteratorType` is `std::counted_iterator`
|
||||
|
||||
## Parameters
|
||||
|
||||
`i` (in)
|
||||
@@ -61,7 +68,7 @@ Unlike the [`parse()`](parse.md) function, this function neither throws an excep
|
||||
: iterator to the start of the character range
|
||||
|
||||
`last` (in)
|
||||
: iterator to the end of the character range
|
||||
: iterator to the end of the character range, or a sentinel value that compares equal to the end iterator with `operator!=`
|
||||
|
||||
## Return value
|
||||
|
||||
@@ -112,6 +119,7 @@ A UTF-8 byte order mark is silently ignored.
|
||||
- Changed [runtime assertion](../../features/assertions.md) in case of `FILE*` null pointers to exception in version 3.12.0.
|
||||
- Added `ignore_trailing_commas` in version 3.13.0.
|
||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
|
||||
!!! warning "Deprecation"
|
||||
|
||||
|
||||
@@ -51,6 +51,38 @@ represent a byte array in modern C++.
|
||||
|
||||
The default values for `BinaryType` is `#!cpp std::vector<std::uint8_t>`.
|
||||
|
||||
#### Custom BinaryType behavior
|
||||
|
||||
When a custom `BinaryType` is configured (other than the default `#!cpp std::vector<std::uint8_t>`), you can assign
|
||||
values of that type directly to a `basic_json` instance, and they will automatically be recognized as binary values
|
||||
rather than arrays:
|
||||
|
||||
```cpp
|
||||
using custom_json = nlohmann::basic_json<
|
||||
nlohmann::ordered_map, // ObjectType
|
||||
std::vector, // ArrayType
|
||||
std::string, // StringType
|
||||
bool, // BooleanType
|
||||
std::int64_t, // NumberIntegerType
|
||||
std::uint64_t, // NumberUnsignedType
|
||||
double, // NumberFloatType
|
||||
std::allocator, // AllocatorType
|
||||
nlohmann::adl_serializer,
|
||||
std::vector<std::byte> // Custom BinaryType
|
||||
>;
|
||||
|
||||
std::vector<std::byte> data{std::byte{1}, std::byte{2}, std::byte{3}};
|
||||
custom_json j = data; // Creates a binary value, not an array
|
||||
assert(j.is_binary());
|
||||
|
||||
// Round-tripping works seamlessly
|
||||
auto extracted = j.get<std::vector<std::byte>>();
|
||||
assert(extracted == data);
|
||||
```
|
||||
|
||||
This automatic type detection is a convenience feature that only applies to custom (non-default) `BinaryType` configurations.
|
||||
The default `nlohmann::json` continues to treat `#!cpp std::vector<std::uint8_t>` as arrays for backward compatibility.
|
||||
|
||||
#### Storage
|
||||
|
||||
Binary Arrays are stored as pointers in a `basic_json` type. That is, for any access to array values, a pointer of the
|
||||
|
||||
@@ -7,8 +7,8 @@ static basic_json from_bjdata(InputType&& i,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true);
|
||||
// (2)
|
||||
template<typename IteratorType>
|
||||
static basic_json from_bjdata(IteratorType first, IteratorType last,
|
||||
template<typename IteratorType, typename SentinelType = IteratorType>
|
||||
static basic_json from_bjdata(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true);
|
||||
```
|
||||
@@ -16,7 +16,7 @@ static basic_json from_bjdata(IteratorType first, IteratorType last,
|
||||
Deserializes a given input to a JSON value using the BJData (Binary JData) serialization format.
|
||||
|
||||
1. Reads from a compatible input.
|
||||
2. Reads from an iterator range.
|
||||
2. Reads from an iterator range, or an iterator and a sentinel of a different type (C++20 ranges support).
|
||||
|
||||
The exact mapping and its limitations are described on a [dedicated page](../../features/binary_formats/bjdata.md).
|
||||
|
||||
@@ -35,6 +35,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
||||
`IteratorType`
|
||||
: a compatible iterator type
|
||||
|
||||
`SentinelType`
|
||||
: defaults to `IteratorType`; may be a different type comparable to `IteratorType` via `operator!=`, for instance.
|
||||
|
||||
- a custom sentinel type for C++20 ranges
|
||||
- `std::default_sentinel_t`, when `IteratorType` is `std::counted_iterator`
|
||||
|
||||
## Parameters
|
||||
|
||||
`i` (in)
|
||||
@@ -44,7 +50,7 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
||||
: iterator to the start of the input
|
||||
|
||||
`last` (in)
|
||||
: iterator to the end of the input
|
||||
: iterator to the end of the input, or a sentinel value that compares equal to the end iterator with `operator!=`
|
||||
|
||||
`strict` (in)
|
||||
: whether to expect the input to be consumed until EOF (`#!cpp true` by default)
|
||||
@@ -103,3 +109,4 @@ Linear in the size of the input.
|
||||
|
||||
- Added in version 3.11.0.
|
||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
|
||||
@@ -7,8 +7,8 @@ static basic_json from_bson(InputType&& i,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true);
|
||||
// (2)
|
||||
template<typename IteratorType>
|
||||
static basic_json from_bson(IteratorType first, IteratorType last,
|
||||
template<typename IteratorType, typename SentinelType = IteratorType>
|
||||
static basic_json from_bson(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true);
|
||||
```
|
||||
@@ -16,7 +16,7 @@ static basic_json from_bson(IteratorType first, IteratorType last,
|
||||
Deserializes a given input to a JSON value using the BSON (Binary JSON) serialization format.
|
||||
|
||||
1. Reads from a compatible input.
|
||||
2. Reads from an iterator range.
|
||||
2. Reads from an iterator range, or an iterator and a sentinel of a different type (C++20 ranges support).
|
||||
|
||||
The exact mapping and its limitations are described on a [dedicated page](../../features/binary_formats/bson.md).
|
||||
|
||||
@@ -35,6 +35,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
||||
`IteratorType`
|
||||
: a compatible iterator type
|
||||
|
||||
`SentinelType`
|
||||
: defaults to `IteratorType`; may be a different type comparable to `IteratorType` via `operator!=`, for instance.
|
||||
|
||||
- a custom sentinel type for C++20 ranges
|
||||
- `std::default_sentinel_t`, when `IteratorType` is `std::counted_iterator`
|
||||
|
||||
## Parameters
|
||||
|
||||
`i` (in)
|
||||
@@ -44,7 +50,7 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
||||
: iterator to the start of the input
|
||||
|
||||
`last` (in)
|
||||
: iterator to the end of the input
|
||||
: iterator to the end of the input, or a sentinel value that compares equal to the end iterator with `operator!=`
|
||||
|
||||
`strict` (in)
|
||||
: whether to expect the input to be consumed until EOF (`#!cpp true` by default)
|
||||
@@ -103,6 +109,7 @@ Linear in the size of the input.
|
||||
|
||||
- Added in version 3.4.0.
|
||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
|
||||
!!! warning "Deprecation"
|
||||
|
||||
|
||||
@@ -9,8 +9,8 @@ static basic_json from_cbor(InputType&& i,
|
||||
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error);
|
||||
|
||||
// (2)
|
||||
template<typename IteratorType>
|
||||
static basic_json from_cbor(IteratorType first, IteratorType last,
|
||||
template<typename IteratorType, typename SentinelType = IteratorType>
|
||||
static basic_json from_cbor(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true,
|
||||
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error);
|
||||
@@ -19,7 +19,7 @@ static basic_json from_cbor(IteratorType first, IteratorType last,
|
||||
Deserializes a given input to a JSON value using the CBOR (Concise Binary Object Representation) serialization format.
|
||||
|
||||
1. Reads from a compatible input.
|
||||
2. Reads from an iterator range.
|
||||
2. Reads from an iterator range, or an iterator and a sentinel of a different type (C++20 ranges support).
|
||||
|
||||
The exact mapping and its limitations are described on a [dedicated page](../../features/binary_formats/cbor.md).
|
||||
|
||||
@@ -38,6 +38,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
||||
`IteratorType`
|
||||
: a compatible iterator type
|
||||
|
||||
`SentinelType`
|
||||
: defaults to `IteratorType`; may be a different type comparable to `IteratorType` via `operator!=`, for instance.
|
||||
|
||||
- a custom sentinel type for C++20 ranges
|
||||
- `std::default_sentinel_t`, when `IteratorType` is `std::counted_iterator`
|
||||
|
||||
## Parameters
|
||||
|
||||
`i` (in)
|
||||
@@ -47,7 +53,7 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
||||
: iterator to the start of the input
|
||||
|
||||
`last` (in)
|
||||
: iterator to the end of the input
|
||||
: iterator to the end of the input, or a sentinel value that compares equal to the end iterator with `operator!=`
|
||||
|
||||
`strict` (in)
|
||||
: whether to expect the input to be consumed until EOF (`#!cpp true` by default)
|
||||
@@ -113,6 +119,7 @@ Linear in the size of the input.
|
||||
- Added `allow_exceptions` parameter in version 3.2.0.
|
||||
- Added `tag_handler` parameter in version 3.9.0.
|
||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
|
||||
!!! warning "Deprecation"
|
||||
|
||||
|
||||
@@ -7,8 +7,8 @@ static basic_json from_msgpack(InputType&& i,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true);
|
||||
// (2)
|
||||
template<typename IteratorType>
|
||||
static basic_json from_msgpack(IteratorType first, IteratorType last,
|
||||
template<typename IteratorType, typename SentinelType = IteratorType>
|
||||
static basic_json from_msgpack(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true);
|
||||
```
|
||||
@@ -16,7 +16,7 @@ static basic_json from_msgpack(IteratorType first, IteratorType last,
|
||||
Deserializes a given input to a JSON value using the MessagePack serialization format.
|
||||
|
||||
1. Reads from a compatible input.
|
||||
2. Reads from an iterator range.
|
||||
2. Reads from an iterator range, or an iterator and a sentinel of a different type (C++20 ranges support).
|
||||
|
||||
The exact mapping and its limitations are described on a [dedicated page](../../features/binary_formats/messagepack.md).
|
||||
|
||||
@@ -35,6 +35,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
||||
`IteratorType`
|
||||
: a compatible iterator type
|
||||
|
||||
`SentinelType`
|
||||
: defaults to `IteratorType`; may be a different type comparable to `IteratorType` via `operator!=`, for instance.
|
||||
|
||||
- a custom sentinel type for C++20 ranges
|
||||
- `std::default_sentinel_t`, when `IteratorType` is `std::counted_iterator`
|
||||
|
||||
## Parameters
|
||||
|
||||
`i` (in)
|
||||
@@ -44,7 +50,7 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
||||
: iterator to the start of the input
|
||||
|
||||
`last` (in)
|
||||
: iterator to the end of the input
|
||||
: iterator to the end of the input, or a sentinel value that compares equal to the end iterator with `operator!=`
|
||||
|
||||
`strict` (in)
|
||||
: whether to expect the input to be consumed until EOF (`#!cpp true` by default)
|
||||
@@ -105,6 +111,7 @@ Linear in the size of the input.
|
||||
- Changed to consume input adapters, removed `start_index` parameter, and added `strict` parameter in version 3.0.0.
|
||||
- Added `allow_exceptions` parameter in version 3.2.0.
|
||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
|
||||
!!! warning "Deprecation"
|
||||
|
||||
|
||||
@@ -7,8 +7,8 @@ static basic_json from_ubjson(InputType&& i,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true);
|
||||
// (2)
|
||||
template<typename IteratorType>
|
||||
static basic_json from_ubjson(IteratorType first, IteratorType last,
|
||||
template<typename IteratorType, typename SentinelType = IteratorType>
|
||||
static basic_json from_ubjson(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true);
|
||||
```
|
||||
@@ -16,7 +16,7 @@ static basic_json from_ubjson(IteratorType first, IteratorType last,
|
||||
Deserializes a given input to a JSON value using the UBJSON (Universal Binary JSON) serialization format.
|
||||
|
||||
1. Reads from a compatible input.
|
||||
2. Reads from an iterator range.
|
||||
2. Reads from an iterator range, or an iterator and a sentinel of a different type (C++20 ranges support).
|
||||
|
||||
The exact mapping and its limitations are described on a [dedicated page](../../features/binary_formats/ubjson.md).
|
||||
|
||||
@@ -35,6 +35,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
||||
`IteratorType`
|
||||
: a compatible iterator type
|
||||
|
||||
`SentinelType`
|
||||
: defaults to `IteratorType`; may be a different type comparable to `IteratorType` via `operator!=`, for instance.
|
||||
|
||||
- a custom sentinel type for C++20 ranges
|
||||
- `std::default_sentinel_t`, when `IteratorType` is `std::counted_iterator`
|
||||
|
||||
## Parameters
|
||||
|
||||
`i` (in)
|
||||
@@ -44,7 +50,7 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
||||
: iterator to the start of the input
|
||||
|
||||
`last` (in)
|
||||
: iterator to the end of the input
|
||||
: iterator to the end of the input, or a sentinel value that compares equal to the end iterator with `operator!=`
|
||||
|
||||
`strict` (in)
|
||||
: whether to expect the input to be consumed until EOF (`#!cpp true` by default)
|
||||
@@ -104,6 +110,7 @@ Linear in the size of the input.
|
||||
- Added in version 3.1.0.
|
||||
- Added `allow_exceptions` parameter in version 3.2.0.
|
||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
|
||||
!!! warning "Deprecation"
|
||||
|
||||
|
||||
@@ -10,8 +10,8 @@ static basic_json parse(InputType&& i,
|
||||
const bool ignore_trailing_commas = false);
|
||||
|
||||
// (2)
|
||||
template<typename IteratorType>
|
||||
static basic_json parse(IteratorType first, IteratorType last,
|
||||
template<typename IteratorType, typename SentinelType = IteratorType>
|
||||
static basic_json parse(IteratorType first, SentinelType last,
|
||||
const parser_callback_t cb = nullptr,
|
||||
const bool allow_exceptions = true,
|
||||
const bool ignore_comments = false,
|
||||
@@ -19,10 +19,11 @@ static basic_json parse(IteratorType first, IteratorType last,
|
||||
```
|
||||
|
||||
1. Deserialize from a compatible input.
|
||||
2. Deserialize from a pair of character iterators
|
||||
2. Deserialize from a pair of character iterators, or an iterator and a sentinel of a different type (C++20 ranges support)
|
||||
|
||||
The `value_type` of the iterator must be an integral type with size of 1, 2, or 4 bytes, which will be interpreted
|
||||
respectively as UTF-8, UTF-16, and UTF-32.
|
||||
respectively as UTF-8, UTF-16, and UTF-32. If `SentinelType` differs from `IteratorType`, it must be comparable to
|
||||
the iterator type with `operator!=`.
|
||||
|
||||
## Template parameters
|
||||
|
||||
@@ -43,6 +44,12 @@ static basic_json parse(IteratorType first, IteratorType last,
|
||||
- a pair of `std::string::iterator` or `std::vector<std::uint8_t>::iterator`
|
||||
- a pair of pointers such as `ptr` and `ptr + len`
|
||||
|
||||
`SentinelType`
|
||||
: defaults to `IteratorType`; may be a different type comparable to `IteratorType` via `operator!=`, for instance.
|
||||
|
||||
- a custom sentinel type for C++20 ranges
|
||||
- `std::default_sentinel_t`, when `IteratorType` is `std::counted_iterator`
|
||||
|
||||
## Parameters
|
||||
|
||||
`i` (in)
|
||||
@@ -67,7 +74,7 @@ static basic_json parse(IteratorType first, IteratorType last,
|
||||
: iterator to the start of a character range
|
||||
|
||||
`last` (in)
|
||||
: iterator to the end of a character range
|
||||
: iterator to the end of a character range, or a sentinel value that compares equal to the end iterator with `operator!=`
|
||||
|
||||
## Return value
|
||||
|
||||
@@ -238,6 +245,7 @@ Invalid Unicode escapes and unpaired surrogates in the input are reported as
|
||||
- Changed [runtime assertion](../../features/assertions.md) in case of `FILE*` null pointers to exception in version 3.12.0.
|
||||
- Added `ignore_trailing_commas` in version 3.13.0.
|
||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
|
||||
!!! warning "Deprecation"
|
||||
|
||||
|
||||
@@ -11,8 +11,8 @@ static bool sax_parse(InputType&& i,
|
||||
const bool ignore_trailing_commas = false);
|
||||
|
||||
// (2)
|
||||
template<class IteratorType, class SAX>
|
||||
static bool sax_parse(IteratorType first, IteratorType last,
|
||||
template<class IteratorType, class SAX, class SentinelType = IteratorType>
|
||||
static bool sax_parse(IteratorType first, SentinelType last,
|
||||
SAX* sax,
|
||||
input_format_t format = input_format_t::json,
|
||||
const bool strict = true,
|
||||
@@ -23,10 +23,11 @@ static bool sax_parse(IteratorType first, IteratorType last,
|
||||
Read from input and generate SAX events
|
||||
|
||||
1. Read from a compatible input.
|
||||
2. Read from a pair of character iterators
|
||||
2. Read from a pair of character iterators, or an iterator and a sentinel of a different type (C++20 ranges support)
|
||||
|
||||
The value_type of the iterator must be an integral type with a size of 1, 2, or 4 bytes, which will be interpreted
|
||||
respectively as UTF-8, UTF-16, and UTF-32.
|
||||
respectively as UTF-8, UTF-16, and UTF-32. If `SentinelType` differs from `IteratorType`, it must be comparable to
|
||||
the iterator type with `operator!=`.
|
||||
|
||||
The SAX event lister must follow the interface of [`json_sax`](../json_sax/index.md).
|
||||
|
||||
@@ -46,6 +47,12 @@ The SAX event lister must follow the interface of [`json_sax`](../json_sax/index
|
||||
: a compatible iterator type for overload (2); a pair of character iterators whose `value_type` is an integral type
|
||||
with a size of 1, 2, or 4 bytes (interpreted respectively as UTF-8, UTF-16, and UTF-32)
|
||||
|
||||
`SentinelType`
|
||||
: defaults to `IteratorType`; may be a different type comparable to `IteratorType` via `operator!=`, for overload (2), for instance.
|
||||
|
||||
- a custom sentinel type for C++20 ranges
|
||||
- `std::default_sentinel_t`, when `IteratorType` is `std::counted_iterator`
|
||||
|
||||
`SAX`
|
||||
: a class fulfilling the SAX event listener interface; see [`json_sax`](../json_sax/index.md)
|
||||
|
||||
@@ -76,7 +83,7 @@ The SAX event lister must follow the interface of [`json_sax`](../json_sax/index
|
||||
: iterator to the start of a character range
|
||||
|
||||
`last` (in)
|
||||
: iterator to the end of a character range
|
||||
: iterator to the end of a character range, or a sentinel value that compares equal to the end iterator with `operator!=`
|
||||
|
||||
## Return value
|
||||
|
||||
@@ -128,6 +135,7 @@ A UTF-8 byte order mark is silently ignored.
|
||||
- Ignoring comments via `ignore_comments` added in version 3.9.0.
|
||||
- Added `ignore_trailing_commas` in version 3.13.0.
|
||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
|
||||
!!! warning "Deprecation"
|
||||
|
||||
|
||||
@@ -24,6 +24,7 @@ header. See also the [macro overview page](../../features/macros.md).
|
||||
- [**JSON_NO_IO**](json_no_io.md) - switch off functions relying on certain C++ I/O headers
|
||||
- [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers
|
||||
- [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace
|
||||
- [**JSON_USE_SIMDUTF**](json_use_simdutf.md) - use the simdutf library to accelerate UTF-8 validation
|
||||
|
||||
## Library version
|
||||
|
||||
|
||||
@@ -38,7 +38,8 @@ When the macro is not defined, the library will define it to its default value.
|
||||
|
||||
Diagnostic messages can also be controlled with the CMake option
|
||||
[`JSON_Diagnostics`](../../integration/cmake.md#json_diagnostics) (`OFF` by default)
|
||||
which defines `JSON_DIAGNOSTICS` accordingly.
|
||||
which defines `JSON_DIAGNOSTICS` accordingly. Note this only applies when building the
|
||||
library from source — see the pre-installed-package caveat on that page.
|
||||
|
||||
## Examples
|
||||
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
# JSON_USE_SIMDUTF
|
||||
|
||||
```cpp
|
||||
#define JSON_USE_SIMDUTF
|
||||
```
|
||||
|
||||
When defined, the parser validates the UTF-8 content of JSON strings that come from a **contiguous byte input**
|
||||
(`std::string`, `std::vector<char>`/`<std::uint8_t>`, string literals, `const char*` ranges, …) using the
|
||||
[simdutf](https://github.com/simdutf/simdutf) library instead of the built-in scalar validator. On text with many
|
||||
non-ASCII characters (e.g. CJK or emoji) this can validate several times faster.
|
||||
|
||||
This is an **opt-in external dependency**. The library itself remains header-only and its behavior is unchanged: the
|
||||
same input is accepted or rejected either way, and every parse error is reported at the same position with the same
|
||||
message (simdutf is only used to fast-path *valid* runs; anything it flags falls back to the scalar path so the exact
|
||||
diagnostic is preserved). Streaming inputs (files, `std::istream`, wide strings, user-defined adapters) always use the
|
||||
scalar path.
|
||||
|
||||
When `JSON_USE_SIMDUTF` is defined you must make the `simdutf.h` header available on the include path and link the
|
||||
simdutf library. When it is not defined, no simdutf header is included and there is no dependency.
|
||||
|
||||
## Default definition
|
||||
|
||||
By default, `#!cpp JSON_USE_SIMDUTF` is not defined and the portable C++11 scalar validator is used.
|
||||
|
||||
```cpp
|
||||
#undef JSON_USE_SIMDUTF
|
||||
```
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
|
||||
The code below enables the simdutf backend for UTF-8 validation.
|
||||
|
||||
```cpp
|
||||
#define JSON_USE_SIMDUTF 1
|
||||
#include <simdutf.h>
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
...
|
||||
```
|
||||
|
||||
The project must also link against simdutf, e.g. with CMake:
|
||||
|
||||
```cmake
|
||||
target_compile_definitions(your_target PRIVATE JSON_USE_SIMDUTF)
|
||||
target_link_libraries(your_target PRIVATE simdutf::simdutf)
|
||||
```
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.12.1.
|
||||
@@ -47,6 +47,28 @@ json j = {{"one", 1}, {"two", 2}};
|
||||
auto m = j.get<std::map<std::string, int>>(); // {{"one", 1}, {"two", 2}}
|
||||
```
|
||||
|
||||
`#!cpp std::pair` and `#!cpp std::tuple` are also supported, converting positionally to and from a JSON array:
|
||||
|
||||
```cpp
|
||||
json j = {1.0, "hello", 42};
|
||||
auto t = j.get<std::tuple<double, std::string, int>>(); // {1.0, "hello", 42}
|
||||
```
|
||||
|
||||
!!! info "Extracting references into a tuple"
|
||||
|
||||
A tuple type may also hold references (e.g. `#!cpp std::tuple<double&, std::string&>`) to avoid copying: `get`
|
||||
then returns a tuple of references pointing directly at the elements stored inside the `basic_json` array,
|
||||
rather than a tuple of copies:
|
||||
|
||||
```cpp
|
||||
json j = {1.0, "hello"};
|
||||
auto refs = j.get<std::tuple<double&, std::string&>>();
|
||||
std::get<1>(refs) = "world"; // modifies j[1] in place
|
||||
```
|
||||
|
||||
A referenced type must be one the library actually stores (or an arithmetic type it can convert to/from);
|
||||
otherwise this is a compile error.
|
||||
|
||||
## Implicit conversions
|
||||
|
||||
By default, a JSON value implicitly converts to a compatible C++ type, so the explicit `get` call can often be omitted:
|
||||
@@ -136,6 +158,20 @@ std::vector<int> numbers = {1, 2, 3};
|
||||
json j = numbers; // [1,2,3]
|
||||
```
|
||||
|
||||
!!! info "Constructing from a C++20 range view"
|
||||
|
||||
A `json` array can also be constructed directly from a C++20 range view (`std::ranges::view`), such as the result
|
||||
of `std::views::filter` or `std::views::transform` -- no intermediate container is needed:
|
||||
|
||||
```cpp
|
||||
std::vector<int> nums{1, 2, 37, 42, 21};
|
||||
auto filtered = nums | std::views::filter([](int i) { return i > 10; });
|
||||
json j(filtered); // [37,42,21]
|
||||
```
|
||||
|
||||
This requires [`JSON_HAS_RANGES`](../api/macros/json_has_ranges.md) to be enabled and is unavailable on MinGW due
|
||||
to incomplete C++20 ranges support there.
|
||||
|
||||
## Your own types
|
||||
|
||||
The conversions above are built in for standard types. To make the same syntax work for **your own** types, provide
|
||||
|
||||
@@ -135,6 +135,31 @@ Enable CI build targets. The exact targets are used during the several CI steps
|
||||
|
||||
Enable [extended diagnostic messages](../home/exceptions.md#extended-diagnostic-messages) by defining macro [`JSON_DIAGNOSTICS`](../api/macros/json_diagnostics.md). This option is `OFF` by default.
|
||||
|
||||
!!! warning "Does not apply to a pre-installed package"
|
||||
|
||||
This option only takes effect when building nlohmann/json from source as part of your own
|
||||
CMake project (e.g. via [`FetchContent`](#fetchcontent) or [`add_subdirectory`](#external)).
|
||||
It has **no effect** on a package that was already built and installed elsewhere (Homebrew,
|
||||
vcpkg, a system package, etc.) — the resulting compile definition is baked into the exported
|
||||
`nlohmann_jsonTargets.cmake` at install time, and `set(JSON_Diagnostics ON)` before
|
||||
`find_package()` does not change it (verified against the Homebrew-installed package: the
|
||||
exported target still carries a fixed `$<$<BOOL:OFF>:JSON_DIAGNOSTICS=1>`, regardless of any
|
||||
variable set in the consuming project).
|
||||
|
||||
To enable extended diagnostics for a pre-installed package, override the imported target's
|
||||
property directly after `find_package()`:
|
||||
|
||||
```cmake
|
||||
find_package(nlohmann_json REQUIRED)
|
||||
set_target_properties(nlohmann_json::nlohmann_json PROPERTIES
|
||||
INTERFACE_COMPILE_DEFINITIONS "JSON_DIAGNOSTICS=1")
|
||||
```
|
||||
|
||||
This only works cleanly when your project is the sole consumer of that imported target. If
|
||||
nlohmann_json is pulled in from more than one place in your dependency graph with different
|
||||
`JSON_DIAGNOSTICS` values, you may see a `"JSON_DIAGNOSTICS" redefined` compiler error, since
|
||||
conflicting `-D` flags can end up on the same compile command line.
|
||||
|
||||
### `JSON_Diagnostic_Positions`
|
||||
|
||||
Enable position diagnostics by defining macro [`JSON_DIAGNOSTIC_POSITIONS`](../api/macros/json_diagnostic_positions.md). This option is `OFF` by default.
|
||||
|
||||
@@ -296,6 +296,7 @@ nav:
|
||||
- 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md
|
||||
- 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md
|
||||
- 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md
|
||||
- 'JSON_USE_SIMDUTF': api/macros/json_use_simdutf.md
|
||||
- 'NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_derived_type.md
|
||||
- 'NLOHMANN_DEFINE_TYPE_INTRUSIVE, NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_intrusive.md
|
||||
- 'NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_non_intrusive.md
|
||||
|
||||
@@ -19,6 +19,7 @@
|
||||
#include <unordered_map> // unordered_map
|
||||
#include <utility> // pair, declval
|
||||
#include <valarray> // valarray
|
||||
#include <vector> // vector
|
||||
|
||||
#include <nlohmann/detail/exceptions.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
@@ -332,6 +333,7 @@ template < typename BasicJsonType, typename ConstructibleArrayType,
|
||||
!is_constructible_object_type<BasicJsonType, ConstructibleArrayType>::value&&
|
||||
!is_constructible_string_type<BasicJsonType, ConstructibleArrayType>::value&&
|
||||
!std::is_same<ConstructibleArrayType, typename BasicJsonType::binary_t>::value&&
|
||||
!is_compatible_binary_type<BasicJsonType, ConstructibleArrayType>::value&&
|
||||
!is_basic_json<ConstructibleArrayType>::value,
|
||||
int > = 0 >
|
||||
auto from_json(const BasicJsonType& j, ConstructibleArrayType& arr)
|
||||
@@ -377,6 +379,25 @@ inline void from_json(const BasicJsonType& j, typename BasicJsonType::binary_t&
|
||||
bin = *j.template get_ptr<const typename BasicJsonType::binary_t*>();
|
||||
}
|
||||
|
||||
template < typename BasicJsonType, typename CompatibleArrayType,
|
||||
enable_if_t < is_compatible_binary_type<BasicJsonType, CompatibleArrayType>::value,
|
||||
int > = 0 >
|
||||
inline void from_json(const BasicJsonType& j, CompatibleArrayType& bin)
|
||||
{
|
||||
if (j.is_binary())
|
||||
{
|
||||
bin = static_cast<CompatibleArrayType>(*j.template get_ptr<const typename BasicJsonType::binary_t*>());
|
||||
}
|
||||
else if (j.is_array())
|
||||
{
|
||||
from_json_array_impl(j, bin, priority_tag<3> {});
|
||||
}
|
||||
else
|
||||
{
|
||||
JSON_THROW(type_error::create(302, concat("type must be binary or array, but is ", j.type_name()), &j));
|
||||
}
|
||||
}
|
||||
|
||||
template<typename BasicJsonType, typename ConstructibleObjectType,
|
||||
enable_if_t<is_constructible_object_type<BasicJsonType, ConstructibleObjectType>::value, int> = 0>
|
||||
inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj)
|
||||
|
||||
@@ -177,8 +177,11 @@ struct external_constructor<value_t::array>
|
||||
}
|
||||
|
||||
template < typename BasicJsonType, typename CompatibleArrayType,
|
||||
enable_if_t < !std::is_same<CompatibleArrayType, typename BasicJsonType::array_t>::value,
|
||||
int > = 0 >
|
||||
enable_if_t < !std::is_same<CompatibleArrayType, typename BasicJsonType::array_t>::value
|
||||
#if JSON_HAS_RANGES && !defined(__MINGW32__)
|
||||
&& !is_compatible_range_view<CompatibleArrayType>::value
|
||||
#endif
|
||||
, int > = 0 >
|
||||
static void construct(BasicJsonType& j, const CompatibleArrayType& arr)
|
||||
{
|
||||
using std::begin;
|
||||
@@ -218,6 +221,25 @@ struct external_constructor<value_t::array>
|
||||
j.set_parents();
|
||||
j.assert_invariant();
|
||||
}
|
||||
|
||||
// std::ranges does not work properly on MinGW due to incomplete C++20 support
|
||||
// see https://github.com/nlohmann/json/issues/4916
|
||||
#if JSON_HAS_RANGES && !defined(__MINGW32__)
|
||||
template<typename BasicJsonType, typename CompatibleArrayType,
|
||||
enable_if_t<is_compatible_range_view<std::remove_cvref_t<CompatibleArrayType>>::value, int> = 0>
|
||||
static void construct(BasicJsonType& j, CompatibleArrayType && arr)
|
||||
{
|
||||
j.m_data.m_value.destroy(j.m_data.m_type);
|
||||
j.m_data.m_type = value_t::array;
|
||||
j.m_data.m_value = value_t::array;
|
||||
for (auto&& x : std::forward<CompatibleArrayType>(arr))
|
||||
{
|
||||
j.m_data.m_value.array->push_back(x);
|
||||
j.set_parent(j.m_data.m_value.array->back());
|
||||
}
|
||||
j.assert_invariant();
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
template<>
|
||||
@@ -355,19 +377,44 @@ template < typename BasicJsonType, typename CompatibleArrayType,
|
||||
!is_compatible_object_type<BasicJsonType, CompatibleArrayType>::value&&
|
||||
!is_compatible_string_type<BasicJsonType, CompatibleArrayType>::value&&
|
||||
!std::is_same<typename BasicJsonType::binary_t, CompatibleArrayType>::value&&
|
||||
!is_basic_json<CompatibleArrayType>::value,
|
||||
!is_compatible_binary_type<BasicJsonType, CompatibleArrayType>::value&&
|
||||
!is_basic_json<CompatibleArrayType>::value
|
||||
#if JSON_HAS_RANGES && !defined(__MINGW32__)
|
||||
&& !is_compatible_range_view<CompatibleArrayType>::value
|
||||
#endif
|
||||
,
|
||||
int > = 0 >
|
||||
inline void to_json(BasicJsonType& j, const CompatibleArrayType& arr)
|
||||
{
|
||||
external_constructor<value_t::array>::construct(j, arr);
|
||||
}
|
||||
|
||||
#if JSON_HAS_RANGES && !defined(__MINGW32__)
|
||||
template < typename BasicJsonType, typename T,
|
||||
enable_if_t < is_compatible_range_view<std::remove_cvref_t<T>>::value
|
||||
&& !is_compatible_string_type<BasicJsonType, std::remove_cvref_t<T>>::value
|
||||
&& !is_compatible_object_type<BasicJsonType, std::remove_cvref_t<T>>::value
|
||||
&& !is_basic_json<std::remove_cvref_t<T>>::value, int > = 0 >
|
||||
inline void to_json(BasicJsonType& j, T && arr)
|
||||
{
|
||||
external_constructor<value_t::array>::construct(j, std::forward<T>(arr));
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename BasicJsonType>
|
||||
inline void to_json(BasicJsonType& j, const typename BasicJsonType::binary_t& bin)
|
||||
{
|
||||
external_constructor<value_t::binary>::construct(j, bin);
|
||||
}
|
||||
|
||||
template < typename BasicJsonType, typename CompatibleArrayType,
|
||||
enable_if_t < is_compatible_binary_type<BasicJsonType, CompatibleArrayType>::value,
|
||||
int > = 0 >
|
||||
inline void to_json(BasicJsonType& j, const CompatibleArrayType& bin)
|
||||
{
|
||||
external_constructor<value_t::binary>::construct(j, typename BasicJsonType::binary_t(bin));
|
||||
}
|
||||
|
||||
template<typename BasicJsonType, typename T,
|
||||
enable_if_t<std::is_convertible<T, BasicJsonType>::value, int> = 0>
|
||||
inline void to_json(BasicJsonType& j, const std::valarray<T>& arr)
|
||||
|
||||
@@ -155,7 +155,9 @@ class input_stream_adapter
|
||||
|
||||
// General-purpose iterator-based adapter. It might not be as fast as
|
||||
// theoretically possible for some containers, but it is extremely versatile.
|
||||
template<typename IteratorType>
|
||||
// SentinelType defaults to IteratorType for backward compatibility, but may
|
||||
// be a different type (e.g., a C++20 sentinel or counted_iterator).
|
||||
template<typename IteratorType, typename SentinelType = IteratorType>
|
||||
class iterator_input_adapter
|
||||
{
|
||||
public:
|
||||
@@ -169,9 +171,10 @@ class iterator_input_adapter
|
||||
// in wide_string_input_adapter, which does not expose this).
|
||||
static constexpr bool supports_seek =
|
||||
std::is_same<typename std::iterator_traits<IteratorType>::iterator_category, std::random_access_iterator_tag>::value
|
||||
&& std::is_same<IteratorType, SentinelType>::value
|
||||
&& sizeof(char_type) == 1;
|
||||
|
||||
iterator_input_adapter(IteratorType first, IteratorType last)
|
||||
iterator_input_adapter(IteratorType first, SentinelType last)
|
||||
: begin(first), current(std::move(first)), end(std::move(last))
|
||||
{}
|
||||
|
||||
@@ -216,19 +219,57 @@ class iterator_input_adapter
|
||||
private:
|
||||
// whether IteratorType refers to a contiguous range and therefore supports
|
||||
// a std::memcpy fast path (pointers always do; in C++20 we can also detect
|
||||
// library iterators such as those of std::vector and std::string)
|
||||
// library iterators such as those of std::vector and std::string).
|
||||
// Computing the available element count needs either same-type iterators
|
||||
// (plain std::distance) or, in C++20, a sized sentinel (std::ranges::distance),
|
||||
// e.g. std::counted_iterator paired with std::default_sentinel_t.
|
||||
static constexpr bool iterator_is_contiguous =
|
||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
std::contiguous_iterator<IteratorType> ||
|
||||
(std::is_same<IteratorType, SentinelType>::value || std::sized_sentinel_for<SentinelType, IteratorType>)
|
||||
&& (std::contiguous_iterator<IteratorType> || std::is_pointer<IteratorType>::value);
|
||||
#else
|
||||
std::is_same<IteratorType, SentinelType>::value && std::is_pointer<IteratorType>::value;
|
||||
#endif
|
||||
std::is_pointer<IteratorType>::value;
|
||||
|
||||
public:
|
||||
// Whether the remaining input is a single contiguous block of 1-byte
|
||||
// elements that the lexer can inspect directly (used for the SWAR string
|
||||
// fast path). Restricted to same-type iterator/sentinel pairs so that plain
|
||||
// std::distance/std::advance are well-defined in all standards.
|
||||
static constexpr bool supports_bulk_scan =
|
||||
iterator_is_contiguous && std::is_same<IteratorType, SentinelType>::value && sizeof(char_type) == 1;
|
||||
|
||||
// Pointer to the next unread element; only valid when bulk_remaining() > 0.
|
||||
const char_type* bulk_data() const
|
||||
{
|
||||
return &*current;
|
||||
}
|
||||
|
||||
// Number of unread elements available as one contiguous block.
|
||||
std::size_t bulk_remaining() const
|
||||
{
|
||||
return static_cast<std::size_t>(std::distance(current, end));
|
||||
}
|
||||
|
||||
// Consume @a n elements previously inspected via bulk_data().
|
||||
void bulk_skip(std::size_t n)
|
||||
{
|
||||
std::advance(current, static_cast<typename std::iterator_traits<IteratorType>::difference_type>(n));
|
||||
}
|
||||
|
||||
private:
|
||||
// contiguous fast path: bulk copy the remaining range with std::memcpy
|
||||
template<class T>
|
||||
std::size_t get_elements_impl(T* dest, std::size_t count, std::true_type /*contiguous*/)
|
||||
{
|
||||
const std::size_t wanted = count * sizeof(T);
|
||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
// std::ranges::distance also supports sized sentinels of a different
|
||||
// type (e.g. std::counted_iterator + std::default_sentinel_t)
|
||||
const std::size_t available = static_cast<std::size_t>(std::ranges::distance(current, end)) * sizeof(char_type);
|
||||
#else
|
||||
const std::size_t available = static_cast<std::size_t>(std::distance(current, end)) * sizeof(char_type);
|
||||
#endif
|
||||
const std::size_t copied = (std::min)(wanted, available);
|
||||
if (JSON_HEDLEY_LIKELY(copied != 0))
|
||||
{
|
||||
@@ -267,7 +308,7 @@ class iterator_input_adapter
|
||||
|
||||
IteratorType begin;
|
||||
IteratorType current;
|
||||
IteratorType end;
|
||||
SentinelType end;
|
||||
|
||||
template<typename BaseInputAdapter, size_t T>
|
||||
friend struct wide_string_input_helper;
|
||||
@@ -381,17 +422,30 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
|
||||
}
|
||||
else
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(!input.empty()))
|
||||
// A supplementary code point is a high surrogate (0xD800..0xDBFF)
|
||||
// followed by a low surrogate (0xDC00..0xDFFF). A lone low
|
||||
// surrogate, a high surrogate at the end of the input, or a high
|
||||
// surrogate followed by any other unit is malformed UTF-16. In
|
||||
// that case the offending unit is passed through unchanged so the
|
||||
// UTF-8 decoder rejects it, matching how \uXXXX surrogate escapes
|
||||
// are handled in the lexer.
|
||||
bool valid_pair = false;
|
||||
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
|
||||
{
|
||||
const auto wc2 = static_cast<unsigned int>(input.get_character());
|
||||
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu));
|
||||
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu));
|
||||
utf8_bytes_filled = 4;
|
||||
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
|
||||
{
|
||||
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu));
|
||||
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu));
|
||||
utf8_bytes_filled = 4;
|
||||
valid_pair = true;
|
||||
}
|
||||
}
|
||||
else
|
||||
|
||||
if (!valid_pair)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
||||
utf8_bytes_filled = 1;
|
||||
@@ -453,19 +507,54 @@ class wide_string_input_adapter
|
||||
std::size_t utf8_bytes_filled = 0;
|
||||
};
|
||||
|
||||
template<typename IteratorType, typename Enable = void>
|
||||
template<typename IteratorType, typename SentinelType = IteratorType, typename Enable = void>
|
||||
struct iterator_input_adapter_factory
|
||||
{
|
||||
using iterator_type = IteratorType;
|
||||
using sentinel_type = SentinelType;
|
||||
using char_type = typename std::iterator_traits<iterator_type>::value_type;
|
||||
using adapter_type = iterator_input_adapter<iterator_type>;
|
||||
using adapter_type = iterator_input_adapter<iterator_type, sentinel_type>;
|
||||
|
||||
static adapter_type create(IteratorType first, IteratorType last)
|
||||
static adapter_type create(IteratorType first, SentinelType last)
|
||||
{
|
||||
return adapter_type(std::move(first), std::move(last));
|
||||
}
|
||||
};
|
||||
|
||||
// Detection: whether IteratorType and SentinelType can be compared with !=
|
||||
template<typename IteratorType, typename SentinelType, typename = void>
|
||||
struct can_compare_ne_impl : std::false_type {};
|
||||
|
||||
template<typename IteratorType, typename SentinelType>
|
||||
struct can_compare_ne_impl < IteratorType, SentinelType,
|
||||
void_t < decltype(std::declval<IteratorType>() != std::declval<SentinelType>()) >>
|
||||
: std::true_type {};
|
||||
|
||||
// Workaround for reversed operator order
|
||||
template<typename IteratorType, typename SentinelType, typename = void>
|
||||
struct can_compare_ne_reversed : std::false_type {};
|
||||
|
||||
template<typename IteratorType, typename SentinelType>
|
||||
struct can_compare_ne_reversed < IteratorType, SentinelType,
|
||||
void_t < decltype(std::declval<SentinelType>() != std::declval<IteratorType>()) >>
|
||||
: std::true_type {};
|
||||
|
||||
template<typename IteratorType, typename SentinelType>
|
||||
struct can_compare_ne_either_order : std::integral_constant < bool,
|
||||
can_compare_ne_impl<IteratorType, SentinelType>::value ||
|
||||
can_compare_ne_reversed<IteratorType, SentinelType>::value > {};
|
||||
|
||||
// std::nullptr_t is excluded explicitly: a literal `nullptr` passed as a
|
||||
// trailing default argument (e.g. parse(s, nullptr, ...)) must never be
|
||||
// mistaken for a sentinel, and some compilers (e.g. GCC 4.8) unreliably
|
||||
// SFINAE the `operator!=` detection above for std::nullptr_t against
|
||||
// container/string types, which would otherwise make such calls ambiguous
|
||||
// with the compatible-input overload.
|
||||
template<typename IteratorType, typename SentinelType>
|
||||
struct can_compare_ne : std::integral_constant < bool,
|
||||
!std::is_same<SentinelType, std::nullptr_t>::value &&
|
||||
can_compare_ne_either_order<IteratorType, SentinelType>::value > {};
|
||||
|
||||
template<typename T>
|
||||
struct is_iterator_of_multibyte
|
||||
{
|
||||
@@ -476,28 +565,52 @@ struct is_iterator_of_multibyte
|
||||
};
|
||||
};
|
||||
|
||||
template<typename IteratorType>
|
||||
struct iterator_input_adapter_factory<IteratorType, enable_if_t<is_iterator_of_multibyte<IteratorType>::value>>
|
||||
template<typename IteratorType, typename SentinelType>
|
||||
struct iterator_input_adapter_factory<IteratorType, SentinelType, enable_if_t<is_iterator_of_multibyte<IteratorType>::value>>
|
||||
{
|
||||
using iterator_type = IteratorType;
|
||||
using sentinel_type = SentinelType;
|
||||
using char_type = typename std::iterator_traits<iterator_type>::value_type;
|
||||
using base_adapter_type = iterator_input_adapter<iterator_type>;
|
||||
using base_adapter_type = iterator_input_adapter<iterator_type, sentinel_type>;
|
||||
using adapter_type = wide_string_input_adapter<base_adapter_type, char_type>;
|
||||
|
||||
static adapter_type create(IteratorType first, IteratorType last)
|
||||
static adapter_type create(IteratorType first, SentinelType last)
|
||||
{
|
||||
return adapter_type(base_adapter_type(std::move(first), std::move(last)));
|
||||
}
|
||||
};
|
||||
|
||||
// General purpose iterator-based input
|
||||
template<typename IteratorType>
|
||||
typename iterator_input_adapter_factory<IteratorType>::adapter_type input_adapter(IteratorType first, IteratorType last)
|
||||
// General purpose iterator-based input (iterator+sentinel pair; SentinelType
|
||||
// defaults to IteratorType for the common same-type case, but may differ for
|
||||
// C++20 ranges-style iterator+sentinel pairs). Only enable for types that can
|
||||
// be compared with !=.
|
||||
template < typename IteratorType, typename SentinelType = IteratorType,
|
||||
typename = typename std::enable_if <
|
||||
can_compare_ne<IteratorType, SentinelType>::value >::type >
|
||||
typename iterator_input_adapter_factory<IteratorType, SentinelType>::adapter_type input_adapter(IteratorType first, SentinelType last)
|
||||
{
|
||||
using factory_type = iterator_input_adapter_factory<IteratorType>;
|
||||
using factory_type = iterator_input_adapter_factory<IteratorType, SentinelType>;
|
||||
return factory_type::create(first, last);
|
||||
}
|
||||
|
||||
// Detect a container that stores its elements contiguously as single bytes
|
||||
// (std::string, std::vector<char/unsigned char>, std::array<char, N>,
|
||||
// std::string_view, ...). Such inputs are wrapped in a pointer-based adapter so
|
||||
// they benefit from the contiguous fast paths (bulk string scanning, memcpy for
|
||||
// binary formats) in every C++ standard - not only in C++20, where the standard
|
||||
// library iterators model std::contiguous_iterator and are detected directly.
|
||||
template<typename ContainerType, typename = void>
|
||||
struct is_contiguous_byte_container : std::false_type {};
|
||||
|
||||
template<typename ContainerType>
|
||||
struct is_contiguous_byte_container < ContainerType, void_t <
|
||||
decltype(std::declval<const ContainerType&>().data()),
|
||||
decltype(std::declval<const ContainerType&>().size()) >>
|
||||
: std::integral_constant < bool,
|
||||
std::is_pointer<decltype(std::declval<const ContainerType&>().data())>::value&&
|
||||
std::is_integral<typename std::remove_pointer<decltype(std::declval<const ContainerType&>().data())>::type>::value&&
|
||||
sizeof(typename std::remove_pointer<decltype(std::declval<const ContainerType&>().data())>::type) == 1 > {};
|
||||
|
||||
// Convenience shorthand from container to iterator
|
||||
// Enables ADL on begin(container) and end(container)
|
||||
// Encloses the using declarations in namespace for not to leak them to outside scope
|
||||
@@ -525,12 +638,32 @@ struct container_input_adapter_factory< ContainerType,
|
||||
|
||||
} // namespace container_input_adapter_factory_impl
|
||||
|
||||
template<typename ContainerType>
|
||||
typename container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::adapter_type input_adapter(ContainerType&& container)
|
||||
// General container path (iterator-based). Contiguous single-byte containers
|
||||
// are excluded here and routed through the pointer-based overload below.
|
||||
template < typename ContainerType,
|
||||
enable_if_t < !is_contiguous_byte_container<ContainerType>::value, int > = 0 >
|
||||
typename container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::adapter_type input_adapter(ContainerType && container)
|
||||
{
|
||||
return container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::create(std::forward<ContainerType>(container));
|
||||
}
|
||||
|
||||
// Contiguous single-byte containers (std::string, std::vector<char>, ...) are
|
||||
// wrapped in a pointer-based adapter so the contiguous fast paths apply in every
|
||||
// standard. The pointer keeps the container's own element type (const char* for
|
||||
// std::string, const std::uint8_t* for std::vector<std::uint8_t>, ...), so the
|
||||
// resulting char_type - and therefore the parsing behavior - is byte-for-byte
|
||||
// identical to the iterator-based path; only the raw pointer additionally
|
||||
// enables the bulk fast paths. The container outlives the adapter for the whole
|
||||
// parse (temporaries live until the end of the full expression), exactly as the
|
||||
// iterators it replaces did.
|
||||
template < typename ContainerType,
|
||||
enable_if_t < is_contiguous_byte_container<ContainerType>::value, int > = 0 >
|
||||
auto input_adapter(const ContainerType& container)
|
||||
-> decltype(input_adapter(container.data(), container.data() + container.size()))
|
||||
{
|
||||
return input_adapter(container.data(), container.data() + container.size());
|
||||
}
|
||||
|
||||
// specialization for std::string
|
||||
using string_input_adapter_type = decltype(input_adapter(std::declval<std::string>()));
|
||||
|
||||
|
||||
@@ -19,7 +19,9 @@
|
||||
#include <vector> // vector
|
||||
|
||||
#include <nlohmann/detail/input/input_adapters.hpp>
|
||||
#include <nlohmann/detail/input/number_parse.hpp>
|
||||
#include <nlohmann/detail/input/position_t.hpp>
|
||||
#include <nlohmann/detail/input/string_scan.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||
|
||||
@@ -125,6 +127,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/)
|
||||
return false;
|
||||
}
|
||||
|
||||
// Detect whether an input adapter exposes a contiguous byte block that the
|
||||
// lexer can scan directly (see iterator_input_adapter::supports_bulk_scan).
|
||||
// Adapters without the flag - file, stream, wide-string, user-defined - fall
|
||||
// back to the character-at-a-time string scanner.
|
||||
template<typename InputAdapterType>
|
||||
using detect_supports_bulk_scan = decltype(InputAdapterType::supports_bulk_scan);
|
||||
|
||||
template<typename InputAdapterType>
|
||||
constexpr bool input_adapter_supports_bulk_scan(std::true_type /*detected*/)
|
||||
{
|
||||
return InputAdapterType::supports_bulk_scan;
|
||||
}
|
||||
|
||||
template<typename InputAdapterType>
|
||||
constexpr bool input_adapter_supports_bulk_scan(std::false_type /*detected*/)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief lexical analysis
|
||||
|
||||
@@ -146,6 +167,14 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
static constexpr bool lazy_token_string =
|
||||
input_adapter_supports_seek<InputAdapterType>(is_detected<detect_supports_seek, InputAdapterType> {});
|
||||
|
||||
/// whether string scanning may bulk-consume runs of ordinary characters
|
||||
/// directly from a contiguous input buffer (SWAR fast path). This requires
|
||||
/// the token to be reconstructible lazily (lazy_token_string), so bypassing
|
||||
/// the per-character capture in get() cannot lose error diagnostics.
|
||||
static constexpr bool bulk_scan =
|
||||
lazy_token_string
|
||||
&& input_adapter_supports_bulk_scan<InputAdapterType>(is_detected<detect_supports_bulk_scan, InputAdapterType> {});
|
||||
|
||||
public:
|
||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||
|
||||
@@ -265,6 +294,40 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
return true;
|
||||
}
|
||||
|
||||
/// contiguous input: bulk-append the run of ordinary characters and complete
|
||||
/// well-formed UTF-8 sequences starting at the current read position, leaving
|
||||
/// the first byte that needs individual handling (the closing quote, an
|
||||
/// escape, a control character, or an ill-formed UTF-8 byte) for get()
|
||||
void scan_string_bulk(std::true_type /*bulk*/)
|
||||
{
|
||||
// a pending unget must be consumed through the normal path first
|
||||
if (next_unget)
|
||||
{
|
||||
return;
|
||||
}
|
||||
const std::size_t remaining = ia.bulk_remaining();
|
||||
if (remaining == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
const auto* const data = reinterpret_cast<const unsigned char*>(ia.bulk_data());
|
||||
|
||||
const std::size_t pos = string_bulk_run(data, remaining);
|
||||
if (pos == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
token_buffer.append(reinterpret_cast<const typename string_t::value_type*>(data), pos);
|
||||
ia.bulk_skip(pos);
|
||||
// the run contains no newline (all bytes < 0x20 are treated as special),
|
||||
// so only the flat character counters advance
|
||||
position.chars_read_total += pos;
|
||||
position.chars_read_current_line += pos;
|
||||
}
|
||||
|
||||
/// streaming input: no bulk fast path
|
||||
void scan_string_bulk(std::false_type /*bulk*/) const noexcept {}
|
||||
|
||||
/*!
|
||||
@brief scan a string literal
|
||||
|
||||
@@ -290,6 +353,10 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
|
||||
while (true)
|
||||
{
|
||||
// bulk-consume ordinary characters from contiguous input, then
|
||||
// handle the next special byte through the switch below
|
||||
scan_string_bulk(std::integral_constant<bool, bulk_scan> {});
|
||||
|
||||
// get the next character
|
||||
switch (get())
|
||||
{
|
||||
@@ -1279,45 +1346,56 @@ scan_number_done:
|
||||
// we are done scanning a number)
|
||||
unget();
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
errno = 0;
|
||||
return convert_number(number_type);
|
||||
}
|
||||
|
||||
// try to parse integers first and fall back to floats
|
||||
/*!
|
||||
@brief convert the number text in token_buffer to its value and token type
|
||||
|
||||
The digit sequence in token_buffer has already been validated (by the
|
||||
scan_number() state machine or by the contiguous fast path) and holds the
|
||||
locale decimal point in place of '.'. Integers are parsed first and fall
|
||||
back to floating point on overflow. This is shared so both scanners produce
|
||||
identical results.
|
||||
*/
|
||||
token_type convert_number(token_type number_type)
|
||||
{
|
||||
const char* const num_begin = token_buffer.data();
|
||||
const char* const num_end = num_begin + token_buffer.size();
|
||||
|
||||
// try to parse integers first and fall back to floats; the digit
|
||||
// sequence has already been validated, so a dedicated parser can avoid
|
||||
// the locale/errno overhead of strtoull
|
||||
if (number_type == token_type::value_unsigned)
|
||||
{
|
||||
const auto x = std::strtoull(token_buffer.data(), &endptr, 10);
|
||||
|
||||
// we checked the number format before
|
||||
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
|
||||
|
||||
if (errno != ERANGE)
|
||||
if (parse_integer_unsigned(num_begin, num_end, value_unsigned))
|
||||
{
|
||||
value_unsigned = static_cast<number_unsigned_t>(x);
|
||||
if (value_unsigned == x)
|
||||
{
|
||||
return token_type::value_unsigned;
|
||||
}
|
||||
return token_type::value_unsigned;
|
||||
}
|
||||
}
|
||||
else if (number_type == token_type::value_integer)
|
||||
{
|
||||
const auto x = std::strtoll(token_buffer.data(), &endptr, 10);
|
||||
|
||||
// we checked the number format before
|
||||
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
|
||||
|
||||
if (errno != ERANGE)
|
||||
if (parse_integer_signed(num_begin, num_end, value_integer))
|
||||
{
|
||||
value_integer = static_cast<number_integer_t>(x);
|
||||
if (value_integer == x)
|
||||
{
|
||||
return token_type::value_integer;
|
||||
}
|
||||
return token_type::value_integer;
|
||||
}
|
||||
}
|
||||
|
||||
// this code is reached if we parse a floating-point number or if an
|
||||
// integer conversion above failed
|
||||
// integer conversion above overflowed. Prefer std::from_chars
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod.
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
if (parse_float_fast(num_begin, num_end, decimal_point_char, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
|
||||
// we checked the number format before
|
||||
@@ -1326,6 +1404,130 @@ scan_number_done:
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief contiguous fast path for scanning a number
|
||||
|
||||
Parses the whole number token straight from the input buffer, avoiding the
|
||||
per-character get()/add() of scan_number(). On success it fills token_buffer
|
||||
(with the locale decimal point substituted, as scan_number() does) and
|
||||
returns the token type. On anything it does not fully recognize as a
|
||||
well-formed number it makes no state change and returns
|
||||
token_type::uninitialized, so the caller falls back to scan_number(), which
|
||||
then produces the exact diagnostic. @a current is the first digit or the
|
||||
leading minus (already read); the remaining bytes are taken from the adapter.
|
||||
*/
|
||||
token_type scan_number_bulk_contiguous()
|
||||
{
|
||||
// a pending unget offsets the buffer position from current; fall back
|
||||
if (next_unget)
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
const std::size_t rem = ia.bulk_remaining();
|
||||
if (rem == 0)
|
||||
{
|
||||
// the first digit is the last input byte; let scan_number() finish
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
// the byte before the next unread one is current (contiguous input)
|
||||
const char* const data = reinterpret_cast<const char*>(ia.bulk_data()) - 1;
|
||||
const std::size_t avail = rem + 1;
|
||||
|
||||
// validate + classify the number extent (mirrors scan_number()'s grammar)
|
||||
std::size_t i = 0;
|
||||
std::size_t dot_index = std::string::npos;
|
||||
token_type number_type = token_type::value_unsigned;
|
||||
if (data[0] == '-')
|
||||
{
|
||||
number_type = token_type::value_integer;
|
||||
i = 1;
|
||||
if (i >= avail)
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
}
|
||||
if (data[i] == '0')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
else if (data[i] >= '1' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
if (i < avail && data[i] == '.')
|
||||
{
|
||||
number_type = token_type::value_float;
|
||||
dot_index = i;
|
||||
++i;
|
||||
if (i >= avail || !(data[i] >= '0' && data[i] <= '9'))
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
}
|
||||
if (i < avail && (data[i] == 'e' || data[i] == 'E'))
|
||||
{
|
||||
number_type = token_type::value_float;
|
||||
++i;
|
||||
if (i < avail && (data[i] == '+' || data[i] == '-'))
|
||||
{
|
||||
++i;
|
||||
}
|
||||
if (i >= avail || !(data[i] >= '0' && data[i] <= '9'))
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
}
|
||||
const std::size_t len = i;
|
||||
|
||||
// materialize the token exactly as scan_number() would, substituting the
|
||||
// locale decimal point so convert_number()'s strtof fallback stays valid.
|
||||
// reset() already cleared token_buffer, so append() fills it (assign() is
|
||||
// avoided because custom string_t types need not provide it)
|
||||
reset();
|
||||
token_buffer.append(reinterpret_cast<const typename string_t::value_type*>(data), len);
|
||||
if (dot_index != std::string::npos)
|
||||
{
|
||||
token_buffer[dot_index] = static_cast<typename string_t::value_type>(decimal_point_char);
|
||||
decimal_point_position = dot_index;
|
||||
}
|
||||
|
||||
// consume the remaining bytes of the number (current was already read)
|
||||
ia.bulk_skip(len - 1);
|
||||
position.chars_read_total += (len - 1);
|
||||
position.chars_read_current_line += (len - 1);
|
||||
|
||||
return convert_number(number_type);
|
||||
}
|
||||
|
||||
/// contiguous input: try the number fast path, else the byte-path scanner
|
||||
token_type scan_number_dispatch(std::true_type /*bulk*/)
|
||||
{
|
||||
const token_type t = scan_number_bulk_contiguous();
|
||||
return (t != token_type::uninitialized) ? t : scan_number();
|
||||
}
|
||||
|
||||
/// streaming input: always use the byte-path scanner
|
||||
token_type scan_number_dispatch(std::false_type /*bulk*/)
|
||||
{
|
||||
return scan_number();
|
||||
}
|
||||
|
||||
/*!
|
||||
@param[in] literal_text the literal text to expect
|
||||
@param[in] length the length of the passed literal text
|
||||
@@ -1680,7 +1882,7 @@ scan_number_done:
|
||||
case '7':
|
||||
case '8':
|
||||
case '9':
|
||||
return scan_number();
|
||||
return scan_number_dispatch(std::integral_constant<bool, bulk_scan> {});
|
||||
|
||||
// end of input (the null byte is needed when parsing from
|
||||
// string literals)
|
||||
|
||||
@@ -0,0 +1,302 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <limits> // numeric_limits
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
// std::from_chars lives in <charconv>, but being in C++17 mode does not
|
||||
// guarantee the header exists: GCC 7 sets __cplusplus to C++17 yet ships no
|
||||
// <charconv> (added in GCC 8; floating-point support in GCC 11). Guard the
|
||||
// include with __has_include so such toolchains fall back to the scalar path.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__has_include)
|
||||
#if __has_include(<charconv>)
|
||||
#include <charconv> // from_chars (only used when __cpp_lib_to_chars is defined)
|
||||
#include <system_error> // errc
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
/*!
|
||||
@brief fast integer parser for an already-validated unsigned integer
|
||||
|
||||
The number scanner has already checked that [first, last) is a valid JSON
|
||||
integer, so this only needs to accumulate the digits and detect overflow. This
|
||||
avoids the locale/errno machinery of std::strtoull, which dominates
|
||||
integer-heavy inputs.
|
||||
|
||||
@param[in] first pointer to the first character (a digit)
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] value the parsed value on success
|
||||
@return true if the value fit into @a NumberUnsignedType; false on overflow, in
|
||||
which case the caller falls back to floating-point parsing (matching the
|
||||
previous std::strtoull behavior)
|
||||
*/
|
||||
template<typename NumberUnsignedType>
|
||||
bool parse_integer_unsigned(const char* first, const char* last, NumberUnsignedType& value) noexcept
|
||||
{
|
||||
// accumulate in the widest unsigned type used by the previous strtoull
|
||||
// path so the overflow behavior is unchanged for custom number types
|
||||
std::uint64_t x = 0;
|
||||
constexpr std::uint64_t cutoff = (std::numeric_limits<std::uint64_t>::max)() / 10u;
|
||||
constexpr std::uint64_t cutlim = (std::numeric_limits<std::uint64_t>::max)() % 10u;
|
||||
for (const char* p = first; p != last; ++p)
|
||||
{
|
||||
const auto digit = static_cast<std::uint64_t>(static_cast<unsigned char>(*p) - static_cast<unsigned char>('0'));
|
||||
if (JSON_HEDLEY_UNLIKELY(x > cutoff || (x == cutoff && digit > cutlim)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
x = (x * 10u) + digit;
|
||||
}
|
||||
value = static_cast<NumberUnsignedType>(x);
|
||||
// reject values that do not round-trip into a narrower NumberUnsignedType
|
||||
return static_cast<std::uint64_t>(value) == x;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief fast integer parser for an already-validated negative integer
|
||||
|
||||
@param[in] first pointer to the leading '-'
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] value the parsed (negative) value on success
|
||||
@return true on success; false on overflow (caller falls back to float)
|
||||
*/
|
||||
template<typename NumberIntegerType>
|
||||
bool parse_integer_signed(const char* first, const char* last, NumberIntegerType& value) noexcept
|
||||
{
|
||||
// the state machine only reaches the signed path via a leading '-'
|
||||
JSON_ASSERT(first != last && *first == '-');
|
||||
std::uint64_t magnitude = 0;
|
||||
// |INT64_MIN| == INT64_MAX + 1; this is the largest admissible magnitude
|
||||
constexpr std::uint64_t limit = static_cast<std::uint64_t>((std::numeric_limits<std::int64_t>::max)()) + 1u;
|
||||
for (const char* p = first + 1; p != last; ++p)
|
||||
{
|
||||
const auto digit = static_cast<std::uint64_t>(static_cast<unsigned char>(*p) - static_cast<unsigned char>('0'));
|
||||
if (JSON_HEDLEY_UNLIKELY(magnitude > (limit - digit) / 10u))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
magnitude = (magnitude * 10u) + digit;
|
||||
}
|
||||
const std::int64_t x = (magnitude == limit)
|
||||
? (std::numeric_limits<std::int64_t>::min)()
|
||||
: -static_cast<std::int64_t>(magnitude);
|
||||
value = static_cast<NumberIntegerType>(x);
|
||||
// reject values that do not round-trip into a narrower NumberIntegerType
|
||||
return static_cast<std::int64_t>(value) == x;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief exact fast path for parsing a `double` (Clinger's algorithm)
|
||||
|
||||
For the common case - at most 19 significant digits, a decimal exponent in
|
||||
[-22, 22], and a significand below 2^53 - the value equals significand *
|
||||
10^exp computed in IEEE-754 double arithmetic, which is exact under
|
||||
round-to-nearest because both operands are exactly representable. This is the
|
||||
same fast path used by fast_float/simdjson; the general cases are left to
|
||||
std::strtod. The parser only activates for number_float_t == double; float and
|
||||
long double keep the std::strtof/std::strtold paths (see the templated overload
|
||||
below).
|
||||
|
||||
@param[in] first pointer to the first character of the number
|
||||
@param[in] last pointer past the last character
|
||||
@param[in] decimal_point the (locale-dependent) decimal point character
|
||||
@param[out] out the parsed value on success
|
||||
@return true if the value was parsed exactly; false to fall back to strtod
|
||||
*/
|
||||
template<typename DecimalPointType>
|
||||
bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept
|
||||
{
|
||||
#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0
|
||||
// Clinger's fast path is only exact when double operations are evaluated in
|
||||
// true double precision. On platforms that keep intermediates in extended
|
||||
// precision (e.g. the x87 FPU on 32-bit x86, where FLT_EVAL_METHOD == 2) the
|
||||
// single significand * 10^scale step is double-rounded and can be 1 ULP off,
|
||||
// so decline and let the caller fall back to the correctly-rounded
|
||||
// std::from_chars / std::strtod path.
|
||||
static_cast<void>(first);
|
||||
static_cast<void>(last);
|
||||
static_cast<void>(decimal_point);
|
||||
static_cast<void>(out);
|
||||
return false;
|
||||
#else
|
||||
static const std::array<double, 23> powers_of_ten =
|
||||
{
|
||||
{
|
||||
1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11,
|
||||
1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22
|
||||
}
|
||||
};
|
||||
|
||||
const char* p = first;
|
||||
bool negative = false;
|
||||
if (p != last && (*p == '-' || *p == '+'))
|
||||
{
|
||||
negative = (*p == '-');
|
||||
++p;
|
||||
}
|
||||
|
||||
std::uint64_t significand = 0;
|
||||
int num_digits = 0;
|
||||
int fractional_digits = 0;
|
||||
bool seen_dot = false;
|
||||
bool any_digit = false;
|
||||
for (; p != last; ++p)
|
||||
{
|
||||
const char c = *p;
|
||||
if (c >= '0' && c <= '9')
|
||||
{
|
||||
any_digit = true;
|
||||
if (JSON_HEDLEY_UNLIKELY(num_digits >= 19))
|
||||
{
|
||||
return false; // significand may not fit into uint64_t
|
||||
}
|
||||
significand = (significand * 10u) + static_cast<std::uint64_t>(c - '0');
|
||||
++num_digits;
|
||||
fractional_digits += static_cast<int>(seen_dot);
|
||||
}
|
||||
else if (static_cast<DecimalPointType>(c) == decimal_point)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(seen_dot))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
seen_dot = true;
|
||||
}
|
||||
else if (c == 'e' || c == 'E')
|
||||
{
|
||||
++p;
|
||||
break;
|
||||
}
|
||||
else
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!any_digit))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
int exponent = 0;
|
||||
if (p != last) // an exponent part remains
|
||||
{
|
||||
bool exp_negative = false;
|
||||
if (p != last && (*p == '-' || *p == '+'))
|
||||
{
|
||||
exp_negative = (*p == '-');
|
||||
++p;
|
||||
}
|
||||
bool any_exp_digit = false;
|
||||
for (; p != last; ++p)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(*p < '0' || *p > '9'))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
exponent = (exponent * 10) + (*p - '0');
|
||||
any_exp_digit = true;
|
||||
if (JSON_HEDLEY_UNLIKELY(exponent > 9999))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!any_exp_digit))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if (exp_negative)
|
||||
{
|
||||
exponent = -exponent;
|
||||
}
|
||||
}
|
||||
|
||||
const int scale = exponent - fractional_digits;
|
||||
if (JSON_HEDLEY_UNLIKELY(significand >= (static_cast<std::uint64_t>(1) << 53)))
|
||||
{
|
||||
return false; // significand not exactly representable as double
|
||||
}
|
||||
|
||||
auto result = static_cast<double>(significand);
|
||||
if (scale >= 0)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(scale > 22))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
result *= powers_of_ten[static_cast<std::size_t>(scale)];
|
||||
}
|
||||
else
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(-scale > 22))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
result /= powers_of_ten[static_cast<std::size_t>(-scale)];
|
||||
}
|
||||
out = negative ? -result : result;
|
||||
return true;
|
||||
#endif
|
||||
}
|
||||
|
||||
/// fast float path is only exact for `double`; decline for float/long double
|
||||
template<typename DecimalPointType, typename FloatType>
|
||||
bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse a float with std::from_chars (Eisel-Lemire) when available
|
||||
|
||||
std::from_chars is locale-independent, correctly rounded, and - via the
|
||||
Eisel-Lemire algorithm in modern standard libraries - much faster than strtod
|
||||
over the whole value range (not just the Clinger subset). It is used only when
|
||||
__cpp_lib_to_chars indicates full floating-point support and only when it
|
||||
consumes the entire token ([first, last)); a partial parse means the buffer
|
||||
uses a non-'.' locale decimal point, in which case the caller falls back to the
|
||||
locale-aware path. An under-/overflow (result_out_of_range) also declines, so
|
||||
the caller's strtod fallback supplies the well-defined ±inf/0 result the parser
|
||||
expects (side-stepping the P4168 divergence between implementations).
|
||||
|
||||
@return true if the value was parsed exactly and fully; false to fall back
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
// JSON_HAS_CPP_17 must gate the use as well as the <charconv> include above:
|
||||
// some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even
|
||||
// in C++14 mode, where <charconv> is not included.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars)
|
||||
const auto result = std::from_chars(first, last, out);
|
||||
return result.ec == std::errc() && result.ptr == last;
|
||||
#else
|
||||
static_cast<void>(first);
|
||||
static_cast<void>(last);
|
||||
static_cast<void>(out);
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -0,0 +1,237 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint64_t
|
||||
#include <cstring> // memcpy
|
||||
|
||||
#if defined(JSON_USE_SIMDUTF)
|
||||
// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in
|
||||
// external dependency: nlohmann/json itself stays header-only and the C++11
|
||||
// scalar validator below is always available; defining JSON_USE_SIMDUTF
|
||||
// additionally requires the simdutf headers on the include path and linking
|
||||
// the simdutf library. See string_bulk_run().
|
||||
#include <simdutf.h>
|
||||
#endif
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
// This file contains the byte-level string-scanning helpers used by the lexer's
|
||||
// contiguous fast path. They operate purely on raw bytes (no dependency on the
|
||||
// lexer's template parameters) so they are free functions, keeping the lexer
|
||||
// itself focused on the state machine; see lexer::scan_string_bulk().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
// classify a single byte as needing individual string handling: the closing
|
||||
// quote, an escape, a control character, or a non-ASCII (UTF-8)
|
||||
// lead/continuation byte. Ordinary bytes (0x20..0x7F except '"' and '\\') are
|
||||
// copied verbatim, which the bulk scanner does 8 bytes at a time.
|
||||
inline bool is_string_special(unsigned char c) noexcept
|
||||
{
|
||||
return c == '\"' || c == '\\' || c < 0x20u || c >= 0x80u;
|
||||
}
|
||||
|
||||
// SWAR helper: return a word whose high bit is set in every byte of @a v that
|
||||
// is_string_special(); zero if the 8 bytes are all ordinary.
|
||||
inline std::uint64_t swar_string_special(std::uint64_t v) noexcept
|
||||
{
|
||||
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
||||
const std::uint64_t has_quote = (q - ones) & ~q & high;
|
||||
const std::uint64_t has_backslash = (b - ones) & ~b & high;
|
||||
const std::uint64_t has_control = (v - 0x2020202020202020ull) & ~v & high; // < 0x20
|
||||
const std::uint64_t has_non_ascii = v & high; // >= 0x80
|
||||
return has_quote | has_backslash | has_control | has_non_ascii;
|
||||
}
|
||||
|
||||
// return the index of the first is_string_special() byte in [data, data+n), or
|
||||
// n if every byte is ordinary; scans 8 bytes at a time
|
||||
inline std::size_t find_string_special(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t word = 0;
|
||||
std::memcpy(&word, data + i, sizeof(word));
|
||||
if (swar_string_special(word) != 0)
|
||||
{
|
||||
// a special byte is in this word; locate it (endian-agnostic)
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
if (is_string_special(data[i + j]))
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
{
|
||||
if (is_string_special(data[i]))
|
||||
{
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its
|
||||
// length (2..4) only when the bytes form a *well-formed* sequence using exactly
|
||||
// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts
|
||||
// precisely what the byte path accepts. Returns 0 for anything that is invalid,
|
||||
// incomplete, or that the byte path must diagnose (the caller then defers to
|
||||
// that path, keeping error messages unchanged). Lead bytes < 0x80 are handled
|
||||
// by the caller and never passed here.
|
||||
inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avail) noexcept
|
||||
{
|
||||
const unsigned char c0 = data[0];
|
||||
if (c0 >= 0xC2 && c0 <= 0xDF) // U+0080..U+07FF
|
||||
{
|
||||
if (avail >= 2 && data[1] >= 0x80 && data[1] <= 0xBF)
|
||||
{
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xE0) // U+0800..U+0FFF
|
||||
{
|
||||
if (avail >= 3 && data[1] >= 0xA0 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||
{
|
||||
return 3;
|
||||
}
|
||||
}
|
||||
else if ((c0 >= 0xE1 && c0 <= 0xEC) || c0 == 0xEE || c0 == 0xEF) // U+1000..U+CFFF, U+E000..U+FFFF
|
||||
{
|
||||
if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||
{
|
||||
return 3;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xED) // U+D000..U+D7FF (excludes surrogates)
|
||||
{
|
||||
if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0x9F && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||
{
|
||||
return 3;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xF0) // U+10000..U+3FFFF
|
||||
{
|
||||
if (avail >= 4 && data[1] >= 0x90 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||
{
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
else if (c0 >= 0xF1 && c0 <= 0xF3) // U+40000..U+FFFFF
|
||||
{
|
||||
if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||
{
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xF4) // U+100000..U+10FFFF
|
||||
{
|
||||
if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0x8F && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||
{
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
return 0; // invalid, incomplete, or must be diagnosed by the byte path
|
||||
}
|
||||
|
||||
// Scalar (C++11) computation of the bulk run length: the number of leading
|
||||
// bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8
|
||||
// sequences, stopping before the first byte that needs individual handling (the
|
||||
// closing quote, an escape, a control character, or an ill-formed/truncated
|
||||
// sequence). ASCII is skipped 8 bytes at a time.
|
||||
inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
std::size_t pos = 0;
|
||||
while (pos < n)
|
||||
{
|
||||
pos += find_string_special(data + pos, n - pos);
|
||||
if (pos >= n || data[pos] < 0x80u)
|
||||
{
|
||||
break; // end of buffer, or a quote/escape/control byte
|
||||
}
|
||||
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
||||
if (seq == 0)
|
||||
{
|
||||
break; // ill-formed or truncated: let the byte path diagnose it
|
||||
}
|
||||
pos += seq;
|
||||
}
|
||||
return pos;
|
||||
}
|
||||
|
||||
#if defined(JSON_USE_SIMDUTF)
|
||||
// Index of the first quote/escape/control byte in [data, data+n) (non-ASCII
|
||||
// bytes are *not* stops here - the whole run is handed to simdutf), or n.
|
||||
inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, data + i, sizeof(v));
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull;
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull;
|
||||
const std::uint64_t hit = ((q - ones) & ~q & high)
|
||||
| ((b - ones) & ~b & high)
|
||||
| ((v - 0x2020202020202020ull) & ~v & high);
|
||||
if (hit != 0)
|
||||
{
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
const unsigned char c = data[i + j];
|
||||
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
{
|
||||
const unsigned char c = data[i];
|
||||
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||
{
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Backend-dispatched bulk run length. With JSON_USE_SIMDUTF the run up to the
|
||||
// next delimiter is validated in one shot by simdutf; on the rare failure the
|
||||
// scalar helper recomputes the exact valid prefix so the byte path still
|
||||
// produces the precise diagnostic. Without it, the pure scalar path is used.
|
||||
inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
#if defined(JSON_USE_SIMDUTF)
|
||||
const std::size_t run = find_string_delimiter(data, n);
|
||||
if (run != 0 && simdutf::validate_utf8(reinterpret_cast<const char*>(data), run))
|
||||
{
|
||||
return run;
|
||||
}
|
||||
return scalar_string_bulk_run(data, n);
|
||||
#else
|
||||
return scalar_string_bulk_run(data, n);
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -13,11 +13,15 @@
|
||||
#include <tuple> // tuple
|
||||
#include <type_traits> // false_type, is_constructible, is_integral, is_same, true_type
|
||||
#include <utility> // declval
|
||||
#include <vector> // vector
|
||||
#if defined(__cpp_lib_byte) && __cpp_lib_byte >= 201603L
|
||||
#include <cstddef> // byte
|
||||
#endif
|
||||
#include <nlohmann/detail/iterators/iterator_traits.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#ifdef JSON_HAS_CPP_17
|
||||
#include <optional> // optional
|
||||
#endif
|
||||
#include <nlohmann/detail/meta/call_std/begin.hpp>
|
||||
#include <nlohmann/detail/meta/call_std/end.hpp>
|
||||
#include <nlohmann/detail/meta/cpp_future.hpp>
|
||||
@@ -450,6 +454,51 @@ struct is_constructible_string_type
|
||||
value_type_t, laundered_type >>::value;
|
||||
};
|
||||
|
||||
// Forward declarations: iteration_proxy.hpp includes this file, so we cannot
|
||||
// include it here.
|
||||
template<typename IteratorType> class iteration_proxy;
|
||||
template<typename IteratorType> class iteration_proxy_value;
|
||||
|
||||
// Identifies nlohmann's internal iteration-proxy types. These must be excluded
|
||||
// before evaluating any std::ranges concept to avoid circular constraints.
|
||||
template<typename T> struct is_iteration_proxy_type : std::false_type {};
|
||||
template<typename T> struct is_iteration_proxy_type<iteration_proxy<T>> : std::true_type {};
|
||||
template<typename T> struct is_iteration_proxy_type<iteration_proxy_value<T>> : std::true_type {};
|
||||
|
||||
// In C++26, std::optional satisfies std::ranges::view; exclude it so the
|
||||
// range-view overload does not hijack the optional serializer.
|
||||
#ifdef JSON_HAS_CPP_17
|
||||
template<typename T> struct is_range_view_optional_type : std::false_type {};
|
||||
template<typename T> struct is_range_view_optional_type<std::optional<T>> : std::true_type {};
|
||||
#else
|
||||
template<typename T> struct is_range_view_optional_type : std::false_type {};
|
||||
#endif
|
||||
|
||||
// std::ranges does not work properly on MinGW due to incomplete C++20 support
|
||||
// see https://github.com/nlohmann/json/issues/4916
|
||||
#if JSON_HAS_RANGES && !defined(__MINGW32__)
|
||||
|
||||
// SafeToCheck guards against types that trigger circular constraints when
|
||||
// std::ranges::view<T> is evaluated on GCC 12 / libstdc++ 12:
|
||||
// - iteration_proxy / iteration_proxy_value directly
|
||||
// - views wrapping the above (e.g. owning_view<iteration_proxy<...>>)
|
||||
// - views wrapping basic_json (e.g. ref_view<json>) — same circularity
|
||||
// via json's constructors → is_compatible_array_type → here
|
||||
// nlohmann's plain range_value_t (iterator_traits-based) is safe to call
|
||||
// before any std::ranges concept is touched, so we use it for the checks.
|
||||
template < typename T, bool SafeToCheck =
|
||||
!is_iteration_proxy_type<T>::value &&
|
||||
!is_iteration_proxy_type<detected_t<range_value_t, T>>::value &&
|
||||
!is_basic_json<detected_t<range_value_t, T>>::value &&
|
||||
!is_range_view_optional_type<T>::value >
|
||||
struct is_compatible_range_view : std::false_type {};
|
||||
|
||||
template<typename T>
|
||||
struct is_compatible_range_view<T, true>
|
||||
: std::bool_constant<std::ranges::view<T>> {};
|
||||
|
||||
#endif
|
||||
|
||||
template<typename BasicJsonType, typename CompatibleArrayType, typename = void>
|
||||
struct is_compatible_array_type_impl : std::false_type {};
|
||||
|
||||
@@ -461,13 +510,38 @@ struct is_compatible_array_type_impl <
|
||||
is_iterator_traits<iterator_traits<detected_t<iterator_t, CompatibleArrayType>>>::value&&
|
||||
// special case for types like std::filesystem::path whose iterator's value_type are themselves
|
||||
// c.f. https://github.com/nlohmann/json/pull/3073
|
||||
!std::is_same<CompatibleArrayType, detected_t<range_value_t, CompatibleArrayType>>::value >>
|
||||
!std::is_same<CompatibleArrayType, detected_t<range_value_t, CompatibleArrayType>>::value
|
||||
// When range-view support is enabled, std::ranges::view types (e.g. std::string_view,
|
||||
// filter_view) can match BOTH this iterator-based specialization AND the view-based one
|
||||
// below, causing ambiguity. Exclude views here so the two specializations are mutually
|
||||
// exclusive: this one handles plain iterable containers, the other handles views.
|
||||
#if JSON_HAS_RANGES && !defined(__MINGW32__)
|
||||
&& !is_compatible_range_view<CompatibleArrayType>::value
|
||||
#endif
|
||||
>>
|
||||
{
|
||||
static constexpr bool value =
|
||||
is_constructible<BasicJsonType,
|
||||
range_value_t<CompatibleArrayType>>::value;
|
||||
};
|
||||
|
||||
#if JSON_HAS_RANGES && !defined(__MINGW32__)
|
||||
template<typename BasicJsonType, typename CompatibleArrayType>
|
||||
struct is_compatible_array_type_impl <
|
||||
BasicJsonType, CompatibleArrayType,
|
||||
enable_if_t < is_compatible_range_view<CompatibleArrayType>::value
|
||||
&& !std::is_same<detected_t<range_value_t, CompatibleArrayType>, char>::value
|
||||
&& !std::is_same<detected_t<range_value_t, CompatibleArrayType>, wchar_t>::value >>
|
||||
{
|
||||
// CompatibleArrayType is a std::ranges::view here, so std::ranges::range_value_t
|
||||
// is safe and correctly handles C++20 iterators that may lack classic iterator_traits.
|
||||
static constexpr bool value =
|
||||
is_constructible<BasicJsonType,
|
||||
std::ranges::range_value_t<CompatibleArrayType>>::value;
|
||||
};
|
||||
#endif
|
||||
|
||||
|
||||
template<typename BasicJsonType, typename CompatibleArrayType>
|
||||
struct is_compatible_array_type
|
||||
: is_compatible_array_type_impl<BasicJsonType, CompatibleArrayType> {};
|
||||
@@ -559,6 +633,14 @@ template<typename BasicJsonType, typename CompatibleType>
|
||||
struct is_compatible_type
|
||||
: is_compatible_type_impl<BasicJsonType, CompatibleType> {};
|
||||
|
||||
template<typename BasicJsonType, typename CompatibleArrayType>
|
||||
struct is_compatible_binary_type
|
||||
{
|
||||
static constexpr bool value =
|
||||
std::is_same<typename BasicJsonType::binary_t::container_type, CompatibleArrayType>::value &&
|
||||
!std::is_same<typename BasicJsonType::binary_t::container_type, std::vector<std::uint8_t>>::value;
|
||||
};
|
||||
|
||||
template<typename BasicJsonType, typename CompatibleReferenceType>
|
||||
struct is_compatible_reference_type_impl
|
||||
{
|
||||
|
||||
+32
-24
@@ -4083,12 +4083,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
return result;
|
||||
}
|
||||
|
||||
/// @brief deserialize from a pair of character iterators
|
||||
/// @brief deserialize from a pair of character iterators (or an iterator+sentinel pair, C++20 ranges support)
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/parse/
|
||||
template<typename IteratorType>
|
||||
template<typename IteratorType, typename SentinelType = IteratorType,
|
||||
detail::enable_if_t<detail::can_compare_ne<IteratorType, SentinelType>::value, int> = 0>
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json parse(IteratorType first,
|
||||
IteratorType last,
|
||||
SentinelType last,
|
||||
parser_callback_t cb = nullptr,
|
||||
const bool allow_exceptions = true,
|
||||
const bool ignore_comments = false,
|
||||
@@ -4122,10 +4123,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||
}
|
||||
|
||||
/// @brief check if the input is valid JSON
|
||||
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/accept/
|
||||
template<typename IteratorType>
|
||||
static bool accept(IteratorType first, IteratorType last,
|
||||
template<typename IteratorType, typename SentinelType = IteratorType,
|
||||
detail::enable_if_t<detail::can_compare_ne<IteratorType, SentinelType>::value, int> = 0>
|
||||
static bool accept(IteratorType first, SentinelType last,
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas = false)
|
||||
{
|
||||
@@ -4157,11 +4159,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX>(std::move(ia), format).sax_parse(format, sax, strict);
|
||||
}
|
||||
|
||||
/// @brief generate SAX events
|
||||
/// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/sax_parse/
|
||||
template<class IteratorType, class SAX>
|
||||
template<class IteratorType, class SAX, class SentinelType = IteratorType,
|
||||
detail::enable_if_t<detail::can_compare_ne<IteratorType, SentinelType>::value, int> = 0>
|
||||
JSON_HEDLEY_NON_NULL(3)
|
||||
static bool sax_parse(IteratorType first, IteratorType last, SAX* sax,
|
||||
static bool sax_parse(IteratorType first, SentinelType last, SAX* sax,
|
||||
input_format_t format = input_format_t::json,
|
||||
const bool strict = true,
|
||||
const bool ignore_comments = false,
|
||||
@@ -4461,11 +4464,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
return res ? result : basic_json(value_t::discarded);
|
||||
}
|
||||
|
||||
/// @brief create a JSON value from an input in CBOR format
|
||||
/// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/from_cbor/
|
||||
template<typename IteratorType>
|
||||
template<typename IteratorType, typename SentinelType = IteratorType,
|
||||
detail::enable_if_t<detail::can_compare_ne<IteratorType, SentinelType>::value, int> = 0>
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_cbor(IteratorType first, IteratorType last,
|
||||
static basic_json from_cbor(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true,
|
||||
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error)
|
||||
@@ -4518,11 +4522,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
return res ? result : basic_json(value_t::discarded);
|
||||
}
|
||||
|
||||
/// @brief create a JSON value from an input in MessagePack format
|
||||
/// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/from_msgpack/
|
||||
template<typename IteratorType>
|
||||
template<typename IteratorType, typename SentinelType = IteratorType,
|
||||
detail::enable_if_t<detail::can_compare_ne<IteratorType, SentinelType>::value, int> = 0>
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_msgpack(IteratorType first, IteratorType last,
|
||||
static basic_json from_msgpack(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
{
|
||||
@@ -4572,11 +4577,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
return res ? result : basic_json(value_t::discarded);
|
||||
}
|
||||
|
||||
/// @brief create a JSON value from an input in UBJSON format
|
||||
/// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/from_ubjson/
|
||||
template<typename IteratorType>
|
||||
template<typename IteratorType, typename SentinelType = IteratorType,
|
||||
detail::enable_if_t<detail::can_compare_ne<IteratorType, SentinelType>::value, int> = 0>
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_ubjson(IteratorType first, IteratorType last,
|
||||
static basic_json from_ubjson(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
{
|
||||
@@ -4626,11 +4632,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
return res ? result : basic_json(value_t::discarded);
|
||||
}
|
||||
|
||||
/// @brief create a JSON value from an input in BJData format
|
||||
/// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/from_bjdata/
|
||||
template<typename IteratorType>
|
||||
template<typename IteratorType, typename SentinelType = IteratorType,
|
||||
detail::enable_if_t<detail::can_compare_ne<IteratorType, SentinelType>::value, int> = 0>
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_bjdata(IteratorType first, IteratorType last,
|
||||
static basic_json from_bjdata(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
{
|
||||
@@ -4656,11 +4663,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
return res ? result : basic_json(value_t::discarded);
|
||||
}
|
||||
|
||||
/// @brief create a JSON value from an input in BSON format
|
||||
/// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/from_bson/
|
||||
template<typename IteratorType>
|
||||
template<typename IteratorType, typename SentinelType = IteratorType,
|
||||
detail::enable_if_t<detail::can_compare_ne<IteratorType, SentinelType>::value, int> = 0>
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_bson(IteratorType first, IteratorType last,
|
||||
static basic_json from_bson(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
{
|
||||
|
||||
+1118
-82
File diff suppressed because it is too large
Load Diff
@@ -30,4 +30,17 @@ inline std::vector<std::uint8_t> read_binary_file(const std::string& filename)
|
||||
return byte_vector;
|
||||
}
|
||||
|
||||
// sentinel for istreambuf_iterator; compares != true until EOF is reached
|
||||
// lets tests read a file directly via the new iterator+sentinel overloads
|
||||
// instead of buffering the whole file into a vector first.
|
||||
// Only the iterator-first direction (it != sentinel) is ever evaluated by
|
||||
// the library's parse loop, so no reversed-order overload is needed.
|
||||
struct istreambuf_sentinel
|
||||
{
|
||||
friend bool operator!=(const std::istreambuf_iterator<char>& it, const istreambuf_sentinel& /*unused*/) noexcept
|
||||
{
|
||||
return it != std::istreambuf_iterator<char>();
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace utils
|
||||
|
||||
@@ -3699,6 +3699,15 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("Parse BJData directly from a file using iterator and sentinel")
|
||||
{
|
||||
std::string const filename = TEST_DATA_DIRECTORY "/json_testsuite/sample.json.bjdata";
|
||||
std::ifstream file(filename, std::ios::binary);
|
||||
const std::istreambuf_iterator<char> first(file);
|
||||
const json parsed = json::from_bjdata(first, utils::istreambuf_sentinel{});
|
||||
CHECK((parsed.is_object() || parsed.is_array()));
|
||||
}
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
TEST_CASE("all BJData first bytes")
|
||||
{
|
||||
|
||||
@@ -1208,6 +1208,19 @@ TEST_CASE("BSON numerical data")
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("Parse BSON directly from a file using iterator and sentinel")
|
||||
{
|
||||
std::string const filename = TEST_DATA_DIRECTORY "/json.org/1.json";
|
||||
|
||||
std::ifstream f_json(filename);
|
||||
const json expected = json::parse(f_json);
|
||||
|
||||
std::ifstream file(filename + ".bson", std::ios::binary);
|
||||
const std::istreambuf_iterator<char> first(file);
|
||||
const json parsed = json::from_bson(first, utils::istreambuf_sentinel{});
|
||||
CHECK(parsed == expected);
|
||||
}
|
||||
|
||||
TEST_CASE("BSON roundtrips" * doctest::skip())
|
||||
{
|
||||
SECTION("reference files")
|
||||
|
||||
@@ -1921,6 +1921,15 @@ TEST_CASE("single CBOR roundtrip")
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("Parse CBOR directly from a file using iterator and sentinel")
|
||||
{
|
||||
std::string const filename = TEST_DATA_DIRECTORY "/json_testsuite/sample.json.cbor";
|
||||
std::ifstream file(filename, std::ios::binary);
|
||||
const std::istreambuf_iterator<char> first(file);
|
||||
const json parsed = json::from_cbor(first, utils::istreambuf_sentinel{});
|
||||
CHECK((parsed.is_object() || parsed.is_array()));
|
||||
}
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
TEST_CASE("CBOR regressions")
|
||||
{
|
||||
|
||||
@@ -12,6 +12,10 @@
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <sstream> // stringstream
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
|
||||
namespace
|
||||
{
|
||||
// shortcut to scan a string literal
|
||||
@@ -224,3 +228,75 @@ TEST_CASE("lexer class")
|
||||
CHECK((scan_string("/**//**//**/", true) == json::lexer::token_type::end_of_input));
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("lexer number fast path")
|
||||
{
|
||||
// The contiguous fast path (used for pointer/string input) must agree with
|
||||
// the streaming byte path (used for std::istream) on token type, numeric
|
||||
// value, and round-trip text for every well-formed number, and reject the
|
||||
// same malformed numbers with the same message.
|
||||
SECTION("contiguous vs streaming parity")
|
||||
{
|
||||
const std::vector<std::string> numbers =
|
||||
{
|
||||
"0", "-0", "1", "-1", "42", "-42", "10", "100", "1234567890",
|
||||
"0.0", "-0.0", "3.14", "-3.14", "0.5", "-0.001", "123.456789",
|
||||
"1e0", "1E0", "1e10", "1e-10", "1e+10", "1.5e3", "-2.5E-4",
|
||||
"9223372036854775807", // INT64_MAX -> unsigned
|
||||
"9223372036854775808", // INT64_MAX + 1 -> unsigned
|
||||
"18446744073709551615", // UINT64_MAX -> unsigned
|
||||
"18446744073709551616", // UINT64_MAX + 1 -> float
|
||||
"-9223372036854775808", // INT64_MIN -> integer
|
||||
"-9223372036854775809", // INT64_MIN - 1 -> float
|
||||
"123456789012345678901234567890", // huge -> float
|
||||
"0.30000000000000004", "2.2250738585072014e-308", "1e308",
|
||||
// high-precision / wide-exponent values that exercise the
|
||||
// std::from_chars (Eisel-Lemire) path beyond the Clinger subset
|
||||
"1.7976931348623157e308", "1.2345678901234567e-250",
|
||||
"9007199254740993", "5e-324", "1e-320"
|
||||
};
|
||||
|
||||
for (const auto& n : numbers)
|
||||
{
|
||||
const std::string doc = "[" + n + "]";
|
||||
|
||||
// contiguous fast path
|
||||
const json a = json::parse(doc);
|
||||
// streaming byte path
|
||||
std::stringstream ss(doc);
|
||||
const json b = json::parse(ss);
|
||||
|
||||
CAPTURE(n);
|
||||
CHECK(a == b);
|
||||
CHECK(a.dump() == b.dump());
|
||||
CHECK(a[0].type() == b[0].type());
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("token type classification")
|
||||
{
|
||||
CHECK((scan_string("0") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("-1") == json::lexer::token_type::value_integer));
|
||||
CHECK((scan_string("1.5") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("1e5") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("18446744073709551615") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("18446744073709551616") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("-9223372036854775808") == json::lexer::token_type::value_integer));
|
||||
CHECK((scan_string("-9223372036854775809") == json::lexer::token_type::value_float));
|
||||
}
|
||||
|
||||
SECTION("malformed numbers are rejected identically")
|
||||
{
|
||||
for (const char* bad :
|
||||
{"-", "1.", "1e", "1e+", "1.2e", "01", "-01", "1..2", "1.2.3"
|
||||
})
|
||||
{
|
||||
CAPTURE(bad);
|
||||
// the contiguous fast path must decline and let the byte path report
|
||||
const std::string doc = std::string("[") + bad + "]";
|
||||
CHECK_FALSE(json::accept(doc));
|
||||
std::stringstream ss(doc);
|
||||
CHECK_FALSE(json::accept(ss));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1782,6 +1782,21 @@ TEST_CASE("std::optional")
|
||||
"[json.exception.type_error.302] type must be string, but is null", json::type_error&);
|
||||
CHECK_THROWS_WITH_AS(std::optional<int>(j_null),
|
||||
"[json.exception.type_error.302] type must be number, but is null", json::type_error&);
|
||||
|
||||
// Assignment goes through the same overload resolution as direct
|
||||
// construction, so it throws for the same reason. This relies on
|
||||
// basic_json's implicit conversion operator, so it only applies
|
||||
// when JSON_USE_IMPLICIT_CONVERSIONS is enabled (the default).
|
||||
#if JSON_USE_IMPLICIT_CONVERSIONS
|
||||
std::optional<std::string> opt_assign;
|
||||
CHECK_THROWS_WITH_AS(opt_assign = j_null,
|
||||
"[json.exception.type_error.302] type must be string, but is null", json::type_error&);
|
||||
#endif
|
||||
|
||||
// get_to() is the correct way to obtain std::nullopt from a JSON null.
|
||||
std::optional<std::string> opt_get_to = "placeholder";
|
||||
j_null.get_to(opt_get_to);
|
||||
CHECK(opt_get_to == std::nullopt);
|
||||
}
|
||||
|
||||
SECTION("string")
|
||||
|
||||
@@ -1641,6 +1641,15 @@ TEST_CASE("single MessagePack roundtrip")
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("Parse MessagePack directly from a file using iterator and sentinel")
|
||||
{
|
||||
std::string const filename = TEST_DATA_DIRECTORY "/json_testsuite/sample.json.msgpack";
|
||||
std::ifstream file(filename, std::ios::binary);
|
||||
const std::istreambuf_iterator<char> first(file);
|
||||
const json parsed = json::from_msgpack(first, utils::istreambuf_sentinel{});
|
||||
CHECK((parsed.is_object() || parsed.is_array()));
|
||||
}
|
||||
|
||||
TEST_CASE("MessagePack roundtrips" * doctest::skip())
|
||||
{
|
||||
SECTION("input from msgpack-python")
|
||||
|
||||
@@ -1136,6 +1136,40 @@ TEST_CASE("regression tests 2")
|
||||
CHECK((decoded == json_4804::array()));
|
||||
}
|
||||
|
||||
SECTION("discussion #4209 - custom BinaryType direct assignment and round-tripping")
|
||||
{
|
||||
// Test that assigning a custom BinaryType directly creates a binary value, not an array
|
||||
const std::vector<std::byte> original{std::byte{1}, std::byte{2}, std::byte{3}};
|
||||
const json_4804 j = original;
|
||||
CHECK(j.is_binary());
|
||||
CHECK(!j.is_array());
|
||||
|
||||
// Test round-tripping: extracting the binary value back as the custom container type
|
||||
const auto extracted = j.get<std::vector<std::byte>>();
|
||||
CHECK(extracted == original);
|
||||
|
||||
// Test that the default json alias behavior is unchanged: std::vector<uint8_t> -> array
|
||||
const json default_json = std::vector<std::uint8_t> {1, 2, 3};
|
||||
CHECK(default_json.is_array());
|
||||
CHECK(!default_json.is_binary());
|
||||
}
|
||||
|
||||
SECTION("discussion #4209 - custom BinaryType extraction from parsed array")
|
||||
{
|
||||
// Test that extracting a custom BinaryType from a parsed JSON array still works
|
||||
// (not just from a binary-typed node)
|
||||
const auto j = json_4804::parse("[1,2,3]");
|
||||
CHECK(j.is_array());
|
||||
CHECK(!j.is_binary());
|
||||
|
||||
// Extracting as custom BinaryType should work from arrays
|
||||
const auto extracted = j.get<std::vector<std::byte>>();
|
||||
CHECK(extracted.size() == 3);
|
||||
CHECK(extracted[0] == std::byte{1});
|
||||
CHECK(extracted[1] == std::byte{2});
|
||||
CHECK(extracted[2] == std::byte{3});
|
||||
}
|
||||
|
||||
SECTION("issue #5046 - implicit conversion of return json to std::optional no longer implicit")
|
||||
{
|
||||
const json jval{};
|
||||
@@ -1165,6 +1199,46 @@ TEST_CASE("regression tests 2")
|
||||
}
|
||||
#endif
|
||||
|
||||
#if JSON_HAS_RANGES && !defined(__MINGW32__)
|
||||
SECTION("issue #4916 - constructing array from C++20 ranges view does not work")
|
||||
{
|
||||
std::vector<int> nums{1, 2, 37, 42, 21};
|
||||
auto filteredNums = nums | std::views::filter([](int i)
|
||||
{
|
||||
return i > 10;
|
||||
});
|
||||
json const j(filteredNums);
|
||||
CHECK(j.type() == json::value_t::array);
|
||||
CHECK(j == json({37, 42, 21}));
|
||||
}
|
||||
#endif
|
||||
|
||||
// owning_view is not available in libstdc++ < 12
|
||||
#if JSON_HAS_RANGES && !defined(__MINGW32__) && !(defined(__GLIBCXX__) && _GLIBCXX_RELEASE < 12)
|
||||
SECTION("issue #4916 - constructing array from prvalue C++20 ranges view (owning_view)")
|
||||
{
|
||||
json const j(std::vector<int> {1, 2, 37, 42, 21} | std::views::filter([](int i)
|
||||
{
|
||||
return i > 10;
|
||||
}));
|
||||
CHECK(j.type() == json::value_t::array);
|
||||
CHECK(j == json({37, 42, 21}));
|
||||
}
|
||||
#endif
|
||||
|
||||
#if JSON_HAS_RANGES && !defined(__MINGW32__)
|
||||
SECTION("issue #4916 - constructing array from C++20 transform view (prvalue elements)")
|
||||
{
|
||||
std::vector<int> nums{1, 2, 3};
|
||||
auto t = nums | std::views::transform([](int i) noexcept
|
||||
{
|
||||
return i * 2;
|
||||
});
|
||||
json const j(t);
|
||||
CHECK(j.type() == json::value_t::array);
|
||||
CHECK(j == json({2, 4, 6}));
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
TEST_CASE_TEMPLATE("issue #4798 - nlohmann::json::to_msgpack() encode float NaN as double", T, double, float) // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization)
|
||||
|
||||
@@ -2393,6 +2393,15 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("Parse UBJSON directly from a file using iterator and sentinel")
|
||||
{
|
||||
std::string const filename = TEST_DATA_DIRECTORY "/json_testsuite/sample.json.ubjson";
|
||||
std::ifstream file(filename, std::ios::binary);
|
||||
const std::istreambuf_iterator<char> first(file);
|
||||
const json parsed = json::from_ubjson(first, utils::istreambuf_sentinel{});
|
||||
CHECK((parsed.is_object() || parsed.is_array()));
|
||||
}
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
TEST_CASE("all UBJSON first bytes")
|
||||
{
|
||||
|
||||
@@ -6,6 +6,13 @@
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
// cmake/test.cmake selects the C++ standard versions with which to build a
|
||||
// unit test based on the presence of JSON_HAS_CPP_<VERSION> macros.
|
||||
// When using macros that are only defined for particular versions of the standard
|
||||
// (e.g., JSON_HAS_FILESYSTEM for C++17 and up), please mention the corresponding
|
||||
// version macro in a comment close by, like this:
|
||||
// JSON_HAS_CPP_<VERSION> (do not remove; see note at top of file)
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
@@ -13,6 +20,10 @@ using nlohmann::json;
|
||||
|
||||
#include <list>
|
||||
|
||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
#include <iterator>
|
||||
#endif
|
||||
|
||||
namespace
|
||||
{
|
||||
TEST_CASE("Use arbitrary stdlib container")
|
||||
@@ -168,4 +179,55 @@ TEST_CASE("Custom iterator")
|
||||
CHECK(as_json.at(3) == 4);
|
||||
}
|
||||
|
||||
// Custom sentinel type for testing heterogeneous iterator+sentinel support
|
||||
struct CustomSentinel
|
||||
{
|
||||
const char* end_ptr;
|
||||
|
||||
// only the iterator-first direction (it != sentinel) is ever evaluated by
|
||||
// the library's parse loop; a reversed-order overload would go unused and
|
||||
// trip -Wunneeded-internal-declaration under -Weverything
|
||||
friend bool operator!=(const char* it, const CustomSentinel& sentinel)
|
||||
{
|
||||
return it != sentinel.end_ptr;
|
||||
}
|
||||
};
|
||||
|
||||
TEST_CASE("Parse with heterogeneous iterator and sentinel types")
|
||||
{
|
||||
const std::string json_str = R"({"key":"value"})";
|
||||
const char* end_ptr = json_str.data() + json_str.size();
|
||||
|
||||
// Parse using pointer and sentinel (different types)
|
||||
json j = json::parse(json_str.data(), CustomSentinel{end_ptr});
|
||||
CHECK(j["key"] == "value");
|
||||
|
||||
// Accept using pointer and sentinel
|
||||
CHECK(json::accept(json_str.data(), CustomSentinel{end_ptr}));
|
||||
|
||||
// Test that the same-type case still works
|
||||
std::string raw_data = R"([1,2,3])";
|
||||
std::list<char> data(raw_data.begin(), raw_data.end());
|
||||
json j2 = json::parse(data.begin(), data.end());
|
||||
CHECK(j2.at(0) == 1);
|
||||
}
|
||||
|
||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
|
||||
TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t")
|
||||
{
|
||||
using iterator_type = std::string::const_iterator;
|
||||
const std::string json_str = R"({"key":"value","array":[1,2,3]})";
|
||||
const auto len = static_cast<std::iter_difference_t<iterator_type>>(json_str.size());
|
||||
|
||||
const std::counted_iterator<iterator_type> first(json_str.begin(), len);
|
||||
const json j = json::parse(first, std::default_sentinel);
|
||||
CHECK(j["key"] == "value");
|
||||
CHECK(j["array"].size() == 3);
|
||||
|
||||
const std::counted_iterator<iterator_type> first2(json_str.begin(), len);
|
||||
CHECK(json::accept(first2, std::default_sentinel));
|
||||
}
|
||||
#endif
|
||||
|
||||
} // namespace
|
||||
|
||||
@@ -53,6 +53,27 @@ TEST_CASE("wide strings")
|
||||
std::wstring const w = L"\"\xDBFF";
|
||||
json _;
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
|
||||
// the exact message depends on the width of wchar_t: a 16-bit
|
||||
// wchar_t passes the lone surrogate to the UTF-8 decoder unchanged
|
||||
// (rejected as a single ill-formed byte at column 2), while a
|
||||
// 32-bit wchar_t first encodes it as an ill-formed three-byte
|
||||
// sequence (rejected one byte later, at column 3)
|
||||
const char* const error_low_surrogate = sizeof(wchar_t) == 2
|
||||
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
||||
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'";
|
||||
const char* const error_high_surrogate = sizeof(wchar_t) == 2
|
||||
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
||||
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'";
|
||||
|
||||
// a lone low surrogate cannot start a pair
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'"'}), error_low_surrogate, json::parse_error&);
|
||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&);
|
||||
// a lone low surrogate must not swallow the following unit: pairing
|
||||
// it with any second unit would produce valid UTF-8, so the error
|
||||
// has to report an ill-formed byte at the surrogate's own position
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -68,11 +89,22 @@ TEST_CASE("wide strings")
|
||||
|
||||
SECTION("invalid std::u16string")
|
||||
{
|
||||
if (wstring_is_utf16())
|
||||
if (u16string_is_utf16())
|
||||
{
|
||||
std::u16string const w = u"\"\xDBFF";
|
||||
json _;
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
|
||||
// a lone low surrogate cannot start a pair
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a lone low surrogate must not swallow the following unit: pairing
|
||||
// it with any second unit would produce valid UTF-8, so the error
|
||||
// has to report an ill-formed byte at the surrogate's own position
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a valid surrogate pair is still decoded (U+1F600)
|
||||
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user