Merge remote-tracking branch 'origin/develop' into claude/fix-issue-3989-db7e45

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

# Conflicts:
#	include/nlohmann/detail/input/binary_reader.hpp
#	include/nlohmann/detail/string_utils.hpp
#	single_include/nlohmann/json.hpp
This commit is contained in:
Niels Lohmann
2026-10-04 12:22:05 +02:00
82 changed files with 7749 additions and 2474 deletions
-9
View File
@@ -1,18 +1,9 @@
# bugprone-use-after-move (hicpp-invalid-access-moved is its alias) still flags
# the basic_json move constructor, which forwards the whole object to its base
# class (#5724), and two forwards in the error-message construction of
# at(KeyType&&) (json.hpp, both overloads: find(std::forward<KeyType>(key))
# followed by string_t(std::forward<KeyType>(key)) in the throw), which #5689
# rewrites. Re-enable both checks once those changes have landed.
# portability-avoid-pragma-once: kept disabled on purpose. #pragma once is accepted
# by every supported compiler, and tools/amalgamate/amalgamate.py strips it from
# single_include, so there is nothing left to fix here.
Checks: '*,
-bugprone-use-after-move,
-hicpp-invalid-access-moved,
-altera-id-dependent-backward-branch,
-altera-struct-pack-align,
-altera-unroll-loops,
+1 -1
View File
@@ -53,11 +53,11 @@ cc_library(
"include/nlohmann/detail/meta/detected.hpp",
"include/nlohmann/detail/meta/identity_tag.hpp",
"include/nlohmann/detail/meta/is_sax.hpp",
"include/nlohmann/detail/meta/logic.hpp",
"include/nlohmann/detail/meta/std_fs.hpp",
"include/nlohmann/detail/meta/type_traits.hpp",
"include/nlohmann/detail/meta/void_t.hpp",
"include/nlohmann/detail/output/binary_writer.hpp",
"include/nlohmann/detail/output/error_handler.hpp",
"include/nlohmann/detail/output/output_adapters.hpp",
"include/nlohmann/detail/output/serializer.hpp",
"include/nlohmann/detail/recursion_depth_limit.hpp",
+6
View File
@@ -61,6 +61,7 @@ option(JSON_Install "Install CMake targets during install
option(JSON_MultipleHeaders "Use non-amalgamated version of the library." ON)
option(JSON_SystemInclude "Include as system headers (skip for clang-tidy)." OFF)
option(JSON_StrictNulHandling "Build with strict NUL-byte handling enabled." OFF)
option(JSON_StrictBinaryUTF8 "Build with UTF-8 checks in the CBOR, UBJSON, BJData, and BSON writers enabled." OFF)
if (JSON_CI)
include(ci)
@@ -118,6 +119,10 @@ if (JSON_StrictNulHandling)
message(STATUS "Strict NUL-byte handling enabled (JSON_STRICT_NUL_HANDLING=1)")
endif()
if (JSON_StrictBinaryUTF8)
message(STATUS "Strict UTF-8 checks in binary writers enabled (JSON_STRICT_BINARY_UTF8=1)")
endif()
if (JSON_Diagnostic_Positions)
message(STATUS "Diagnostic positions enabled (JSON_DIAGNOSTIC_POSITIONS=1)")
endif()
@@ -153,6 +158,7 @@ target_compile_definitions(
$<$<BOOL:${JSON_Diagnostic_Positions}>:JSON_DIAGNOSTIC_POSITIONS=1>
$<$<BOOL:${JSON_LegacyDiscardedValueComparison}>:JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1>
$<$<BOOL:${JSON_StrictNulHandling}>:JSON_STRICT_NUL_HANDLING=1>
$<$<BOOL:${JSON_StrictBinaryUTF8}>:JSON_STRICT_BINARY_UTF8=1>
)
target_include_directories(
+1 -1
View File
@@ -701,7 +701,7 @@ ci_get_cmake(4.0.0 CMAKE_4_0_0_BINARY)
# the tests require CMake 3.13 or later, so they are excluded for CMake 3.5.0
set(JSON_CMAKE_FLAGS_3_5_0 JSON_Diagnostics JSON_Diagnostic_Positions JSON_GlobalUDLs JSON_ImplicitConversions JSON_DisableEnumSerialization
JSON_LegacyDiscardedValueComparison JSON_Install JSON_MultipleHeaders JSON_SystemInclude JSON_Valgrind
JSON_StrictNulHandling)
JSON_StrictNulHandling JSON_StrictBinaryUTF8)
set(JSON_CMAKE_FLAGS_3_31_6 JSON_BuildTests ${JSON_CMAKE_FLAGS_3_5_0})
set(JSON_CMAKE_FLAGS_4_0_0 JSON_BuildTests ${JSON_CMAKE_FLAGS_3_5_0})
+1
View File
@@ -19,6 +19,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('format_as', 'Function', 'api/
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::accept', 'Function', 'api/basic_json/accept/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::array', 'Function', 'api/basic_json/array/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::array_t', 'Type', 'api/basic_json/array_t/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::as_base_class', 'Method', 'api/basic_json/as_base_class/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::at', 'Method', 'api/basic_json/at/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::back', 'Method', 'api/basic_json/back/index.html');
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::basic_json', 'Constructor', 'api/basic_json/basic_json/index.html');
@@ -0,0 +1,53 @@
# <small>nlohmann::basic_json::</small>as_base_class
```cpp
json_base_class_t& as_base_class() noexcept;
const json_base_class_t& as_base_class() const noexcept;
```
Returns a reference to this object as its custom base class [`json_base_class_t`](json_base_class_t.md). No copy is
made.
Since `basic_json` derives from `json_base_class_t`, a member of `basic_json` hides any member of the custom base class
with the same name. This function makes such hidden members accessible again.
## Return value
reference to this object as [`json_base_class_t`](json_base_class_t.md)
## Exception safety
No-throw guarantee: this function never throws exceptions.
## Complexity
Constant.
## Notes
The function is equivalent to `static_cast<json_base_class_t&>(j)` (or `static_cast<const json_base_class_t&>(j)`).
## Examples
??? example
The example shows how to use `as_base_class` to access members of the custom base class that are hidden by members
of `basic_json`.
```cpp
--8<-- "examples/as_base_class.cpp"
```
Output:
```json
--8<-- "examples/as_base_class.output"
```
## See also
- [json_base_class_t](json_base_class_t.md) - type of the custom base class
## Version history
- Added in version 3.13.0.
+6 -3
View File
@@ -25,10 +25,12 @@ and `ensure_ascii` parameters.
result consists of ASCII characters only.
`error_handler` (in)
: how to react on decoding errors; there are three possible values (see [`error_handler_t`](error_handler_t.md):
: how to react on decoding errors; there are four possible values (see [`error_handler_t`](error_handler_t.md):
`strict` (throws an exception in case a decoding error occurs; default), `replace` (replace invalid UTF-8 sequences
with U+FFFD), and `ignore` (ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the
output unchanged, and invalid bytes are dropped)).
with U+FFFD), `ignore` (ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the
output unchanged, and invalid bytes are dropped), and `keep` (write the ill-formed bytes to the output as is,
without escaping them, even if `ensure_ascii` is `#!cpp true`; the result is then not valid UTF-8, but equals the
input bytes exactly, and well-formed characters around the ill-formed bytes are still escaped as usual)).
## Return value
@@ -94,3 +96,4 @@ Binary values are serialized as an object containing two keys:
- Indentation character `indent_char`, option `ensure_ascii` and exceptions added in version 3.0.0.
- Error handlers added in version 3.4.0.
- Serialization of binary values added in version 3.8.0.
- Error handler `keep` added in version 3.13.0.
@@ -4,15 +4,31 @@
enum class error_handler_t {
strict,
replace,
ignore
ignore,
keep
};
```
This enumeration is used in the [`dump`](dump.md) function to choose how to treat decoding errors while serializing a
`basic_json` value. Three values are differentiated:
This enumeration is used to choose how to treat ill-formed UTF-8 in a string value or object key:
- [`dump`](dump.md) uses it while serializing a `basic_json` value to text.
- [`to_cbor`](to_cbor.md), [`to_msgpack`](to_msgpack.md), [`to_ubjson`](to_ubjson.md), [`to_bjdata`](to_bjdata.md),
and [`to_bson`](to_bson.md) use it while serializing a `basic_json` value to that binary format. Their default is
`keep`, as no binary writer checked before this parameter was added. CBOR, UBJSON, BJData, and BSON require valid
UTF-8, so for these four the default is `strict` if [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md)
is enabled; MessagePack's specification explicitly allows a string to contain ill-formed UTF-8, so `to_msgpack`
stays at `keep`. `to_bon8` does not take this parameter: BON8 always validates, since UTF-8 lead bytes are
structural to that format.
- [`from_cbor`](from_cbor.md), [`from_msgpack`](from_msgpack.md), [`from_ubjson`](from_ubjson.md),
[`from_bjdata`](from_bjdata.md), and [`from_bson`](from_bson.md) use it while parsing that binary format, to decide
whether to check a string value or object key for well-formed UTF-8 at all; by default (`keep`) they do not, as no
binary reader did before this parameter was added. `from_bon8` does not take this parameter, for the same reason
`to_bon8` does not.
Four values are differentiated:
strict
: throw a `type_error` exception in case of invalid UTF-8
: throw a `type_error`/`parse_error` exception in case of invalid UTF-8
replace
: replace invalid UTF-8 sequences with U+FFFD (� REPLACEMENT CHARACTER)
@@ -20,6 +36,12 @@ replace
ignore
: ignore invalid UTF-8 sequences; all valid bytes are copied to the output unchanged, and invalid bytes are dropped
keep
: keep invalid UTF-8 sequences unchanged; only meaningful for the binary formats mentioned above, since [`dump`]
(dump.md) itself must produce text, and `keep` there writes the ill-formed bytes to the output as is, so the
result is then not valid UTF-8 (but still equals the input bytes exactly, including around any well-formed
characters, which are still escaped as usual)
## Examples
??? example
@@ -45,3 +67,5 @@ ignore
## Version history
- Added in version 3.4.0.
- Added `keep`, and made this enumeration apply to the binary readers and writers in addition to `dump`, in version
3.13.0.
+12 -3
View File
@@ -5,12 +5,14 @@
template<typename InputType>
static basic_json from_bjdata(InputType&& i,
const bool strict = true,
const bool allow_exceptions = true);
const bool allow_exceptions = true,
const error_handler_t error_handler = error_handler_t::keep);
// (2)
template<typename IteratorType, typename SentinelType = IteratorType>
static basic_json from_bjdata(IteratorType first, SentinelType last,
const bool strict = true,
const bool allow_exceptions = true);
const bool allow_exceptions = true,
const error_handler_t error_handler = error_handler_t::keep);
```
Deserializes a given input to a JSON value using the BJData (Binary JData) serialization format.
@@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
`allow_exceptions` (in)
: whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default)
`error_handler` (in)
: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md).
BJData does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not
check at all, as every binary reader did before this parameter was added; `strict` checks and throws;
`replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would
## Return value
deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be
@@ -73,7 +81,7 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
the end of the file was not reached when `strict` was set to true
- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if a parse error occurs
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string could not be parsed
successfully
successfully, or if a string value or object key is not valid UTF-8 and `error_handler` is `strict`
- Throws [out_of_range.408](../../home/exceptions.md#jsonexceptionout_of_range408) if the size of an optimized container
or n-dimensional array cannot be represented by `std::size_t`
@@ -111,3 +119,4 @@ Linear in the size of the input.
- Added in version 3.11.0.
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
- Added `error_handler` parameter in version 3.13.0.
+13 -2
View File
@@ -5,12 +5,14 @@
template<typename InputType>
static basic_json from_bson(InputType&& i,
const bool strict = true,
const bool allow_exceptions = true);
const bool allow_exceptions = true,
const error_handler_t error_handler = error_handler_t::keep);
// (2)
template<typename IteratorType, typename SentinelType = IteratorType>
static basic_json from_bson(IteratorType first, SentinelType last,
const bool strict = true,
const bool allow_exceptions = true);
const bool allow_exceptions = true,
const error_handler_t error_handler = error_handler_t::keep);
```
Deserializes a given input to a JSON value using the BSON (Binary JSON) serialization format.
@@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
`allow_exceptions` (in)
: whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default)
`error_handler` (in)
: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md).
BSON does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not
check at all, as every binary reader did before this parameter was added; `strict` checks and throws;
`replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would
## Return value
deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be
@@ -75,6 +83,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
invalid string or byte array length)
- Throws [`parse_error.114`](../../home/exceptions.md#jsonexceptionparse_error114) if an unsupported BSON record type is
encountered
- Throws [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) if a string value or object key is
not valid UTF-8 and `error_handler` is `strict`
## Complexity
@@ -111,6 +121,7 @@ Linear in the size of the input.
- Added in version 3.4.0.
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
- Added `error_handler` parameter in version 3.13.0.
!!! warning "Deprecation"
+14 -4
View File
@@ -6,14 +6,16 @@ template<typename InputType>
static basic_json from_cbor(InputType&& i,
const bool strict = true,
const bool allow_exceptions = true,
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error);
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error,
const error_handler_t error_handler = error_handler_t::keep);
// (2)
template<typename IteratorType, typename SentinelType = IteratorType>
static basic_json from_cbor(IteratorType first, SentinelType last,
const bool strict = true,
const bool allow_exceptions = true,
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error);
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error,
const error_handler_t error_handler = error_handler_t::keep);
```
Deserializes a given input to a JSON value using the CBOR (Concise Binary Object Representation) serialization format.
@@ -65,6 +67,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
: how to treat CBOR tags (optional, `error` by default); see [`cbor_tag_handler_t`](cbor_tag_handler_t.md) for more
information
`error_handler` (in)
: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md).
CBOR does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not
check at all, as every binary reader did before this parameter was added; `strict` checks and throws;
`replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would
## Return value
deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be
@@ -80,8 +88,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
the end of the file was not reached when `strict` was set to true
- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from CBOR were
used in the given input or if the input is not valid CBOR
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other
types are not supported, as JSON object keys are always strings) or a string is malformed
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of
other types are not supported, as JSON object keys are always strings), or if a string value or object key is not
valid UTF-8 and `error_handler` is `strict`
## Complexity
@@ -121,6 +130,7 @@ Linear in the size of the input.
- Added `tag_handler` parameter in version 3.9.0.
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
- Added `error_handler` parameter in version 3.13.0.
!!! warning "Deprecation"
@@ -5,12 +5,14 @@
template<typename InputType>
static basic_json from_msgpack(InputType&& i,
const bool strict = true,
const bool allow_exceptions = true);
const bool allow_exceptions = true,
const error_handler_t error_handler = error_handler_t::keep);
// (2)
template<typename IteratorType, typename SentinelType = IteratorType>
static basic_json from_msgpack(IteratorType first, SentinelType last,
const bool strict = true,
const bool allow_exceptions = true);
const bool allow_exceptions = true,
const error_handler_t error_handler = error_handler_t::keep);
```
Deserializes a given input to a JSON value using the MessagePack serialization format.
@@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
`allow_exceptions` (in)
: whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default)
`error_handler` (in)
: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md).
MessagePack's specification explicitly allows ill-formed UTF-8, so checking is opt-in: the default, `keep`, does
not check at all, as every binary reader did before this parameter was added; `strict` checks and throws;
`replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would
## Return value
deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be
@@ -73,8 +81,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
the end of the file was not reached when `strict` was set to true
- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from
MessagePack were used in the given input or if the input is not valid MessagePack
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other
types are not supported, as JSON object keys are always strings) or a string is malformed
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of
other types are not supported, as JSON object keys are always strings), or if a string value or object key is not
valid UTF-8 and `error_handler` is `strict`
## Complexity
@@ -113,6 +122,7 @@ Linear in the size of the input.
- Added `allow_exceptions` parameter in version 3.2.0.
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
- Added `error_handler` parameter in version 3.13.0.
!!! warning "Deprecation"
+12 -3
View File
@@ -5,12 +5,14 @@
template<typename InputType>
static basic_json from_ubjson(InputType&& i,
const bool strict = true,
const bool allow_exceptions = true);
const bool allow_exceptions = true,
const error_handler_t error_handler = error_handler_t::keep);
// (2)
template<typename IteratorType, typename SentinelType = IteratorType>
static basic_json from_ubjson(IteratorType first, SentinelType last,
const bool strict = true,
const bool allow_exceptions = true);
const bool allow_exceptions = true,
const error_handler_t error_handler = error_handler_t::keep);
```
Deserializes a given input to a JSON value using the UBJSON (Universal Binary JSON) serialization format.
@@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
`allow_exceptions` (in)
: whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default)
`error_handler` (in)
: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md).
UBJSON does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not
check at all, as every binary reader did before this parameter was added; `strict` checks and throws;
`replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would
## Return value
deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be
@@ -73,7 +81,7 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
the end of the file was not reached when `strict` was set to true
- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if a parse error occurs
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string could not be parsed
successfully
successfully, or if a string value or object key is not valid UTF-8 and `error_handler` is `strict`
- Throws [out_of_range.408](../../home/exceptions.md#jsonexceptionout_of_range408) if the size of an optimized container
or n-dimensional array cannot be represented by `std::size_t`
@@ -112,6 +120,7 @@ Linear in the size of the input.
- Added `allow_exceptions` parameter in version 3.2.0.
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
- Added `error_handler` parameter in version 3.13.0.
!!! warning "Deprecation"
+1
View File
@@ -200,6 +200,7 @@ Direct access to the stored value of a JSON value.
- [**get_ref**](get_ref.md) - get a reference value
- [**operator ValueType**](operator_ValueType.md) - get a value
- [**get_binary**](get_binary.md) - get a binary value
- [**as_base_class**](as_base_class.md) - access the custom base class
### Element access
@@ -27,6 +27,18 @@ A `CustomBaseClass` with non-static data members forfeits `basic_json`'s
[standard layout](https://en.cppreference.com/w/cpp/named_req/StandardLayoutType) guarantee. See
[Template Parameter Requirements](../../features/types/template_parameters.md#custombaseclass).
#### Name conflicts
Since `basic_json` derives from `CustomBaseClass`, members of `basic_json` hide members of `CustomBaseClass` with the
same name. Hidden members remain accessible via [`as_base_class`](as_base_class.md) or by casting the value to
`json_base_class_t`.
!!! warning "Avoid generic member names"
Future versions of the library may add members to `basic_json` that hide members of `CustomBaseClass` that are
accessible today. To reduce the risk of such conflicts, avoid generic names for the members of `CustomBaseClass`,
for instance by using a distinctive prefix.
## Examples
??? example
@@ -45,8 +57,10 @@ A `CustomBaseClass` with non-static data members forfeits `basic_json`'s
## See also
- [as_base_class](as_base_class.md) - access the custom base class
- [Template Parameter Requirements](../../features/types/template_parameters.md#custombaseclass) - the requirements for `CustomBaseClass`
## Version history
- Added in version 3.12.0.
- Made a public member type in version 3.13.0; it was private before, so it could not be named outside the class.
+1 -1
View File
@@ -13,7 +13,7 @@ JSON object holding version information
| key | description |
|-------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
| `compiler` | Information on the used compiler. It is an object with the following keys: `c++` (the used C++ standard), `family` (the compiler family; possible values are `clang`, `icc`, `gcc`, `ilecpp`, `msvc`, `pgcpp`, `sunpro`, and `unknown`), and `version` (the compiler version). On HP aCC compilers, `compiler` is instead the plain string `hp`. |
| `compiler` | Information on the used compiler. It is an object with the following keys: `c++` (the used C++ standard), `family` (the compiler family; possible values are `clang`, `icc`, `gcc`, `hp`, `ilecpp`, `msvc`, `pgcpp`, `sunpro`, and `unknown`), and `version` (the compiler version). |
| `copyright` | The copyright line for the library as string. |
| `name` | The name of the library as string. |
| `platform` | The used platform as string. Possible values are `win32`, `linux`, `apple`, `unix`, and `unknown`. |
+19 -4
View File
@@ -5,15 +5,18 @@
static std::vector<std::uint8_t> to_bjdata(const basic_json& j,
const bool use_size = false,
const bool use_type = false,
const bjdata_version_t version = bjdata_version_t::draft2);
const bjdata_version_t version = bjdata_version_t::draft2,
const error_handler_t error_handler = error_handler_t::keep);
// (2)
static void to_bjdata(const basic_json& j, detail::output_adapter<std::uint8_t> o,
const bool use_size = false, const bool use_type = false,
const bjdata_version_t version = bjdata_version_t::draft2);
const bjdata_version_t version = bjdata_version_t::draft2,
const error_handler_t error_handler = error_handler_t::keep);
static void to_bjdata(const basic_json& j, detail::output_adapter<char> o,
const bool use_size = false, const bool use_type = false,
const bjdata_version_t version = bjdata_version_t::draft2);
const bjdata_version_t version = bjdata_version_t::draft2,
const error_handler_t error_handler = error_handler_t::keep);
```
Serializes a given JSON value `j` to a byte vector using the BJData (Binary JData) serialization format. BJData aims to
@@ -43,6 +46,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
: which version of BJData to use (see note on "Binary values" on [BJData](../../features/binary_formats/bjdata.md));
optional, `#!cpp bjdata_version_t::draft2` by default.
`error_handler` (in)
: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md).
The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_bjdata` did before
this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would.
If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead.
## Return value
1. BJData serialization as byte vector
@@ -56,6 +65,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
- Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size`
is false, and `j` contains a non-empty array, object, or binary value.
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is
not valid UTF-8 and `error_handler` is `strict` (the default only if
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled)
## Complexity
@@ -104,4 +116,7 @@ Linear in the size of the JSON value `j`.
## Version history
- Added in version 3.11.0.
- BJData version parameter (for draft3 binary encoding) added in version 3.12.0.
- BJData version parameter (for draft3 binary encoding) added in version 3.12.0.
- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key
that is not valid UTF-8 unchanged, as before; `strict` (the default if
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`.
+19 -3
View File
@@ -2,11 +2,14 @@
```cpp
// (1)
static std::vector<std::uint8_t> to_bson(const basic_json& j);
static std::vector<std::uint8_t> to_bson(const basic_json& j,
const error_handler_t error_handler = error_handler_t::keep);
// (2)
static void to_bson(const basic_json& j, detail::output_adapter<std::uint8_t> o);
static void to_bson(const basic_json& j, detail::output_adapter<char> o);
static void to_bson(const basic_json& j, detail::output_adapter<std::uint8_t> o,
const error_handler_t error_handler = error_handler_t::keep);
static void to_bson(const basic_json& j, detail::output_adapter<char> o,
const error_handler_t error_handler = error_handler_t::keep);
```
BSON (Binary JSON) is a binary format in which zero or more ordered key/value pairs are stored as a single entity (a
@@ -25,6 +28,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
`o` (in)
: output adapter to write serialization to
`error_handler` (in)
: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md).
The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_bson` did before
this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would.
If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead.
## Return value
1. BSON serialization as a byte vector
@@ -46,6 +55,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
- Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value
exceeds 255, the maximum of the BSON binary subtype; example:
`"subtype 70000 is too large for the BSON binary subtype (max 255)"`
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is
not valid UTF-8 and `error_handler` is `strict` (the default only if
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled)
## Complexity
@@ -98,3 +110,7 @@ pass before anything is written.
- Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0.
- Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0.
- `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0.
- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key
that is not valid UTF-8 unchanged, as before; `strict` (the default if
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316` before anything
is written.
+21 -3
View File
@@ -2,11 +2,14 @@
```cpp
// (1)
static std::vector<std::uint8_t> to_cbor(const basic_json& j);
static std::vector<std::uint8_t> to_cbor(const basic_json& j,
const error_handler_t error_handler = error_handler_t::keep);
// (2)
static void to_cbor(const basic_json& j, detail::output_adapter<std::uint8_t> o);
static void to_cbor(const basic_json& j, detail::output_adapter<char> o);
static void to_cbor(const basic_json& j, detail::output_adapter<std::uint8_t> o,
const error_handler_t error_handler = error_handler_t::keep);
static void to_cbor(const basic_json& j, detail::output_adapter<char> o,
const error_handler_t error_handler = error_handler_t::keep);
```
Serializes a given JSON value `j` to a byte vector using the CBOR (Concise Binary Object Representation) serialization
@@ -26,6 +29,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
`o` (in)
: output adapter to write serialization to
`error_handler` (in)
: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md).
The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_cbor` did before
this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would.
If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead.
## Return value
1. CBOR serialization as a byte vector
@@ -35,6 +44,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
Strong guarantee: if an exception is thrown, there are no changes in the JSON value.
## Exceptions
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is
not valid UTF-8 and `error_handler` is `strict` (the default only if
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled)
## Complexity
Linear in the size of the JSON value `j`.
@@ -68,3 +83,6 @@ Linear in the size of the JSON value `j`.
- Added in version 2.0.9.
- Compact representation of floating-point numbers added in version 3.8.0.
- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key
that is not valid UTF-8 unchanged, as before; `strict` (the default if
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`.
+17 -3
View File
@@ -2,11 +2,14 @@
```cpp
// (1)
static std::vector<std::uint8_t> to_msgpack(const basic_json& j);
static std::vector<std::uint8_t> to_msgpack(const basic_json& j,
const error_handler_t error_handler = error_handler_t::keep);
// (2)
static void to_msgpack(const basic_json& j, detail::output_adapter<std::uint8_t> o);
static void to_msgpack(const basic_json& j, detail::output_adapter<char> o);
static void to_msgpack(const basic_json& j, detail::output_adapter<std::uint8_t> o,
const error_handler_t error_handler = error_handler_t::keep);
static void to_msgpack(const basic_json& j, detail::output_adapter<char> o,
const error_handler_t error_handler = error_handler_t::keep);
```
Serializes a given JSON value `j` to a byte vector using the MessagePack serialization format. MessagePack is a binary
@@ -25,6 +28,13 @@ The exact mapping and its limitations are described on a [dedicated page](../../
`o` (in)
: output adapter to write serialization to
`error_handler` (in)
: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md).
The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_msgpack` did before
this parameter was added and as the MessagePack specification allows; `strict` throws; `replace`/`ignore` sanitize
it the same way [`dump`](dump.md) would. Unlike the other binary writers, the default stays `keep` even if
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled.
## Return value
1. MessagePack serialization as a byte vector
@@ -42,6 +52,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
- Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value
exceeds 255, the maximum of the MessagePack ext type; example:
`"subtype 70000 is too large for the MessagePack ext type (max 255)"`
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is
not valid UTF-8 and `error_handler` is `strict`
## Complexity
@@ -91,6 +103,8 @@ Linear in the size of the JSON value `j`.
- Added in version 2.0.9.
- Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0.
- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key
that is not valid UTF-8 unchanged, as before.
- Fixed in version 3.13.0 to serialize `number_integer_t`/`number_unsigned_t` pairs of different width correctly;
before, integers could be serialized with the wrong value if `number_integer_t` was narrower than
`number_unsigned_t`.
+18 -3
View File
@@ -4,13 +4,16 @@
// (1)
static std::vector<std::uint8_t> to_ubjson(const basic_json& j,
const bool use_size = false,
const bool use_type = false);
const bool use_type = false,
const error_handler_t error_handler = error_handler_t::keep);
// (2)
static void to_ubjson(const basic_json& j, detail::output_adapter<std::uint8_t> o,
const bool use_size = false, const bool use_type = false);
const bool use_size = false, const bool use_type = false,
const error_handler_t error_handler = error_handler_t::keep);
static void to_ubjson(const basic_json& j, detail::output_adapter<char> o,
const bool use_size = false, const bool use_type = false);
const bool use_size = false, const bool use_type = false,
const error_handler_t error_handler = error_handler_t::keep);
```
Serializes a given JSON value `j` to a byte vector using the UBJSON (Universal Binary JSON) serialization format. UBJSON
@@ -36,6 +39,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../
: whether to add type annotations to container types (must be combined with `#!cpp use_size = true`); optional,
`#!cpp false` by default.
`error_handler` (in)
: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md).
The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_ubjson` did before
this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would.
If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead.
## Return value
1. UBJSON serialization as a byte vector
@@ -49,6 +58,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
- Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size`
is false, and `j` contains a non-empty array, object, or binary value.
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is
not valid UTF-8 and `error_handler` is `strict` (the default only if
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled)
## Complexity
@@ -97,3 +109,6 @@ Linear in the size of the JSON value `j`.
## Version history
- Added in version 3.1.0.
- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key
that is not valid UTF-8 unchanged, as before; `strict` (the default if
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`.
+2
View File
@@ -18,6 +18,8 @@ header. See also the [macro overview page](../../features/macros.md).
- [**JSON_PRECISE_STREAM_POSITION**](json_precise_stream_position.md) - opt in to leaving an input stream positioned
right after a parsed number
- [**JSON_STRICT_BINARY_UTF8**](json_strict_binary_utf8.md) - opt in to checking strings for valid UTF-8 in the CBOR,
UBJSON, BJData, and BSON writers
- [**JSON_STRICT_NUL_HANDLING**](json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of
treating it as end of input
@@ -115,3 +115,4 @@ The default value is `0` (disabled — existing behavior is preserved).
## Version history
- Added in version 3.13.0.
- Planned to become the default (with the macro removed) in version 4.0.0.
@@ -0,0 +1,101 @@
# JSON_STRICT_BINARY_UTF8
```cpp
#define JSON_STRICT_BINARY_UTF8 /* value */
```
When defined to `1`, the `error_handler` parameter of the binary writers [`to_cbor`](../basic_json/to_cbor.md),
[`to_ubjson`](../basic_json/to_ubjson.md), [`to_bjdata`](../basic_json/to_bjdata.md), and
[`to_bson`](../basic_json/to_bson.md) defaults to [`error_handler_t::strict`](../basic_json/error_handler_t.md) instead
of `error_handler_t::keep`. These writers then check every string value and object key for valid UTF-8 and throw
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, like
[`dump`](../basic_json/dump.md) does. Without it, they write the bytes unchanged. An `error_handler` passed explicitly
always takes precedence.
The macro does not affect:
- [`to_msgpack`](../basic_json/to_msgpack.md): the MessagePack specification allows a `str` value to contain bytes that
are not valid UTF-8, so its `error_handler` always defaults to `keep`.
- [`to_bon8`](../basic_json/to_bon8.md): BON8 always checks, because the UTF-8 lead bytes mark where a string ends.
- The binary readers ([`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md),
[`from_ubjson`](../basic_json/from_ubjson.md), [`from_bjdata`](../basic_json/from_bjdata.md),
[`from_bson`](../basic_json/from_bson.md)): none of these formats requires a decoder to reject ill-formed UTF-8, so
they always return the bytes unchanged.
## Default definition
The default value is `0` (disabled, the behavior of version 3.12.0 and earlier is preserved).
```cpp
#define JSON_STRICT_BINARY_UTF8 0
```
## Notes
!!! note "Background"
CBOR, UBJSON, BJData, and BSON all require strings to be UTF-8. Up to version 3.12.0, the writers did not check
this, so they could produce output that other decoders reject. Checking by default would break code that stores
other encodings (for instance ISO 8859-1) in a string and only ever writes it to a binary format. You can pass
`error_handler_t::strict` to each call, or use this macro to check by default ahead of version 4.0.0, where
`strict` is planned to become the default (see
[#5529](https://github.com/nlohmann/json/issues/5529) and [#5651](https://github.com/nlohmann/json/issues/5651)).
!!! warning "Opt-in only"
This macro must be defined **before** including `<nlohmann/json.hpp>`. Defining it after the include has no
effect.
!!! note "ABI compatibility"
The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_sbu8`), resulting in
distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program
without One Definition Rule (ODR) violations, but they cannot exchange instances of library types.
## Examples
??? example "Default behavior (macro not defined)"
Without the macro, the bytes are written unchanged:
```cpp
#include <nlohmann/json.hpp>
using json = nlohmann::json;
int main()
{
auto v = json::to_cbor(json("\xFF"));
// v is {0x61, 0xFF}
}
```
??? example "Opt-in check (macro defined to 1)"
With the macro, ill-formed UTF-8 is rejected:
```cpp
#define JSON_STRICT_BINARY_UTF8 1
#include <nlohmann/json.hpp>
using json = nlohmann::json;
int main()
{
auto v = json::to_cbor(json("\xFF"));
// throws type_error.316: invalid UTF-8 byte at index 0: 0xFF
}
```
## See also
- [**to_cbor**](../basic_json/to_cbor.md) - create a CBOR serialization of a JSON value
- [**to_ubjson**](../basic_json/to_ubjson.md) - create a UBJSON serialization of a JSON value
- [**to_bjdata**](../basic_json/to_bjdata.md) - create a BJData serialization of a JSON value
- [**to_bson**](../basic_json/to_bson.md) - create a BSON serialization of a JSON value
- [**error_handler_t**](../basic_json/error_handler_t.md) - how [`dump`](../basic_json/dump.md) treats ill-formed UTF-8
## Version history
- Added in version 3.13.0.
- Planned to become the default (with the macro removed) in version 4.0.0.
+65 -5
View File
@@ -14,7 +14,7 @@ work items are tracked in the [GitHub milestones](https://github.com/nlohmann/js
opt-in.
- **Keep the 3.x public API stable.** Releases follow [semantic versioning](https://semver.org). Changes that would
break existing code are only added behind a feature macro, so users can opt in and test their code before a next
major release.
major release, see [Version 4.0](#version-40).
- **Support a broad range of compilers and platforms.** The [CI](quality_assurance.md) keeps testing old and new
versions of GCC, Clang, MSVC, and other compilers on Linux, macOS, and Windows.
- **Keep the quality assurance up.** Every change keeps the test coverage at 100%, passes the static and dynamic
@@ -37,7 +37,67 @@ work items are tracked in the [GitHub milestones](https://github.com/nlohmann/js
## Version 4.0
There is no decision yet on whether or when a version 4.0 with breaking changes will be released. Proposals that need
a major version, for instance stricter type conversions, are collected in issue
[#3453](https://github.com/nlohmann/json/issues/3453). Until then, such changes are only added as opt-in behavior
behind feature macros.
There is no release date for version 4.0 yet. Proposals that need a major version, for instance stricter type
conversions, are collected in issue [#3453](https://github.com/nlohmann/json/issues/3453).
!!! note "Not final"
The plan for version 4.0 described below is not final and may still change: macros may be added to or removed from
the list, and planned defaults may be revised. Any such change will be documented on this page.
### Trying out 4.0 today
Version 4.0 will not be developed on a separate branch. Instead, every breaking change is first added to a 3.x release
behind a macro whose default keeps the 3.x behavior. Version 4.0 then switches the defaults and removes the macros.
Version 4.0 is therefore the sum of these macros: you can try it on the 3.x release train today by defining each macro
to its 4.0 value and fixing what no longer compiles or behaves differently. Once your code works with all of them, it
is ready for version 4.0.
The following macros guard changes that are planned to become the default in version 4.0:
| Macro | 3.x default | 4.0 behavior | CMake option | Added |
|------------------------------------------------------------------------------------------------------------------|-------------|-------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------|--------|
| [`JSON_USE_IMPLICIT_CONVERSIONS`](../api/macros/json_use_implicit_conversions.md) | `1` | `0`: no implicit conversions from `basic_json` to other types; use [`get`](../api/basic_json/get.md) instead | [`JSON_ImplicitConversions`](../integration/cmake.md#json_implicitconversions) | 3.9.0 |
| [`JSON_USE_GLOBAL_UDLS`](../api/macros/json_use_global_udls.md) | `1` | `0`: the string literals `_json` and `_json_pointer` are only available in namespace `nlohmann::literals` | [`JSON_GlobalUDLs`](../integration/cmake.md#json_globaludls) | 3.11.0 |
| [`JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md) | `0` | removed: the deprecated legacy comparison of discarded values can no longer be enabled | [`JSON_LegacyDiscardedValueComparison`](../integration/cmake.md#json_legacydiscardedvaluecomparison) | 3.11.0 |
| [`JSON_BRACE_INIT_COPY_SEMANTICS`](../api/macros/json_brace_init_copy_semantics.md) | `0` | `1`: single-element brace initialization such as `#!cpp json j{obj};` copies the element instead of creating an array | – | 3.13.0 |
| [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) | `0` | `1`: reading from a stream does not consume the character after a number | – | 3.13.0 |
| [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) | `0` | `1`: a NUL byte in the input is a parse error instead of the end of input | [`JSON_StrictNulHandling`](../integration/cmake.md#json_strictnulhandling) | 3.13.0 |
| [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) | `0` | `1`: `to_cbor`, `to_ubjson`, `to_bjdata`, and `to_bson` throw for strings that are not valid UTF-8 by default | [`JSON_StrictBinaryUTF8`](../integration/cmake.md#json_strictbinaryutf8) | 3.13.0 |
For example, the following makes a 3.x release behave like version 4.0 with respect to these changes:
```cpp
#define JSON_USE_IMPLICIT_CONVERSIONS 0
#define JSON_USE_GLOBAL_UDLS 0
#define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0
#define JSON_BRACE_INIT_COPY_SEMANTICS 1
#define JSON_PRECISE_STREAM_POSITION 1
#define JSON_STRICT_NUL_HANDLING 1
#define JSON_STRICT_BINARY_UTF8 1
#include <nlohmann/json.hpp>
```
The macros must be defined before the library header is included; setting them once in the build system is the easiest
way to achieve this.
### Removal of deprecated functions
Version 4.0 will remove all deprecated functions. Compiling with deprecation warnings enabled shows which of them your
code still uses. The [migration guide](../integration/migration_guide.md#replace-deprecated-functions) shows how to
replace each of them.
| Deprecated | Since | Migration |
|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------|----------------------------------------------------------------------------------|
| `#!cpp operator<<(basic_json&, std::istream&)` | 3.0.0 | [Parsing](../integration/migration_guide.md#parsing) |
| `#!cpp operator>>(const basic_json&, std::ostream&)` | 3.0.0 | [Miscellaneous functions](../integration/migration_guide.md#miscellaneous-functions) |
| `iterator_wrapper` | 3.1.0 | [Miscellaneous functions](../integration/migration_guide.md#miscellaneous-functions) |
| [`parse`](../api/basic_json/parse.md), [`accept`](../api/basic_json/accept.md), and [`sax_parse`](../api/basic_json/sax_parse.md) with an initializer list `{ptr, len}` or `{first, last}` | 3.8.0 | [Parsing](../integration/migration_guide.md#parsing) |
| [`from_bson`](../api/basic_json/from_bson.md), [`from_cbor`](../api/basic_json/from_cbor.md), [`from_msgpack`](../api/basic_json/from_msgpack.md), and [`from_ubjson`](../api/basic_json/from_ubjson.md) with `(ptr, len)` or an initializer list | 3.8.0 | [Parsing](../integration/migration_guide.md#parsing) |
| [`json_pointer::operator string_t`](../api/json_pointer/operator_string_t.md) | 3.11.0 | [JSON Pointers](../integration/migration_guide.md#json-pointers) |
| [`json_pointer`](../api/json_pointer/index.md) with a `basic_json` type as template argument, and the overloads of `value`, `contains`, `operator[]`, and `at` accepting such a pointer | 3.11.0 | [JSON Pointers](../integration/migration_guide.md#json-pointers) |
| Comparing a [`json_pointer`](../api/json_pointer/index.md) with a string via [`operator==`](../api/json_pointer/operator_eq.md) or [`operator!=`](../api/json_pointer/operator_ne.md) | 3.11.2 | [JSON Pointers](../integration/migration_guide.md#json-pointers) |
The deprecated legacy comparison of discarded values is controlled by a macro and therefore listed in the table above.
New breaking changes will follow the same path: they are added to these tables when they land in a 3.x release.
@@ -0,0 +1,41 @@
#include <iostream>
#include <nlohmann/json.hpp>
class base_class_with_hidden_members
{
public:
const char* type_name() const noexcept
{
return "my_type_name";
}
std::size_t size() const noexcept
{
return 42;
}
};
using json = nlohmann::basic_json <
std::map,
std::vector,
std::string,
bool,
std::int64_t,
std::uint64_t,
double,
std::allocator,
nlohmann::adl_serializer,
std::vector<std::uint8_t>,
base_class_with_hidden_members
>;
int main()
{
json j = {1, 2, 3};
// the members of basic_json hide the members of the base class
std::cout << j.type_name() << ' ' << j.size() << '\n';
// access the hidden members of the base class
std::cout << j.as_base_class().type_name() << ' ' << j.as_base_class().size() << '\n';
}
@@ -0,0 +1,2 @@
array 3
my_type_name 42
@@ -20,5 +20,6 @@ int main()
<< j_invalid.dump(-1, ' ', false, json::error_handler_t::replace)
<< "\nstring with ignored invalid characters: "
<< j_invalid.dump(-1, ' ', false, json::error_handler_t::ignore)
<< '\n';
<< "\nstring with the invalid byte kept as is (" << j_invalid.dump(-1, ' ', false, json::error_handler_t::keep).size()
<< " bytes, not valid UTF-8 itself)\n";
}
@@ -1,3 +1,4 @@
[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9
string with replaced invalid characters: "ä�ü"
string with ignored invalid characters: "äü"
string with the invalid byte kept as is (7 bytes, not valid UTF-8 itself)
@@ -63,6 +63,15 @@ The library uses the following mapping from JSON values types to BJData types ac
- strings with more than 18446744073709551615 bytes, i.e., 2<sup>64</sup>-1 bytes (theoretical)
!!! warning "UTF-8 validation of string values and object keys"
BJData strings must use UTF-8 encoding. By default (the [`error_handler`](../../api/basic_json/to_bjdata.md)
parameter left at `keep`), `to_bjdata()` writes the bytes of string values and object keys unchanged, even if they
are not valid UTF-8. With `error_handler_t::strict`, it throws
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead;
`replace`/`ignore` sanitize the string. [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md)
makes `strict` the default.
!!! info "Unused BJData markers"
The following markers are not used in the conversion:
@@ -208,6 +217,19 @@ The library maps BJData types to JSON value types as follows:
The mapping is **complete** in the sense that any BJData value can be converted to a JSON value.
!!! warning "Ill-formed UTF-8 in string values and object keys"
BJData strings must use UTF-8 encoding, but checking it on read is opt-in: with the
[`error_handler`](../../api/basic_json/from_bjdata.md) parameter left at `keep` (the default), `from_bjdata()`
accepts a string value or object key whose bytes are not valid UTF-8 and hands them back unchanged. Passing
`error_handler_t::strict` makes `from_bjdata()` check and throw
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and
`replace`/`ignore` sanitize the string instead of keeping it. However,
[`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read with the default
`keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. `to_bjdata()`'s
own `error_handler` parameter defaults to `keep` (see above), so such a value is written back unchanged.
!!! info "Round trips"
A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with
@@ -109,14 +109,21 @@ The library maps BSON record types to JSON value types as follows:
If BSON input must be validated for strict specification compliance, validate it separately before passing it to
`from_bson()`.
!!! warning "UTF-8 validation of string values"
!!! warning "Ill-formed UTF-8 in string values"
The BSON specification requires `string` values (type `0x02`) to be valid UTF-8. This library validates the
bytes of every such string at decode time and rejects ill-formed UTF-8 with a
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions`
set to `false`, a discarded value), rather than only failing later when the resulting value is dumped. Element
(key) names and `binary` values (type `0x05`) are unaffected and are never validated, since they are read
byte-by-byte as a C string, or are not required to hold text, respectively.
The BSON specification requires `string` values (type `0x02`) to be valid UTF-8, but this is not required of a
decoder, so checking is opt-in: with the [`error_handler`](../../api/basic_json/from_bson.md) parameter left at
`keep` (the default), `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back
unchanged. Passing `error_handler_t::strict` makes `from_bson()` check and throw
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and
`replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md)
still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a
value read with the default `keep` handler, unless an error handler is passed that replaces or ignores the
ill-formed bytes. `to_bson()`'s own `error_handler` parameter defaults to `keep`, so such a string value or element
(key) name is written unchanged; with `strict` (the default if
[`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled), it throws the same exception
instead. Element (key) names are never validated on read, since they are read byte-by-byte as a C string. `binary`
values (type `0x05`) are unaffected, since they are not required to hold text.
??? example "Example: deserialize a JSON value from BSON"
@@ -189,15 +189,21 @@ The library maps CBOR types to JSON value types as follows:
([RFC 8392](https://www.rfc-editor.org/rfc/rfc8392.html)), cannot be read with this library and need a
general-purpose CBOR library instead.
!!! warning "UTF-8 validation of text strings"
!!! warning "Ill-formed UTF-8 in text strings"
[RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings
(major type 3) to be valid UTF-8. This library validates the bytes of every text string (object keys included) at
decode time and rejects ill-formed UTF-8 with a
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with
`allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value is
dumped. Byte strings (major type 2) are unaffected and are never validated, since they are not required to hold
text.
[RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings (major
type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this, so checking is opt-in: with the
[`error_handler`](../../api/basic_json/from_cbor.md) parameter left at `keep` (the default), `from_cbor()` accepts a
text string (object keys included) whose bytes are not valid UTF-8 and hands them back unchanged. Passing
`error_handler_t::strict` makes `from_cbor()` check and throw
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and
`replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md)
still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a
value read with the default `keep` handler, unless an error handler is passed that replaces or ignores the
ill-formed bytes. `to_cbor()`'s own [`error_handler`](../../api/basic_json/to_cbor.md) parameter defaults to `keep`,
so such a value is written back unchanged; with `strict` (the default if
[`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled), it throws the same exception
instead. Byte strings (major type 2) are unaffected, since they are not required to hold text.
!!! warning "Tagged items"
@@ -153,14 +153,23 @@ The library maps MessagePack types to JSON value types as follows:
This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed
on. Such input needs a general-purpose MessagePack library instead.
!!! warning "UTF-8 validation of string values"
!!! warning "Ill-formed UTF-8 in string values"
The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8.
This library validates the bytes of every such string (object keys included) at decode time and rejects
ill-formed UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or,
with `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting
value is dumped. `bin`/`ext`/`fixext` values are unaffected and are never validated, since they are not required
to hold text.
The MessagePack specification explicitly allows a `str` value (`fixstr`, `str 8`, `str 16`, `str 32`) to contain
a byte sequence that is not valid UTF-8, and expects a deserializer to hand the original bytes back unchanged.
This library follows that by default: with its
[`error_handler`](../../api/basic_json/from_msgpack.md) parameter left at `keep` (the default),
`from_msgpack()` reads `str` bytes (object keys included) as-is, without validating them, so such a value
round-trips through `from_msgpack(to_msgpack(j))` byte for byte. Passing `error_handler_t::strict` makes
`from_msgpack()` check anyway and throw
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and
`replace`/`ignore` sanitize the string instead of keeping it. `to_msgpack()` also writes `str` bytes as-is by
default, since the specification permits it; its [`error_handler`](../../api/basic_json/to_msgpack.md) parameter
can be set to `strict` to throw [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) instead, or
to `replace`/`ignore` to sanitize the string, for instance for a decoder that rejects ill-formed UTF-8. However,
[`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way with the
default `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes.
??? example "Example: deserialize a JSON value from MessagePack"
@@ -47,6 +47,15 @@ The library uses the following mapping from JSON values types to UBJSON types ac
- strings with more than 9223372036854775807 bytes (theoretical)
!!! warning "UTF-8 validation of string values and object keys"
UBJSON's required string encoding is UTF-8. By default (the [`error_handler`](../../api/basic_json/to_ubjson.md)
parameter left at `keep`), `to_ubjson()` writes the bytes of string values and object keys unchanged, even if they
are not valid UTF-8. With `error_handler_t::strict`, it throws
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead;
`replace`/`ignore` sanitize the string. [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md)
makes `strict` the default.
!!! info "Unused UBJSON markers"
The following markers are not used in the conversion:
@@ -120,6 +129,19 @@ The library maps UBJSON types to JSON value types as follows:
The mapping is **complete** in the sense that any UBJSON value can be converted to a JSON value.
!!! warning "Ill-formed UTF-8 in string values and object keys"
UBJSON's required string encoding is UTF-8, but checking it on read is opt-in: with the
[`error_handler`](../../api/basic_json/from_ubjson.md) parameter left at `keep` (the default), `from_ubjson()`
accepts a string value or object key whose bytes are not valid UTF-8 and hands them back unchanged. Passing
`error_handler_t::strict` makes `from_ubjson()` check and throw
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and
`replace`/`ignore` sanitize the string instead of keeping it. However,
[`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read with the default
`keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. `to_ubjson()`'s
own `error_handler` parameter defaults to `keep` (see above), so such a value is written back unchanged.
??? example "Example: deserialize a JSON value from UBJSON"
```cpp
+14
View File
@@ -138,6 +138,20 @@ using the library with compilers that do not fully support C++11 and may only wo
See [full documentation of `JSON_SKIP_UNSUPPORTED_COMPILER_CHECK`](../api/macros/json_skip_unsupported_compiler_check.md).
## `JSON_STRICT_BINARY_UTF8`
When defined to `1`, [`to_cbor`](../api/basic_json/to_cbor.md), [`to_ubjson`](../api/basic_json/to_ubjson.md),
[`to_bjdata`](../api/basic_json/to_bjdata.md), and [`to_bson`](../api/basic_json/to_bson.md) throw
[`type_error.316`](../home/exceptions.md#jsonexceptiontype_error316) for a string value or object key that is not
valid UTF-8. The default value is `0`, which writes the bytes unchanged as before version 3.13.0; this is planned to
become the default in version 4.0.0.
The check can also be enabled with the CMake option
[`JSON_StrictBinaryUTF8`](../integration/cmake.md#json_strictbinaryutf8) (`OFF` by default) which sets
`JSON_STRICT_BINARY_UTF8` accordingly.
See [full documentation of `JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md).
## `JSON_STRICT_NUL_HANDLING`
When defined to `1`, a `'\0'` (NUL) byte anywhere in the input is rejected with `parse_error.101`, like any other
+1
View File
@@ -20,6 +20,7 @@ The complete default namespace name is derived as follows:
`_bics`.
- [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) defined non-zero appends `_psp`.
- [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) defined non-zero appends `_snul`.
- [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) defined non-zero appends `_sbu8`.
- The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by
underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component)
below.
@@ -71,24 +71,25 @@ on whether the end of the item with the error is known, a distinction that
If the item is complete, but cannot be passed on as it is, it is replaced, and parsing continues after it:
| Mistake | Formats | Repair |
|---------------------------------------------------------------------|-----------------------------------------|-------------------------------------------------------------------------|
| tag | CBOR | ignored |
| simple value other than `false`, `true`, and `null`, like undefined | CBOR | `#!json null` |
| negative integer below the range of `number_integer_t` | CBOR | the nearest floating-point number |
| string that is not valid UTF-8 | BJData, BSON, CBOR, MessagePack, UBJSON | each ill-formed sequence becomes U+FFFD |
| character (`C`) that is not ASCII | BJData, UBJSON | U+FFFD |
| invalid high-precision number (`H`) | BJData, UBJSON | the longest valid beginning is kept, as for JSON text, or `#!json null` |
| high-precision number too large | BJData, UBJSON | passed as infinity, together with its text |
| object key that is not a string | BON8, CBOR, MessagePack | the member is skipped |
| element of a type the library does not read, like ObjectId or date | BSON | `#!json null` |
| string without its terminator | BSON | kept |
| document whose size does not match its content | BSON | kept |
| Mistake | Formats | Repair |
|---------------------------------------------------------------------|-------------------------|-------------------------------------------------------------------------|
| tag | CBOR | ignored |
| simple value other than `false`, `true`, and `null`, like undefined | CBOR | `#!json null` |
| negative integer below the range of `number_integer_t` | CBOR | the nearest floating-point number |
| character (`C`) that is not ASCII | BJData, UBJSON | U+FFFD |
| invalid high-precision number (`H`) | BJData, UBJSON | the longest valid beginning is kept, as for JSON text, or `#!json null` |
| high-precision number too large | BJData, UBJSON | passed as infinity, together with its text |
| object key that is not a string | BON8, CBOR, MessagePack | the member is skipped |
| element of a type the library does not read, like ObjectId or date | BSON | `#!json null` |
| string without its terminator | BSON | kept |
| document whose size does not match its content | BSON | kept |
CBOR tags and simple values are repaired as [RFC 8949, Section 6.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-6.1)
suggests for converting CBOR to JSON. Note that [`sax_parse`](../../api/basic_json/sax_parse.md) has no parameter for
CBOR tags, so every tag is an error there; when recovering, tags are ignored like with
[`cbor_tag_handler_t::ignore`](../../api/basic_json/cbor_tag_handler_t.md).
[`cbor_tag_handler_t::ignore`](../../api/basic_json/cbor_tag_handler_t.md). Strings that are not valid UTF-8 are no
error: like [`from_cbor`](../../api/basic_json/from_cbor.md) and the other functions by default, `sax_parse` passes
them on as they are.
After any other error, the end of the item is unknown: the input ended, a byte is not a valid type marker, or a size
cannot be right. Parsing then stops, and the value read so far is completed: a key that waits for its value gets
+10 -1
View File
@@ -341,7 +341,10 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde
A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a
string was read where one was required (for instance as a map key), the string's length specification is invalid, or
the string's bytes are not valid UTF-8.
the string's bytes are not valid UTF-8 and the `error_handler` parameter of the corresponding `from_*` function is
set to `strict`. By default (`error_handler_t::keep`), the bytes of a string are not checked for valid UTF-8 on read;
see the ill-formed UTF-8 notes on the individual [binary format](../features/binary_formats/index.md) pages for how
such a string is handled depending on `error_handler`.
CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other
type (for instance integers or `null`) are therefore not supported; see the notes on
@@ -749,6 +752,12 @@ The [`unflatten()`](../api/basic_json/unflatten.md) function only works for an o
The [`dump()`](../api/basic_json/dump.md) function only works with UTF-8 encoded strings; that is, if you assign a `std::string` to a JSON value, make sure it is UTF-8 encoded. See the FAQ entry on [serializing untrusted or invalid UTF-8](faq.md#serializing-untrusted-or-invalid-utf-8) for background and the recommended fix.
The binary writers [`to_cbor()`](../api/basic_json/to_cbor.md), [`to_ubjson()`](../api/basic_json/to_ubjson.md),
[`to_bjdata()`](../api/basic_json/to_bjdata.md), and [`to_bson()`](../api/basic_json/to_bson.md) throw this exception
as well for a string value or object key that is not valid UTF-8 if their `error_handler` is `strict` (the default if
[`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled). So does
[`to_msgpack()`](../api/basic_json/to_msgpack.md) if `error_handler_t::strict` is passed.
!!! failure "Example message"
Calling `dump()` on a JSON value containing an ISO 8859-1 encoded string:
+5
View File
@@ -212,6 +212,11 @@ Use the non-amalgamated version of the library. This option is `ON` by default.
Treat the library headers like system headers (i.e., adding `SYSTEM` to the [`target_include_directories`](https://cmake.org/cmake/help/latest/command/target_include_directories.html) call) to check for this library by tools like Clang-Tidy. This option is `OFF` by default.
### `JSON_StrictBinaryUTF8`
Check string values and object keys for valid UTF-8 in the CBOR, UBJSON, BJData, and BSON writers, by defining the
macro [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md). This option is `OFF` by default.
### `JSON_StrictNulHandling`
Reject a `'\0'` (NUL) byte in the input instead of treating it as end of input, by defining the macro
@@ -2,11 +2,14 @@
This page collects some guidelines on how to future-proof your code for future versions of this library. For how to
add the library to your project in the first place, see [Integration](index.md), [CMake](cmake.md), or
[Package Managers](package_managers.md).
[Package Managers](package_managers.md). The [roadmap](../community/roadmap.md#version-40) lists what will change in
version 4.0, including the macros that let you try its behavior with a 3.x release; this page describes how to adjust
your code.
## Replace deprecated functions
The following functions have been deprecated and will be removed in the next major version (i.e., 4.0.0). All
The following functions have been deprecated and will be removed in the next major version (i.e., 4.0.0), see the
[roadmap](../community/roadmap.md#removal-of-deprecated-functions) for an overview. All
deprecations are annotated with
[`HEDLEY_DEPRECATED_FOR`](https://nemequ.github.io/hedley/api-reference.html#HEDLEY_DEPRECATED_FOR) to report which
function to use instead.
+2
View File
@@ -117,6 +117,7 @@ nav:
- 'accept': api/basic_json/accept.md
- 'array': api/basic_json/array.md
- 'array_t': api/basic_json/array_t.md
- 'as_base_class': api/basic_json/as_base_class.md
- 'at': api/basic_json/at.md
- 'back': api/basic_json/back.md
- 'begin': api/basic_json/begin.md
@@ -307,6 +308,7 @@ nav:
- 'JSON_PRECISE_STREAM_POSITION': api/macros/json_precise_stream_position.md
- 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md
- 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md
- 'JSON_STRICT_BINARY_UTF8': api/macros/json_strict_binary_utf8.md
- 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md
- 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md
- 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md
+15 -4
View File
@@ -46,6 +46,10 @@
#define JSON_STRICT_NUL_HANDLING 0
#endif
#ifndef JSON_STRICT_BINARY_UTF8
#define JSON_STRICT_BINARY_UTF8 0
#endif
#if JSON_DIAGNOSTICS
#define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag
#else
@@ -82,14 +86,20 @@
#define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING
#endif
#if JSON_STRICT_BINARY_UTF8
#define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8
#else
#define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8
#endif
#ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION
#define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0
#endif
// Construct the namespace ABI tags component
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f)
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g)
#define NLOHMANN_JSON_ABI_TAGS \
NLOHMANN_JSON_ABI_TAGS_CONCAT( \
@@ -98,7 +108,8 @@
NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \
NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \
NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \
NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING)
NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \
NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8)
// Construct the namespace version component
#define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \
@@ -27,7 +27,6 @@
#include <nlohmann/detail/meta/identity_tag.hpp>
#include <nlohmann/detail/meta/std_fs.hpp>
#include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/detail/meta/logic.hpp>
#include <nlohmann/detail/string_concat.hpp>
#include <nlohmann/detail/value_t.hpp>
@@ -211,62 +210,29 @@ inline void from_json(const BasicJsonType& j, std::valarray<T>& l)
});
}
// element is not itself a C array: read it directly
template<typename BasicJsonType, typename T>
auto from_json_c_array_element(const BasicJsonType& j, T& e)
-> decltype(e = j.template get<T>(), void())
{
e = j.template get<T>();
}
// element is itself a C array: recurse one dimension at a time, so any rank is supported
template<typename BasicJsonType, typename T, std::size_t N>
auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
-> decltype(j.template get<T>(), void())
void from_json_c_array_element(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
{
for (std::size_t i = 0; i < N; ++i)
{
arr[i] = j.at(i).template get<T>();
from_json_c_array_element(j.at(i), arr[i]);
}
}
template<typename BasicJsonType, typename T, std::size_t N1, std::size_t N2>
auto from_json(const BasicJsonType& j, T (&arr)[N1][N2]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
-> decltype(j.template get<T>(), void())
template<typename BasicJsonType, typename T, std::size_t N>
auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
-> decltype(j.template get<typename std::remove_all_extents<T>::type>(), void())
{
for (std::size_t i1 = 0; i1 < N1; ++i1)
{
for (std::size_t i2 = 0; i2 < N2; ++i2)
{
arr[i1][i2] = j.at(i1).at(i2).template get<T>();
}
}
}
template<typename BasicJsonType, typename T, std::size_t N1, std::size_t N2, std::size_t N3>
auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
-> decltype(j.template get<T>(), void())
{
for (std::size_t i1 = 0; i1 < N1; ++i1)
{
for (std::size_t i2 = 0; i2 < N2; ++i2)
{
for (std::size_t i3 = 0; i3 < N3; ++i3)
{
arr[i1][i2][i3] = j.at(i1).at(i2).at(i3).template get<T>();
}
}
}
}
template<typename BasicJsonType, typename T, std::size_t N1, std::size_t N2, std::size_t N3, std::size_t N4>
auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3][N4]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
-> decltype(j.template get<T>(), void())
{
for (std::size_t i1 = 0; i1 < N1; ++i1)
{
for (std::size_t i2 = 0; i2 < N2; ++i2)
{
for (std::size_t i3 = 0; i3 < N3; ++i3)
{
for (std::size_t i4 = 0; i4 < N4; ++i4)
{
arr[i1][i2][i3][i4] = j.at(i1).at(i2).at(i3).at(i4).template get<T>();
}
}
}
}
from_json_c_array_element(j, arr);
}
template<typename BasicJsonType>
@@ -286,20 +252,33 @@ auto from_json_array_impl(const BasicJsonType& j, std::array<T, N>& arr,
}
}
// reserve() is called through this pair (modeled on from_json_object_reserve)
// so from_json_array_impl below has a single body for both ConstructibleArrayType
// that support reserve() and those that don't.
template<typename ConstructibleArrayType>
auto from_json_array_reserve(ConstructibleArrayType& arr, typename ConstructibleArrayType::size_type size, priority_tag<1> /*unused*/)
-> decltype(arr.reserve(size), void())
{
arr.reserve(size);
}
template<typename ConstructibleArrayType>
void from_json_array_reserve(ConstructibleArrayType& /*arr*/, std::size_t /*size*/, priority_tag<0> /*unused*/)
{}
template<typename BasicJsonType, typename ConstructibleArrayType,
enable_if_t<
std::is_assignable<ConstructibleArrayType&, ConstructibleArrayType>::value,
int> = 0>
auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, priority_tag<1> /*unused*/)
-> decltype(
arr.reserve(std::declval<typename ConstructibleArrayType::size_type>()),
j.template get<typename ConstructibleArrayType::value_type>(),
void())
{
using std::end;
ConstructibleArrayType ret;
ret.reserve(j.size());
from_json_array_reserve(ret, j.size(), priority_tag<1> {});
std::transform(j.begin(), j.end(),
std::inserter(ret, end(ret)), [](const BasicJsonType & i)
{
@@ -310,27 +289,6 @@ auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, p
arr = std::move(ret);
}
template<typename BasicJsonType, typename ConstructibleArrayType,
enable_if_t<
std::is_assignable<ConstructibleArrayType&, ConstructibleArrayType>::value,
int> = 0>
inline void from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr,
priority_tag<0> /*unused*/)
{
using std::end;
ConstructibleArrayType ret;
std::transform(
j.begin(), j.end(), std::inserter(ret, end(ret)),
[](const BasicJsonType & i)
{
// get<BasicJsonType>() returns *this, this won't call a from_json
// method when value_type is BasicJsonType
return i.template get<typename ConstructibleArrayType::value_type>();
});
arr = std::move(ret);
}
template < typename BasicJsonType, typename ConstructibleArrayType,
enable_if_t <
is_constructible_array_type<BasicJsonType, ConstructibleArrayType>::value&&
@@ -433,9 +391,7 @@ inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj)
}
// overload for arithmetic types, not chosen for basic_json template arguments
// (BooleanType, etc.); note: Is it really necessary to provide explicit
// overloads for boolean_t etc. in case of a custom BooleanType which is not
// an arithmetic type?
// (BooleanType, etc.)
template < typename BasicJsonType, typename ArithmeticType,
enable_if_t <
std::is_arithmetic<ArithmeticType>::value&&
@@ -531,7 +487,7 @@ inline void from_json_tuple_impl(const BasicJsonType& j, std::pair<A1, A2>& p, p
template<typename BasicJsonType, typename... Args>
std::tuple<Args...> from_json_tuple_impl(const BasicJsonType& j, identity_tag<std::tuple<Args...>> /*unused*/, priority_tag<2> /*unused*/)
{
static_assert(cxpr_and<cxpr_or<cxpr_not<std::is_reference<Args>>, is_compatible_reference_type<const BasicJsonType&, Args>>...>::value,
static_assert(conjunction<disjunction<negation<std::is_reference<Args>>, is_compatible_reference_type<const BasicJsonType&, Args>>...>::value,
"Can not return a tuple containing references to types not contained in a Json, try Json::get_to()");
return from_json_tuple_impl_base<1, Args...>(j, index_sequence_for<Args...> {});
}
@@ -554,10 +510,10 @@ auto from_json(const BasicJsonType& j, TupleRelated&& t)
return from_json_tuple_impl(j, std::forward<TupleRelated>(t), priority_tag<3> {});
}
template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator,
typename = enable_if_t < !std::is_constructible <
typename BasicJsonType::string_t, Key >::value >>
inline void from_json(const BasicJsonType& j, std::map<Key, Value, Compare, Allocator>& m)
// shared body for std::map/std::unordered_map with a non-string Key: both
// containers are read from an array of [key, value] pairs the same way
template<typename BasicJsonType, typename MapType>
void from_json_pair_array_to_map(const BasicJsonType& j, MapType& m)
{
if (JSON_HEDLEY_UNLIKELY(!j.is_array()))
{
@@ -570,33 +526,29 @@ inline void from_json(const BasicJsonType& j, std::map<Key, Value, Compare, Allo
{
JSON_THROW(type_error::create(302, concat("type must be array, but is ", p.type_name()), &p));
}
m.emplace(p.at(0).template get<Key>(), p.at(1).template get<Value>());
m.emplace(p.at(0).template get<typename MapType::key_type>(), p.at(1).template get<typename MapType::mapped_type>());
}
}
template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator,
typename = enable_if_t < !std::is_constructible <
typename BasicJsonType::string_t, Key >::value >>
void from_json(const BasicJsonType& j, std::map<Key, Value, Compare, Allocator>& m)
{
from_json_pair_array_to_map(j, m);
}
template < typename BasicJsonType, typename Key, typename Value, typename Hash, typename KeyEqual, typename Allocator,
typename = enable_if_t < !std::is_constructible <
typename BasicJsonType::string_t, Key >::value >>
inline void from_json(const BasicJsonType& j, std::unordered_map<Key, Value, Hash, KeyEqual, Allocator>& m)
void from_json(const BasicJsonType& j, std::unordered_map<Key, Value, Hash, KeyEqual, Allocator>& m)
{
if (JSON_HEDLEY_UNLIKELY(!j.is_array()))
{
JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j));
}
m.clear();
for (const auto& p : j)
{
if (JSON_HEDLEY_UNLIKELY(!p.is_array()))
{
JSON_THROW(type_error::create(302, concat("type must be array, but is ", p.type_name()), &p));
}
m.emplace(p.at(0).template get<Key>(), p.at(1).template get<Value>());
}
from_json_pair_array_to_map(j, m);
}
#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM
// Workaround for MSVC 19.51 (and possibly later): in large in large cpp files, the compiler may fail to resolve with generic has_from_json (issue #4996)
// Workaround for MSVC 19.51 (and possibly later): in large cpp files, the compiler may fail to resolve with generic has_from_json (issue #4996)
template<typename BasicJsonType>
struct has_from_json<BasicJsonType, std_fs::path, void> : std::true_type {};
@@ -178,7 +178,7 @@ struct external_constructor<value_t::array>
template < typename BasicJsonType, typename CompatibleArrayType,
enable_if_t < !std::is_same<CompatibleArrayType, typename BasicJsonType::array_t>::value
#if JSON_HAS_RANGES && !defined(__MINGW32__)
#if JSON_HAS_RANGE_VIEW_CONVERSION
&& !is_compatible_range_view<CompatibleArrayType>::value
#endif
, int > = 0 >
@@ -222,9 +222,7 @@ struct external_constructor<value_t::array>
j.assert_invariant();
}
// std::ranges does not work properly on MinGW due to incomplete C++20 support
// see https://github.com/nlohmann/json/issues/4916
#if JSON_HAS_RANGES && !defined(__MINGW32__)
#if JSON_HAS_RANGE_VIEW_CONVERSION
template<typename BasicJsonType, typename CompatibleArrayType,
enable_if_t<is_compatible_range_view<std::remove_cvref_t<CompatibleArrayType>>::value, int> = 0>
static void construct(BasicJsonType& j, CompatibleArrayType && arr)
@@ -384,7 +382,7 @@ template < typename BasicJsonType, typename CompatibleArrayType,
!std::is_same<typename BasicJsonType::binary_t, CompatibleArrayType>::value&&
!is_compatible_binary_type<BasicJsonType, CompatibleArrayType>::value&&
!is_basic_json<CompatibleArrayType>::value
#if JSON_HAS_RANGES && !defined(__MINGW32__)
#if JSON_HAS_RANGE_VIEW_CONVERSION
&& !is_compatible_range_view<CompatibleArrayType>::value
#endif
,
@@ -394,7 +392,7 @@ inline void to_json(BasicJsonType& j, const CompatibleArrayType& arr)
external_constructor<value_t::array>::construct(j, arr);
}
#if JSON_HAS_RANGES && !defined(__MINGW32__)
#if JSON_HAS_RANGE_VIEW_CONVERSION
template < typename BasicJsonType, typename T,
enable_if_t < is_compatible_range_view<std::remove_cvref_t<T>>::value
&& !is_compatible_string_type<BasicJsonType, std::remove_cvref_t<T>>::value
+21
View File
@@ -286,6 +286,27 @@ class other_error : public exception
other_error(int id_, const char* what_arg) : exception(id_, what_arg) {}
};
/*!
@brief helper function to call JSON_THROW from a template
@note JSON_THROW is a macro that, depending on the JSON_THROW_USER /
JSON_TRY_USER / JSON_NOEXCEPTION configuration, may expand to code
that does not reference its argument (e.g. `std::abort()`), which
would trigger a compilation error if the argument's type depends on
a template parameter that is otherwise unused. Wrapping the call in
a templated function avoids this and gives the compiler a single
place to see the (possibly unused) parameter.
*/
template<typename ExceptionType>
void templated_json_throw(ExceptionType exception)
{
JSON_THROW(exception);
// JSON_THROW may expand to code that discards its argument (e.g. when
// exceptions are disabled) - the cast below avoids an unused-parameter
// warning with -Werror in that case
(void)exception;
}
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
+83 -48
View File
@@ -31,6 +31,7 @@
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/is_sax.hpp>
#include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/detail/output/error_handler.hpp>
#include <nlohmann/detail/string_concat.hpp>
#include <nlohmann/detail/string_utils.hpp>
#include <nlohmann/detail/value_t.hpp>
@@ -130,8 +131,16 @@ class binary_reader
@brief create a binary reader
@param[in] adapter input adapter to read from
@param[in] format the binary format to parse
@param[in] error_handler how to treat text strings and object keys that
are not well-formed UTF-8; none of the supported formats
requires a decoder to reject those, so the default is to
@ref error_handler_t::keep them unchanged, as every binary
reader did before this parameter existed
*/
explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json) noexcept : ia(std::move(adapter)), input_format(format)
explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json,
const error_handler_t error_handler = error_handler_t::keep) noexcept
: ia(std::move(adapter)), input_format(format), error_handler(error_handler)
{
(void)detail::is_sax_static_asserts<SAX, BasicJsonType> {};
}
@@ -594,7 +603,7 @@ class binary_reader
{
if (get_bson_cstr_bulk(result, std::integral_constant<bool, bulk_scan> {}))
{
return true;
return check_string_utf8(result, "key");
}
auto out = std::back_inserter(result);
@@ -607,7 +616,7 @@ class binary_reader
}
if (current == 0x00)
{
return true;
return check_string_utf8(result, "key");
}
*out++ = static_cast<typename string_t::value_type>(current);
}
@@ -693,7 +702,7 @@ class binary_reader
return current != char_traits<char_type>::eof() && repair_requested();
}
return true;
return check_string_utf8(result, "string");
}
/*!
@@ -1353,7 +1362,7 @@ class binary_reader
@return whether string creation completed
*/
bool get_cbor_string(string_t& result)
bool get_cbor_string(string_t& result, const char* context = "string")
{
// number of indefinite-length strings that have been opened and not
// closed yet. RFC 8949, Section 3.2.3 does not permit nesting them,
@@ -1383,7 +1392,7 @@ class binary_reader
{
if (--open == 0)
{
return true;
return check_string_utf8(result, context);
}
get();
continue;
@@ -1396,7 +1405,7 @@ class binary_reader
if (open == 0)
{
return true;
return check_string_utf8(result, context);
}
get();
@@ -1421,7 +1430,7 @@ class binary_reader
// EOF and major type 3 (text string) are left to get_cbor_string
if (current == char_traits<char_type>::eof() || (static_cast<unsigned int>(current) & 0xE0u) == 0x60u)
{
return get_cbor_string(result);
return get_cbor_string(result, "key");
}
const char* found = nullptr;
@@ -2334,7 +2343,7 @@ class binary_reader
@return whether string creation completed
*/
bool get_msgpack_string(string_t& result)
bool get_msgpack_string(string_t& result, const char* context = "string")
{
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "string")))
{
@@ -2377,25 +2386,25 @@ class binary_reader
case 0xBE:
case 0xBF:
{
return get_string(input_format_t::msgpack, static_cast<unsigned int>(current) & 0x1Fu, result);
return get_string(input_format_t::msgpack, static_cast<unsigned int>(current) & 0x1Fu, result) && check_string_utf8(result, context);
}
case 0xD9: // str 8
{
std::uint8_t len{};
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result);
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context);
}
case 0xDA: // str 16
{
std::uint16_t len{};
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result);
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context);
}
case 0xDB: // str 32
{
std::uint32_t len{};
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result);
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context);
}
default:
@@ -2474,7 +2483,7 @@ class binary_reader
// byte 0xC1 are left to get_msgpack_string
if (current == char_traits<char_type>::eof())
{
return get_msgpack_string(result);
return get_msgpack_string(result, "key");
}
if (current <= 0x7F || current >= 0xE0)
{
@@ -2490,7 +2499,7 @@ class binary_reader
}
else
{
return get_msgpack_string(result);
return get_msgpack_string(result, "key");
}
break;
}
@@ -2887,7 +2896,7 @@ class binary_reader
if (top.is_object)
{
key.clear();
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key)))
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, true, "key") || !sax->key(key)))
{
return false;
}
@@ -2909,7 +2918,7 @@ class binary_reader
if (top.is_object)
{
key.clear();
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key)))
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false, "key") || !sax->key(key)))
{
return false;
}
@@ -2977,7 +2986,7 @@ class binary_reader
@return whether string creation completed
*/
bool get_ubjson_string(string_t& result, const bool get_char = true)
bool get_ubjson_string(string_t& result, const bool get_char = true, const char* context = "string")
{
if (get_char)
{
@@ -2998,31 +3007,31 @@ class binary_reader
case 'U':
{
std::uint8_t len{};
return get_number(input_format, len) && get_string(input_format, len, result);
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'i':
{
std::int8_t len{};
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'I':
{
std::int16_t len{};
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'l':
{
std::int32_t len{};
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'L':
{
std::int64_t len{};
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'u':
@@ -3032,7 +3041,7 @@ class binary_reader
break;
}
std::uint16_t len{};
return get_number(input_format, len) && get_string(input_format, len, result);
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'm':
@@ -3042,7 +3051,7 @@ class binary_reader
break;
}
std::uint32_t len{};
return get_number(input_format, len) && get_string(input_format, len, result);
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
case 'M':
@@ -3052,7 +3061,7 @@ class binary_reader
break;
}
std::uint64_t len{};
return get_number(input_format, len) && get_string(input_format, len, result);
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
}
default:
@@ -3557,10 +3566,9 @@ class binary_reader
{
return false;
}
// when recovering, the character becomes U+FFFD, as an
// invalid byte in a string does
string_t replacement;
append_replacement_character(replacement);
// when recovering, the character becomes U+FFFD, as with
// error_handler_t::replace
string_t replacement = sanitize_utf8(string_t(1, static_cast<typename string_t::value_type>(current)), error_handler_t::replace);
return sax->string(replacement);
}
string_t s(1, static_cast<typename string_t::value_type>(current));
@@ -4731,34 +4739,58 @@ class binary_reader
const NumberType len,
string_t& result)
{
// get_bytes() appends to result, and CBOR indefinite-length strings
// collect all their chunks in the same result; validating only the
// newly read bytes keeps the check linear in the input size
const std::size_t old_size = result.size();
if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result)))
// Strings are taken as is by default: none of CBOR (RFC 8949 §3.1
// leaves the choice to the decoder), MessagePack (whose spec
// explicitly allows a str object to contain an invalid byte
// sequence), UBJSON, BJData, or BSON requires a decoder to reject
// ill-formed UTF-8. Checking (and, with @ref error_handler_t::strict,
// rejecting, or with `replace`/`ignore`, sanitizing) is opt-in via
// @ref error_handler, applied once the whole string (all chunks of
// an indefinite-length CBOR string included) has been assembled, by
// @ref check_string_utf8 at the call site.
return get_bytes(format, len, "string", result);
}
/*!
@brief validate a decoded text string (value or object key) against @ref error_handler
None of the binary formats requires a decoder to reject ill-formed UTF-8
in a text string (see @ref get_string), so by default
(@ref error_handler_t::keep) this does nothing. A stricter
@ref error_handler opts into the same well-formedness check @ref
serializer::dump_escaped_impl applies when dumping a string:
@ref error_handler_t::strict rejects ill-formed input with
parse_error.113 (honoring `allow_exceptions` via @a sax), while
@ref error_handler_t::replace / @ref error_handler_t::ignore sanitize
@a result in place, using the exact same rules.
@param[in,out] result the already assembled string to check
@param[in] context further context information (for diagnostics)
@return whether @a result is acceptable (always true for `keep`)
*/
bool check_string_utf8(string_t& result, const char* context)
{
if (error_handler == error_handler_t::keep || is_valid_utf8(result))
{
return false;
return true;
}
// RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications
// all require text strings to be valid UTF-8; reject anything else
// right here so malformed input is caught at decode time instead of
// only surfacing later as a type_error.316 when the value is dumped
// (which would defeat allow_exceptions=false / strict discarding).
if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size)))
if (error_handler == error_handler_t::strict)
{
static_cast<void>(report_error(chars_read, get_token_string(),
parse_error::create(113, chars_read,
exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)));
auto last_token = get_token_string();
static_cast<void>(report_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(input_format, "invalid string: ill-formed UTF-8 byte", context), nullptr)));
// when recovering, the string is repaired as with
// error_handler_t::replace
if (!repair_requested())
{
return false;
}
// when recovering, each ill-formed sequence becomes U+FFFD, as it
// does in JSON text
replace_invalid_utf8(result, old_size);
result = sanitize_utf8(result, error_handler_t::replace);
return true;
}
result = sanitize_utf8(result, error_handler);
return true;
}
@@ -5222,6 +5254,9 @@ class binary_reader
/// input format
const input_format_t input_format = input_format_t::json;
/// how to treat text strings/object keys that are not well-formed UTF-8
const error_handler_t error_handler = error_handler_t::keep;
/// the SAX parser
json_sax_t* sax = nullptr;
@@ -453,8 +453,10 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
}
else
{
// get the current character
const auto wc = input.get_character();
// get the current character; converted to an unsigned type so that
// a negative unit (wint_t is signed on some platforms) is not
// mistaken for an ASCII character or for EOF
const auto wc = static_cast<std::uint32_t>(input.get_character());
if (wc <= 0x10FFFF)
{
@@ -522,9 +524,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
bool valid_pair = false;
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
{
const auto wc2 = static_cast<unsigned int>(input.get_character());
// only consume the next unit if it completes the pair
const auto wc2 = static_cast<unsigned int>(*input.current);
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{
input.get_character();
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes_filled = 0;
encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
@@ -537,7 +541,8 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (!valid_pair)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
utf8_bytes[0] = 0xFF;
utf8_bytes_filled = 1;
}
}
@@ -746,7 +751,7 @@ struct container_input_adapter_factory< ContainerType,
{
// container is forwarded twice on purpose: the resulting begin/end
// iterator types must match adapter_type, computed the same way
// NOLINTNEXTLINE(bugprone-use-after-move)
// NOLINTNEXTLINE(bugprone-use-after-move,hicpp-invalid-access-moved)
return input_adapter(begin(std::forward<ContainerType>(container)), end(std::forward<ContainerType>(container)));
}
};
@@ -60,9 +60,11 @@ class iter_impl // NOLINT(cppcoreguidelines-special-member-functions,hicpp-speci
static_assert(is_basic_json<typename std::remove_const<BasicJsonType>::type>::value,
"iter_impl only accepts (const) basic_json");
// superficial check for the LegacyBidirectionalIterator named requirement
static_assert(std::is_base_of<std::bidirectional_iterator_tag, std::bidirectional_iterator_tag>::value
&& std::is_base_of<std::bidirectional_iterator_tag, typename std::iterator_traits<typename array_t::iterator>::iterator_category>::value,
"basic_json iterator assumes array and object type iterators satisfy the LegacyBidirectionalIterator named requirement.");
// note: only array_t::iterator is checked here; object_t::iterator may be
// a forward-only iterator as long as reverse iteration and operator--
// are never used on it
static_assert(std::is_base_of<std::bidirectional_iterator_tag, typename std::iterator_traits<typename array_t::iterator>::iterator_category>::value,
"basic_json iterator assumes array type iterators satisfy the LegacyBidirectionalIterator named requirement.");
public:
/// The std::iterator class template (used as a base class to provide typedefs) is deprecated in C++17.
+104 -148
View File
@@ -240,6 +240,72 @@ class json_pointer
}
private:
/*!
@brief result of @ref parse_array_index
@ref array_index maps each value to the corresponding parse_error/out_of_range
exception; @ref contains and @ref get_checked_or_null, which must not throw for
an out-of-range or unrepresentable index, switch on it directly instead.
*/
enum class array_index_status
{
ok, ///< @a s is a valid, representable array index
leading_zero, ///< @a s begins with '0' but has more than one character
not_a_number, ///< @a s does not begin with a digit
unresolved, ///< @a s could not be converted to an integer
exceeds_size_type ///< @a s converts to an integer that exceeds size_type
};
/*!
@param[in] s reference token to be converted into an array index
@param[out] idx the integer representation of @a s if @ref array_index_status::ok
is returned; left unchanged otherwise
@return whether @a s is a valid array index, and if not, why
@note this function never throws; @ref array_index and the callers that must not
throw (@ref contains, @ref get_checked_or_null) build on it instead of each
re-implementing the RFC 6901 digit rules and the @a size_type range check
*/
template<typename BasicJsonType>
static array_index_status parse_array_index(const string_t& s, typename BasicJsonType::size_type& idx) noexcept
{
using size_type = typename BasicJsonType::size_type;
// error condition (cf. RFC 6901, Sect. 4)
if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0'))
{
return array_index_status::leading_zero;
}
// error condition (cf. RFC 6901, Sect. 4)
if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9')))
{
return array_index_status::not_a_number;
}
const char* p = s.data();
char* p_end = nullptr; // NOLINT(misc-const-correctness)
errno = 0; // strtoull doesn't reset errno
const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int)
if (p == p_end // invalid input or empty string
|| errno == ERANGE // out of range
|| JSON_HEDLEY_UNLIKELY(static_cast<std::size_t>(p_end - p) != s.size())) // incomplete read
{
return array_index_status::unresolved;
}
// the index does not fit into size_type; on 64-bit platforms this is
// only SIZE_MAX itself (see #2203 and #5395)
if (res >= static_cast<unsigned long long>((std::numeric_limits<size_type>::max)())) // NOLINT(runtime/int)
{
return array_index_status::exceeds_size_type;
}
idx = static_cast<size_type>(res);
return array_index_status::ok;
}
/*!
@param[in] s reference token to be converted into an array index
@@ -253,39 +319,23 @@ class json_pointer
template<typename BasicJsonType>
static typename BasicJsonType::size_type array_index(const string_t& s)
{
using size_type = typename BasicJsonType::size_type;
// error condition (cf. RFC 6901, Sect. 4)
if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0'))
typename BasicJsonType::size_type idx{};
switch (parse_array_index<BasicJsonType>(s, idx))
{
JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr));
case array_index_status::leading_zero:
JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr));
case array_index_status::not_a_number:
JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr));
case array_index_status::unresolved:
JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr));
case array_index_status::exceeds_size_type:
JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr));
case array_index_status::ok:
default:
break;
}
// error condition (cf. RFC 6901, Sect. 4)
if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9')))
{
JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr));
}
const char* p = s.data();
char* p_end = nullptr; // NOLINT(misc-const-correctness)
errno = 0; // strtoull doesn't reset errno
const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int)
if (p == p_end // invalid input or empty string
|| errno == ERANGE // out of range
|| JSON_HEDLEY_UNLIKELY(static_cast<std::size_t>(p_end - p) != s.size())) // incomplete read
{
JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr));
}
// the index does not fit into size_type; on 64-bit platforms this is
// only SIZE_MAX itself (see #2203 and #5395)
if (res >= static_cast<unsigned long long>((std::numeric_limits<size_type>::max)())) // NOLINT(runtime/int)
{
JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr));
}
return static_cast<size_type>(res);
return idx;
}
JSON_PRIVATE_UNLESS_TESTED:
@@ -590,63 +640,6 @@ class json_pointer
return *ptr;
}
/*!
@throw parse_error.106 if an array index begins with '0'
@throw parse_error.109 if an array index was not a number
@throw out_of_range.402 if the array index '-' is used
@throw out_of_range.404 if the JSON pointer can not be resolved
*/
template<typename BasicJsonType>
const BasicJsonType& get_checked(const BasicJsonType* ptr) const
{
for (const auto& reference_token : reference_tokens)
{
switch (ptr->type())
{
case detail::value_t::object:
{
// note: at performs range check
ptr = &ptr->at(reference_token);
break;
}
case detail::value_t::array:
{
if (JSON_HEDLEY_UNLIKELY(reference_token == "-"))
{
// "-" always fails the range check
JSON_THROW(detail::out_of_range::create(402, detail::concat(
"array index '-' (", std::to_string(ptr->m_data.m_value.array->size()),
") is out of range"), ptr));
}
const auto idx = array_index<BasicJsonType>(reference_token);
// Bounds check before access to avoid exception with JSON_NOEXCEPTION
if (JSON_HEDLEY_UNLIKELY(idx >= ptr->m_data.m_value.array->size()))
{
JSON_THROW(detail::out_of_range::create(401, detail::concat(
"array index ", std::to_string(idx), " is out of range"), ptr));
}
ptr = &ptr->operator[](idx);
break;
}
case detail::value_t::null:
case detail::value_t::string:
case detail::value_t::boolean:
case detail::value_t::number_integer:
case detail::value_t::number_unsigned:
case detail::value_t::number_float:
case detail::value_t::binary:
case detail::value_t::discarded:
default:
JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", reference_token, "'"), ptr));
}
}
return *ptr;
}
/*!
@brief return a pointer to the pointed to value, or `nullptr` if the
pointer cannot be resolved because a key is missing, an array
@@ -685,29 +678,24 @@ class json_pointer
return nullptr;
}
// tokens that array_index() rejects with parse_error.106/109
// are passed on to it; all other tokens that it would reject
// with out_of_range.404/410 are detected here, so that this
// also works without exceptions
if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1 && !(reference_token[0] >= '1' && reference_token[0] <= '9')))
// a malformed index throws parse_error.106/109; an
// index that is syntactically valid but cannot be
// represented (out_of_range.404/410) is treated like an
// out-of-range index below
typename BasicJsonType::size_type idx{};
switch (parse_array_index<BasicJsonType>(reference_token, idx))
{
static_cast<void>(array_index<BasicJsonType>(reference_token)); // throws parse_error.106/109
case array_index_status::leading_zero:
JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", reference_token, "' must not begin with '0'"), nullptr));
case array_index_status::not_a_number:
JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", reference_token, "' is not a number"), nullptr));
case array_index_status::unresolved:
case array_index_status::exceeds_size_type:
return nullptr;
case array_index_status::ok:
default:
break;
}
if (JSON_HEDLEY_UNLIKELY(reference_token.empty() || !std::all_of(reference_token.begin(), reference_token.end(), [](const char c)
{
return c >= '0' && c <= '9';
})))
{
return nullptr;
}
errno = 0; // strtoull() does not reset errno on success
char* p_end = nullptr; // NOLINT(misc-const-correctness)
const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int)
if (JSON_HEDLEY_UNLIKELY(errno == ERANGE || magnitude >= static_cast<unsigned long long>((std::numeric_limits<typename BasicJsonType::size_type>::max)()))) // NOLINT(runtime/int)
{
return nullptr;
}
const auto idx = static_cast<typename BasicJsonType::size_type>(magnitude);
if (JSON_HEDLEY_UNLIKELY(idx >= ptr->m_data.m_value.array->size()))
{
@@ -734,8 +722,8 @@ class json_pointer
}
/*!
@throw parse_error.106 if an array index begins with '0'
@throw parse_error.109 if an array index was not a number
@note unlike array_index(), this never throws: a malformed or unrepresentable
array index reference token is treated like a missing key (see #5395)
*/
template<typename BasicJsonType>
bool contains(const BasicJsonType* ptr) const
@@ -763,49 +751,17 @@ class json_pointer
// "-" always fails the range check
return false;
}
if (JSON_HEDLEY_UNLIKELY(reference_token.empty()))
{
// an empty reference token is not an array index; array_index()
// would throw out_of_range.404 -- contains() must not throw (see #5395)
return false;
}
if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9')))
{
// invalid char
return false;
}
if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1))
{
if (JSON_HEDLEY_UNLIKELY(!('1' <= reference_token[0] && reference_token[0] <= '9')))
{
// the first char should be between '1' and '9'
return false;
}
for (std::size_t i = 1; i < reference_token.size(); i++)
{
if (JSON_HEDLEY_UNLIKELY(!('0' <= reference_token[i] && reference_token[i] <= '9')))
{
// other char should be between '0' and '9'
return false;
}
}
}
// the reference token consists only of digits at this point (cf. checks
// above); however, its numeric value might not be representable, in which
// case array_index() would throw out_of_range.404/410 -- contains() must
// not throw (see #5395), so such a reference token is treated as "not found"
errno = 0; // strtoull() does not reset errno on success
char* p_end = nullptr; // NOLINT(misc-const-correctness)
const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int)
if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX
|| magnitude >= static_cast<unsigned long long>((std::numeric_limits<typename BasicJsonType::size_type>::max)()))) // NOLINT(runtime/int)
// any parse failure (malformed index, or one that is syntactically
// valid but not representable as size_type) means the reference
// token cannot denote an existing array element -- contains() must
// not throw (see #5395), so it is treated as "not found"
typename BasicJsonType::size_type idx{};
if (JSON_HEDLEY_UNLIKELY(parse_array_index<BasicJsonType>(reference_token, idx) != array_index_status::ok))
{
// the array index cannot be represented as size_type
return false;
}
const auto idx = array_index<BasicJsonType>(reference_token);
if (idx >= ptr->size())
{
// index out of range
+17 -43
View File
@@ -9,7 +9,6 @@
#pragma once
#include <utility> // declval, pair
#include <nlohmann/detail/meta/detected.hpp>
#include <nlohmann/thirdparty/hedley/hedley.hpp>
// This file contains all internal macro definitions (except those affecting ABI)
@@ -140,10 +139,12 @@
// libstdc++ < 11 has incomplete C++20 ranges (issue #4440)
#elif defined(_GLIBCXX_RELEASE) && _GLIBCXX_RELEASE < 11
#define JSON_HAS_RANGES 0
// libc++ < 16 has incomplete C++20 ranges (issue #4440)
// clang < 16 with libstdc++ does not implement the ranges customization
// points libstdc++ declares, so its C++20 ranges support is incomplete (issue #5161)
#elif defined(__clang__) && !defined(__apple_build_version__) \
&& __clang_major__ < 16 && defined(__GLIBCXX__)
#define JSON_HAS_RANGES 0
// libc++ < 16 has incomplete C++20 ranges (issue #4440)
#elif defined(_LIBCPP_VERSION) && _LIBCPP_VERSION < 160000
#define JSON_HAS_RANGES 0
// nvcc CUDA 12.0/12.1 chokes on the enable_borrowed_range variable-template
@@ -158,6 +159,18 @@
#endif
#endif
// std::ranges view conversion (to_json/is_compatible_array_type_impl) additionally
// needs to be disabled on MinGW, whose std::ranges support is incomplete
// (issue #4916); this macro combines both conditions so the check and its
// reason are not duplicated at every use site.
#ifndef JSON_HAS_RANGE_VIEW_CONVERSION
#if JSON_HAS_RANGES && !defined(__MINGW32__)
#define JSON_HAS_RANGE_VIEW_CONVERSION 1
#else
#define JSON_HAS_RANGE_VIEW_CONVERSION 0
#endif
#endif
#ifndef JSON_HAS_STD_FORMAT
#if defined(JSON_HAS_CPP_20) && defined(__cpp_lib_format)
#define JSON_HAS_STD_FORMAT 1
@@ -279,21 +292,6 @@
/*!
@brief function to wrap JSON_THROW_MACRO - there can be compilation errors about
there being no arguments to JSON_THROW that depend on template arguments
if this is not used to call JSON_THROW
*/
template<typename ExceptionType>
void templated_json_throw(ExceptionType exception)
{
JSON_THROW(exception);
/* JSON_THROW(exception) discards exception and aborts - void cast needed to supress
compilation error if compiled with -Werror and Wunused-parameter */
(void)exception;
}
/*!
@brief macro to briefly define a mapping between an enum and JSON with exception
on invalid input
@@ -314,7 +312,7 @@ void templated_json_throw(ExceptionType exception)
return ej_pair.first == e; \
}); \
if (it != std::end(m)) j = it->second; \
else templated_json_throw<nlohmann::detail::out_of_range>(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE, nullptr)); \
else ::nlohmann::detail::templated_json_throw<nlohmann::detail::out_of_range>(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE, nullptr)); \
} \
template<typename BasicJsonType> \
inline void from_json(const BasicJsonType& j, ENUM_TYPE& e) \
@@ -329,7 +327,7 @@ void templated_json_throw(ExceptionType exception)
return ej_pair.second == j; \
}); \
if (it != std::end(m)) e = it->first; \
else templated_json_throw<nlohmann::detail::out_of_range>(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \
else ::nlohmann::detail::templated_json_throw<nlohmann::detail::out_of_range>(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \
}
// Ugly macros to avoid uglier copy-paste when specializing basic_json. They
@@ -874,30 +872,6 @@ void templated_json_throw(ExceptionType exception)
\
template<typename... T> \
using result_of_##std_name = decltype(std_name(std::declval<T>()...)); \
} \
\
namespace detail2 { \
struct std_name##_tag \
{ \
}; \
\
template<typename... T> \
std_name##_tag std_name(T&&...); \
\
template<typename... T> \
using result_of_##std_name = decltype(std_name(std::declval<T>()...)); \
\
template<typename... T> \
struct would_call_std_##std_name \
{ \
static constexpr auto const value = ::nlohmann::detail:: \
is_detected_exact<std_name##_tag, result_of_##std_name, T...>::value; \
}; \
} /* namespace detail2 */ \
\
template<typename... T> \
struct would_call_std_##std_name : detail2::would_call_std_##std_name<T...> \
{ \
}
#ifndef JSON_USE_IMPLICIT_CONVERSIONS
@@ -35,12 +35,14 @@
#undef JSON_HAS_EXPERIMENTAL_FILESYSTEM
#undef JSON_HAS_THREE_WAY_COMPARISON
#undef JSON_HAS_RANGES
#undef JSON_HAS_RANGE_VIEW_CONVERSION
#undef JSON_HAS_STD_FORMAT
#undef JSON_HAS_STATIC_RTTI
#undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
#undef JSON_BRACE_INIT_COPY_SEMANTICS
#undef JSON_PRECISE_STREAM_POSITION
#undef JSON_STRICT_NUL_HANDLING
#undef JSON_STRICT_BINARY_UTF8
#endif
#include <nlohmann/thirdparty/hedley/hedley_undef.hpp>
@@ -12,6 +12,6 @@
NLOHMANN_JSON_NAMESPACE_BEGIN
NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin);
NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin)
NLOHMANN_JSON_NAMESPACE_END
@@ -12,6 +12,6 @@
NLOHMANN_JSON_NAMESPACE_BEGIN
NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end);
NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end)
NLOHMANN_JSON_NAMESPACE_END
+1 -34
View File
@@ -8,7 +8,7 @@
#pragma once
#include <cstdint> // size_t
#include <cstddef> // size_t
#include <utility> // declval
#include <string> // string
@@ -70,37 +70,6 @@ using parse_error_function_t = decltype(std::declval<T&>().parse_error(
std::declval<std::size_t>(), std::declval<const std::string&>(),
std::declval<const Exception&>()));
template<typename SAX, typename BasicJsonType>
struct is_sax
{
private:
static_assert(is_basic_json<BasicJsonType>::value,
"BasicJsonType must be of type basic_json<...>");
using number_integer_t = typename BasicJsonType::number_integer_t;
using number_unsigned_t = typename BasicJsonType::number_unsigned_t;
using number_float_t = typename BasicJsonType::number_float_t;
using string_t = typename BasicJsonType::string_t;
using binary_t = typename BasicJsonType::binary_t;
using exception_t = typename BasicJsonType::exception;
public:
static constexpr bool value =
is_detected_exact<bool, null_function_t, SAX>::value &&
is_detected_exact<bool, boolean_function_t, SAX>::value &&
is_detected_exact<bool, number_integer_function_t, SAX, number_integer_t>::value &&
is_detected_exact<bool, number_unsigned_function_t, SAX, number_unsigned_t>::value &&
is_detected_exact<bool, number_float_function_t, SAX, number_float_t, string_t>::value &&
is_detected_exact<bool, string_function_t, SAX, string_t>::value &&
is_detected_exact<bool, binary_function_t, SAX, binary_t>::value &&
is_detected_exact<bool, start_object_function_t, SAX>::value &&
is_detected_exact<bool, key_function_t, SAX, string_t>::value &&
is_detected_exact<bool, end_object_function_t, SAX>::value &&
is_detected_exact<bool, start_array_function_t, SAX>::value &&
is_detected_exact<bool, end_array_function_t, SAX>::value &&
is_detected_exact<bool, parse_error_function_t, SAX, exception_t>::value;
};
template<typename SAX, typename BasicJsonType>
struct is_sax_static_asserts
{
@@ -120,8 +89,6 @@ struct is_sax_static_asserts
"Missing/invalid function: bool null()");
static_assert(is_detected_exact<bool, boolean_function_t, SAX>::value,
"Missing/invalid function: bool boolean(bool)");
static_assert(is_detected_exact<bool, boolean_function_t, SAX>::value,
"Missing/invalid function: bool boolean(bool)");
static_assert(
is_detected_exact<bool, number_integer_function_t, SAX,
number_integer_t>::value,
-54
View File
@@ -1,54 +0,0 @@
#pragma once
#include <nlohmann/detail/macro_scope.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
{
#ifdef JSON_HAS_CPP_17
template<bool... Booleans>
struct cxpr_or_impl : std::integral_constant < bool, (Booleans || ...) > {};
template<bool... Booleans>
struct cxpr_and_impl : std::integral_constant < bool, (Booleans &&...) > {};
#else
template<bool... Booleans>
struct cxpr_or_impl : std::false_type {};
template<bool... Booleans>
struct cxpr_or_impl<true, Booleans...> : std::true_type {};
template<bool... Booleans>
struct cxpr_or_impl<false, Booleans...> : cxpr_or_impl<Booleans...> {};
template<bool... Booleans>
struct cxpr_and_impl : std::true_type {};
template<bool... Booleans>
struct cxpr_and_impl<true, Booleans...> : cxpr_and_impl<Booleans...> {};
template<bool... Booleans>
struct cxpr_and_impl<false, Booleans...> : std::false_type {};
#endif
template<class Boolean>
struct cxpr_not : std::integral_constant < bool, !Boolean::value > {};
template<class... Booleans>
struct cxpr_or : cxpr_or_impl<Booleans::value...> {};
template<bool... Booleans>
struct cxpr_or_c : cxpr_or_impl<Booleans...> {};
template<class... Booleans>
struct cxpr_and : cxpr_and_impl<Booleans::value...> {};
template<bool... Booleans>
struct cxpr_and_c : cxpr_and_impl<Booleans...> {};
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
+12 -23
View File
@@ -314,6 +314,13 @@ template<class B, class... Bn>
struct conjunction<B, Bn...>
: std::conditional<static_cast<bool>(B::value), conjunction<Bn...>, B>::type {};
// https://en.cppreference.com/w/cpp/types/disjunction
template<class...> struct disjunction : std::false_type { };
template<class B> struct disjunction<B> : B { };
template<class B, class... Bn>
struct disjunction<B, Bn...>
: std::conditional<static_cast<bool>(B::value), B, disjunction<Bn...>>::type {};
// https://en.cppreference.com/w/cpp/types/negation
template<class B> struct negation : std::integral_constant < bool, !B::value > { };
@@ -508,9 +515,7 @@ template<typename T> struct is_range_view_optional_type<std::optional<T>> : std:
template<typename T> struct is_range_view_optional_type : std::false_type {};
#endif
// std::ranges does not work properly on MinGW due to incomplete C++20 support
// see https://github.com/nlohmann/json/issues/4916
#if JSON_HAS_RANGES && !defined(__MINGW32__)
#if JSON_HAS_RANGE_VIEW_CONVERSION
// SafeToCheck guards against types that trigger circular constraints when
// std::ranges::view<T> is evaluated on GCC 12 / libstdc++ 12:
@@ -549,7 +554,7 @@ struct is_compatible_array_type_impl <
// filter_view) can match BOTH this iterator-based specialization AND the view-based one
// below, causing ambiguity. Exclude views here so the two specializations are mutually
// exclusive: this one handles plain iterable containers, the other handles views.
#if JSON_HAS_RANGES && !defined(__MINGW32__)
#if JSON_HAS_RANGE_VIEW_CONVERSION
&& !is_compatible_range_view<CompatibleArrayType>::value
#endif
>>
@@ -559,7 +564,7 @@ struct is_compatible_array_type_impl <
range_value_t<CompatibleArrayType>>::value;
};
#if JSON_HAS_RANGES && !defined(__MINGW32__)
#if JSON_HAS_RANGE_VIEW_CONVERSION
template<typename BasicJsonType, typename CompatibleArrayType>
struct is_compatible_array_type_impl <
BasicJsonType, CompatibleArrayType,
@@ -635,7 +640,6 @@ struct is_compatible_integer_type_impl <
std::is_integral<CompatibleNumberIntegerType>::value&&
!std::is_same<bool, CompatibleNumberIntegerType>::value >>
{
// is there an assert somewhere on overflows?
using RealLimits = std::numeric_limits<RealIntegerType>;
using CompatibleLimits = std::numeric_limits<CompatibleNumberIntegerType>;
@@ -863,20 +867,7 @@ struct has_capacity : std::integral_constant<bool, is_detected<detect_capacity,
// a naive helper to check if a type is an ordered_map (exploits the fact that
// ordered_map inherits capacity() from std::vector)
template <typename T>
struct is_ordered_map
{
using one = char;
struct two
{
char x[2]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
};
template <typename C> static one test( decltype(&C::capacity) ) ;
template <typename C> static two test(...);
enum { value = sizeof(test<T>(nullptr)) == sizeof(char) }; // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg,cppcoreguidelines-use-enum-class)
};
struct is_ordered_map : has_capacity<T> {};
// to avoid useless casts (see https://github.com/nlohmann/json/issues/2893#issuecomment-889152324)
template < typename T, typename U, enable_if_t < !std::is_same<T, U>::value, int > = 0 >
@@ -900,10 +891,8 @@ using all_signed = conjunction<std::is_signed<Types>...>;
template<typename... Types>
using all_unsigned = conjunction<std::is_unsigned<Types>...>;
// there's a disjunction trait in another PR; replace when merged
template<typename... Types>
using same_sign = std::integral_constant < bool,
all_signed<Types...>::value || all_unsigned<Types...>::value >;
using same_sign = disjunction<all_signed<Types...>, all_unsigned<Types...>>;
template<typename OfType, typename T>
using never_out_of_range = std::integral_constant < bool,
+178 -29
View File
@@ -26,6 +26,7 @@
#include <nlohmann/detail/input/binary_reader.hpp>
#include <nlohmann/detail/input/string_scan.hpp>
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/output/error_handler.hpp>
#include <nlohmann/detail/output/output_adapters.hpp>
#include <nlohmann/detail/string_concat.hpp>
#include <nlohmann/detail/string_utils.hpp>
@@ -93,8 +94,12 @@ class binary_writer
@param[in] sink output sink to write to (a value-type sink such as
output_vector_sink, or output_adapter_sink wrapping a
type-erased output adapter)
@param[in] error_handler_ how to treat a string value or object key that
is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON;
never consulted by @ref write_bon8)
*/
explicit binary_writer(OutputSinkType sink) : oa(std::move(sink))
explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = binary_writer_default_error_handler())
: oa(std::move(sink)), error_handler(error_handler_)
{}
/*!
@@ -107,14 +112,20 @@ class binary_writer
from one.
@param[in] adapter output adapter to write to
@param[in] error_handler_ how to treat a string value or object key that
is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON;
never consulted by @ref write_bon8)
*/
template < typename SinkType = OutputSinkType,
typename std::enable_if < std::is_constructible<SinkType, output_adapter_t<CharType>>::value, int >::type = 0 >
explicit binary_writer(output_adapter_t<CharType> adapter) : oa(SinkType(std::move(adapter)))
explicit binary_writer(output_adapter_t<CharType> adapter, const error_handler_t error_handler_ = binary_writer_default_error_handler())
: oa(SinkType(std::move(adapter))), error_handler(error_handler_)
{}
/*!
@param[in] j JSON value to serialize
@throw type_error.316 if a string value or an object key is not valid
UTF-8
@throw type_error.317 if @a j is not an object
*/
void write_bson(const BasicJsonType& j)
@@ -145,6 +156,8 @@ class binary_writer
/*!
@param[in] j JSON value to serialize
@throw type_error.316 if a string value or an object key is not valid
UTF-8
*/
void write_cbor(const BasicJsonType& j)
{
@@ -211,13 +224,16 @@ class binary_writer
case value_t::string:
{
string_t storage;
const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage);
// step 1: write control byte and the string length
write_cbor_head(0x60, j.m_data.m_value.string->size());
write_cbor_head(0x60, value.size());
// step 2: write the string
oa.write_characters(
reinterpret_cast<const CharType*>(j.m_data.m_value.string->data()),
j.m_data.m_value.string->size());
reinterpret_cast<const CharType*>(value.data()),
value.size());
break;
}
@@ -287,6 +303,17 @@ class binary_writer
// step 2: write each element
for (const auto& el : *j.m_data.m_value.object)
{
// el.first is checked here, against the object as
// diagnostics context, because write_cbor(el.first)
// converts it to a temporary basic_json that would be
// used as the context instead; for error_handler_t::keep
// and ::replace/::ignore the recursive write_cbor(el.first)
// call below handles the key like any other string, so no
// separate check is needed here for those
if (error_handler == error_handler_t::strict)
{
check_utf8(el.first, j);
}
write_cbor(el.first);
write_cbor(el.second);
}
@@ -434,8 +461,11 @@ class binary_writer
case value_t::string:
{
string_t storage;
const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage);
// step 1: write control byte and the string length
const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j);
const auto N = to_msgpack_length(value.size(), j);
if (N <= 31)
{
// fixstr
@@ -462,8 +492,8 @@ class binary_writer
// step 2: write the string
oa.write_characters(
reinterpret_cast<const CharType*>(j.m_data.m_value.string->data()),
j.m_data.m_value.string->size());
reinterpret_cast<const CharType*>(value.data()),
value.size());
break;
}
@@ -610,6 +640,13 @@ class binary_writer
// step 2: write each element
for (const auto& el : *j.m_data.m_value.object)
{
// as in write_cbor, el.first is checked here against the
// object as diagnostics context; the recursive call below
// handles keep/replace/ignore like any other string
if (error_handler == error_handler_t::strict)
{
check_utf8(el.first, j);
}
write_msgpack(el.first);
write_msgpack(el.second);
}
@@ -629,6 +666,8 @@ class binary_writer
@param[in] add_prefix whether prefixes need to be used for this value
@param[in] use_bjdata whether write in BJData format, default is false
@param[in] bjdata_version which BJData version to use, default is draft2
@throw type_error.316 if a string value or an object key is not valid
UTF-8
*/
void write_ubjson(const BasicJsonType& j, const bool use_count,
const bool use_type, const bool add_prefix = true,
@@ -678,14 +717,17 @@ class binary_writer
case value_t::string:
{
string_t storage;
const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage);
if (add_prefix)
{
oa.write_character(to_char_type('S'));
}
write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata);
write_number_with_ubjson_prefix(value.size(), true, use_bjdata);
oa.write_characters(
reinterpret_cast<const CharType*>(j.m_data.m_value.string->data()),
j.m_data.m_value.string->size());
reinterpret_cast<const CharType*>(value.data()),
value.size());
break;
}
@@ -840,10 +882,12 @@ class binary_writer
for (const auto& el : *j.m_data.m_value.object)
{
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
string_t storage;
const string_t& key = sanitize_utf8_for_write(el.first, j, storage);
write_number_with_ubjson_prefix(key.size(), true, use_bjdata);
oa.write_characters(
reinterpret_cast<const CharType*>(el.first.data()),
el.first.size());
reinterpret_cast<const CharType*>(key.data()),
key.size());
write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version);
}
@@ -884,8 +928,12 @@ class binary_writer
/*!
@return The size of a BSON document entry header, including the id marker
and the entry name size (and its null-terminator).
@throw out_of_range.409 if @a name contains U+0000, before anything is
written
@throw type_error.316 if @a name is not valid UTF-8, before anything is
written
*/
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
{
const auto it = name.find(static_cast<typename string_t::value_type>(0));
if (JSON_HEDLEY_UNLIKELY(it != BasicJsonType::string_t::npos))
@@ -893,8 +941,10 @@ class binary_writer
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
}
static_cast<void>(j);
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
string_t storage;
const string_t& sanitized = sanitize_utf8_for_write(name, j, storage);
return /*id*/ 1ul + sanitized.size() + /*zero-terminator*/1u;
}
/*!
@@ -914,14 +964,28 @@ class binary_writer
/*!
@brief Writes the given @a element_type and @a name to the output adapter
@a name has already been validated (and, for @ref error_handler_t::strict,
found well-formed) by @ref calc_bson_entry_header_size during the earlier
size pass, so only @ref error_handler_t::replace / @ref
error_handler_t::ignore need to sanitize it again here, to actually write
the bytes that size was computed from.
*/
void write_bson_entry_header(const string_t& name,
const std::uint8_t element_type)
{
oa.write_character(to_char_type(element_type));
oa.write_characters(
reinterpret_cast<const CharType*>(name.data()),
name.size());
if (error_handler == error_handler_t::keep || error_handler == error_handler_t::strict || is_valid_utf8(name))
{
oa.write_characters(reinterpret_cast<const CharType*>(name.data()), name.size());
}
else
{
const string_t sanitized = sanitize_utf8(name, error_handler);
oa.write_characters(reinterpret_cast<const CharType*>(sanitized.data()), sanitized.size());
}
// the terminating null byte is written explicitly rather than taken
// from the buffer, so that string_t::data() need not be null-terminated
oa.write_character(to_char_type(0x00));
@@ -949,24 +1013,50 @@ class binary_writer
/*!
@return The size of the BSON-encoded string in @a value
@throw type_error.316 if @a value is not valid UTF-8, before anything is
written
@note The UTF-8 check is skipped if @a value is already too long for the
32-bit BSON length field (@ref to_bson_length rejects it later, once
the size of the whole document is known); this also keeps the check
from reading past a StringType that reports a size larger than what
it actually holds.
*/
static std::size_t calc_bson_string_size(const string_t& value)
std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j)
{
if (JSON_HEDLEY_LIKELY(value_in_range_of<std::int32_t>(value.size())))
{
string_t storage;
const string_t& sanitized = sanitize_utf8_for_write(value, j, storage);
return sizeof(std::int32_t) + sanitized.size() + 1ul;
}
return sizeof(std::int32_t) + value.size() + 1ul;
}
/*!
@brief Writes a BSON element with key @a name and string value @a value
@a value has already been validated (and, for @ref error_handler_t::strict,
found well-formed) by @ref calc_bson_string_size during the earlier size
pass, so only @ref error_handler_t::replace / @ref error_handler_t::ignore
need to sanitize it again here, to actually write the bytes that size was
computed from.
*/
void write_bson_string(const string_t& name,
const string_t& value)
{
write_bson_entry_header(name, 0x02);
write_number<std::int32_t>(to_bson_length(value.size() + 1ul), true);
const bool sanitize = error_handler != error_handler_t::keep
&& error_handler != error_handler_t::strict
&& !is_valid_utf8(value);
const string_t sanitized = sanitize ? sanitize_utf8(value, error_handler) : string_t{};
const string_t& written = sanitize ? sanitized : value;
write_number<std::int32_t>(to_bson_length(written.size() + 1ul), true);
oa.write_characters(
reinterpret_cast<const CharType*>(value.data()),
value.size());
reinterpret_cast<const CharType*>(written.data()),
written.size());
// the terminating null byte is written explicitly rather than taken
// from the buffer, so that string_t::data() need not be null-terminated
oa.write_character(to_char_type(0x00));
@@ -1080,8 +1170,10 @@ class binary_writer
is neither an object nor an array
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
into a byte, before anything is written
@throw type_error.316 if @a j is a string that is not valid UTF-8, before
anything is written
*/
static std::size_t calc_bson_value_size(const BasicJsonType& j)
std::size_t calc_bson_value_size(const BasicJsonType& j)
{
switch (j.type())
{
@@ -1101,7 +1193,7 @@ class binary_writer
return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned);
case value_t::string:
return calc_bson_string_size(*j.m_data.m_value.string);
return calc_bson_string_size(*j.m_data.m_value.string, j);
case value_t::null:
return 0ul;
@@ -1214,8 +1306,10 @@ class binary_writer
written
@throw out_of_range.415 if a binary value's subtype does not fit into a
byte, before anything is written
@throw type_error.316 if a string value or a key is not valid UTF-8,
before anything is written
*/
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
{
// the object or array whose entries are being sized, and the ones it
// is in; nothing is allocated unless the document nests
@@ -2092,7 +2186,7 @@ class binary_writer
*/
void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context)
{
check_bon8_utf8(s, context);
check_utf8(s, context);
// a string that follows another string terminates it
if (string_open)
@@ -2122,7 +2216,7 @@ class binary_writer
@throw type_error.316 if @a s is not valid UTF-8; the message names the
first byte of the first invalid or incomplete sequence
*/
static void check_bon8_utf8(const string_t& s, const BasicJsonType& context)
static void check_utf8(const string_t& s, const BasicJsonType& context)
{
static_cast<void>(context); // only used when exceptions are enabled
const auto* data = reinterpret_cast<const unsigned char*>(s.data());
@@ -2133,6 +2227,57 @@ class binary_writer
}
}
/*!
@brief return @a s as it should be written, honoring @ref error_handler
Used by @ref write_cbor, @ref write_msgpack, @ref write_ubjson (and so
@ref write_bjdata), and the BSON writing functions for string values and
object keys; never by @ref write_bon8, which always validates, since UTF-8
lead bytes are structural there.
- @ref error_handler_t::keep: @a s is returned unchanged, without even
checking it (the behavior of release 3.12.0 and earlier).
- @ref error_handler_t::strict: @ref check_utf8 is called, which throws
type_error.316 if @a s is not valid UTF-8.
- @ref error_handler_t::replace / @ref error_handler_t::ignore: @a s is
sanitized into @a storage with exactly the rules @ref
serializer::dump_escaped_impl uses, so that parsing what @ref
basic_json::dump produces for the same string and the same handler
yields the same result.
Well-formed input is never copied: this returns a reference to @a s
itself in every case but a sanitized `replace`/`ignore` one, so @a
storage must outlive the returned reference only then.
@param[in] s the string (value or object key) to write
@param[in] context the value @a s belongs to (for diagnostics)
@param[out] storage backing storage for a sanitized copy
@return a reference to @a s, or to @a storage once it holds a sanitized copy
*/
const string_t& sanitize_utf8_for_write(const string_t& s, const BasicJsonType& context, string_t& storage) const
{
switch (error_handler)
{
case error_handler_t::keep:
return s;
case error_handler_t::strict:
check_utf8(s, context);
return s;
case error_handler_t::replace:
case error_handler_t::ignore:
default:
if (is_valid_utf8(s))
{
return s;
}
storage = sanitize_utf8(s, error_handler);
return storage;
}
}
/*!
@brief write an integer in the shortest encoding
@@ -2461,6 +2606,10 @@ class binary_writer
/// the output
OutputSinkType oa;
/// how to treat a string value or object key that is not valid UTF-8
/// (CBOR, MessagePack, UBJSON, BJData, and BSON; not BON8)
const error_handler_t error_handler = binary_writer_default_error_handler();
};
} // namespace detail
@@ -0,0 +1,50 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <nlohmann/detail/abi_macros.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
{
/// how to treat decoding errors
///
/// @ref basic_json::dump uses this to decide what to do with ill-formed
/// UTF-8 while escaping a string, and the binary writers (@ref
/// basic_json::to_cbor, @ref basic_json::to_ubjson, @ref
/// basic_json::to_bjdata, @ref basic_json::to_bson) use it the same way for
/// string values and object keys. The binary readers (@ref
/// basic_json::from_cbor, @ref basic_json::from_msgpack, @ref
/// basic_json::from_ubjson, @ref basic_json::from_bjdata, @ref
/// basic_json::from_bson) use it to decide whether to check text strings
/// and object keys for well-formed UTF-8 at all, since none of those
/// formats requires a decoder to do so.
enum class error_handler_t
{
strict, ///< throw a type_error/parse_error exception in case of invalid UTF-8
replace, ///< replace invalid UTF-8 sequences with U+FFFD
ignore, ///< ignore invalid UTF-8 sequences
keep ///< keep invalid UTF-8 sequences unchanged
};
/// the default error handler of the CBOR, UBJSON, BJData, and BSON writers:
/// error_handler_t::strict if JSON_STRICT_BINARY_UTF8 is enabled, otherwise
/// error_handler_t::keep (the behavior before version 3.13.0)
constexpr error_handler_t binary_writer_default_error_handler() noexcept
{
#if JSON_STRICT_BINARY_UTF8
return error_handler_t::strict;
#else
return error_handler_t::keep;
#endif
}
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
+62 -10
View File
@@ -27,6 +27,7 @@
#include <nlohmann/detail/input/string_scan.hpp>
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/cpp_future.hpp>
#include <nlohmann/detail/output/error_handler.hpp>
#include <nlohmann/detail/output/output_adapters.hpp>
#include <nlohmann/detail/recursion_depth_limit.hpp>
#include <nlohmann/detail/string_concat.hpp>
@@ -41,14 +42,6 @@ namespace detail
// serialization //
///////////////////
/// how to treat decoding errors
enum class error_handler_t
{
strict, ///< throw a type_error exception in case of invalid UTF-8
replace, ///< replace invalid UTF-8 sequences with U+FFFD
ignore ///< ignore invalid UTF-8 sequences
};
template<typename BasicJsonType>
class serializer
{
@@ -839,6 +832,16 @@ class serializer
// EnsureAscii parameter is used, non-ASCII characters
if ((codepoint <= 0x1F) || (EnsureAscii && (codepoint >= 0x7F)))
{
if (EnsureAscii && error_handler == error_handler_t::keep)
{
// this character was buffered as raw bytes
// below in case it turned out to be part of
// an ill-formed sequence (which is kept as
// is); now that it decoded to a well-formed
// code point, undo that and \u-escape it
// like any other character instead
bytes = bytes_after_last_accept;
}
if (codepoint <= 0xFFFF)
{
write_u_escape(bytes, static_cast<std::uint16_t>(codepoint));
@@ -937,6 +940,44 @@ class serializer
break;
}
case error_handler_t::keep:
{
// the bytes of this (now abandoned) ill-formed
// sequence seen so far are already buffered below
// and are kept unchanged in the output
if (undumped_chars > 0)
{
// the byte that ended the sequence may be OK
// for itself (e.g., a quote that must still be
// escaped, or the lead byte of a well-formed
// code point), so read it again
--i;
}
else
{
// a byte that cannot start a sequence (e.g.,
// 0xFF or a stray continuation byte) is kept
// as well
string_buffer[bytes++] = s[i];
}
// write buffer and reset index; there must be 13 bytes
// left, as this is the maximal number of bytes to be
// written ("\uxxxx\uxxxx\0") for one code point
if (string_buffer.size() - bytes < 13)
{
put_buffer(string_buffer, bytes);
bytes = 0;
}
bytes_after_last_accept = bytes;
undumped_chars = 0;
// continue processing the string
state = UTF8_ACCEPT;
break;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
@@ -945,9 +986,12 @@ class serializer
default: // decode found yet incomplete multibyte code point
{
if (!EnsureAscii)
if (!EnsureAscii || error_handler == error_handler_t::keep)
{
// code point will not be escaped - copy byte to buffer
// code point will not be escaped (or will be kept as
// is if it turns out to be ill-formed) - copy byte to
// buffer; dropped again above if it decodes to a
// well-formed code point that needs \u-escaping
string_buffer[bytes++] = s[i];
}
++undumped_chars;
@@ -998,6 +1042,14 @@ class serializer
break;
}
case error_handler_t::keep:
{
// write the ill-formed trailing bytes as is; they were
// buffered above regardless of EnsureAscii
put_buffer(string_buffer, bytes);
break;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
@@ -39,7 +39,6 @@ inline std::size_t concat_length(const char /*c*/, const Args& ... rest)
template<typename... Args>
inline std::size_t concat_length(const char* cstr, const Args& ... rest)
{
// cppcheck-suppress ignoredReturnValue
return ::strlen(cstr) + concat_length(rest...);
}
+85 -60
View File
@@ -17,6 +17,7 @@
#include <nlohmann/detail/abi_macros.hpp>
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/output/error_handler.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -118,13 +119,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
written by Björn Hoehrmann. See
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four
The library checks UTF-8 well-formedness (RFC 3629, section 4) in three
places, which differ in speed, diagnostics, and how they read the input:
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR,
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
text strings at decode time).
- decode() below: the serializer, to escape and, in strict mode, reject
ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON,
UBJSON and BJData readers do not use it: none of those specs requires a
decoder to reject ill-formed UTF-8 in text strings, so the readers keep
the bytes as is and leave the check to dump() and the binary writers.
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
diagnostic for each kind of error.
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
@@ -180,19 +182,19 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std:
}
/*!
@brief check whether a string consists solely of valid UTF-8
@brief check a string for well-formed UTF-8 (RFC 3629, section 4)
Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text
strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the
MessagePack/BSON specifications all require text strings to be UTF-8), so
that malformed input is caught immediately instead of only surfacing later
as a type_error.316 when the resulting value is dumped.
Used by the binary readers (CBOR, MessagePack, UBJSON, BJData, BSON) when an
@ref error_handler_t other than `keep` is requested for a text string value
or object key: none of those formats requires a decoder to reject ill-formed
UTF-8 on its own, so the check is opt-in there, unlike the JSON lexer and the
serializer's @ref decode -based escaping, which always run it.
@param[in] s the string to check
@param[in] first index of the first byte to check; the bytes before it are
assumed to have been validated already and to end on a
code point boundary
@return whether @a s (from index @a first on) is valid UTF-8
@param[in] first the index to start checking at
@return whether `s.substr(first)` is well-formed UTF-8
@sa @ref decode
*/
template<typename StringType>
inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept
@@ -213,76 +215,99 @@ inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noex
}
/*!
@brief append U+FFFD REPLACEMENT CHARACTER, encoded in UTF-8
@param[in,out] s the string to append to
@brief sanitize a string with ill-formed UTF-8 for @ref error_handler_t::replace or @ref error_handler_t::ignore
Replaces every maximal ill-formed subsequence with U+FFFD (`replace`) or
drops it (`ignore`), using exactly the same boundaries @ref
serializer::dump_escaped_impl uses while escaping a string: a byte that does
not extend the sequence started by the previous byte(s) is reread as the
start of a new one, instead of being swallowed along with them.
@pre @a error_handler is @ref error_handler_t::replace or @ref error_handler_t::ignore
@note Well-formed input is copied through unchanged, including bytes (e.g.
control characters or quotes) that @ref serializer::dump_escaped_impl
would itself escape; this function only concerns itself with
well-formedness, not with producing valid JSON text.
@param[in] s the string to sanitize
@param[in] error_handler @ref error_handler_t::replace or @ref error_handler_t::ignore
@return @a s with every ill-formed subsequence replaced or removed
@sa @ref decode
*/
template<typename StringType>
inline void append_replacement_character(StringType& s)
inline StringType sanitize_utf8(const StringType& s, const error_handler_t error_handler)
{
s.push_back(static_cast<typename StringType::value_type>(0xEFu));
s.push_back(static_cast<typename StringType::value_type>(0xBFu));
s.push_back(static_cast<typename StringType::value_type>(0xBDu));
}
JSON_ASSERT(error_handler == error_handler_t::replace || error_handler == error_handler_t::ignore);
/*!
@brief replace ill-formed UTF-8 with U+FFFD REPLACEMENT CHARACTER
StringType result;
result.reserve(s.size());
Each maximal subpart of an ill-formed sequence becomes one U+FFFD, as the
Unicode Standard recommends (Section 3.9, "U+FFFD Substitution of Maximal
Subparts"), and as the parser for JSON text does when it recovers from errors.
@param[in,out] s the string to repair
@param[in] first index of the first byte to repair; the bytes before it are
assumed to be valid UTF-8 that ends on a code point boundary
*/
template<typename StringType>
inline void replace_invalid_utf8(StringType& s, const std::size_t first = 0)
{
StringType result = s;
result.resize(first);
std::uint8_t state = UTF8_ACCEPT;
std::uint32_t codepoint = 0;
// the first byte of the sequence being decoded
std::size_t sequence_start = first;
std::uint8_t state = UTF8_ACCEPT;
// length of result after the last accepted code point
std::size_t result_len_after_last_accept = 0;
// whether bytes of an as yet unresolved sequence were already appended
bool pending = false;
std::size_t i = first;
while (i < s.size())
for (std::size_t i = 0; i < s.size(); ++i)
{
switch (decode(state, codepoint, static_cast<std::uint8_t>(s[i])))
{
case UTF8_ACCEPT:
for (++i; sequence_start < i; ++sequence_start)
{
result.push_back(s[sequence_start]);
}
case UTF8_ACCEPT: // decode found a well-formed code point
{
result.push_back(s[i]);
result_len_after_last_accept = result.size();
pending = false;
break;
}
case UTF8_REJECT:
append_replacement_character(result);
// the byte that made the sequence ill-formed begins the next
// one, unless it began this one
if (i == sequence_start)
case UTF8_REJECT: // decode found an ill-formed byte
{
// in case we saw this byte for the first time, read it again,
// because it may be fine for itself, just not for the
// sequence that came before it
if (pending)
{
++i;
--i;
}
// drop the bytes of the ill-formed sequence buffered below
result.resize(result_len_after_last_accept);
if (error_handler == error_handler_t::replace)
{
result.append("\xEF\xBF\xBD");
result_len_after_last_accept = result.size();
}
pending = false;
state = UTF8_ACCEPT;
sequence_start = i;
break;
}
default: // in the middle of a sequence
++i;
default: // decode found yet incomplete multibyte code point
{
result.push_back(s[i]);
pending = true;
break;
}
}
}
// a sequence that the string ends in the middle of
// the string ended with an incomplete sequence
if (state != UTF8_ACCEPT)
{
append_replacement_character(result);
result.resize(result_len_after_last_accept);
if (error_handler == error_handler_t::replace)
{
result.append("\xEF\xBF\xBD");
}
}
s = std::move(result);
return result;
}
} // namespace detail
+336 -437
View File
File diff suppressed because it is too large Load Diff
+68 -115
View File
@@ -74,16 +74,43 @@ template <class Key, class T, class IgnoredLess = std::less<Key>,
return *this;
}
private:
/// @brief find the entry for @a key, for either constness of @a self
/// @note the single place that performs the linear key search
template<typename Self, typename KeyType>
static auto find_impl(Self& self, KeyType&& key) -> decltype(self.begin())
{
for (auto it = self.begin(); it != self.end(); ++it)
{
if (self.m_compare(it->first, key))
{
return it;
}
}
return self.end();
}
/// @brief remove the entry @a it points to, preserving order
/// @note keys are not movable, so the tail is destroyed and re-constructed in place
void erase_at(iterator it)
{
for (auto next = it; ++next != this->end(); ++it)
{
it->~value_type(); // Destroy but keep allocation
new (&*it) value_type{std::move(*next)};
}
Container::pop_back();
}
public:
template<class V, detail::enable_if_t<
detail::is_constructible<T, V>::value, int> = 0>
std::pair<iterator, bool> emplace(const key_type& key, V && t)
{
for (auto it = this->begin(); it != this->end(); ++it)
const auto it = find_impl(*this, key);
if (it != this->end())
{
if (m_compare(it->first, key))
{
return {it, false};
}
return {it, false};
}
append(key, std::forward<V>(t));
return {std::prev(this->end()), true};
@@ -94,12 +121,10 @@ template <class Key, class T, class IgnoredLess = std::less<Key>,
detail::is_constructible<T, V>>::value, int> = 0>
std::pair<iterator, bool> emplace(KeyType && key, V && t)
{
for (auto it = this->begin(); it != this->end(); ++it)
const auto it = find_impl(*this, key);
if (it != this->end())
{
if (m_compare(it->first, key))
{
return {it, false};
}
return {it, false};
}
append(std::forward<KeyType>(key), std::forward<V>(t));
return {std::prev(this->end()), true};
@@ -131,75 +156,55 @@ template <class Key, class T, class IgnoredLess = std::less<Key>,
T& at(const key_type& key)
{
for (auto it = this->begin(); it != this->end(); ++it)
const auto it = find_impl(*this, key);
if (it == this->end())
{
if (m_compare(it->first, key))
{
return it->second;
}
JSON_THROW(std::out_of_range("key not found"));
}
JSON_THROW(std::out_of_range("key not found"));
return it->second;
}
template<class KeyType, detail::enable_if_t<
detail::is_usable_as_key_type<key_compare, key_type, KeyType>::value, int> = 0>
T & at(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward)
{
for (auto it = this->begin(); it != this->end(); ++it)
const auto it = find_impl(*this, key);
if (it == this->end())
{
if (m_compare(it->first, key))
{
return it->second;
}
JSON_THROW(std::out_of_range("key not found"));
}
JSON_THROW(std::out_of_range("key not found"));
return it->second;
}
const T& at(const key_type& key) const
{
for (auto it = this->begin(); it != this->end(); ++it)
const auto it = find_impl(*this, key);
if (it == this->end())
{
if (m_compare(it->first, key))
{
return it->second;
}
JSON_THROW(std::out_of_range("key not found"));
}
JSON_THROW(std::out_of_range("key not found"));
return it->second;
}
template<class KeyType, detail::enable_if_t<
detail::is_usable_as_key_type<key_compare, key_type, KeyType>::value, int> = 0>
const T & at(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward)
{
for (auto it = this->begin(); it != this->end(); ++it)
const auto it = find_impl(*this, key);
if (it == this->end())
{
if (m_compare(it->first, key))
{
return it->second;
}
JSON_THROW(std::out_of_range("key not found"));
}
JSON_THROW(std::out_of_range("key not found"));
return it->second;
}
size_type erase(const key_type& key)
{
for (auto it = this->begin(); it != this->end(); ++it)
const auto it = find_impl(*this, key);
if (it != this->end())
{
if (m_compare(it->first, key))
{
// Since we cannot move const Keys, re-construct them in place
for (auto next = it; ++next != this->end(); ++it)
{
it->~value_type(); // Destroy but keep allocation
new (&*it) value_type{std::move(*next)};
}
Container::pop_back();
return 1;
}
erase_at(it);
return 1;
}
return 0;
}
@@ -208,19 +213,11 @@ template <class Key, class T, class IgnoredLess = std::less<Key>,
detail::is_usable_as_key_type<key_compare, key_type, KeyType>::value, int> = 0>
size_type erase(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward)
{
for (auto it = this->begin(); it != this->end(); ++it)
const auto it = find_impl(*this, key);
if (it != this->end())
{
if (m_compare(it->first, key))
{
// Since we cannot move const Keys, re-construct them in place
for (auto next = it; ++next != this->end(); ++it)
{
it->~value_type(); // Destroy but keep allocation
new (&*it) value_type{std::move(*next)};
}
Container::pop_back();
return 1;
}
erase_at(it);
return 1;
}
return 0;
}
@@ -285,80 +282,38 @@ template <class Key, class T, class IgnoredLess = std::less<Key>,
size_type count(const key_type& key) const
{
for (auto it = this->begin(); it != this->end(); ++it)
{
if (m_compare(it->first, key))
{
return 1;
}
}
return 0;
return find_impl(*this, key) != this->end() ? 1 : 0;
}
template<class KeyType, detail::enable_if_t<
detail::is_usable_as_key_type<key_compare, key_type, KeyType>::value, int> = 0>
size_type count(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward)
{
for (auto it = this->begin(); it != this->end(); ++it)
{
if (m_compare(it->first, key))
{
return 1;
}
}
return 0;
return find_impl(*this, key) != this->end() ? 1 : 0;
}
iterator find(const key_type& key)
{
for (auto it = this->begin(); it != this->end(); ++it)
{
if (m_compare(it->first, key))
{
return it;
}
}
return Container::end();
return find_impl(*this, key);
}
template<class KeyType, detail::enable_if_t<
detail::is_usable_as_key_type<key_compare, key_type, KeyType>::value, int> = 0>
iterator find(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward)
{
for (auto it = this->begin(); it != this->end(); ++it)
{
if (m_compare(it->first, key))
{
return it;
}
}
return Container::end();
return find_impl(*this, key);
}
const_iterator find(const key_type& key) const
{
for (auto it = this->begin(); it != this->end(); ++it)
{
if (m_compare(it->first, key))
{
return it;
}
}
return Container::end();
return find_impl(*this, key);
}
template<class KeyType, detail::enable_if_t<
detail::is_usable_as_key_type<key_compare, key_type, KeyType>::value, int> = 0>
const_iterator find(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward)
{
for (auto it = this->begin(); it != this->end(); ++it)
{
if (m_compare(it->first, key))
{
return it;
}
}
return Container::end();
return find_impl(*this, key);
}
std::pair<iterator, bool> insert( value_type&& value )
@@ -368,12 +323,10 @@ template <class Key, class T, class IgnoredLess = std::less<Key>,
std::pair<iterator, bool> insert( const value_type& value )
{
for (auto it = this->begin(); it != this->end(); ++it)
const auto it = find_impl(*this, value.first);
if (it != this->end())
{
if (m_compare(it->first, value.first))
{
return {it, false};
}
return {it, false};
}
append(value);
return {--this->end(), true};
+3840
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+15 -4
View File
@@ -63,6 +63,10 @@
#define JSON_STRICT_NUL_HANDLING 0
#endif
#ifndef JSON_STRICT_BINARY_UTF8
#define JSON_STRICT_BINARY_UTF8 0
#endif
#if JSON_DIAGNOSTICS
#define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag
#else
@@ -99,14 +103,20 @@
#define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING
#endif
#if JSON_STRICT_BINARY_UTF8
#define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8
#else
#define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8
#endif
#ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION
#define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0
#endif
// Construct the namespace ABI tags component
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f)
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g)
#define NLOHMANN_JSON_ABI_TAGS \
NLOHMANN_JSON_ABI_TAGS_CONCAT( \
@@ -115,7 +125,8 @@
NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \
NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \
NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \
NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING)
NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \
NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8)
// Construct the namespace version component
#define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \
+4
View File
@@ -44,6 +44,10 @@ TEST_CASE("default namespace")
expected += "_snul";
#endif
#if JSON_STRICT_BINARY_UTF8
expected += "_sbu8";
#endif
expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR);
expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR);
expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json";
+4
View File
@@ -45,6 +45,10 @@ TEST_CASE("default namespace without version component")
expected += "_snul";
#endif
#if JSON_STRICT_BINARY_UTF8
expected += "_sbu8";
#endif
expected += "::basic_json";
// fallback for Clang
@@ -0,0 +1,362 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <string>
#include <vector>
namespace
{
struct ill_formed_case
{
const char* name;
std::string bytes;
};
// RFC 3629 ill-formed sequences used throughout this file, plus one
// well-formed sequence for contrast
const std::vector<ill_formed_case> ill_formed_cases =
{
{"overlong", "\xC0\xAE"},
{"lone_0xFF", "\xFF"},
{"truncated", "\xE2\x82"},
{"surrogate", "\xED\xA0\x80"},
};
const std::string valid_sequence = "\xC3\xA9"; // U+00E9, "é"
using eh = json::error_handler_t;
const std::vector<eh> all_handlers = {eh::strict, eh::replace, eh::ignore, eh::keep};
// what dump()+parse() produces for a sanitizing error_handler; this is the
// ground truth every binary writer/reader is checked against
std::string dump_and_parse(const std::string& raw, eh error_handler)
{
return json::parse(json(raw).dump(-1, ' ', false, error_handler)).get<std::string>();
}
} // namespace
TEST_CASE("UTF-8 error_handler for the binary readers and writers")
{
SECTION("writers: string value")
{
for (const auto& c : ill_formed_cases)
{
CAPTURE(c.name);
const json jval = c.bytes;
CHECK_THROWS_AS(json::to_cbor(jval, eh::strict), json::type_error&);
CHECK_THROWS_AS(json::to_msgpack(jval, eh::strict), json::type_error&);
CHECK_THROWS_AS(json::to_ubjson(jval, false, false, eh::strict), json::type_error&);
CHECK_THROWS_AS(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&);
{
json jobj;
jobj["k"] = jval;
CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&);
}
for (const auto h :
{
eh::replace, eh::ignore
})
{
CAPTURE(static_cast<int>(h));
const std::string expected = dump_and_parse(c.bytes, h);
CHECK(json::from_cbor(json::to_cbor(jval, h)).get<std::string>() == expected);
CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get<std::string>() == expected);
CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get<std::string>() == expected);
CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get<std::string>() == expected);
{
json jobj;
jobj["k"] = jval;
const auto bytes = json::to_bson(jobj, h);
CHECK(json::from_bson(bytes)["k"].get<std::string>() == expected);
}
}
// keep: the writer passes the ill-formed bytes through unchanged,
// exactly as every binary writer did before this parameter existed
CHECK(json::from_cbor(json::to_cbor(jval, eh::keep)).get<std::string>() == c.bytes);
CHECK(json::from_msgpack(json::to_msgpack(jval, eh::keep)).get<std::string>() == c.bytes);
CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, eh::keep)).get<std::string>() == c.bytes);
CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep)).get<std::string>() == c.bytes);
{
json jobj;
jobj["k"] = jval;
const auto bytes = json::to_bson(jobj, eh::keep);
CHECK(json::from_bson(bytes)["k"].get<std::string>() == c.bytes);
}
}
}
SECTION("writers: object key")
{
for (const auto& c : ill_formed_cases)
{
CAPTURE(c.name);
json jobj;
jobj[c.bytes] = 1;
CHECK_THROWS_AS(json::to_cbor(jobj, eh::strict), json::type_error&);
CHECK_THROWS_AS(json::to_msgpack(jobj, eh::strict), json::type_error&);
CHECK_THROWS_AS(json::to_ubjson(jobj, false, false, eh::strict), json::type_error&);
CHECK_THROWS_AS(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&);
CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&);
for (const auto h :
{
eh::replace, eh::ignore
})
{
CAPTURE(static_cast<int>(h));
const std::string expected = dump_and_parse(c.bytes, h);
CHECK(json::from_cbor(json::to_cbor(jobj, h)).begin().key() == expected);
CHECK(json::from_msgpack(json::to_msgpack(jobj, h)).begin().key() == expected);
CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, h)).begin().key() == expected);
CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, h)).begin().key() == expected);
CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == expected);
}
// keep: object keys round-trip unchanged too
CHECK(json::from_cbor(json::to_cbor(jobj, eh::keep)).begin().key() == c.bytes);
CHECK(json::from_msgpack(json::to_msgpack(jobj, eh::keep)).begin().key() == c.bytes);
CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, eh::keep)).begin().key() == c.bytes);
CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep)).begin().key() == c.bytes);
CHECK(json::from_bson(json::to_bson(jobj, eh::keep)).begin().key() == c.bytes);
}
}
SECTION("readers: string value")
{
for (const auto& c : ill_formed_cases)
{
CAPTURE(c.name);
// bytes produced the lenient (keep) way, as any binary reader
// accepted them before this parameter existed
const auto cbor_bytes = json::to_cbor(json(c.bytes), eh::keep);
const auto msgpack_bytes = json::to_msgpack(json(c.bytes)); // to_msgpack has no error_handler; always pass-through
const auto ubjson_bytes = json::to_ubjson(json(c.bytes), false, false, eh::keep);
const auto bjdata_bytes = json::to_bjdata(json(c.bytes), false, false, json::bjdata_version_t::draft2, eh::keep);
const auto bson_bytes = [&c]
{
json jobj;
jobj["k"] = c.bytes;
return json::to_bson(jobj, eh::keep);
}();
// keep (the default): bytes are kept unchanged
CHECK(json::from_cbor(cbor_bytes).get<std::string>() == c.bytes);
CHECK(json::from_msgpack(msgpack_bytes).get<std::string>() == c.bytes);
CHECK(json::from_ubjson(ubjson_bytes).get<std::string>() == c.bytes);
CHECK(json::from_bjdata(bjdata_bytes).get<std::string>() == c.bytes);
CHECK(json::from_bson(bson_bytes)["k"].get<std::string>() == c.bytes);
// strict: parse_error.113, discarded (not thrown) when allow_exceptions is false
CHECK_THROWS_AS(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, eh::strict), json::parse_error&);
CHECK(json::from_cbor(cbor_bytes, true, false, json::cbor_tag_handler_t::error, eh::strict).is_discarded());
CHECK_THROWS_AS(json::from_msgpack(msgpack_bytes, true, true, eh::strict), json::parse_error&);
CHECK(json::from_msgpack(msgpack_bytes, true, false, eh::strict).is_discarded());
CHECK_THROWS_AS(json::from_ubjson(ubjson_bytes, true, true, eh::strict), json::parse_error&);
CHECK(json::from_ubjson(ubjson_bytes, true, false, eh::strict).is_discarded());
CHECK_THROWS_AS(json::from_bjdata(bjdata_bytes, true, true, eh::strict), json::parse_error&);
CHECK(json::from_bjdata(bjdata_bytes, true, false, eh::strict).is_discarded());
CHECK_THROWS_AS(json::from_bson(bson_bytes, true, true, eh::strict), json::parse_error&);
CHECK(json::from_bson(bson_bytes, true, false, eh::strict).is_discarded());
// replace / ignore: match what dump() would have sanitized the same bytes to
for (const auto h :
{
eh::replace, eh::ignore
})
{
CAPTURE(static_cast<int>(h));
const std::string expected = dump_and_parse(c.bytes, h);
CHECK(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, h).get<std::string>() == expected);
CHECK(json::from_msgpack(msgpack_bytes, true, true, h).get<std::string>() == expected);
CHECK(json::from_ubjson(ubjson_bytes, true, true, h).get<std::string>() == expected);
CHECK(json::from_bjdata(bjdata_bytes, true, true, h).get<std::string>() == expected);
CHECK(json::from_bson(bson_bytes, true, true, h)["k"].get<std::string>() == expected);
}
}
}
SECTION("readers: object key")
{
for (const auto& c : ill_formed_cases)
{
CAPTURE(c.name);
json jobj;
jobj[c.bytes] = 1;
const auto cbor_bytes = json::to_cbor(jobj, eh::keep);
const auto msgpack_bytes = json::to_msgpack(jobj);
const auto ubjson_bytes = json::to_ubjson(jobj, false, false, eh::keep);
const auto bjdata_bytes = json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep);
const auto bson_bytes = json::to_bson(jobj, eh::keep);
CHECK(json::from_cbor(cbor_bytes).begin().key() == c.bytes);
CHECK(json::from_msgpack(msgpack_bytes).begin().key() == c.bytes);
CHECK(json::from_ubjson(ubjson_bytes).begin().key() == c.bytes);
CHECK(json::from_bjdata(bjdata_bytes).begin().key() == c.bytes);
CHECK(json::from_bson(bson_bytes).begin().key() == c.bytes);
CHECK_THROWS_AS(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, eh::strict), json::parse_error&);
CHECK_THROWS_AS(json::from_msgpack(msgpack_bytes, true, true, eh::strict), json::parse_error&);
CHECK_THROWS_AS(json::from_ubjson(ubjson_bytes, true, true, eh::strict), json::parse_error&);
CHECK_THROWS_AS(json::from_bjdata(bjdata_bytes, true, true, eh::strict), json::parse_error&);
CHECK_THROWS_AS(json::from_bson(bson_bytes, true, true, eh::strict), json::parse_error&);
for (const auto h :
{
eh::replace, eh::ignore
})
{
CAPTURE(static_cast<int>(h));
const std::string expected = dump_and_parse(c.bytes, h);
CHECK(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, h).begin().key() == expected);
CHECK(json::from_msgpack(msgpack_bytes, true, true, h).begin().key() == expected);
CHECK(json::from_ubjson(ubjson_bytes, true, true, h).begin().key() == expected);
CHECK(json::from_bjdata(bjdata_bytes, true, true, h).begin().key() == expected);
CHECK(json::from_bson(bson_bytes, true, true, h).begin().key() == expected);
}
}
}
SECTION("well-formed UTF-8 is unaffected by error_handler")
{
const json jval = valid_sequence;
json jobj;
jobj[valid_sequence] = valid_sequence;
for (const auto h : all_handlers)
{
CAPTURE(static_cast<int>(h));
CHECK(json::from_cbor(json::to_cbor(jval, h)).get<std::string>() == valid_sequence);
CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get<std::string>() == valid_sequence);
CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get<std::string>() == valid_sequence);
CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get<std::string>() == valid_sequence);
CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == valid_sequence);
CHECK(json::from_cbor(json::to_cbor(jval, eh::keep), true, true, json::cbor_tag_handler_t::error, h).get<std::string>() == valid_sequence);
CHECK(json::from_msgpack(json::to_msgpack(jval), true, true, h).get<std::string>() == valid_sequence);
}
}
SECTION("dump() with error_handler_t::keep writes raw bytes as is")
{
for (const auto& c : ill_formed_cases)
{
CAPTURE(c.name);
const json jval = c.bytes;
const std::string dumped = jval.dump(-1, ' ', false, eh::keep);
CHECK(dumped.find(c.bytes) != std::string::npos);
// even with ensure_ascii, the ill-formed bytes are written as is
const std::string dumped_ascii = jval.dump(-1, ' ', true, eh::keep);
CHECK(dumped_ascii.find(c.bytes) != std::string::npos);
}
// well-formed characters around an ill-formed sequence are still
// escaped as usual under ensure_ascii
const json mixed = valid_sequence + ill_formed_cases[1].bytes; // "é" + lone 0xFF
const std::string dumped_mixed = mixed.dump(-1, ' ', true, eh::keep);
CHECK(dumped_mixed.find("\\u00e9") != std::string::npos);
CHECK(dumped_mixed.find(ill_formed_cases[1].bytes) != std::string::npos);
// the byte that ends an ill-formed sequence is read again, so a quote,
// a backslash, or a control character after it is still escaped, and
// a well-formed code point after it is escaped under ensure_ascii
for (const bool ensure_ascii :
{
false, true
})
{
CAPTURE(ensure_ascii);
CHECK(json("\xC3\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\\"\"");
CHECK(json("\xC3\\").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\\\\"");
CHECK(json("\xC3\n").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\n\"");
CHECK(json("\xE2\x82\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xE2\x82\\\"\"");
CHECK(json("\xFF\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xFF\\\"\"");
CHECK(json("a\xE2\x82").dump(-1, ' ', ensure_ascii, eh::keep) == "\"a\xE2\x82\"");
}
CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', false, eh::keep) == "\"\xC3\xC3\xA9\"");
CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', true, eh::keep) == "\"\xC3\\u00e9\"");
}
SECTION("to_msgpack defaults to keep; to_bon8 is not affected by error_handler")
{
const json jval = ill_formed_cases[1].bytes; // lone 0xFF
// to_msgpack's error_handler defaults to keep, as MessagePack's spec
// allows any bytes in a str, so the bytes are passed through
CHECK(json::to_msgpack(jval) == json::to_msgpack(jval, eh::keep));
CHECK(json::from_msgpack(json::to_msgpack(jval)).get<std::string>() == ill_formed_cases[1].bytes);
// the diagnostics context of an ill-formed key is the object
json jobj;
jobj["\xFF"] = 1;
CHECK_THROWS_WITH_AS(json::to_msgpack(jobj, eh::strict), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// to_bon8 has no error_handler parameter; UTF-8 is structural for
// BON8, so it always rejects ill-formed input
CHECK_THROWS_AS(json::to_bon8(jval), json::type_error&);
}
SECTION("allow_exceptions=false with error_handler_t::strict discards the value")
{
const auto bytes = json::to_cbor(json(ill_formed_cases[0].bytes), eh::keep);
const json result = json::from_cbor(bytes, true, false, json::cbor_tag_handler_t::error, eh::strict);
CHECK(result.is_discarded());
}
SECTION("default parameters are unchanged")
{
const json jval = ill_formed_cases[0].bytes;
// to_*: the default error_handler is keep, so ill-formed bytes are
// written unchanged, exactly as in release 3.12.0 (it is strict only
// if JSON_STRICT_BINARY_UTF8 is enabled, see
// unit-binary_utf8_strict.cpp)
CHECK(json::to_cbor(jval) == json::to_cbor(jval, eh::keep));
CHECK(json::to_ubjson(jval) == json::to_ubjson(jval, false, false, eh::keep));
CHECK(json::to_bjdata(jval) == json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep));
{
json jobj;
jobj["k"] = jval;
CHECK(json::to_bson(jobj) == json::to_bson(jobj, eh::keep));
}
// from_*: the default error_handler is keep, so ill-formed bytes are
// still accepted unchanged, exactly as in release 3.12.0
const auto cbor_bytes = json::to_cbor(jval, eh::keep);
CHECK(json::from_cbor(cbor_bytes).get<std::string>() == ill_formed_cases[0].bytes);
const auto ubjson_bytes = json::to_ubjson(jval, false, false, eh::keep);
CHECK(json::from_ubjson(ubjson_bytes).get<std::string>() == ill_formed_cases[0].bytes);
const auto bjdata_bytes = json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep);
CHECK(json::from_bjdata(bjdata_bytes).get<std::string>() == ill_formed_cases[0].bytes);
const auto msgpack_bytes = json::to_msgpack(jval);
CHECK(json::from_msgpack(msgpack_bytes).get<std::string>() == ill_formed_cases[0].bytes);
json bson_obj;
bson_obj["k"] = jval;
const auto bson_bytes = json::to_bson(bson_obj, eh::keep);
CHECK(json::from_bson(bson_bytes)["k"].get<std::string>() == ill_formed_cases[0].bytes);
}
}
+122
View File
@@ -0,0 +1,122 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
// The binary writers check strings and object keys for valid UTF-8 only if
// JSON_STRICT_BINARY_UTF8 is enabled (planned to be the default in 4.0.0).
// Without it, they write the bytes unchanged, as before version 3.13.0; the
// tests for that are next to the other tests of each format.
#ifdef JSON_STRICT_BINARY_UTF8
#undef JSON_STRICT_BINARY_UTF8
#endif
#define JSON_STRICT_BINARY_UTF8 1
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <cstdint>
#include <vector>
TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)")
{
SECTION("CBOR")
{
// a string value with ill-formed UTF-8 is rejected
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
// an object key with ill-formed UTF-8 is rejected the same way
CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// binary values are not text and are unaffected
CHECK_NOTHROW(json::to_cbor(json::binary(std::vector<std::uint8_t>({0xFF}))));
// a value read back from CBOR with ill-formed bytes cannot be written
// back either (the reader is lenient regardless of the macro)
const json j = json::from_cbor(std::vector<std::uint8_t>({0x62, 0xc0, 0xae}));
CHECK_THROWS_WITH_AS(json::to_cbor(j), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
}
SECTION("UBJSON")
{
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
// an object key with ill-formed UTF-8 is rejected the same way
CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
}
SECTION("BJData")
{
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
// an object key with ill-formed UTF-8 is rejected the same way
CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
}
SECTION("BSON")
{
// to_bson() rejects the same kind of ill-formed string value, before
// any bytes reach the output adapter (the BSON document length
// prefix must be known up front, so nothing is written incrementally)
std::vector<std::uint8_t> out{0x42}; // a sentinel byte the writer must not touch
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter<std::uint8_t>(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
CHECK(out == std::vector<std::uint8_t> {0x42});
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
// a truncated multi-byte sequence
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
// an encoded surrogate half (U+D800)
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
// an overlong encoding of '.'
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
// an object key with ill-formed UTF-8 is rejected as well; unlike
// the reader (which never validates element names), the writer
// checks both string values and object keys
CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
}
SECTION("an explicit error_handler overrides the default")
{
// the macro only changes the default of the error_handler parameter
CHECK(json::to_cbor(json("\xFF"), json::error_handler_t::keep) == std::vector<std::uint8_t>({0x61, 0xff}));
CHECK(json::to_ubjson(json("\xFF"), false, false, json::error_handler_t::keep) == std::vector<std::uint8_t>({'S', 'i', 1, 0xff}));
CHECK(json::to_bjdata(json("\xFF"), false, false, json::bjdata_version_t::draft2, json::error_handler_t::keep) == std::vector<std::uint8_t>({'S', 'i', 1, 0xff}));
CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}}, json::error_handler_t::keep)) == json{{"s", "\xFF"}});
CHECK(json::to_cbor(json("\xFF"), json::error_handler_t::replace) == std::vector<std::uint8_t>({0x63, 0xef, 0xbf, 0xbd}));
}
SECTION("MessagePack and BON8 are unaffected")
{
// MessagePack allows any bytes in a str, so to_msgpack() still
// defaults to keep (strict only if passed explicitly); BON8 always
// checks, because the lead bytes mark where strings end
CHECK(json::to_msgpack(json("\xFF")) == std::vector<std::uint8_t>({0xa1, 0xff}));
CHECK_THROWS_AS(json::to_msgpack(json("\xFF"), json::error_handler_t::strict), json::type_error&);
CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&);
}
}
+37
View File
@@ -3906,6 +3906,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
CHECK(json::to_bjdata(j) == v);
CHECK(json::from_bjdata(v) == j);
}
SECTION("ill-formed UTF-8 (see #5529, #5651)")
{
// none of the binary format specs requires a decoder to reject
// ill-formed UTF-8 in a text string, so a value whose bytes are
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
// round-trips byte for byte as a string value; to_bjdata() writes
// the bytes unchanged, as before 3.13.0, unless
// JSON_STRICT_BINARY_UTF8 is enabled (see
// unit-binary_utf8_strict.cpp)
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
json j;
CHECK_NOTHROW(j = json::from_bjdata(v));
REQUIRE(j.is_string());
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
CHECK_THROWS_AS(j.dump(), json::type_error&);
CHECK(json::from_bjdata(json::to_bjdata(j)) == j);
// the same bytes as an object key round-trip as well
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
json j_key;
CHECK_NOTHROW(j_key = json::from_bjdata(v_key));
REQUIRE(j_key.is_object());
CHECK(j_key.contains(std::string("\xc0\xae")));
CHECK(json::from_bjdata(json::to_bjdata(j_key)) == j_key);
CHECK(json::from_bjdata(json::to_bjdata(json("\xFF"))) == json("\xFF"));
// a truncated multi-byte sequence
CHECK(json::from_bjdata(json::to_bjdata(json("\xC3"))) == json("\xC3"));
// an encoded surrogate half (U+D800)
CHECK(json::from_bjdata(json::to_bjdata(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
// an overlong encoding of '.'
CHECK(json::from_bjdata(json::to_bjdata(json("\xC0\xAF"))) == json("\xC0\xAF"));
// an object key with ill-formed UTF-8 is kept the same way
CHECK(json::from_bjdata(json::to_bjdata(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
}
}
SECTION("Array Type")
+37
View File
@@ -154,6 +154,43 @@ TEST_CASE("BSON")
#endif
}
SECTION("ill-formed UTF-8 (see #5529, #5651)")
{
// a BSON document {"s": "\xC0\xAE"} (0xC0 0xAE is an overlong
// encoding of '.'); the BSON spec does not require a decoder to
// reject ill-formed UTF-8 in a string value, so the reader hands the
// bytes back unchanged
const std::vector<uint8_t> v =
{
0x0F, 0x00, 0x00, 0x00, // document length
0x02, 's', 0x00, // type 0x02 (string), key "s"
0x03, 0x00, 0x00, 0x00, // string length (including null)
0xc0, 0xae, 0x00, // string content and its null terminator
0x00 // document terminator
};
json j;
CHECK_NOTHROW(j = json::from_bson(v));
REQUIRE(j.is_object());
REQUIRE(j.contains("s"));
CHECK(j["s"].get_ref<const json::string_t&>() == std::string("\xc0\xae"));
// dump() still requires valid UTF-8 and throws for such a value
CHECK_THROWS_AS(j.dump(), json::type_error&);
// to_bson() writes the bytes back unchanged, as before 3.13.0,
// unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp)
CHECK(json::from_bson(json::to_bson(j)) == j);
CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}})) == json{{"s", "\xFF"}});
// a truncated multi-byte sequence
CHECK(json::from_bson(json::to_bson(json{{"s", "\xC3"}})) == json{{"s", "\xC3"}});
// an encoded surrogate half (U+D800)
CHECK(json::from_bson(json::to_bson(json{{"s", "\xED\xA0\x80"}})) == json{{"s", "\xED\xA0\x80"}});
// an overlong encoding of '.'
CHECK(json::from_bson(json::to_bson(json{{"s", "\xC0\xAF"}})) == json{{"s", "\xC0\xAF"}});
// an object key with ill-formed UTF-8 is kept as well
CHECK(json::from_bson(json::to_bson(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
}
SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON")
{
// out_of_range.412 is thrown from a single shared helper
+67 -18
View File
@@ -1801,19 +1801,41 @@ TEST_CASE("CBOR")
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&);
}
SECTION("invalid UTF-8 in string (see #5529)")
SECTION("ill-formed UTF-8 in string (see #5529, #5651)")
{
// RFC 8949 §3.1 leaves it up to the decoder whether to reject
// ill-formed UTF-8 in a text string; this library does not, and
// hands the original bytes back unchanged, matching the
// MessagePack reader and the behavior before #5185/#5531 (not in
// any release)
// a two-character text string (major type 3) whose bytes are not
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be
// rejected at decode time, matching every other kind of
// malformed binary input, rather than only failing later when
// the resulting value is dumped
json _;
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
CHECK(json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae}), true, false).is_discarded());
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') round-trips
// byte for byte as a string value
const std::vector<uint8_t> ill_formed_value = {0x62, 0xc0, 0xae};
json j_value;
CHECK_NOTHROW(j_value = json::from_cbor(ill_formed_value));
REQUIRE(j_value.is_string());
CHECK(j_value.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
// dump() still requires valid UTF-8 and throws for such a value,
// unless an error handler that replaces or ignores the bytes is
// passed
CHECK_THROWS_AS(j_value.dump(), json::type_error&);
// to_cbor() writes the bytes back unchanged, as before 3.13.0,
// unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp)
CHECK(json::from_cbor(json::to_cbor(j_value)) == j_value);
// the same bytes as an object key round-trip as well
const std::vector<uint8_t> ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01};
json j_key;
CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key));
REQUIRE(j_key.is_object());
CHECK(j_key.contains(std::string("\xc0\xae")));
CHECK(json::from_cbor(json::to_cbor(j_key)) == j_key);
// a CBOR byte string (major type 2) with the very same bytes is
// NOT text and must still be accepted as-is
json _;
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x42, 0xc0, 0xae})));
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
@@ -1822,17 +1844,47 @@ TEST_CASE("CBOR")
CHECK(json::from_cbor(json::to_cbor(j)) == j);
}
SECTION("invalid UTF-8 in indefinite-length string")
SECTION("to_cbor keeps ill-formed UTF-8 (see #5651)")
{
// to_cbor() writes the bytes unchanged, as before 3.13.0, unless
// JSON_STRICT_BINARY_UTF8 is enabled (see
// unit-binary_utf8_strict.cpp); from_cbor() reads them back as is
CHECK(json::from_cbor(json::to_cbor(json("\xFF"))) == json("\xFF"));
// a truncated multi-byte sequence
CHECK(json::from_cbor(json::to_cbor(json("\xC3"))) == json("\xC3"));
// an encoded surrogate half (U+D800)
CHECK(json::from_cbor(json::to_cbor(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
// an overlong encoding of '.'
CHECK(json::from_cbor(json::to_cbor(json("\xC0\xAF"))) == json("\xC0\xAF"));
// an object key with ill-formed UTF-8 is kept the same way
CHECK(json::from_cbor(json::to_cbor(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
// binary values are not text and are unaffected
CHECK_NOTHROW(json::to_cbor(json::binary(std::vector<std::uint8_t>({0xFF}))));
}
SECTION("ill-formed UTF-8 in indefinite-length string")
{
json _;
// every chunk must be valid UTF-8 on its own (RFC 8949, Section
// 3.2.3), so a code point split across two chunks is rejected
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded());
// the chunks are concatenated as is, without checking that each
// chunk is valid UTF-8 on its own (RFC 8949, Section 3.2.3), so
// a code point split across two chunks yields a valid string
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})));
CHECK(_ == "\xc3\xa9");
CHECK(_.dump() == "\"\xc3\xa9\"");
// an ill-formed later chunk is rejected after valid ones
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
// a truncated code point is kept as is
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0xff})));
CHECK(_ == "\xc3");
CHECK_THROWS_AS(_.dump(), json::type_error&);
CHECK(json::from_cbor(json::to_cbor(_)) == _);
// an ill-formed later chunk is kept after valid ones
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})));
CHECK(_ == "\xc3\xa9\xc0\xae");
CHECK_THROWS_AS(_.dump(), json::type_error&);
// valid multi-byte chunks are accepted
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6");
@@ -1840,9 +1892,6 @@ TEST_CASE("CBOR")
SECTION("many chunks in indefinite-length string")
{
// only the newly read chunk is validated, not the whole string
// collected so far; validating the latter made this input take
// quadratic time (about ten seconds for 100000 chunks)
constexpr std::size_t chunks = 100000;
std::vector<uint8_t> v{0x7f};
for (std::size_t i = 0; i < chunks; ++i)
+3 -5
View File
@@ -2686,12 +2686,10 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX")
SECTION("move constructor resets the moved-from value to npos")
{
// basic_json(basic_json&&) (json.hpp, around line 1951) copies
// basic_json(basic_json&&) copies
// other's start_position/end_position into *this and then resets
// other's to npos (see the cppcheck-suppress[accessForwarded]
// annotation there, which flags this reset as worth a second
// look). Only the top-level moved-from value is affected; its
// (moved-away) children are gone along with it.
// other's to npos. Only the top-level moved-from value is
// affected; its (moved-away) children are gone along with it.
const std::string s = R"({"a":1,"b":[1,2,3]})";
json a = json::parse(s);
const auto a_start = a.start_pos();
+59
View File
@@ -430,6 +430,37 @@ TEST_CASE("value conversion")
CHECK(std::equal(std::begin(nbs[0][0][0]), std::end(nbs[1][1][1]), std::begin(nbs2[0][0][0])));
}
SECTION("built-in arrays: 5D")
{
// NOLINTBEGIN(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
const int nbs[1][1][1][2][2] = {{{{{0, 1}, {2, 3}}}}};
int nbs2[1][1][1][2][2] = {{{{{0, 0}, {0, 0}}}}};
// NOLINTEND(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
const json j2 = nbs;
j2.get_to(nbs2);
CHECK(std::equal(std::begin(nbs[0][0][0][0]), std::end(nbs[0][0][0][1]), std::begin(nbs2[0][0][0][0])));
}
SECTION("built-in arrays: mismatched shape")
{
// NOLINTBEGIN(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
int nbs2[2][3] = {{0, 0, 0}, {0, 0, 0}};
// NOLINTEND(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
SECTION("not an array")
{
const json j2 = 42;
CHECK_THROWS_WITH_AS(j2.get_to(nbs2), "[json.exception.type_error.304] cannot use at() with number", json::type_error&);
}
SECTION("too few elements")
{
const json j2 = {{0, 1, 2}};
CHECK_THROWS_WITH_AS(j2.get_to(nbs2), "[json.exception.out_of_range.401] array index 1 is out of range", json::out_of_range&);
}
}
SECTION("std::deque<json>")
{
std::deque<json> a{"previous", "value"};
@@ -1748,6 +1779,34 @@ NLOHMANN_JSON_SERIALIZE_ENUM_STRICT(StrictTaskState,
{STRICT_TS_COMPLETED, "completed"},
})
// regression test for #5708 item 2: NLOHMANN_JSON_SERIALIZE_ENUM_STRICT must not rely on
// unqualified lookup of a helper name that a user's own namespace may also declare
namespace ns_with_colliding_name
{
// NOLINTNEXTLINE(misc-use-internal-linkage) - used to shadow the library's internal helper name
inline void templated_json_throw(int /*unused*/) {}
enum class colliding_enum { a, b };
// NOLINTNEXTLINE(misc-use-internal-linkage,misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - false positive
NLOHMANN_JSON_SERIALIZE_ENUM_STRICT(colliding_enum,
{
{colliding_enum::a, "a"},
{colliding_enum::b, "b"}
})
} // namespace ns_with_colliding_name
TEST_CASE("NLOHMANN_JSON_SERIALIZE_ENUM_STRICT in a namespace with a colliding name")
{
using ns_with_colliding_name::colliding_enum;
CHECK(json(colliding_enum::a) == "a");
CHECK(colliding_enum::b == json("b"));
json _;
CHECK_THROWS_WITH_AS(_ = json("nope").get<colliding_enum>(), "[json.exception.out_of_range.410] enum value out of range for colliding_enum: \"nope\"", json::out_of_range&);
}
TEST_CASE("Strict JSON to enum mapping")
{
SECTION("enum class")
+70
View File
@@ -10,6 +10,8 @@
#include <set>
#include <sstream>
#include <string>
#include <type_traits>
#include <utility>
#include <vector>
#include "doctest_compatibility.h"
@@ -406,6 +408,74 @@ TEST_CASE("JSON Visit Node")
CHECK(expected.empty());
}
// Test accessing members of a custom base class that are hidden by members of nlohmann::basic_json
class base_class_with_hidden_members
{
public:
const char* type_name() const noexcept // NOLINT(readability-convert-member-functions-to-static)
{
return "custom type_name";
}
std::size_t size() const noexcept
{
return m_size;
}
std::size_t m_size = 42;
};
using json_with_hidden_base_members =
nlohmann::basic_json <
std::map,
std::vector,
std::string,
bool,
std::int64_t,
std::uint64_t,
double,
std::allocator,
nlohmann::adl_serializer,
std::vector<std::uint8_t>,
base_class_with_hidden_members
>;
TEST_CASE("JSON Node as_base_class")
{
using json = json_with_hidden_base_members;
static_assert(std::is_same<decltype(std::declval<json&>().as_base_class()), json::json_base_class_t&>::value, "");
static_assert(std::is_same<decltype(std::declval<const json&>().as_base_class()), const json::json_base_class_t&>::value, "");
static_assert(noexcept(std::declval<json&>().as_base_class()), "");
static_assert(noexcept(std::declval<const json&>().as_base_class()), "");
SECTION("non-const")
{
json j = {1, 2, 3};
CHECK(std::string(j.type_name()) == "array");
CHECK(j.size() == 3);
CHECK(std::string(j.as_base_class().type_name()) == "custom type_name");
CHECK(j.as_base_class().size() == 42);
CHECK(&j.as_base_class() == &static_cast<json::json_base_class_t&>(j));
j.as_base_class().m_size = 7;
CHECK(j.as_base_class().size() == 7);
CHECK(j.size() == 3);
}
SECTION("const")
{
const json j = {1, 2, 3};
CHECK(std::string(j.type_name()) == "array");
CHECK(j.size() == 3);
CHECK(std::string(j.as_base_class().type_name()) == "custom type_name");
CHECK(j.as_base_class().size() == 42);
CHECK(&j.as_base_class() == &static_cast<const json::json_base_class_t&>(j));
}
}
// A custom base class with a const member: copy-constructible (initializing a
// const member works fine), but not copy-/move-assignable (assigning one does
// not). Used to check that copy construction never requires more than that.
+43
View File
@@ -630,6 +630,49 @@ TEST_CASE("modifiers")
}
}
SECTION("rvalue at position moves rather than copies")
{
// regression test: insert(pos, basic_json&&) used to forward to
// insert(pos, const basic_json&) because the named rvalue
// reference parameter is itself an lvalue, so it always
// deep-copied its argument instead of moving it
json j_big = std::string(1000, 'x');
const auto* const original_buffer = j_big.get_ref<const std::string&>().data();
auto it = j_array.insert(j_array.begin(), std::move(j_big));
CHECK(j_array.size() == 5);
CHECK(*it == json(std::string(1000, 'x')));
CHECK((*it).get_ref<const std::string&>().data() == original_buffer);
// the moved-from value is null, the same as after push_back(&&)
CHECK(j_big.is_null()); // NOLINT(bugprone-use-after-move,hicpp-invalid-access-moved)
}
SECTION("self-aliasing insertion")
{
SECTION("without reallocation")
{
json j_self = {1, 2, 3, 4};
j_self.get_ref<json::array_t&>().reserve(j_self.size() + 1);
auto it = j_self.insert(j_self.begin(), std::move(j_self[1]));
CHECK(j_self.size() == 5);
CHECK(*it == json(2));
CHECK(j_self == json({2, 1, nullptr, 3, 4}));
}
SECTION("with reallocation")
{
json j_self = {1, 2, 3, 4};
j_self.get_ref<json::array_t&>().shrink_to_fit();
auto it = j_self.insert(j_self.begin(), std::move(j_self[1]));
CHECK(j_self.size() == 5);
CHECK(*it == json(2));
CHECK(j_self == json({2, 1, nullptr, 3, 4}));
}
}
SECTION("copies at position")
{
SECTION("insert before begin()")
+28 -8
View File
@@ -1540,19 +1540,39 @@ TEST_CASE("MessagePack")
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&);
}
SECTION("invalid UTF-8 in string (see #5529)")
SECTION("ill-formed UTF-8 in string (see #5529, #5651)")
{
// the MessagePack specification explicitly allows a str object to
// contain a byte sequence that is not valid UTF-8 and expects a
// deserializer to hand the original bytes back unchanged; this
// library follows that, unlike CBOR/UBJSON/BJData/BSON, whose
// specifications require text strings to be valid UTF-8
// a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8
// (0xC0 0xAE is an overlong encoding of '.') must be rejected at
// decode time, matching every other kind of malformed binary
// input, rather than only failing later when the resulting
// value is dumped
json _;
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0xa2, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing MessagePack string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
CHECK(json::from_msgpack(std::vector<uint8_t>({0xa2, 0xc0, 0xae}), true, false).is_discarded());
// (0xC0 0xAE is an overlong encoding of '.') round-trips byte for
// byte as a string value
const std::vector<uint8_t> ill_formed_value = {0xa2, 0xc0, 0xae};
json j_value;
CHECK_NOTHROW(j_value = json::from_msgpack(ill_formed_value));
REQUIRE(j_value.is_string());
CHECK(j_value.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
CHECK(json::from_msgpack(json::to_msgpack(j_value)) == j_value);
// dump() still requires valid UTF-8 and throws for such a value,
// unless an error handler that replaces or ignores the bytes is
// passed
CHECK_THROWS_AS(j_value.dump(), json::type_error&);
// the same bytes as an object key round-trip as well
const std::vector<uint8_t> ill_formed_key = {0x81, 0xa2, 0xc0, 0xae, 0x01};
json j_key;
CHECK_NOTHROW(j_key = json::from_msgpack(ill_formed_key));
REQUIRE(j_key.is_object());
CHECK(j_key.contains(std::string("\xc0\xae")));
CHECK(json::from_msgpack(json::to_msgpack(j_key)) == j_key);
// a MessagePack bin8 blob with the very same bytes is NOT text
// and must still be accepted as-is
json _;
CHECK_NOTHROW(_ = json::from_msgpack(std::vector<uint8_t>({0xc4, 0x02, 0xc0, 0xae})));
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
+20 -4
View File
@@ -808,6 +808,26 @@ TEST_CASE("regression tests 2")
CHECK(j == k);
}
#ifdef JSON_HAS_CPP_17
SECTION("issue #5066 - MSVC converts json to std::variant<json> via the conversion operator")
{
// std::variant<json> must not be retrievable via get<>(), because otherwise the
// implicit conversion operator becomes a candidate that MSVC picks over the variant's
// converting constructor, routing a number through the string from_json overload
static_assert(!nlohmann::detail::is_detected<nlohmann::detail::get_template_function, const json&, std::variant<json>>::value,
"std::variant<json> must not be retrievable via get<>()");
// clang before 7 cannot instantiate libstdc++'s std::variant<json>
#if !(defined(__clang__) && __clang_major__ < 7)
// push_back, not emplace_back: #5066 needs the implicit conversion
// from json to the vector's value type
std::vector<std::variant<json>> v;
v.push_back(json(1)); // NOLINT(hicpp-use-emplace,modernize-use-emplace)
CHECK(std::get<0>(v[0]) == 1);
#endif
}
#endif
SECTION("issue #3669 - invalid use of incomplete type with optional member and to_json")
{
const Issue3669Holder h{};
@@ -1286,15 +1306,11 @@ TEST_CASE("regression test - #3989 SAX parse_error() returning true")
{json::input_format_t::cbor, {0x82, 0xC1, 0x05, 0xD9, 0xD9, 0xF7, 0x06}, {5, 6}, 2},
// CBOR: undefined and other simple values become null
{json::input_format_t::cbor, {0x84, 0xF7, 0xE0, 0xF8, 0x20, 0x01}, {nullptr, nullptr, nullptr, 1}, 3},
// CBOR: ill-formed UTF-8 becomes U+FFFD, also in keys
{json::input_format_t::cbor, {0xA1, 0x61, 0xFF, 0x62, 0xC3, 0x28}, {{replacement_character(), replacement_character() + "("}}, 2},
// CBOR: members whose key is not a string are skipped, whatever their key and value
{json::input_format_t::cbor, {0xA4, 0x01, 0x02, 0x82, 0x01, 0x02, 0xA1, 0x61, 'x', 0x9F, 0xFF, 0xC1, 0x01, 0x5F, 0x41, 0x00, 0xFF, 0x61, 'a', 0x03}, {{"a", 3}}, 3},
{json::input_format_t::cbor, {0xBF, 0xF5, 0xBF, 0x61, 'x', 0x7F, 0x61, 'y', 0xFF, 0xFF, 0x61, 'a', 0x03, 0xFF}, {{"a", 3}}, 1},
// MessagePack: members whose key is not a string are skipped
{json::input_format_t::msgpack, {0x84, 0x01, 0x02, 0x81, 0xA1, 'x', 0x01, 0x92, 0x01, 0x02, 0xD4, 0x01, 0x02, 0xC0, 0xA1, 'a', 0x04}, {{"a", 4}}, 3},
// MessagePack: ill-formed UTF-8 becomes U+FFFD
{json::input_format_t::msgpack, {0x92, 0xA2, 0xC3, 0x28, 0xA3, 0xE2, 0x82, 'x'}, {replacement_character() + "(", replacement_character() + "x"}, 2},
// UBJSON: a char that is not ASCII becomes U+FFFD
{json::input_format_t::ubjson, {'[', 'C', 0x80, 'C', 'A', ']'}, {replacement_character(), "A"}, 1},
// UBJSON: the longest beginning of a high-precision number is kept
+12
View File
@@ -10,10 +10,14 @@
#if JSON_TEST_USING_MULTIPLE_HEADERS
#include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/ordered_map.hpp>
#else
#include <nlohmann/json.hpp>
#endif
#include <map>
#include <string>
TEST_CASE("type traits")
{
SECTION("is_c_string")
@@ -83,4 +87,12 @@ TEST_CASE("type traits")
}
}
}
SECTION("is_ordered_map")
{
using nlohmann::detail::is_ordered_map;
CHECK(is_ordered_map<nlohmann::ordered_map<std::string, int>>::value);
CHECK_FALSE(is_ordered_map<std::map<std::string, int>>::value);
}
}
+37
View File
@@ -2505,6 +2505,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
CHECK(json::to_ubjson(j) == v);
CHECK(json::from_ubjson(v) == j);
}
SECTION("ill-formed UTF-8 (see #5529, #5651)")
{
// none of the binary format specs requires a decoder to reject
// ill-formed UTF-8 in a text string, so a value whose bytes are
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
// round-trips byte for byte as a string value; to_ubjson() writes
// the bytes unchanged, as before 3.13.0, unless
// JSON_STRICT_BINARY_UTF8 is enabled (see
// unit-binary_utf8_strict.cpp)
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
json j;
CHECK_NOTHROW(j = json::from_ubjson(v));
REQUIRE(j.is_string());
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
CHECK_THROWS_AS(j.dump(), json::type_error&);
CHECK(json::from_ubjson(json::to_ubjson(j)) == j);
// the same bytes as an object key round-trip as well
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
json j_key;
CHECK_NOTHROW(j_key = json::from_ubjson(v_key));
REQUIRE(j_key.is_object());
CHECK(j_key.contains(std::string("\xc0\xae")));
CHECK(json::from_ubjson(json::to_ubjson(j_key)) == j_key);
CHECK(json::from_ubjson(json::to_ubjson(json("\xFF"))) == json("\xFF"));
// a truncated multi-byte sequence
CHECK(json::from_ubjson(json::to_ubjson(json("\xC3"))) == json("\xC3"));
// an encoded surrogate half (U+D800)
CHECK(json::from_ubjson(json::to_ubjson(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
// an overlong encoding of '.'
CHECK(json::from_ubjson(json::to_ubjson(json("\xC0\xAF"))) == json("\xC0\xAF"));
// an object key with ill-formed UTF-8 is kept the same way
CHECK(json::from_ubjson(json::to_ubjson(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
}
}
SECTION("Array Type")
+56 -4
View File
@@ -8,6 +8,7 @@
#include "doctest_compatibility.h"
#include <cwchar>
#include <nlohmann/json.hpp>
using nlohmann::json;
@@ -68,15 +69,15 @@ TEST_CASE("wide strings")
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// a valid surrogate pair is still decoded (U+1F600)
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
}
@@ -104,4 +105,55 @@ TEST_CASE("wide strings")
// the same unit inside a string is reported as an ill-formed byte
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
}
SECTION("malformed wide-string input outside strings (#5645)")
{
json _;
// a lone low surrogate inside a literal must not be truncated to its
// low byte and mistaken for the letter the literal expects next
// (0xDC72 truncates to 'r', which is what "true" expects after 't')
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xDC72), u'u', u'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
// ... also when the lone surrogate is the last unit of the input
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'f', u'a', u'l', u's', static_cast<char16_t>(0xDD65)}),
"[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'fals\xFF'", json::parse_error&);
// a high surrogate followed by a unit that is not its low surrogate
// must not silently swallow that unit
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xD872), u'X', u'u', u'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
// ... in particular, if the swallowed unit is the newline that ends a
// // comment, the comment must not extend over the following line
CHECK(json::parse(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
u',', u'2', u' ', u'/', u'/', u'\n', u']'},
nullptr, true, /*ignore_comments*/true) == json::parse("[1,2]"));
CHECK(json::accept(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
u',', u'2', u' ', u'/', u'/', u'\n', u']'}, /*ignore_comments*/true));
// cases 5 and 6 use a 32-bit wchar_t (Linux, macOS, the BSDs) to reach
// the UTF-32 helper tested above via u32string; the 16-bit wchar_t of
// Windows goes through the UTF-16 helper instead, already covered by
// the u16string cases above
#if WCHAR_MAX > 0xFFFFu
// a negative wchar_t must not be mistaken for
// char_traits<char>::eof() and silently end the input, letting
// trailing garbage pass the strict end-of-input check (only observable
// where wint_t is signed, e.g. macOS/the BSDs; on Linux wint_t is
// unsigned and this was already handled by #5348)
std::wstring w = L"[1]";
w.push_back(static_cast<wchar_t>(-1));
w += L"garbage";
CHECK(!json::accept(w));
CHECK_THROWS_WITH_AS(_ = json::parse(w),
"[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
// other negative wchar_t units must not be truncated to their low
// byte (0xFFFFFF72 truncates to 'r', as in the u16string case above)
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L't', static_cast<wchar_t>(0xFFFFFF72), L'u', L'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
#endif
}
}