diff --git a/.clang-tidy b/.clang-tidy index fa3e03ae3..a16136785 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -1,18 +1,9 @@ -# bugprone-use-after-move (hicpp-invalid-access-moved is its alias) still flags -# the basic_json move constructor, which forwards the whole object to its base -# class (#5724), and two forwards in the error-message construction of -# at(KeyType&&) (json.hpp, both overloads: find(std::forward(key)) -# followed by string_t(std::forward(key)) in the throw), which #5689 -# rewrites. Re-enable both checks once those changes have landed. # portability-avoid-pragma-once: kept disabled on purpose. #pragma once is accepted # by every supported compiler, and tools/amalgamate/amalgamate.py strips it from # single_include, so there is nothing left to fix here. Checks: '*, - -bugprone-use-after-move, - -hicpp-invalid-access-moved, - -altera-id-dependent-backward-branch, -altera-struct-pack-align, -altera-unroll-loops, diff --git a/BUILD.bazel b/BUILD.bazel index 03f73fa92..3dfffad1e 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -53,11 +53,11 @@ cc_library( "include/nlohmann/detail/meta/detected.hpp", "include/nlohmann/detail/meta/identity_tag.hpp", "include/nlohmann/detail/meta/is_sax.hpp", - "include/nlohmann/detail/meta/logic.hpp", "include/nlohmann/detail/meta/std_fs.hpp", "include/nlohmann/detail/meta/type_traits.hpp", "include/nlohmann/detail/meta/void_t.hpp", "include/nlohmann/detail/output/binary_writer.hpp", + "include/nlohmann/detail/output/error_handler.hpp", "include/nlohmann/detail/output/output_adapters.hpp", "include/nlohmann/detail/output/serializer.hpp", "include/nlohmann/detail/recursion_depth_limit.hpp", diff --git a/CMakeLists.txt b/CMakeLists.txt index f0a6771dc..b5092ae49 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -61,6 +61,7 @@ option(JSON_Install "Install CMake targets during install option(JSON_MultipleHeaders "Use non-amalgamated version of the library." ON) option(JSON_SystemInclude "Include as system headers (skip for clang-tidy)." OFF) option(JSON_StrictNulHandling "Build with strict NUL-byte handling enabled." OFF) +option(JSON_StrictBinaryUTF8 "Build with UTF-8 checks in the CBOR, UBJSON, BJData, and BSON writers enabled." OFF) if (JSON_CI) include(ci) @@ -118,6 +119,10 @@ if (JSON_StrictNulHandling) message(STATUS "Strict NUL-byte handling enabled (JSON_STRICT_NUL_HANDLING=1)") endif() +if (JSON_StrictBinaryUTF8) + message(STATUS "Strict UTF-8 checks in binary writers enabled (JSON_STRICT_BINARY_UTF8=1)") +endif() + if (JSON_Diagnostic_Positions) message(STATUS "Diagnostic positions enabled (JSON_DIAGNOSTIC_POSITIONS=1)") endif() @@ -153,6 +158,7 @@ target_compile_definitions( $<$:JSON_DIAGNOSTIC_POSITIONS=1> $<$:JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1> $<$:JSON_STRICT_NUL_HANDLING=1> + $<$:JSON_STRICT_BINARY_UTF8=1> ) target_include_directories( diff --git a/cmake/ci.cmake b/cmake/ci.cmake index e67aba6d3..2aceb5afd 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -701,7 +701,7 @@ ci_get_cmake(4.0.0 CMAKE_4_0_0_BINARY) # the tests require CMake 3.13 or later, so they are excluded for CMake 3.5.0 set(JSON_CMAKE_FLAGS_3_5_0 JSON_Diagnostics JSON_Diagnostic_Positions JSON_GlobalUDLs JSON_ImplicitConversions JSON_DisableEnumSerialization JSON_LegacyDiscardedValueComparison JSON_Install JSON_MultipleHeaders JSON_SystemInclude JSON_Valgrind - JSON_StrictNulHandling) + JSON_StrictNulHandling JSON_StrictBinaryUTF8) set(JSON_CMAKE_FLAGS_3_31_6 JSON_BuildTests ${JSON_CMAKE_FLAGS_3_5_0}) set(JSON_CMAKE_FLAGS_4_0_0 JSON_BuildTests ${JSON_CMAKE_FLAGS_3_5_0}) diff --git a/docs/docset/docSet.sql b/docs/docset/docSet.sql index 477b2f9fe..09802e43e 100644 --- a/docs/docset/docSet.sql +++ b/docs/docset/docSet.sql @@ -19,6 +19,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('format_as', 'Function', 'api/ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::accept', 'Function', 'api/basic_json/accept/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::array', 'Function', 'api/basic_json/array/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::array_t', 'Type', 'api/basic_json/array_t/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::as_base_class', 'Method', 'api/basic_json/as_base_class/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::at', 'Method', 'api/basic_json/at/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::back', 'Method', 'api/basic_json/back/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::basic_json', 'Constructor', 'api/basic_json/basic_json/index.html'); diff --git a/docs/mkdocs/docs/api/basic_json/as_base_class.md b/docs/mkdocs/docs/api/basic_json/as_base_class.md new file mode 100644 index 000000000..769707140 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json/as_base_class.md @@ -0,0 +1,53 @@ +# nlohmann::basic_json::as_base_class + +```cpp +json_base_class_t& as_base_class() noexcept; +const json_base_class_t& as_base_class() const noexcept; +``` + +Returns a reference to this object as its custom base class [`json_base_class_t`](json_base_class_t.md). No copy is +made. + +Since `basic_json` derives from `json_base_class_t`, a member of `basic_json` hides any member of the custom base class +with the same name. This function makes such hidden members accessible again. + +## Return value + +reference to this object as [`json_base_class_t`](json_base_class_t.md) + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +The function is equivalent to `static_cast(j)` (or `static_cast(j)`). + +## Examples + +??? example + + The example shows how to use `as_base_class` to access members of the custom base class that are hidden by members + of `basic_json`. + + ```cpp + --8<-- "examples/as_base_class.cpp" + ``` + + Output: + + ```json + --8<-- "examples/as_base_class.output" + ``` + +## See also + +- [json_base_class_t](json_base_class_t.md) - type of the custom base class + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/dump.md b/docs/mkdocs/docs/api/basic_json/dump.md index a3c9db3a8..d805f3db1 100644 --- a/docs/mkdocs/docs/api/basic_json/dump.md +++ b/docs/mkdocs/docs/api/basic_json/dump.md @@ -25,10 +25,12 @@ and `ensure_ascii` parameters. result consists of ASCII characters only. `error_handler` (in) -: how to react on decoding errors; there are three possible values (see [`error_handler_t`](error_handler_t.md): +: how to react on decoding errors; there are four possible values (see [`error_handler_t`](error_handler_t.md): `strict` (throws an exception in case a decoding error occurs; default), `replace` (replace invalid UTF-8 sequences - with U+FFFD), and `ignore` (ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the - output unchanged, and invalid bytes are dropped)). + with U+FFFD), `ignore` (ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the + output unchanged, and invalid bytes are dropped), and `keep` (write the ill-formed bytes to the output as is, + without escaping them, even if `ensure_ascii` is `#!cpp true`; the result is then not valid UTF-8, but equals the + input bytes exactly, and well-formed characters around the ill-formed bytes are still escaped as usual)). ## Return value @@ -94,3 +96,4 @@ Binary values are serialized as an object containing two keys: - Indentation character `indent_char`, option `ensure_ascii` and exceptions added in version 3.0.0. - Error handlers added in version 3.4.0. - Serialization of binary values added in version 3.8.0. +- Error handler `keep` added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/error_handler_t.md b/docs/mkdocs/docs/api/basic_json/error_handler_t.md index 51dc6510f..17327fe26 100644 --- a/docs/mkdocs/docs/api/basic_json/error_handler_t.md +++ b/docs/mkdocs/docs/api/basic_json/error_handler_t.md @@ -4,15 +4,31 @@ enum class error_handler_t { strict, replace, - ignore + ignore, + keep }; ``` -This enumeration is used in the [`dump`](dump.md) function to choose how to treat decoding errors while serializing a -`basic_json` value. Three values are differentiated: +This enumeration is used to choose how to treat ill-formed UTF-8 in a string value or object key: + +- [`dump`](dump.md) uses it while serializing a `basic_json` value to text. +- [`to_cbor`](to_cbor.md), [`to_msgpack`](to_msgpack.md), [`to_ubjson`](to_ubjson.md), [`to_bjdata`](to_bjdata.md), + and [`to_bson`](to_bson.md) use it while serializing a `basic_json` value to that binary format. Their default is + `keep`, as no binary writer checked before this parameter was added. CBOR, UBJSON, BJData, and BSON require valid + UTF-8, so for these four the default is `strict` if [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) + is enabled; MessagePack's specification explicitly allows a string to contain ill-formed UTF-8, so `to_msgpack` + stays at `keep`. `to_bon8` does not take this parameter: BON8 always validates, since UTF-8 lead bytes are + structural to that format. +- [`from_cbor`](from_cbor.md), [`from_msgpack`](from_msgpack.md), [`from_ubjson`](from_ubjson.md), + [`from_bjdata`](from_bjdata.md), and [`from_bson`](from_bson.md) use it while parsing that binary format, to decide + whether to check a string value or object key for well-formed UTF-8 at all; by default (`keep`) they do not, as no + binary reader did before this parameter was added. `from_bon8` does not take this parameter, for the same reason + `to_bon8` does not. + +Four values are differentiated: strict -: throw a `type_error` exception in case of invalid UTF-8 +: throw a `type_error`/`parse_error` exception in case of invalid UTF-8 replace : replace invalid UTF-8 sequences with U+FFFD (� REPLACEMENT CHARACTER) @@ -20,6 +36,12 @@ replace ignore : ignore invalid UTF-8 sequences; all valid bytes are copied to the output unchanged, and invalid bytes are dropped +keep +: keep invalid UTF-8 sequences unchanged; only meaningful for the binary formats mentioned above, since [`dump`] + (dump.md) itself must produce text, and `keep` there writes the ill-formed bytes to the output as is, so the + result is then not valid UTF-8 (but still equals the input bytes exactly, including around any well-formed + characters, which are still escaped as usual) + ## Examples ??? example @@ -45,3 +67,5 @@ ignore ## Version history - Added in version 3.4.0. +- Added `keep`, and made this enumeration apply to the binary readers and writers in addition to `dump`, in version + 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/from_bjdata.md b/docs/mkdocs/docs/api/basic_json/from_bjdata.md index 9df21ab30..a45e00ad5 100644 --- a/docs/mkdocs/docs/api/basic_json/from_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/from_bjdata.md @@ -5,12 +5,14 @@ template static basic_json from_bjdata(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_bjdata(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the BJData (Binary JData) serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + BJData does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -73,7 +81,7 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if a parse error occurs - Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string could not be parsed - successfully + successfully, or if a string value or object key is not valid UTF-8 and `error_handler` is `strict` - Throws [out_of_range.408](../../home/exceptions.md#jsonexceptionout_of_range408) if the size of an optimized container or n-dimensional array cannot be represented by `std::size_t` @@ -111,3 +119,4 @@ Linear in the size of the input. - Added in version 3.11.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/from_bson.md b/docs/mkdocs/docs/api/basic_json/from_bson.md index 9dfd9dc18..b037e8e07 100644 --- a/docs/mkdocs/docs/api/basic_json/from_bson.md +++ b/docs/mkdocs/docs/api/basic_json/from_bson.md @@ -5,12 +5,14 @@ template static basic_json from_bson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_bson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the BSON (Binary JSON) serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + BSON does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -75,6 +83,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va invalid string or byte array length) - Throws [`parse_error.114`](../../home/exceptions.md#jsonexceptionparse_error114) if an unsupported BSON record type is encountered +- Throws [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) if a string value or object key is + not valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -111,6 +121,7 @@ Linear in the size of the input. - Added in version 3.4.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/from_cbor.md b/docs/mkdocs/docs/api/basic_json/from_cbor.md index 791c183bb..7d7df08a5 100644 --- a/docs/mkdocs/docs/api/basic_json/from_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/from_cbor.md @@ -6,14 +6,16 @@ template static basic_json from_cbor(InputType&& i, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error); + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_cbor(IteratorType first, SentinelType last, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error); + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the CBOR (Concise Binary Object Representation) serialization format. @@ -65,6 +67,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ : how to treat CBOR tags (optional, `error` by default); see [`cbor_tag_handler_t`](cbor_tag_handler_t.md) for more information +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + CBOR does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -80,8 +88,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from CBOR were used in the given input or if the input is not valid CBOR -- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other - types are not supported, as JSON object keys are always strings) or a string is malformed +- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of + other types are not supported, as JSON object keys are always strings), or if a string value or object key is not + valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -121,6 +130,7 @@ Linear in the size of the input. - Added `tag_handler` parameter in version 3.9.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/from_msgpack.md b/docs/mkdocs/docs/api/basic_json/from_msgpack.md index 395512acb..2e7fa5062 100644 --- a/docs/mkdocs/docs/api/basic_json/from_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/from_msgpack.md @@ -5,12 +5,14 @@ template static basic_json from_msgpack(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_msgpack(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the MessagePack serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + MessagePack's specification explicitly allows ill-formed UTF-8, so checking is opt-in: the default, `keep`, does + not check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -73,8 +81,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from MessagePack were used in the given input or if the input is not valid MessagePack -- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other - types are not supported, as JSON object keys are always strings) or a string is malformed +- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of + other types are not supported, as JSON object keys are always strings), or if a string value or object key is not + valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -113,6 +122,7 @@ Linear in the size of the input. - Added `allow_exceptions` parameter in version 3.2.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/from_ubjson.md b/docs/mkdocs/docs/api/basic_json/from_ubjson.md index 1ad076588..ac9404d4c 100644 --- a/docs/mkdocs/docs/api/basic_json/from_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/from_ubjson.md @@ -5,12 +5,14 @@ template static basic_json from_ubjson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_ubjson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the UBJSON (Universal Binary JSON) serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + UBJSON does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -73,7 +81,7 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if a parse error occurs - Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string could not be parsed - successfully + successfully, or if a string value or object key is not valid UTF-8 and `error_handler` is `strict` - Throws [out_of_range.408](../../home/exceptions.md#jsonexceptionout_of_range408) if the size of an optimized container or n-dimensional array cannot be represented by `std::size_t` @@ -112,6 +120,7 @@ Linear in the size of the input. - Added `allow_exceptions` parameter in version 3.2.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/index.md b/docs/mkdocs/docs/api/basic_json/index.md index 866bda67a..fc934ca93 100644 --- a/docs/mkdocs/docs/api/basic_json/index.md +++ b/docs/mkdocs/docs/api/basic_json/index.md @@ -200,6 +200,7 @@ Direct access to the stored value of a JSON value. - [**get_ref**](get_ref.md) - get a reference value - [**operator ValueType**](operator_ValueType.md) - get a value - [**get_binary**](get_binary.md) - get a binary value +- [**as_base_class**](as_base_class.md) - access the custom base class ### Element access diff --git a/docs/mkdocs/docs/api/basic_json/json_base_class_t.md b/docs/mkdocs/docs/api/basic_json/json_base_class_t.md index 0d1abc9d4..6f0026558 100644 --- a/docs/mkdocs/docs/api/basic_json/json_base_class_t.md +++ b/docs/mkdocs/docs/api/basic_json/json_base_class_t.md @@ -27,6 +27,18 @@ A `CustomBaseClass` with non-static data members forfeits `basic_json`'s [standard layout](https://en.cppreference.com/w/cpp/named_req/StandardLayoutType) guarantee. See [Template Parameter Requirements](../../features/types/template_parameters.md#custombaseclass). +#### Name conflicts + +Since `basic_json` derives from `CustomBaseClass`, members of `basic_json` hide members of `CustomBaseClass` with the +same name. Hidden members remain accessible via [`as_base_class`](as_base_class.md) or by casting the value to +`json_base_class_t`. + +!!! warning "Avoid generic member names" + + Future versions of the library may add members to `basic_json` that hide members of `CustomBaseClass` that are + accessible today. To reduce the risk of such conflicts, avoid generic names for the members of `CustomBaseClass`, + for instance by using a distinctive prefix. + ## Examples ??? example @@ -45,8 +57,10 @@ A `CustomBaseClass` with non-static data members forfeits `basic_json`'s ## See also +- [as_base_class](as_base_class.md) - access the custom base class - [Template Parameter Requirements](../../features/types/template_parameters.md#custombaseclass) - the requirements for `CustomBaseClass` ## Version history - Added in version 3.12.0. +- Made a public member type in version 3.13.0; it was private before, so it could not be named outside the class. diff --git a/docs/mkdocs/docs/api/basic_json/meta.md b/docs/mkdocs/docs/api/basic_json/meta.md index 55476c5f3..9b1ac2a4e 100644 --- a/docs/mkdocs/docs/api/basic_json/meta.md +++ b/docs/mkdocs/docs/api/basic_json/meta.md @@ -13,7 +13,7 @@ JSON object holding version information | key | description | |-------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| -| `compiler` | Information on the used compiler. It is an object with the following keys: `c++` (the used C++ standard), `family` (the compiler family; possible values are `clang`, `icc`, `gcc`, `ilecpp`, `msvc`, `pgcpp`, `sunpro`, and `unknown`), and `version` (the compiler version). On HP aCC compilers, `compiler` is instead the plain string `hp`. | +| `compiler` | Information on the used compiler. It is an object with the following keys: `c++` (the used C++ standard), `family` (the compiler family; possible values are `clang`, `icc`, `gcc`, `hp`, `ilecpp`, `msvc`, `pgcpp`, `sunpro`, and `unknown`), and `version` (the compiler version). | | `copyright` | The copyright line for the library as string. | | `name` | The name of the library as string. | | `platform` | The used platform as string. Possible values are `win32`, `linux`, `apple`, `unix`, and `unknown`. | diff --git a/docs/mkdocs/docs/api/basic_json/to_bjdata.md b/docs/mkdocs/docs/api/basic_json/to_bjdata.md index 44cc399e1..df3b69004 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/to_bjdata.md @@ -5,15 +5,18 @@ static std::vector to_bjdata(const basic_json& j, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2); + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = error_handler_t::keep); // (2) static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2); + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = error_handler_t::keep); static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2); + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the BJData (Binary JData) serialization format. BJData aims to @@ -43,6 +46,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ : which version of BJData to use (see note on "Binary values" on [BJData](../../features/binary_formats/bjdata.md)); optional, `#!cpp bjdata_version_t::draft2` by default. +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_bjdata` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. BJData serialization as byte vector @@ -56,6 +65,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false, and `j` contains a non-empty array, object, or binary value. +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -104,4 +116,7 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.11.0. -- BJData version parameter (for draft3 binary encoding) added in version 3.12.0. \ No newline at end of file +- BJData version parameter (for draft3 binary encoding) added in version 3.12.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`. \ No newline at end of file diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index c79f39ffc..ad974e048 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -2,11 +2,14 @@ ```cpp // (1) -static std::vector to_bson(const basic_json& j); +static std::vector to_bson(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep); // (2) -static void to_bson(const basic_json& j, detail::output_adapter o); -static void to_bson(const basic_json& j, detail::output_adapter o); +static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); +static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); ``` BSON (Binary JSON) is a binary format in which zero or more ordered key/value pairs are stored as a single entity (a @@ -25,6 +28,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `o` (in) : output adapter to write serialization to +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_bson` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. BSON serialization as a byte vector @@ -46,6 +55,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the BSON binary subtype; example: `"subtype 70000 is too large for the BSON binary subtype (max 255)"` +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -98,3 +110,7 @@ pass before anything is written. - Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. - Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0. - `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316` before anything + is written. diff --git a/docs/mkdocs/docs/api/basic_json/to_cbor.md b/docs/mkdocs/docs/api/basic_json/to_cbor.md index 3bbd9c7d3..c72f310bb 100644 --- a/docs/mkdocs/docs/api/basic_json/to_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/to_cbor.md @@ -2,11 +2,14 @@ ```cpp // (1) -static std::vector to_cbor(const basic_json& j); +static std::vector to_cbor(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep); // (2) -static void to_cbor(const basic_json& j, detail::output_adapter o); -static void to_cbor(const basic_json& j, detail::output_adapter o); +static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); +static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the CBOR (Concise Binary Object Representation) serialization @@ -26,6 +29,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `o` (in) : output adapter to write serialization to +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_cbor` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. CBOR serialization as a byte vector @@ -35,6 +44,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ Strong guarantee: if an exception is thrown, there are no changes in the JSON value. +## Exceptions + +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) + ## Complexity Linear in the size of the JSON value `j`. @@ -68,3 +83,6 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Compact representation of floating-point numbers added in version 3.8.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`. diff --git a/docs/mkdocs/docs/api/basic_json/to_msgpack.md b/docs/mkdocs/docs/api/basic_json/to_msgpack.md index 707de9514..17b9fe29c 100644 --- a/docs/mkdocs/docs/api/basic_json/to_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/to_msgpack.md @@ -2,11 +2,14 @@ ```cpp // (1) -static std::vector to_msgpack(const basic_json& j); +static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep); // (2) -static void to_msgpack(const basic_json& j, detail::output_adapter o); -static void to_msgpack(const basic_json& j, detail::output_adapter o); +static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); +static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the MessagePack serialization format. MessagePack is a binary @@ -25,6 +28,13 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `o` (in) : output adapter to write serialization to +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_msgpack` did before + this parameter was added and as the MessagePack specification allows; `strict` throws; `replace`/`ignore` sanitize + it the same way [`dump`](dump.md) would. Unlike the other binary writers, the default stays `keep` even if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled. + ## Return value 1. MessagePack serialization as a byte vector @@ -42,6 +52,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the MessagePack ext type; example: `"subtype 70000 is too large for the MessagePack ext type (max 255)"` +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -91,6 +103,8 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before. - Fixed in version 3.13.0 to serialize `number_integer_t`/`number_unsigned_t` pairs of different width correctly; before, integers could be serialized with the wrong value if `number_integer_t` was narrower than `number_unsigned_t`. diff --git a/docs/mkdocs/docs/api/basic_json/to_ubjson.md b/docs/mkdocs/docs/api/basic_json/to_ubjson.md index 1b7f7767e..6860c01c6 100644 --- a/docs/mkdocs/docs/api/basic_json/to_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/to_ubjson.md @@ -4,13 +4,16 @@ // (1) static std::vector to_ubjson(const basic_json& j, const bool use_size = false, - const bool use_type = false); + const bool use_type = false, + const error_handler_t error_handler = error_handler_t::keep); // (2) static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false); + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = error_handler_t::keep); static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false); + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the UBJSON (Universal Binary JSON) serialization format. UBJSON @@ -36,6 +39,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ : whether to add type annotations to container types (must be combined with `#!cpp use_size = true`); optional, `#!cpp false` by default. +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_ubjson` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. UBJSON serialization as a byte vector @@ -49,6 +58,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false, and `j` contains a non-empty array, object, or binary value. +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -97,3 +109,6 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.1.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`. diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index 3c9b42af2..872d6cb8c 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -18,6 +18,8 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_PRECISE_STREAM_POSITION**](json_precise_stream_position.md) - opt in to leaving an input stream positioned right after a parsed number +- [**JSON_STRICT_BINARY_UTF8**](json_strict_binary_utf8.md) - opt in to checking strings for valid UTF-8 in the CBOR, + UBJSON, BJData, and BSON writers - [**JSON_STRICT_NUL_HANDLING**](json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of treating it as end of input diff --git a/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md b/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md index 459818c87..71b8bb28f 100644 --- a/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md +++ b/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md @@ -115,3 +115,4 @@ The default value is `0` (disabled — existing behavior is preserved). ## Version history - Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md new file mode 100644 index 000000000..ed7bbe6b3 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md @@ -0,0 +1,101 @@ +# JSON_STRICT_BINARY_UTF8 + +```cpp +#define JSON_STRICT_BINARY_UTF8 /* value */ +``` + +When defined to `1`, the `error_handler` parameter of the binary writers [`to_cbor`](../basic_json/to_cbor.md), +[`to_ubjson`](../basic_json/to_ubjson.md), [`to_bjdata`](../basic_json/to_bjdata.md), and +[`to_bson`](../basic_json/to_bson.md) defaults to [`error_handler_t::strict`](../basic_json/error_handler_t.md) instead +of `error_handler_t::keep`. These writers then check every string value and object key for valid UTF-8 and throw +[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, like +[`dump`](../basic_json/dump.md) does. Without it, they write the bytes unchanged. An `error_handler` passed explicitly +always takes precedence. + +The macro does not affect: + +- [`to_msgpack`](../basic_json/to_msgpack.md): the MessagePack specification allows a `str` value to contain bytes that + are not valid UTF-8, so its `error_handler` always defaults to `keep`. +- [`to_bon8`](../basic_json/to_bon8.md): BON8 always checks, because the UTF-8 lead bytes mark where a string ends. +- The binary readers ([`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md), + [`from_ubjson`](../basic_json/from_ubjson.md), [`from_bjdata`](../basic_json/from_bjdata.md), + [`from_bson`](../basic_json/from_bson.md)): none of these formats requires a decoder to reject ill-formed UTF-8, so + they always return the bytes unchanged. + +## Default definition + +The default value is `0` (disabled, the behavior of version 3.12.0 and earlier is preserved). + +```cpp +#define JSON_STRICT_BINARY_UTF8 0 +``` + +## Notes + +!!! note "Background" + + CBOR, UBJSON, BJData, and BSON all require strings to be UTF-8. Up to version 3.12.0, the writers did not check + this, so they could produce output that other decoders reject. Checking by default would break code that stores + other encodings (for instance ISO 8859-1) in a string and only ever writes it to a binary format. You can pass + `error_handler_t::strict` to each call, or use this macro to check by default ahead of version 4.0.0, where + `strict` is planned to become the default (see + [#5529](https://github.com/nlohmann/json/issues/5529) and [#5651](https://github.com/nlohmann/json/issues/5651)). + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no + effect. + +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_sbu8`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + +## Examples + +??? example "Default behavior (macro not defined)" + + Without the macro, the bytes are written unchanged: + + ```cpp + #include + + using json = nlohmann::json; + + int main() + { + auto v = json::to_cbor(json("\xFF")); + // v is {0x61, 0xFF} + } + ``` + +??? example "Opt-in check (macro defined to 1)" + + With the macro, ill-formed UTF-8 is rejected: + + ```cpp + #define JSON_STRICT_BINARY_UTF8 1 + #include + + using json = nlohmann::json; + + int main() + { + auto v = json::to_cbor(json("\xFF")); + // throws type_error.316: invalid UTF-8 byte at index 0: 0xFF + } + ``` + +## See also + +- [**to_cbor**](../basic_json/to_cbor.md) - create a CBOR serialization of a JSON value +- [**to_ubjson**](../basic_json/to_ubjson.md) - create a UBJSON serialization of a JSON value +- [**to_bjdata**](../basic_json/to_bjdata.md) - create a BJData serialization of a JSON value +- [**to_bson**](../basic_json/to_bson.md) - create a BSON serialization of a JSON value +- [**error_handler_t**](../basic_json/error_handler_t.md) - how [`dump`](../basic_json/dump.md) treats ill-formed UTF-8 + +## Version history + +- Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/community/roadmap.md b/docs/mkdocs/docs/community/roadmap.md index e8c407d3f..48977b295 100644 --- a/docs/mkdocs/docs/community/roadmap.md +++ b/docs/mkdocs/docs/community/roadmap.md @@ -14,7 +14,7 @@ work items are tracked in the [GitHub milestones](https://github.com/nlohmann/js opt-in. - **Keep the 3.x public API stable.** Releases follow [semantic versioning](https://semver.org). Changes that would break existing code are only added behind a feature macro, so users can opt in and test their code before a next - major release. + major release, see [Version 4.0](#version-40). - **Support a broad range of compilers and platforms.** The [CI](quality_assurance.md) keeps testing old and new versions of GCC, Clang, MSVC, and other compilers on Linux, macOS, and Windows. - **Keep the quality assurance up.** Every change keeps the test coverage at 100%, passes the static and dynamic @@ -37,7 +37,67 @@ work items are tracked in the [GitHub milestones](https://github.com/nlohmann/js ## Version 4.0 -There is no decision yet on whether or when a version 4.0 with breaking changes will be released. Proposals that need -a major version, for instance stricter type conversions, are collected in issue -[#3453](https://github.com/nlohmann/json/issues/3453). Until then, such changes are only added as opt-in behavior -behind feature macros. +There is no release date for version 4.0 yet. Proposals that need a major version, for instance stricter type +conversions, are collected in issue [#3453](https://github.com/nlohmann/json/issues/3453). + +!!! note "Not final" + + The plan for version 4.0 described below is not final and may still change: macros may be added to or removed from + the list, and planned defaults may be revised. Any such change will be documented on this page. + +### Trying out 4.0 today + +Version 4.0 will not be developed on a separate branch. Instead, every breaking change is first added to a 3.x release +behind a macro whose default keeps the 3.x behavior. Version 4.0 then switches the defaults and removes the macros. +Version 4.0 is therefore the sum of these macros: you can try it on the 3.x release train today by defining each macro +to its 4.0 value and fixing what no longer compiles or behaves differently. Once your code works with all of them, it +is ready for version 4.0. + +The following macros guard changes that are planned to become the default in version 4.0: + +| Macro | 3.x default | 4.0 behavior | CMake option | Added | +|------------------------------------------------------------------------------------------------------------------|-------------|-------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------|--------| +| [`JSON_USE_IMPLICIT_CONVERSIONS`](../api/macros/json_use_implicit_conversions.md) | `1` | `0`: no implicit conversions from `basic_json` to other types; use [`get`](../api/basic_json/get.md) instead | [`JSON_ImplicitConversions`](../integration/cmake.md#json_implicitconversions) | 3.9.0 | +| [`JSON_USE_GLOBAL_UDLS`](../api/macros/json_use_global_udls.md) | `1` | `0`: the string literals `_json` and `_json_pointer` are only available in namespace `nlohmann::literals` | [`JSON_GlobalUDLs`](../integration/cmake.md#json_globaludls) | 3.11.0 | +| [`JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md) | `0` | removed: the deprecated legacy comparison of discarded values can no longer be enabled | [`JSON_LegacyDiscardedValueComparison`](../integration/cmake.md#json_legacydiscardedvaluecomparison) | 3.11.0 | +| [`JSON_BRACE_INIT_COPY_SEMANTICS`](../api/macros/json_brace_init_copy_semantics.md) | `0` | `1`: single-element brace initialization such as `#!cpp json j{obj};` copies the element instead of creating an array | – | 3.13.0 | +| [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) | `0` | `1`: reading from a stream does not consume the character after a number | – | 3.13.0 | +| [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) | `0` | `1`: a NUL byte in the input is a parse error instead of the end of input | [`JSON_StrictNulHandling`](../integration/cmake.md#json_strictnulhandling) | 3.13.0 | +| [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) | `0` | `1`: `to_cbor`, `to_ubjson`, `to_bjdata`, and `to_bson` throw for strings that are not valid UTF-8 by default | [`JSON_StrictBinaryUTF8`](../integration/cmake.md#json_strictbinaryutf8) | 3.13.0 | + +For example, the following makes a 3.x release behave like version 4.0 with respect to these changes: + +```cpp +#define JSON_USE_IMPLICIT_CONVERSIONS 0 +#define JSON_USE_GLOBAL_UDLS 0 +#define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 +#define JSON_BRACE_INIT_COPY_SEMANTICS 1 +#define JSON_PRECISE_STREAM_POSITION 1 +#define JSON_STRICT_NUL_HANDLING 1 +#define JSON_STRICT_BINARY_UTF8 1 +#include +``` + +The macros must be defined before the library header is included; setting them once in the build system is the easiest +way to achieve this. + +### Removal of deprecated functions + +Version 4.0 will remove all deprecated functions. Compiling with deprecation warnings enabled shows which of them your +code still uses. The [migration guide](../integration/migration_guide.md#replace-deprecated-functions) shows how to +replace each of them. + +| Deprecated | Since | Migration | +|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------|----------------------------------------------------------------------------------| +| `#!cpp operator<<(basic_json&, std::istream&)` | 3.0.0 | [Parsing](../integration/migration_guide.md#parsing) | +| `#!cpp operator>>(const basic_json&, std::ostream&)` | 3.0.0 | [Miscellaneous functions](../integration/migration_guide.md#miscellaneous-functions) | +| `iterator_wrapper` | 3.1.0 | [Miscellaneous functions](../integration/migration_guide.md#miscellaneous-functions) | +| [`parse`](../api/basic_json/parse.md), [`accept`](../api/basic_json/accept.md), and [`sax_parse`](../api/basic_json/sax_parse.md) with an initializer list `{ptr, len}` or `{first, last}` | 3.8.0 | [Parsing](../integration/migration_guide.md#parsing) | +| [`from_bson`](../api/basic_json/from_bson.md), [`from_cbor`](../api/basic_json/from_cbor.md), [`from_msgpack`](../api/basic_json/from_msgpack.md), and [`from_ubjson`](../api/basic_json/from_ubjson.md) with `(ptr, len)` or an initializer list | 3.8.0 | [Parsing](../integration/migration_guide.md#parsing) | +| [`json_pointer::operator string_t`](../api/json_pointer/operator_string_t.md) | 3.11.0 | [JSON Pointers](../integration/migration_guide.md#json-pointers) | +| [`json_pointer`](../api/json_pointer/index.md) with a `basic_json` type as template argument, and the overloads of `value`, `contains`, `operator[]`, and `at` accepting such a pointer | 3.11.0 | [JSON Pointers](../integration/migration_guide.md#json-pointers) | +| Comparing a [`json_pointer`](../api/json_pointer/index.md) with a string via [`operator==`](../api/json_pointer/operator_eq.md) or [`operator!=`](../api/json_pointer/operator_ne.md) | 3.11.2 | [JSON Pointers](../integration/migration_guide.md#json-pointers) | + +The deprecated legacy comparison of discarded values is controlled by a macro and therefore listed in the table above. + +New breaking changes will follow the same path: they are added to these tables when they land in a 3.x release. diff --git a/docs/mkdocs/docs/examples/as_base_class.cpp b/docs/mkdocs/docs/examples/as_base_class.cpp new file mode 100644 index 000000000..48357b045 --- /dev/null +++ b/docs/mkdocs/docs/examples/as_base_class.cpp @@ -0,0 +1,41 @@ +#include +#include + +class base_class_with_hidden_members +{ + public: + const char* type_name() const noexcept + { + return "my_type_name"; + } + + std::size_t size() const noexcept + { + return 42; + } +}; + +using json = nlohmann::basic_json < + std::map, + std::vector, + std::string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer, + std::vector, + base_class_with_hidden_members + >; + +int main() +{ + json j = {1, 2, 3}; + + // the members of basic_json hide the members of the base class + std::cout << j.type_name() << ' ' << j.size() << '\n'; + + // access the hidden members of the base class + std::cout << j.as_base_class().type_name() << ' ' << j.as_base_class().size() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/as_base_class.output b/docs/mkdocs/docs/examples/as_base_class.output new file mode 100644 index 000000000..5ca62a673 --- /dev/null +++ b/docs/mkdocs/docs/examples/as_base_class.output @@ -0,0 +1,2 @@ +array 3 +my_type_name 42 diff --git a/docs/mkdocs/docs/examples/error_handler_t.cpp b/docs/mkdocs/docs/examples/error_handler_t.cpp index b4718d7e6..bf035cea3 100644 --- a/docs/mkdocs/docs/examples/error_handler_t.cpp +++ b/docs/mkdocs/docs/examples/error_handler_t.cpp @@ -20,5 +20,6 @@ int main() << j_invalid.dump(-1, ' ', false, json::error_handler_t::replace) << "\nstring with ignored invalid characters: " << j_invalid.dump(-1, ' ', false, json::error_handler_t::ignore) - << '\n'; + << "\nstring with the invalid byte kept as is (" << j_invalid.dump(-1, ' ', false, json::error_handler_t::keep).size() + << " bytes, not valid UTF-8 itself)\n"; } diff --git a/docs/mkdocs/docs/examples/error_handler_t.output b/docs/mkdocs/docs/examples/error_handler_t.output index 718d62bee..37cae62a2 100644 --- a/docs/mkdocs/docs/examples/error_handler_t.output +++ b/docs/mkdocs/docs/examples/error_handler_t.output @@ -1,3 +1,4 @@ [json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9 string with replaced invalid characters: "ä�ü" string with ignored invalid characters: "äü" +string with the invalid byte kept as is (7 bytes, not valid UTF-8 itself) diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index 8e3ed41f6..3e3d8e885 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -63,6 +63,15 @@ The library uses the following mapping from JSON values types to BJData types ac - strings with more than 18446744073709551615 bytes, i.e., 264-1 bytes (theoretical) +!!! warning "UTF-8 validation of string values and object keys" + + BJData strings must use UTF-8 encoding. By default (the [`error_handler`](../../api/basic_json/to_bjdata.md) + parameter left at `keep`), `to_bjdata()` writes the bytes of string values and object keys unchanged, even if they + are not valid UTF-8. With `error_handler_t::strict`, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead; + `replace`/`ignore` sanitize the string. [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) + makes `strict` the default. + !!! info "Unused BJData markers" The following markers are not used in the conversion: @@ -208,6 +217,19 @@ The library maps BJData types to JSON value types as follows: The mapping is **complete** in the sense that any BJData value can be converted to a JSON value. +!!! warning "Ill-formed UTF-8 in string values and object keys" + + BJData strings must use UTF-8 encoding, but checking it on read is opt-in: with the + [`error_handler`](../../api/basic_json/from_bjdata.md) parameter left at `keep` (the default), `from_bjdata()` + accepts a string value or object key whose bytes are not valid UTF-8 and hands them back unchanged. Passing + `error_handler_t::strict` makes `from_bjdata()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read with the default + `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. `to_bjdata()`'s + own `error_handler` parameter defaults to `keep` (see above), so such a value is written back unchanged. + !!! info "Round trips" A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with diff --git a/docs/mkdocs/docs/features/binary_formats/bson.md b/docs/mkdocs/docs/features/binary_formats/bson.md index cca11451e..f205cba17 100644 --- a/docs/mkdocs/docs/features/binary_formats/bson.md +++ b/docs/mkdocs/docs/features/binary_formats/bson.md @@ -109,14 +109,21 @@ The library maps BSON record types to JSON value types as follows: If BSON input must be validated for strict specification compliance, validate it separately before passing it to `from_bson()`. -!!! warning "UTF-8 validation of string values" +!!! warning "Ill-formed UTF-8 in string values" - The BSON specification requires `string` values (type `0x02`) to be valid UTF-8. This library validates the - bytes of every such string at decode time and rejects ill-formed UTF-8 with a - [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions` - set to `false`, a discarded value), rather than only failing later when the resulting value is dumped. Element - (key) names and `binary` values (type `0x05`) are unaffected and are never validated, since they are read - byte-by-byte as a C string, or are not required to hold text, respectively. + The BSON specification requires `string` values (type `0x02`) to be valid UTF-8, but this is not required of a + decoder, so checking is opt-in: with the [`error_handler`](../../api/basic_json/from_bson.md) parameter left at + `keep` (the default), `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back + unchanged. Passing `error_handler_t::strict` makes `from_bson()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md) + still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a + value read with the default `keep` handler, unless an error handler is passed that replaces or ignores the + ill-formed bytes. `to_bson()`'s own `error_handler` parameter defaults to `keep`, so such a string value or element + (key) name is written unchanged; with `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled), it throws the same exception + instead. Element (key) names are never validated on read, since they are read byte-by-byte as a C string. `binary` + values (type `0x05`) are unaffected, since they are not required to hold text. ??? example "Example: deserialize a JSON value from BSON" diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index eb7bc7f41..7b5be4631 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -189,15 +189,21 @@ The library maps CBOR types to JSON value types as follows: ([RFC 8392](https://www.rfc-editor.org/rfc/rfc8392.html)), cannot be read with this library and need a general-purpose CBOR library instead. -!!! warning "UTF-8 validation of text strings" +!!! warning "Ill-formed UTF-8 in text strings" - [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings - (major type 3) to be valid UTF-8. This library validates the bytes of every text string (object keys included) at - decode time and rejects ill-formed UTF-8 with a - [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with - `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value is - dumped. Byte strings (major type 2) are unaffected and are never validated, since they are not required to hold - text. + [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings (major + type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this, so checking is opt-in: with the + [`error_handler`](../../api/basic_json/from_cbor.md) parameter left at `keep` (the default), `from_cbor()` accepts a + text string (object keys included) whose bytes are not valid UTF-8 and hands them back unchanged. Passing + `error_handler_t::strict` makes `from_cbor()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md) + still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a + value read with the default `keep` handler, unless an error handler is passed that replaces or ignores the + ill-formed bytes. `to_cbor()`'s own [`error_handler`](../../api/basic_json/to_cbor.md) parameter defaults to `keep`, + so such a value is written back unchanged; with `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled), it throws the same exception + instead. Byte strings (major type 2) are unaffected, since they are not required to hold text. !!! warning "Tagged items" diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index 2e674252a..24eeab139 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -153,14 +153,23 @@ The library maps MessagePack types to JSON value types as follows: This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed on. Such input needs a general-purpose MessagePack library instead. -!!! warning "UTF-8 validation of string values" +!!! warning "Ill-formed UTF-8 in string values" - The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8. - This library validates the bytes of every such string (object keys included) at decode time and rejects - ill-formed UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, - with `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting - value is dumped. `bin`/`ext`/`fixext` values are unaffected and are never validated, since they are not required - to hold text. + The MessagePack specification explicitly allows a `str` value (`fixstr`, `str 8`, `str 16`, `str 32`) to contain + a byte sequence that is not valid UTF-8, and expects a deserializer to hand the original bytes back unchanged. + This library follows that by default: with its + [`error_handler`](../../api/basic_json/from_msgpack.md) parameter left at `keep` (the default), + `from_msgpack()` reads `str` bytes (object keys included) as-is, without validating them, so such a value + round-trips through `from_msgpack(to_msgpack(j))` byte for byte. Passing `error_handler_t::strict` makes + `from_msgpack()` check anyway and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. `to_msgpack()` also writes `str` bytes as-is by + default, since the specification permits it; its [`error_handler`](../../api/basic_json/to_msgpack.md) parameter + can be set to `strict` to throw [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) instead, or + to `replace`/`ignore` to sanitize the string, for instance for a decoder that rejects ill-formed UTF-8. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way with the + default `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. ??? example "Example: deserialize a JSON value from MessagePack" diff --git a/docs/mkdocs/docs/features/binary_formats/ubjson.md b/docs/mkdocs/docs/features/binary_formats/ubjson.md index 37aa069e2..f7a14d855 100644 --- a/docs/mkdocs/docs/features/binary_formats/ubjson.md +++ b/docs/mkdocs/docs/features/binary_formats/ubjson.md @@ -47,6 +47,15 @@ The library uses the following mapping from JSON values types to UBJSON types ac - strings with more than 9223372036854775807 bytes (theoretical) +!!! warning "UTF-8 validation of string values and object keys" + + UBJSON's required string encoding is UTF-8. By default (the [`error_handler`](../../api/basic_json/to_ubjson.md) + parameter left at `keep`), `to_ubjson()` writes the bytes of string values and object keys unchanged, even if they + are not valid UTF-8. With `error_handler_t::strict`, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead; + `replace`/`ignore` sanitize the string. [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) + makes `strict` the default. + !!! info "Unused UBJSON markers" The following markers are not used in the conversion: @@ -120,6 +129,19 @@ The library maps UBJSON types to JSON value types as follows: The mapping is **complete** in the sense that any UBJSON value can be converted to a JSON value. +!!! warning "Ill-formed UTF-8 in string values and object keys" + + UBJSON's required string encoding is UTF-8, but checking it on read is opt-in: with the + [`error_handler`](../../api/basic_json/from_ubjson.md) parameter left at `keep` (the default), `from_ubjson()` + accepts a string value or object key whose bytes are not valid UTF-8 and hands them back unchanged. Passing + `error_handler_t::strict` makes `from_ubjson()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read with the default + `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. `to_ubjson()`'s + own `error_handler` parameter defaults to `keep` (see above), so such a value is written back unchanged. + ??? example "Example: deserialize a JSON value from UBJSON" ```cpp diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index 90bb352c3..681980315 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -138,6 +138,20 @@ using the library with compilers that do not fully support C++11 and may only wo See [full documentation of `JSON_SKIP_UNSUPPORTED_COMPILER_CHECK`](../api/macros/json_skip_unsupported_compiler_check.md). +## `JSON_STRICT_BINARY_UTF8` + +When defined to `1`, [`to_cbor`](../api/basic_json/to_cbor.md), [`to_ubjson`](../api/basic_json/to_ubjson.md), +[`to_bjdata`](../api/basic_json/to_bjdata.md), and [`to_bson`](../api/basic_json/to_bson.md) throw +[`type_error.316`](../home/exceptions.md#jsonexceptiontype_error316) for a string value or object key that is not +valid UTF-8. The default value is `0`, which writes the bytes unchanged as before version 3.13.0; this is planned to +become the default in version 4.0.0. + +The check can also be enabled with the CMake option +[`JSON_StrictBinaryUTF8`](../integration/cmake.md#json_strictbinaryutf8) (`OFF` by default) which sets +`JSON_STRICT_BINARY_UTF8` accordingly. + +See [full documentation of `JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md). + ## `JSON_STRICT_NUL_HANDLING` When defined to `1`, a `'\0'` (NUL) byte anywhere in the input is rejected with `parse_error.101`, like any other diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 577f5e221..dbddac13d 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -20,6 +20,7 @@ The complete default namespace name is derived as follows: `_bics`. - [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) defined non-zero appends `_psp`. - [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) defined non-zero appends `_snul`. + - [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) defined non-zero appends `_sbu8`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/docs/mkdocs/docs/features/parsing/error_recovery.md b/docs/mkdocs/docs/features/parsing/error_recovery.md index de41206ee..7dc09e9ed 100644 --- a/docs/mkdocs/docs/features/parsing/error_recovery.md +++ b/docs/mkdocs/docs/features/parsing/error_recovery.md @@ -71,24 +71,25 @@ on whether the end of the item with the error is known, a distinction that If the item is complete, but cannot be passed on as it is, it is replaced, and parsing continues after it: -| Mistake | Formats | Repair | -|---------------------------------------------------------------------|-----------------------------------------|-------------------------------------------------------------------------| -| tag | CBOR | ignored | -| simple value other than `false`, `true`, and `null`, like undefined | CBOR | `#!json null` | -| negative integer below the range of `number_integer_t` | CBOR | the nearest floating-point number | -| string that is not valid UTF-8 | BJData, BSON, CBOR, MessagePack, UBJSON | each ill-formed sequence becomes U+FFFD | -| character (`C`) that is not ASCII | BJData, UBJSON | U+FFFD | -| invalid high-precision number (`H`) | BJData, UBJSON | the longest valid beginning is kept, as for JSON text, or `#!json null` | -| high-precision number too large | BJData, UBJSON | passed as infinity, together with its text | -| object key that is not a string | BON8, CBOR, MessagePack | the member is skipped | -| element of a type the library does not read, like ObjectId or date | BSON | `#!json null` | -| string without its terminator | BSON | kept | -| document whose size does not match its content | BSON | kept | +| Mistake | Formats | Repair | +|---------------------------------------------------------------------|-------------------------|-------------------------------------------------------------------------| +| tag | CBOR | ignored | +| simple value other than `false`, `true`, and `null`, like undefined | CBOR | `#!json null` | +| negative integer below the range of `number_integer_t` | CBOR | the nearest floating-point number | +| character (`C`) that is not ASCII | BJData, UBJSON | U+FFFD | +| invalid high-precision number (`H`) | BJData, UBJSON | the longest valid beginning is kept, as for JSON text, or `#!json null` | +| high-precision number too large | BJData, UBJSON | passed as infinity, together with its text | +| object key that is not a string | BON8, CBOR, MessagePack | the member is skipped | +| element of a type the library does not read, like ObjectId or date | BSON | `#!json null` | +| string without its terminator | BSON | kept | +| document whose size does not match its content | BSON | kept | CBOR tags and simple values are repaired as [RFC 8949, Section 6.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-6.1) suggests for converting CBOR to JSON. Note that [`sax_parse`](../../api/basic_json/sax_parse.md) has no parameter for CBOR tags, so every tag is an error there; when recovering, tags are ignored like with -[`cbor_tag_handler_t::ignore`](../../api/basic_json/cbor_tag_handler_t.md). +[`cbor_tag_handler_t::ignore`](../../api/basic_json/cbor_tag_handler_t.md). Strings that are not valid UTF-8 are no +error: like [`from_cbor`](../../api/basic_json/from_cbor.md) and the other functions by default, `sax_parse` passes +them on as they are. After any other error, the end of the item is unknown: the input ended, a byte is not a valid type marker, or a size cannot be right. Parsing then stops, and the value read so far is completed: a key that waits for its value gets diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index c78aeaa68..e7eb8fd03 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -341,7 +341,10 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a string was read where one was required (for instance as a map key), the string's length specification is invalid, or -the string's bytes are not valid UTF-8. +the string's bytes are not valid UTF-8 and the `error_handler` parameter of the corresponding `from_*` function is +set to `strict`. By default (`error_handler_t::keep`), the bytes of a string are not checked for valid UTF-8 on read; +see the ill-formed UTF-8 notes on the individual [binary format](../features/binary_formats/index.md) pages for how +such a string is handled depending on `error_handler`. CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other type (for instance integers or `null`) are therefore not supported; see the notes on @@ -749,6 +752,12 @@ The [`unflatten()`](../api/basic_json/unflatten.md) function only works for an o The [`dump()`](../api/basic_json/dump.md) function only works with UTF-8 encoded strings; that is, if you assign a `std::string` to a JSON value, make sure it is UTF-8 encoded. See the FAQ entry on [serializing untrusted or invalid UTF-8](faq.md#serializing-untrusted-or-invalid-utf-8) for background and the recommended fix. +The binary writers [`to_cbor()`](../api/basic_json/to_cbor.md), [`to_ubjson()`](../api/basic_json/to_ubjson.md), +[`to_bjdata()`](../api/basic_json/to_bjdata.md), and [`to_bson()`](../api/basic_json/to_bson.md) throw this exception +as well for a string value or object key that is not valid UTF-8 if their `error_handler` is `strict` (the default if +[`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled). So does +[`to_msgpack()`](../api/basic_json/to_msgpack.md) if `error_handler_t::strict` is passed. + !!! failure "Example message" Calling `dump()` on a JSON value containing an ISO 8859-1 encoded string: diff --git a/docs/mkdocs/docs/integration/cmake.md b/docs/mkdocs/docs/integration/cmake.md index 785fc3ed6..9151d12be 100644 --- a/docs/mkdocs/docs/integration/cmake.md +++ b/docs/mkdocs/docs/integration/cmake.md @@ -212,6 +212,11 @@ Use the non-amalgamated version of the library. This option is `ON` by default. Treat the library headers like system headers (i.e., adding `SYSTEM` to the [`target_include_directories`](https://cmake.org/cmake/help/latest/command/target_include_directories.html) call) to check for this library by tools like Clang-Tidy. This option is `OFF` by default. +### `JSON_StrictBinaryUTF8` + +Check string values and object keys for valid UTF-8 in the CBOR, UBJSON, BJData, and BSON writers, by defining the +macro [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md). This option is `OFF` by default. + ### `JSON_StrictNulHandling` Reject a `'\0'` (NUL) byte in the input instead of treating it as end of input, by defining the macro diff --git a/docs/mkdocs/docs/integration/migration_guide.md b/docs/mkdocs/docs/integration/migration_guide.md index 29d54b6df..8a07d48a3 100644 --- a/docs/mkdocs/docs/integration/migration_guide.md +++ b/docs/mkdocs/docs/integration/migration_guide.md @@ -2,11 +2,14 @@ This page collects some guidelines on how to future-proof your code for future versions of this library. For how to add the library to your project in the first place, see [Integration](index.md), [CMake](cmake.md), or -[Package Managers](package_managers.md). +[Package Managers](package_managers.md). The [roadmap](../community/roadmap.md#version-40) lists what will change in +version 4.0, including the macros that let you try its behavior with a 3.x release; this page describes how to adjust +your code. ## Replace deprecated functions -The following functions have been deprecated and will be removed in the next major version (i.e., 4.0.0). All +The following functions have been deprecated and will be removed in the next major version (i.e., 4.0.0), see the +[roadmap](../community/roadmap.md#removal-of-deprecated-functions) for an overview. All deprecations are annotated with [`HEDLEY_DEPRECATED_FOR`](https://nemequ.github.io/hedley/api-reference.html#HEDLEY_DEPRECATED_FOR) to report which function to use instead. diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 7f043aea8..1c6e6866b 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -117,6 +117,7 @@ nav: - 'accept': api/basic_json/accept.md - 'array': api/basic_json/array.md - 'array_t': api/basic_json/array_t.md + - 'as_base_class': api/basic_json/as_base_class.md - 'at': api/basic_json/at.md - 'back': api/basic_json/back.md - 'begin': api/basic_json/begin.md @@ -307,6 +308,7 @@ nav: - 'JSON_PRECISE_STREAM_POSITION': api/macros/json_precise_stream_position.md - 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md - 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md + - 'JSON_STRICT_BINARY_UTF8': api/macros/json_strict_binary_utf8.md - 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md - 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md - 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index 0bace616a..0153c8706 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -46,6 +46,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -82,14 +86,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -98,7 +108,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/conversions/from_json.hpp b/include/nlohmann/detail/conversions/from_json.hpp index d5c3328ca..5274b9857 100644 --- a/include/nlohmann/detail/conversions/from_json.hpp +++ b/include/nlohmann/detail/conversions/from_json.hpp @@ -27,7 +27,6 @@ #include #include #include -#include #include #include @@ -211,62 +210,29 @@ inline void from_json(const BasicJsonType& j, std::valarray& l) }); } +// element is not itself a C array: read it directly +template +auto from_json_c_array_element(const BasicJsonType& j, T& e) +-> decltype(e = j.template get(), void()) +{ + e = j.template get(); +} + +// element is itself a C array: recurse one dimension at a time, so any rank is supported template -auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) +void from_json_c_array_element(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { for (std::size_t i = 0; i < N; ++i) { - arr[i] = j.at(i).template get(); + from_json_c_array_element(j.at(i), arr[i]); } } -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) +template +auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) +-> decltype(j.template get::type>(), void()) { - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - arr[i1][i2] = j.at(i1).at(i2).template get(); - } - } -} - -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) -{ - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - for (std::size_t i3 = 0; i3 < N3; ++i3) - { - arr[i1][i2][i3] = j.at(i1).at(i2).at(i3).template get(); - } - } - } -} - -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3][N4]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) -{ - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - for (std::size_t i3 = 0; i3 < N3; ++i3) - { - for (std::size_t i4 = 0; i4 < N4; ++i4) - { - arr[i1][i2][i3][i4] = j.at(i1).at(i2).at(i3).at(i4).template get(); - } - } - } - } + from_json_c_array_element(j, arr); } template @@ -286,20 +252,33 @@ auto from_json_array_impl(const BasicJsonType& j, std::array& arr, } } +// reserve() is called through this pair (modeled on from_json_object_reserve) +// so from_json_array_impl below has a single body for both ConstructibleArrayType +// that support reserve() and those that don't. +template +auto from_json_array_reserve(ConstructibleArrayType& arr, typename ConstructibleArrayType::size_type size, priority_tag<1> /*unused*/) +-> decltype(arr.reserve(size), void()) +{ + arr.reserve(size); +} + +template +void from_json_array_reserve(ConstructibleArrayType& /*arr*/, std::size_t /*size*/, priority_tag<0> /*unused*/) +{} + template::value, int> = 0> auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, priority_tag<1> /*unused*/) -> decltype( - arr.reserve(std::declval()), j.template get(), void()) { using std::end; ConstructibleArrayType ret; - ret.reserve(j.size()); + from_json_array_reserve(ret, j.size(), priority_tag<1> {}); std::transform(j.begin(), j.end(), std::inserter(ret, end(ret)), [](const BasicJsonType & i) { @@ -310,27 +289,6 @@ auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, p arr = std::move(ret); } -template::value, - int> = 0> -inline void from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, - priority_tag<0> /*unused*/) -{ - using std::end; - - ConstructibleArrayType ret; - std::transform( - j.begin(), j.end(), std::inserter(ret, end(ret)), - [](const BasicJsonType & i) - { - // get() returns *this, this won't call a from_json - // method when value_type is BasicJsonType - return i.template get(); - }); - arr = std::move(ret); -} - template < typename BasicJsonType, typename ConstructibleArrayType, enable_if_t < is_constructible_array_type::value&& @@ -433,9 +391,7 @@ inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj) } // overload for arithmetic types, not chosen for basic_json template arguments -// (BooleanType, etc.); note: Is it really necessary to provide explicit -// overloads for boolean_t etc. in case of a custom BooleanType which is not -// an arithmetic type? +// (BooleanType, etc.) template < typename BasicJsonType, typename ArithmeticType, enable_if_t < std::is_arithmetic::value&& @@ -531,7 +487,7 @@ inline void from_json_tuple_impl(const BasicJsonType& j, std::pair& p, p template std::tuple from_json_tuple_impl(const BasicJsonType& j, identity_tag> /*unused*/, priority_tag<2> /*unused*/) { - static_assert(cxpr_and>, is_compatible_reference_type>...>::value, + static_assert(conjunction>, is_compatible_reference_type>...>::value, "Can not return a tuple containing references to types not contained in a Json, try Json::get_to()"); return from_json_tuple_impl_base<1, Args...>(j, index_sequence_for {}); } @@ -554,10 +510,10 @@ auto from_json(const BasicJsonType& j, TupleRelated&& t) return from_json_tuple_impl(j, std::forward(t), priority_tag<3> {}); } -template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator, - typename = enable_if_t < !std::is_constructible < - typename BasicJsonType::string_t, Key >::value >> -inline void from_json(const BasicJsonType& j, std::map& m) +// shared body for std::map/std::unordered_map with a non-string Key: both +// containers are read from an array of [key, value] pairs the same way +template +void from_json_pair_array_to_map(const BasicJsonType& j, MapType& m) { if (JSON_HEDLEY_UNLIKELY(!j.is_array())) { @@ -570,33 +526,29 @@ inline void from_json(const BasicJsonType& j, std::map(), p.at(1).template get()); + m.emplace(p.at(0).template get(), p.at(1).template get()); } } +template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator, + typename = enable_if_t < !std::is_constructible < + typename BasicJsonType::string_t, Key >::value >> +void from_json(const BasicJsonType& j, std::map& m) +{ + from_json_pair_array_to_map(j, m); +} + template < typename BasicJsonType, typename Key, typename Value, typename Hash, typename KeyEqual, typename Allocator, typename = enable_if_t < !std::is_constructible < typename BasicJsonType::string_t, Key >::value >> -inline void from_json(const BasicJsonType& j, std::unordered_map& m) +void from_json(const BasicJsonType& j, std::unordered_map& m) { - if (JSON_HEDLEY_UNLIKELY(!j.is_array())) - { - JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); - } - m.clear(); - for (const auto& p : j) - { - if (JSON_HEDLEY_UNLIKELY(!p.is_array())) - { - JSON_THROW(type_error::create(302, concat("type must be array, but is ", p.type_name()), &p)); - } - m.emplace(p.at(0).template get(), p.at(1).template get()); - } + from_json_pair_array_to_map(j, m); } #if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM -// Workaround for MSVC 19.51 (and possibly later): in large in large cpp files, the compiler may fail to resolve with generic has_from_json (issue #4996) +// Workaround for MSVC 19.51 (and possibly later): in large cpp files, the compiler may fail to resolve with generic has_from_json (issue #4996) template struct has_from_json : std::true_type {}; diff --git a/include/nlohmann/detail/conversions/to_json.hpp b/include/nlohmann/detail/conversions/to_json.hpp index 7b97068f3..2dba6c163 100644 --- a/include/nlohmann/detail/conversions/to_json.hpp +++ b/include/nlohmann/detail/conversions/to_json.hpp @@ -178,7 +178,7 @@ struct external_constructor template < typename BasicJsonType, typename CompatibleArrayType, enable_if_t < !std::is_same::value -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif , int > = 0 > @@ -222,9 +222,7 @@ struct external_constructor j.assert_invariant(); } - // std::ranges does not work properly on MinGW due to incomplete C++20 support - // see https://github.com/nlohmann/json/issues/4916 -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template>::value, int> = 0> static void construct(BasicJsonType& j, CompatibleArrayType && arr) @@ -384,7 +382,7 @@ template < typename BasicJsonType, typename CompatibleArrayType, !std::is_same::value&& !is_compatible_binary_type::value&& !is_basic_json::value -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif , @@ -394,7 +392,7 @@ inline void to_json(BasicJsonType& j, const CompatibleArrayType& arr) external_constructor::construct(j, arr); } -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template < typename BasicJsonType, typename T, enable_if_t < is_compatible_range_view>::value && !is_compatible_string_type>::value diff --git a/include/nlohmann/detail/exceptions.hpp b/include/nlohmann/detail/exceptions.hpp index 808a3b6bf..155f4a4e5 100644 --- a/include/nlohmann/detail/exceptions.hpp +++ b/include/nlohmann/detail/exceptions.hpp @@ -286,6 +286,27 @@ class other_error : public exception other_error(int id_, const char* what_arg) : exception(id_, what_arg) {} }; +/*! +@brief helper function to call JSON_THROW from a template +@note JSON_THROW is a macro that, depending on the JSON_THROW_USER / + JSON_TRY_USER / JSON_NOEXCEPTION configuration, may expand to code + that does not reference its argument (e.g. `std::abort()`), which + would trigger a compilation error if the argument's type depends on + a template parameter that is otherwise unused. Wrapping the call in + a templated function avoids this and gives the compiler a single + place to see the (possibly unused) parameter. +*/ +template +void templated_json_throw(ExceptionType exception) +{ + JSON_THROW(exception); + + // JSON_THROW may expand to code that discards its argument (e.g. when + // exceptions are disabled) - the cast below avoids an unused-parameter + // warning with -Werror in that case + (void)exception; +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 294cf83ff..14c26df86 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -31,6 +31,7 @@ #include #include #include +#include #include #include #include @@ -130,8 +131,16 @@ class binary_reader @brief create a binary reader @param[in] adapter input adapter to read from + @param[in] format the binary format to parse + @param[in] error_handler how to treat text strings and object keys that + are not well-formed UTF-8; none of the supported formats + requires a decoder to reject those, so the default is to + @ref error_handler_t::keep them unchanged, as every binary + reader did before this parameter existed */ - explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json) noexcept : ia(std::move(adapter)), input_format(format) + explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json, + const error_handler_t error_handler = error_handler_t::keep) noexcept + : ia(std::move(adapter)), input_format(format), error_handler(error_handler) { (void)detail::is_sax_static_asserts {}; } @@ -594,7 +603,7 @@ class binary_reader { if (get_bson_cstr_bulk(result, std::integral_constant {})) { - return true; + return check_string_utf8(result, "key"); } auto out = std::back_inserter(result); @@ -607,7 +616,7 @@ class binary_reader } if (current == 0x00) { - return true; + return check_string_utf8(result, "key"); } *out++ = static_cast(current); } @@ -693,7 +702,7 @@ class binary_reader return current != char_traits::eof() && repair_requested(); } - return true; + return check_string_utf8(result, "string"); } /*! @@ -1353,7 +1362,7 @@ class binary_reader @return whether string creation completed */ - bool get_cbor_string(string_t& result) + bool get_cbor_string(string_t& result, const char* context = "string") { // number of indefinite-length strings that have been opened and not // closed yet. RFC 8949, Section 3.2.3 does not permit nesting them, @@ -1383,7 +1392,7 @@ class binary_reader { if (--open == 0) { - return true; + return check_string_utf8(result, context); } get(); continue; @@ -1396,7 +1405,7 @@ class binary_reader if (open == 0) { - return true; + return check_string_utf8(result, context); } get(); @@ -1421,7 +1430,7 @@ class binary_reader // EOF and major type 3 (text string) are left to get_cbor_string if (current == char_traits::eof() || (static_cast(current) & 0xE0u) == 0x60u) { - return get_cbor_string(result); + return get_cbor_string(result, "key"); } const char* found = nullptr; @@ -2334,7 +2343,7 @@ class binary_reader @return whether string creation completed */ - bool get_msgpack_string(string_t& result) + bool get_msgpack_string(string_t& result, const char* context = "string") { if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "string"))) { @@ -2377,25 +2386,25 @@ class binary_reader case 0xBE: case 0xBF: { - return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result); + return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result) && check_string_utf8(result, context); } case 0xD9: // str 8 { std::uint8_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDA: // str 16 { std::uint16_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDB: // str 32 { std::uint32_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } default: @@ -2474,7 +2483,7 @@ class binary_reader // byte 0xC1 are left to get_msgpack_string if (current == char_traits::eof()) { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } if (current <= 0x7F || current >= 0xE0) { @@ -2490,7 +2499,7 @@ class binary_reader } else { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } break; } @@ -2887,7 +2896,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, true, "key") || !sax->key(key))) { return false; } @@ -2909,7 +2918,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false, "key") || !sax->key(key))) { return false; } @@ -2977,7 +2986,7 @@ class binary_reader @return whether string creation completed */ - bool get_ubjson_string(string_t& result, const bool get_char = true) + bool get_ubjson_string(string_t& result, const bool get_char = true, const char* context = "string") { if (get_char) { @@ -2998,31 +3007,31 @@ class binary_reader case 'U': { std::uint8_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'i': { std::int8_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'I': { std::int16_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'l': { std::int32_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'L': { std::int64_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'u': @@ -3032,7 +3041,7 @@ class binary_reader break; } std::uint16_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'm': @@ -3042,7 +3051,7 @@ class binary_reader break; } std::uint32_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'M': @@ -3052,7 +3061,7 @@ class binary_reader break; } std::uint64_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } default: @@ -3557,10 +3566,9 @@ class binary_reader { return false; } - // when recovering, the character becomes U+FFFD, as an - // invalid byte in a string does - string_t replacement; - append_replacement_character(replacement); + // when recovering, the character becomes U+FFFD, as with + // error_handler_t::replace + string_t replacement = sanitize_utf8(string_t(1, static_cast(current)), error_handler_t::replace); return sax->string(replacement); } string_t s(1, static_cast(current)); @@ -4731,34 +4739,58 @@ class binary_reader const NumberType len, string_t& result) { - // get_bytes() appends to result, and CBOR indefinite-length strings - // collect all their chunks in the same result; validating only the - // newly read bytes keeps the check linear in the input size - const std::size_t old_size = result.size(); - if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) + // Strings are taken as is by default: none of CBOR (RFC 8949 §3.1 + // leaves the choice to the decoder), MessagePack (whose spec + // explicitly allows a str object to contain an invalid byte + // sequence), UBJSON, BJData, or BSON requires a decoder to reject + // ill-formed UTF-8. Checking (and, with @ref error_handler_t::strict, + // rejecting, or with `replace`/`ignore`, sanitizing) is opt-in via + // @ref error_handler, applied once the whole string (all chunks of + // an indefinite-length CBOR string included) has been assembled, by + // @ref check_string_utf8 at the call site. + return get_bytes(format, len, "string", result); + } + + /*! + @brief validate a decoded text string (value or object key) against @ref error_handler + + None of the binary formats requires a decoder to reject ill-formed UTF-8 + in a text string (see @ref get_string), so by default + (@ref error_handler_t::keep) this does nothing. A stricter + @ref error_handler opts into the same well-formedness check @ref + serializer::dump_escaped_impl applies when dumping a string: + @ref error_handler_t::strict rejects ill-formed input with + parse_error.113 (honoring `allow_exceptions` via @a sax), while + @ref error_handler_t::replace / @ref error_handler_t::ignore sanitize + @a result in place, using the exact same rules. + + @param[in,out] result the already assembled string to check + @param[in] context further context information (for diagnostics) + @return whether @a result is acceptable (always true for `keep`) + */ + bool check_string_utf8(string_t& result, const char* context) + { + if (error_handler == error_handler_t::keep || is_valid_utf8(result)) { - return false; + return true; } - // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications - // all require text strings to be valid UTF-8; reject anything else - // right here so malformed input is caught at decode time instead of - // only surfacing later as a type_error.316 when the value is dumped - // (which would defeat allow_exceptions=false / strict discarding). - if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) + if (error_handler == error_handler_t::strict) { - static_cast(report_error(chars_read, get_token_string(), - parse_error::create(113, chars_read, - exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr))); + auto last_token = get_token_string(); + static_cast(report_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format, "invalid string: ill-formed UTF-8 byte", context), nullptr))); + // when recovering, the string is repaired as with + // error_handler_t::replace if (!repair_requested()) { return false; } - // when recovering, each ill-formed sequence becomes U+FFFD, as it - // does in JSON text - replace_invalid_utf8(result, old_size); + result = sanitize_utf8(result, error_handler_t::replace); + return true; } + result = sanitize_utf8(result, error_handler); return true; } @@ -5222,6 +5254,9 @@ class binary_reader /// input format const input_format_t input_format = input_format_t::json; + /// how to treat text strings/object keys that are not well-formed UTF-8 + const error_handler_t error_handler = error_handler_t::keep; + /// the SAX parser json_sax_t* sax = nullptr; diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index 09f0ded14..9578e275a 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -453,8 +453,10 @@ struct wide_string_input_helper } else { - // get the current character - const auto wc = input.get_character(); + // get the current character; converted to an unsigned type so that + // a negative unit (wint_t is signed on some platforms) is not + // mistaken for an ASCII character or for EOF + const auto wc = static_cast(input.get_character()); if (wc <= 0x10FFFF) { @@ -522,9 +524,11 @@ struct wide_string_input_helper bool valid_pair = false; if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty())) { - const auto wc2 = static_cast(input.get_character()); + // only consume the next unit if it completes the pair + const auto wc2 = static_cast(*input.current); if (0xDC00 <= wc2 && wc2 <= 0xDFFF) { + input.get_character(); const auto charcode = 0x10000u + (((static_cast(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); utf8_bytes_filled = 0; encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) @@ -537,7 +541,8 @@ struct wide_string_input_helper if (!valid_pair) { - utf8_bytes[0] = static_cast::int_type>(wc); + // emit a byte that is never valid UTF-8 (see the UTF-32 case) + utf8_bytes[0] = 0xFF; utf8_bytes_filled = 1; } } @@ -746,7 +751,7 @@ struct container_input_adapter_factory< ContainerType, { // container is forwarded twice on purpose: the resulting begin/end // iterator types must match adapter_type, computed the same way - // NOLINTNEXTLINE(bugprone-use-after-move) + // NOLINTNEXTLINE(bugprone-use-after-move,hicpp-invalid-access-moved) return input_adapter(begin(std::forward(container)), end(std::forward(container))); } }; diff --git a/include/nlohmann/detail/iterators/iter_impl.hpp b/include/nlohmann/detail/iterators/iter_impl.hpp index 2115b6aeb..84fdd07f8 100644 --- a/include/nlohmann/detail/iterators/iter_impl.hpp +++ b/include/nlohmann/detail/iterators/iter_impl.hpp @@ -60,9 +60,11 @@ class iter_impl // NOLINT(cppcoreguidelines-special-member-functions,hicpp-speci static_assert(is_basic_json::type>::value, "iter_impl only accepts (const) basic_json"); // superficial check for the LegacyBidirectionalIterator named requirement - static_assert(std::is_base_of::value - && std::is_base_of::iterator_category>::value, - "basic_json iterator assumes array and object type iterators satisfy the LegacyBidirectionalIterator named requirement."); + // note: only array_t::iterator is checked here; object_t::iterator may be + // a forward-only iterator as long as reverse iteration and operator-- + // are never used on it + static_assert(std::is_base_of::iterator_category>::value, + "basic_json iterator assumes array type iterators satisfy the LegacyBidirectionalIterator named requirement."); public: /// The std::iterator class template (used as a base class to provide typedefs) is deprecated in C++17. diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index a10f4f39f..788af1047 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -240,6 +240,72 @@ class json_pointer } private: + /*! + @brief result of @ref parse_array_index + + @ref array_index maps each value to the corresponding parse_error/out_of_range + exception; @ref contains and @ref get_checked_or_null, which must not throw for + an out-of-range or unrepresentable index, switch on it directly instead. + */ + enum class array_index_status + { + ok, ///< @a s is a valid, representable array index + leading_zero, ///< @a s begins with '0' but has more than one character + not_a_number, ///< @a s does not begin with a digit + unresolved, ///< @a s could not be converted to an integer + exceeds_size_type ///< @a s converts to an integer that exceeds size_type + }; + + /*! + @param[in] s reference token to be converted into an array index + @param[out] idx the integer representation of @a s if @ref array_index_status::ok + is returned; left unchanged otherwise + + @return whether @a s is a valid array index, and if not, why + + @note this function never throws; @ref array_index and the callers that must not + throw (@ref contains, @ref get_checked_or_null) build on it instead of each + re-implementing the RFC 6901 digit rules and the @a size_type range check + */ + template + static array_index_status parse_array_index(const string_t& s, typename BasicJsonType::size_type& idx) noexcept + { + using size_type = typename BasicJsonType::size_type; + + // error condition (cf. RFC 6901, Sect. 4) + if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0')) + { + return array_index_status::leading_zero; + } + + // error condition (cf. RFC 6901, Sect. 4) + if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9'))) + { + return array_index_status::not_a_number; + } + + const char* p = s.data(); + char* p_end = nullptr; // NOLINT(misc-const-correctness) + errno = 0; // strtoull doesn't reset errno + const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) + if (p == p_end // invalid input or empty string + || errno == ERANGE // out of range + || JSON_HEDLEY_UNLIKELY(static_cast(p_end - p) != s.size())) // incomplete read + { + return array_index_status::unresolved; + } + + // the index does not fit into size_type; on 64-bit platforms this is + // only SIZE_MAX itself (see #2203 and #5395) + if (res >= static_cast((std::numeric_limits::max)())) // NOLINT(runtime/int) + { + return array_index_status::exceeds_size_type; + } + + idx = static_cast(res); + return array_index_status::ok; + } + /*! @param[in] s reference token to be converted into an array index @@ -253,39 +319,23 @@ class json_pointer template static typename BasicJsonType::size_type array_index(const string_t& s) { - using size_type = typename BasicJsonType::size_type; - - // error condition (cf. RFC 6901, Sect. 4) - if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0')) + typename BasicJsonType::size_type idx{}; + switch (parse_array_index(s, idx)) { - JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr)); + case array_index_status::leading_zero: + JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr)); + case array_index_status::not_a_number: + JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); + case array_index_status::unresolved: + JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr)); + case array_index_status::exceeds_size_type: + JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); + case array_index_status::ok: + default: + break; } - // error condition (cf. RFC 6901, Sect. 4) - if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9'))) - { - JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); - } - - const char* p = s.data(); - char* p_end = nullptr; // NOLINT(misc-const-correctness) - errno = 0; // strtoull doesn't reset errno - const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) - if (p == p_end // invalid input or empty string - || errno == ERANGE // out of range - || JSON_HEDLEY_UNLIKELY(static_cast(p_end - p) != s.size())) // incomplete read - { - JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr)); - } - - // the index does not fit into size_type; on 64-bit platforms this is - // only SIZE_MAX itself (see #2203 and #5395) - if (res >= static_cast((std::numeric_limits::max)())) // NOLINT(runtime/int) - { - JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); - } - - return static_cast(res); + return idx; } JSON_PRIVATE_UNLESS_TESTED: @@ -590,63 +640,6 @@ class json_pointer return *ptr; } - /*! - @throw parse_error.106 if an array index begins with '0' - @throw parse_error.109 if an array index was not a number - @throw out_of_range.402 if the array index '-' is used - @throw out_of_range.404 if the JSON pointer can not be resolved - */ - template - const BasicJsonType& get_checked(const BasicJsonType* ptr) const - { - for (const auto& reference_token : reference_tokens) - { - switch (ptr->type()) - { - case detail::value_t::object: - { - // note: at performs range check - ptr = &ptr->at(reference_token); - break; - } - - case detail::value_t::array: - { - if (JSON_HEDLEY_UNLIKELY(reference_token == "-")) - { - // "-" always fails the range check - JSON_THROW(detail::out_of_range::create(402, detail::concat( - "array index '-' (", std::to_string(ptr->m_data.m_value.array->size()), - ") is out of range"), ptr)); - } - - const auto idx = array_index(reference_token); - // Bounds check before access to avoid exception with JSON_NOEXCEPTION - if (JSON_HEDLEY_UNLIKELY(idx >= ptr->m_data.m_value.array->size())) - { - JSON_THROW(detail::out_of_range::create(401, detail::concat( - "array index ", std::to_string(idx), " is out of range"), ptr)); - } - ptr = &ptr->operator[](idx); - break; - } - - case detail::value_t::null: - case detail::value_t::string: - case detail::value_t::boolean: - case detail::value_t::number_integer: - case detail::value_t::number_unsigned: - case detail::value_t::number_float: - case detail::value_t::binary: - case detail::value_t::discarded: - default: - JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", reference_token, "'"), ptr)); - } - } - - return *ptr; - } - /*! @brief return a pointer to the pointed to value, or `nullptr` if the pointer cannot be resolved because a key is missing, an array @@ -685,29 +678,24 @@ class json_pointer return nullptr; } - // tokens that array_index() rejects with parse_error.106/109 - // are passed on to it; all other tokens that it would reject - // with out_of_range.404/410 are detected here, so that this - // also works without exceptions - if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1 && !(reference_token[0] >= '1' && reference_token[0] <= '9'))) + // a malformed index throws parse_error.106/109; an + // index that is syntactically valid but cannot be + // represented (out_of_range.404/410) is treated like an + // out-of-range index below + typename BasicJsonType::size_type idx{}; + switch (parse_array_index(reference_token, idx)) { - static_cast(array_index(reference_token)); // throws parse_error.106/109 + case array_index_status::leading_zero: + JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", reference_token, "' must not begin with '0'"), nullptr)); + case array_index_status::not_a_number: + JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", reference_token, "' is not a number"), nullptr)); + case array_index_status::unresolved: + case array_index_status::exceeds_size_type: + return nullptr; + case array_index_status::ok: + default: + break; } - if (JSON_HEDLEY_UNLIKELY(reference_token.empty() || !std::all_of(reference_token.begin(), reference_token.end(), [](const char c) - { - return c >= '0' && c <= '9'; - }))) - { - return nullptr; - } - errno = 0; // strtoull() does not reset errno on success - char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) - if (JSON_HEDLEY_UNLIKELY(errno == ERANGE || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) - { - return nullptr; - } - const auto idx = static_cast(magnitude); if (JSON_HEDLEY_UNLIKELY(idx >= ptr->m_data.m_value.array->size())) { @@ -734,8 +722,8 @@ class json_pointer } /*! - @throw parse_error.106 if an array index begins with '0' - @throw parse_error.109 if an array index was not a number + @note unlike array_index(), this never throws: a malformed or unrepresentable + array index reference token is treated like a missing key (see #5395) */ template bool contains(const BasicJsonType* ptr) const @@ -763,49 +751,17 @@ class json_pointer // "-" always fails the range check return false; } - if (JSON_HEDLEY_UNLIKELY(reference_token.empty())) - { - // an empty reference token is not an array index; array_index() - // would throw out_of_range.404 -- contains() must not throw (see #5395) - return false; - } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9'))) - { - // invalid char - return false; - } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1)) - { - if (JSON_HEDLEY_UNLIKELY(!('1' <= reference_token[0] && reference_token[0] <= '9'))) - { - // the first char should be between '1' and '9' - return false; - } - for (std::size_t i = 1; i < reference_token.size(); i++) - { - if (JSON_HEDLEY_UNLIKELY(!('0' <= reference_token[i] && reference_token[i] <= '9'))) - { - // other char should be between '0' and '9' - return false; - } - } - } - // the reference token consists only of digits at this point (cf. checks - // above); however, its numeric value might not be representable, in which - // case array_index() would throw out_of_range.404/410 -- contains() must - // not throw (see #5395), so such a reference token is treated as "not found" - errno = 0; // strtoull() does not reset errno on success - char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) - if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX - || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) + // any parse failure (malformed index, or one that is syntactically + // valid but not representable as size_type) means the reference + // token cannot denote an existing array element -- contains() must + // not throw (see #5395), so it is treated as "not found" + typename BasicJsonType::size_type idx{}; + if (JSON_HEDLEY_UNLIKELY(parse_array_index(reference_token, idx) != array_index_status::ok)) { - // the array index cannot be represented as size_type return false; } - const auto idx = array_index(reference_token); if (idx >= ptr->size()) { // index out of range diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 3a5f7e606..17118927c 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -9,7 +9,6 @@ #pragma once #include // declval, pair -#include #include // This file contains all internal macro definitions (except those affecting ABI) @@ -140,10 +139,12 @@ // libstdc++ < 11 has incomplete C++20 ranges (issue #4440) #elif defined(_GLIBCXX_RELEASE) && _GLIBCXX_RELEASE < 11 #define JSON_HAS_RANGES 0 - // libc++ < 16 has incomplete C++20 ranges (issue #4440) + // clang < 16 with libstdc++ does not implement the ranges customization + // points libstdc++ declares, so its C++20 ranges support is incomplete (issue #5161) #elif defined(__clang__) && !defined(__apple_build_version__) \ && __clang_major__ < 16 && defined(__GLIBCXX__) #define JSON_HAS_RANGES 0 + // libc++ < 16 has incomplete C++20 ranges (issue #4440) #elif defined(_LIBCPP_VERSION) && _LIBCPP_VERSION < 160000 #define JSON_HAS_RANGES 0 // nvcc CUDA 12.0/12.1 chokes on the enable_borrowed_range variable-template @@ -158,6 +159,18 @@ #endif #endif +// std::ranges view conversion (to_json/is_compatible_array_type_impl) additionally +// needs to be disabled on MinGW, whose std::ranges support is incomplete +// (issue #4916); this macro combines both conditions so the check and its +// reason are not duplicated at every use site. +#ifndef JSON_HAS_RANGE_VIEW_CONVERSION + #if JSON_HAS_RANGES && !defined(__MINGW32__) + #define JSON_HAS_RANGE_VIEW_CONVERSION 1 + #else + #define JSON_HAS_RANGE_VIEW_CONVERSION 0 + #endif +#endif + #ifndef JSON_HAS_STD_FORMAT #if defined(JSON_HAS_CPP_20) && defined(__cpp_lib_format) #define JSON_HAS_STD_FORMAT 1 @@ -279,21 +292,6 @@ -/*! -@brief function to wrap JSON_THROW_MACRO - there can be compilation errors about - there being no arguments to JSON_THROW that depend on template arguments - if this is not used to call JSON_THROW -*/ -template -void templated_json_throw(ExceptionType exception) -{ - JSON_THROW(exception); - - /* JSON_THROW(exception) discards exception and aborts - void cast needed to supress - compilation error if compiled with -Werror and Wunused-parameter */ - (void)exception; -} - /*! @brief macro to briefly define a mapping between an enum and JSON with exception on invalid input @@ -314,7 +312,7 @@ void templated_json_throw(ExceptionType exception) return ej_pair.first == e; \ }); \ if (it != std::end(m)) j = it->second; \ - else templated_json_throw(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE, nullptr)); \ + else ::nlohmann::detail::templated_json_throw(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE, nullptr)); \ } \ template \ inline void from_json(const BasicJsonType& j, ENUM_TYPE& e) \ @@ -329,7 +327,7 @@ void templated_json_throw(ExceptionType exception) return ej_pair.second == j; \ }); \ if (it != std::end(m)) e = it->first; \ - else templated_json_throw(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \ + else ::nlohmann::detail::templated_json_throw(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \ } // Ugly macros to avoid uglier copy-paste when specializing basic_json. They @@ -874,30 +872,6 @@ void templated_json_throw(ExceptionType exception) \ template \ using result_of_##std_name = decltype(std_name(std::declval()...)); \ - } \ - \ - namespace detail2 { \ - struct std_name##_tag \ - { \ - }; \ - \ - template \ - std_name##_tag std_name(T&&...); \ - \ - template \ - using result_of_##std_name = decltype(std_name(std::declval()...)); \ - \ - template \ - struct would_call_std_##std_name \ - { \ - static constexpr auto const value = ::nlohmann::detail:: \ - is_detected_exact::value; \ - }; \ - } /* namespace detail2 */ \ - \ - template \ - struct would_call_std_##std_name : detail2::would_call_std_##std_name \ - { \ } #ifndef JSON_USE_IMPLICIT_CONVERSIONS diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index 1e6e6cce6..55b2ac99e 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -35,12 +35,14 @@ #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM #undef JSON_HAS_THREE_WAY_COMPARISON #undef JSON_HAS_RANGES + #undef JSON_HAS_RANGE_VIEW_CONVERSION #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION #undef JSON_STRICT_NUL_HANDLING + #undef JSON_STRICT_BINARY_UTF8 #endif #include diff --git a/include/nlohmann/detail/meta/call_std/begin.hpp b/include/nlohmann/detail/meta/call_std/begin.hpp index a086a5f3d..d12bbf7da 100644 --- a/include/nlohmann/detail/meta/call_std/begin.hpp +++ b/include/nlohmann/detail/meta/call_std/begin.hpp @@ -12,6 +12,6 @@ NLOHMANN_JSON_NAMESPACE_BEGIN -NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin); +NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin) NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/meta/call_std/end.hpp b/include/nlohmann/detail/meta/call_std/end.hpp index 40c9942b9..dc1895fc4 100644 --- a/include/nlohmann/detail/meta/call_std/end.hpp +++ b/include/nlohmann/detail/meta/call_std/end.hpp @@ -12,6 +12,6 @@ NLOHMANN_JSON_NAMESPACE_BEGIN -NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end); +NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end) NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/meta/is_sax.hpp b/include/nlohmann/detail/meta/is_sax.hpp index 8e8a0de24..ac238b0c2 100644 --- a/include/nlohmann/detail/meta/is_sax.hpp +++ b/include/nlohmann/detail/meta/is_sax.hpp @@ -8,7 +8,7 @@ #pragma once -#include // size_t +#include // size_t #include // declval #include // string @@ -70,37 +70,6 @@ using parse_error_function_t = decltype(std::declval().parse_error( std::declval(), std::declval(), std::declval())); -template -struct is_sax -{ - private: - static_assert(is_basic_json::value, - "BasicJsonType must be of type basic_json<...>"); - - using number_integer_t = typename BasicJsonType::number_integer_t; - using number_unsigned_t = typename BasicJsonType::number_unsigned_t; - using number_float_t = typename BasicJsonType::number_float_t; - using string_t = typename BasicJsonType::string_t; - using binary_t = typename BasicJsonType::binary_t; - using exception_t = typename BasicJsonType::exception; - - public: - static constexpr bool value = - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value; -}; - template struct is_sax_static_asserts { @@ -120,8 +89,6 @@ struct is_sax_static_asserts "Missing/invalid function: bool null()"); static_assert(is_detected_exact::value, "Missing/invalid function: bool boolean(bool)"); - static_assert(is_detected_exact::value, - "Missing/invalid function: bool boolean(bool)"); static_assert( is_detected_exact::value, diff --git a/include/nlohmann/detail/meta/logic.hpp b/include/nlohmann/detail/meta/logic.hpp deleted file mode 100644 index cb50a5d19..000000000 --- a/include/nlohmann/detail/meta/logic.hpp +++ /dev/null @@ -1,54 +0,0 @@ -#pragma once - -#include - -NLOHMANN_JSON_NAMESPACE_BEGIN -namespace detail -{ -#ifdef JSON_HAS_CPP_17 - -template -struct cxpr_or_impl : std::integral_constant < bool, (Booleans || ...) > {}; - -template -struct cxpr_and_impl : std::integral_constant < bool, (Booleans &&...) > {}; - -#else - -template -struct cxpr_or_impl : std::false_type {}; - -template -struct cxpr_or_impl : std::true_type {}; - -template -struct cxpr_or_impl : cxpr_or_impl {}; - -template -struct cxpr_and_impl : std::true_type {}; - -template -struct cxpr_and_impl : cxpr_and_impl {}; - -template -struct cxpr_and_impl : std::false_type {}; - -#endif - -template -struct cxpr_not : std::integral_constant < bool, !Boolean::value > {}; - -template -struct cxpr_or : cxpr_or_impl {}; - -template -struct cxpr_or_c : cxpr_or_impl {}; - -template -struct cxpr_and : cxpr_and_impl {}; - -template -struct cxpr_and_c : cxpr_and_impl {}; - -} // namespace detail -NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/meta/type_traits.hpp b/include/nlohmann/detail/meta/type_traits.hpp index 14029c0fc..68bf4ad2c 100644 --- a/include/nlohmann/detail/meta/type_traits.hpp +++ b/include/nlohmann/detail/meta/type_traits.hpp @@ -314,6 +314,13 @@ template struct conjunction : std::conditional(B::value), conjunction, B>::type {}; +// https://en.cppreference.com/w/cpp/types/disjunction +template struct disjunction : std::false_type { }; +template struct disjunction : B { }; +template +struct disjunction +: std::conditional(B::value), B, disjunction>::type {}; + // https://en.cppreference.com/w/cpp/types/negation template struct negation : std::integral_constant < bool, !B::value > { }; @@ -508,9 +515,7 @@ template struct is_range_view_optional_type> : std: template struct is_range_view_optional_type : std::false_type {}; #endif -// std::ranges does not work properly on MinGW due to incomplete C++20 support -// see https://github.com/nlohmann/json/issues/4916 -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION // SafeToCheck guards against types that trigger circular constraints when // std::ranges::view is evaluated on GCC 12 / libstdc++ 12: @@ -549,7 +554,7 @@ struct is_compatible_array_type_impl < // filter_view) can match BOTH this iterator-based specialization AND the view-based one // below, causing ambiguity. Exclude views here so the two specializations are mutually // exclusive: this one handles plain iterable containers, the other handles views. -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif >> @@ -559,7 +564,7 @@ struct is_compatible_array_type_impl < range_value_t>::value; }; -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template struct is_compatible_array_type_impl < BasicJsonType, CompatibleArrayType, @@ -635,7 +640,6 @@ struct is_compatible_integer_type_impl < std::is_integral::value&& !std::is_same::value >> { - // is there an assert somewhere on overflows? using RealLimits = std::numeric_limits; using CompatibleLimits = std::numeric_limits; @@ -863,20 +867,7 @@ struct has_capacity : std::integral_constant -struct is_ordered_map -{ - using one = char; - - struct two - { - char x[2]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - }; - - template static one test( decltype(&C::capacity) ) ; - template static two test(...); - - enum { value = sizeof(test(nullptr)) == sizeof(char) }; // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg,cppcoreguidelines-use-enum-class) -}; +struct is_ordered_map : has_capacity {}; // to avoid useless casts (see https://github.com/nlohmann/json/issues/2893#issuecomment-889152324) template < typename T, typename U, enable_if_t < !std::is_same::value, int > = 0 > @@ -900,10 +891,8 @@ using all_signed = conjunction...>; template using all_unsigned = conjunction...>; -// there's a disjunction trait in another PR; replace when merged template -using same_sign = std::integral_constant < bool, - all_signed::value || all_unsigned::value >; +using same_sign = disjunction, all_unsigned>; template using never_out_of_range = std::integral_constant < bool, diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 9d844d85e..e74671db2 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -26,6 +26,7 @@ #include #include #include +#include #include #include #include @@ -93,8 +94,12 @@ class binary_writer @param[in] sink output sink to write to (a value-type sink such as output_vector_sink, or output_adapter_sink wrapping a type-erased output adapter) + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ - explicit binary_writer(OutputSinkType sink) : oa(std::move(sink)) + explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(std::move(sink)), error_handler(error_handler_) {} /*! @@ -107,14 +112,20 @@ class binary_writer from one. @param[in] adapter output adapter to write to + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ template < typename SinkType = OutputSinkType, typename std::enable_if < std::is_constructible>::value, int >::type = 0 > - explicit binary_writer(output_adapter_t adapter) : oa(SinkType(std::move(adapter))) + explicit binary_writer(output_adapter_t adapter, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(SinkType(std::move(adapter))), error_handler(error_handler_) {} /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -145,6 +156,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -211,13 +224,16 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - write_cbor_head(0x60, j.m_data.m_value.string->size()); + write_cbor_head(0x60, value.size()); // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -287,6 +303,17 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // el.first is checked here, against the object as + // diagnostics context, because write_cbor(el.first) + // converts it to a temporary basic_json that would be + // used as the context instead; for error_handler_t::keep + // and ::replace/::ignore the recursive write_cbor(el.first) + // call below handles the key like any other string, so no + // separate check is needed here for those + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_cbor(el.first); write_cbor(el.second); } @@ -434,8 +461,11 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); + const auto N = to_msgpack_length(value.size(), j); if (N <= 31) { // fixstr @@ -462,8 +492,8 @@ class binary_writer // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -610,6 +640,13 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // as in write_cbor, el.first is checked here against the + // object as diagnostics context; the recursive call below + // handles keep/replace/ignore like any other string + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_msgpack(el.first); write_msgpack(el.second); } @@ -629,6 +666,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -678,14 +717,17 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + if (add_prefix) { oa.write_character(to_char_type('S')); } - write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); + write_number_with_ubjson_prefix(value.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -840,10 +882,12 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { - write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); + string_t storage; + const string_t& key = sanitize_utf8_for_write(el.first, j, storage); + write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(el.first.data()), - el.first.size()); + reinterpret_cast(key.data()), + key.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } @@ -884,8 +928,12 @@ class binary_writer /*! @return The size of a BSON document entry header, including the id marker and the entry name size (and its null-terminator). + @throw out_of_range.409 if @a name contains U+0000, before anything is + written + @throw type_error.316 if @a name is not valid UTF-8, before anything is + written */ - static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) + std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { const auto it = name.find(static_cast(0)); if (JSON_HEDLEY_UNLIKELY(it != BasicJsonType::string_t::npos)) @@ -893,8 +941,10 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - static_cast(j); - return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(name, j, storage); + + return /*id*/ 1ul + sanitized.size() + /*zero-terminator*/1u; } /*! @@ -914,14 +964,28 @@ class binary_writer /*! @brief Writes the given @a element_type and @a name to the output adapter + + @a name has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_entry_header_size during the earlier + size pass, so only @ref error_handler_t::replace / @ref + error_handler_t::ignore need to sanitize it again here, to actually write + the bytes that size was computed from. */ void write_bson_entry_header(const string_t& name, const std::uint8_t element_type) { oa.write_character(to_char_type(element_type)); - oa.write_characters( - reinterpret_cast(name.data()), - name.size()); + + if (error_handler == error_handler_t::keep || error_handler == error_handler_t::strict || is_valid_utf8(name)) + { + oa.write_characters(reinterpret_cast(name.data()), name.size()); + } + else + { + const string_t sanitized = sanitize_utf8(name, error_handler); + oa.write_characters(reinterpret_cast(sanitized.data()), sanitized.size()); + } + // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -949,24 +1013,50 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value + @throw type_error.316 if @a value is not valid UTF-8, before anything is + written + + @note The UTF-8 check is skipped if @a value is already too long for the + 32-bit BSON length field (@ref to_bson_length rejects it later, once + the size of the whole document is known); this also keeps the check + from reading past a StringType that reports a size larger than what + it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value) + std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { + if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, j, storage); + return sizeof(std::int32_t) + sanitized.size() + 1ul; + } return sizeof(std::int32_t) + value.size() + 1ul; } /*! @brief Writes a BSON element with key @a name and string value @a value + + @a value has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_string_size during the earlier size + pass, so only @ref error_handler_t::replace / @ref error_handler_t::ignore + need to sanitize it again here, to actually write the bytes that size was + computed from. */ void write_bson_string(const string_t& name, const string_t& value) { write_bson_entry_header(name, 0x02); - write_number(to_bson_length(value.size() + 1ul), true); + const bool sanitize = error_handler != error_handler_t::keep + && error_handler != error_handler_t::strict + && !is_valid_utf8(value); + const string_t sanitized = sanitize ? sanitize_utf8(value, error_handler) : string_t{}; + const string_t& written = sanitize ? sanitized : value; + + write_number(to_bson_length(written.size() + 1ul), true); oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + reinterpret_cast(written.data()), + written.size()); // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -1080,8 +1170,10 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written + @throw type_error.316 if @a j is a string that is not valid UTF-8, before + anything is written */ - static std::size_t calc_bson_value_size(const BasicJsonType& j) + std::size_t calc_bson_value_size(const BasicJsonType& j) { switch (j.type()) { @@ -1101,7 +1193,7 @@ class binary_writer return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string, j); case value_t::null: return 0ul; @@ -1214,8 +1306,10 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written + @throw type_error.316 if a string value or a key is not valid UTF-8, + before anything is written */ - static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) + std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { // the object or array whose entries are being sized, and the ones it // is in; nothing is allocated unless the document nests @@ -2092,7 +2186,7 @@ class binary_writer */ void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) { - check_bon8_utf8(s, context); + check_utf8(s, context); // a string that follows another string terminates it if (string_open) @@ -2122,7 +2216,7 @@ class binary_writer @throw type_error.316 if @a s is not valid UTF-8; the message names the first byte of the first invalid or incomplete sequence */ - static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + static void check_utf8(const string_t& s, const BasicJsonType& context) { static_cast(context); // only used when exceptions are enabled const auto* data = reinterpret_cast(s.data()); @@ -2133,6 +2227,57 @@ class binary_writer } } + /*! + @brief return @a s as it should be written, honoring @ref error_handler + + Used by @ref write_cbor, @ref write_msgpack, @ref write_ubjson (and so + @ref write_bjdata), and the BSON writing functions for string values and + object keys; never by @ref write_bon8, which always validates, since UTF-8 + lead bytes are structural there. + + - @ref error_handler_t::keep: @a s is returned unchanged, without even + checking it (the behavior of release 3.12.0 and earlier). + - @ref error_handler_t::strict: @ref check_utf8 is called, which throws + type_error.316 if @a s is not valid UTF-8. + - @ref error_handler_t::replace / @ref error_handler_t::ignore: @a s is + sanitized into @a storage with exactly the rules @ref + serializer::dump_escaped_impl uses, so that parsing what @ref + basic_json::dump produces for the same string and the same handler + yields the same result. + + Well-formed input is never copied: this returns a reference to @a s + itself in every case but a sanitized `replace`/`ignore` one, so @a + storage must outlive the returned reference only then. + + @param[in] s the string (value or object key) to write + @param[in] context the value @a s belongs to (for diagnostics) + @param[out] storage backing storage for a sanitized copy + + @return a reference to @a s, or to @a storage once it holds a sanitized copy + */ + const string_t& sanitize_utf8_for_write(const string_t& s, const BasicJsonType& context, string_t& storage) const + { + switch (error_handler) + { + case error_handler_t::keep: + return s; + + case error_handler_t::strict: + check_utf8(s, context); + return s; + + case error_handler_t::replace: + case error_handler_t::ignore: + default: + if (is_valid_utf8(s)) + { + return s; + } + storage = sanitize_utf8(s, error_handler); + return storage; + } + } + /*! @brief write an integer in the shortest encoding @@ -2461,6 +2606,10 @@ class binary_writer /// the output OutputSinkType oa; + + /// how to treat a string value or object key that is not valid UTF-8 + /// (CBOR, MessagePack, UBJSON, BJData, and BSON; not BON8) + const error_handler_t error_handler = binary_writer_default_error_handler(); }; } // namespace detail diff --git a/include/nlohmann/detail/output/error_handler.hpp b/include/nlohmann/detail/output/error_handler.hpp new file mode 100644 index 000000000..b95d70fce --- /dev/null +++ b/include/nlohmann/detail/output/error_handler.hpp @@ -0,0 +1,50 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// how to treat decoding errors +/// +/// @ref basic_json::dump uses this to decide what to do with ill-formed +/// UTF-8 while escaping a string, and the binary writers (@ref +/// basic_json::to_cbor, @ref basic_json::to_ubjson, @ref +/// basic_json::to_bjdata, @ref basic_json::to_bson) use it the same way for +/// string values and object keys. The binary readers (@ref +/// basic_json::from_cbor, @ref basic_json::from_msgpack, @ref +/// basic_json::from_ubjson, @ref basic_json::from_bjdata, @ref +/// basic_json::from_bson) use it to decide whether to check text strings +/// and object keys for well-formed UTF-8 at all, since none of those +/// formats requires a decoder to do so. +enum class error_handler_t +{ + strict, ///< throw a type_error/parse_error exception in case of invalid UTF-8 + replace, ///< replace invalid UTF-8 sequences with U+FFFD + ignore, ///< ignore invalid UTF-8 sequences + keep ///< keep invalid UTF-8 sequences unchanged +}; + +/// the default error handler of the CBOR, UBJSON, BJData, and BSON writers: +/// error_handler_t::strict if JSON_STRICT_BINARY_UTF8 is enabled, otherwise +/// error_handler_t::keep (the behavior before version 3.13.0) +constexpr error_handler_t binary_writer_default_error_handler() noexcept +{ +#if JSON_STRICT_BINARY_UTF8 + return error_handler_t::strict; +#else + return error_handler_t::keep; +#endif +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index 1c519d4c7..dcaae8055 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -27,6 +27,7 @@ #include #include #include +#include #include #include #include @@ -41,14 +42,6 @@ namespace detail // serialization // /////////////////// -/// how to treat decoding errors -enum class error_handler_t -{ - strict, ///< throw a type_error exception in case of invalid UTF-8 - replace, ///< replace invalid UTF-8 sequences with U+FFFD - ignore ///< ignore invalid UTF-8 sequences -}; - template class serializer { @@ -839,6 +832,16 @@ class serializer // EnsureAscii parameter is used, non-ASCII characters if ((codepoint <= 0x1F) || (EnsureAscii && (codepoint >= 0x7F))) { + if (EnsureAscii && error_handler == error_handler_t::keep) + { + // this character was buffered as raw bytes + // below in case it turned out to be part of + // an ill-formed sequence (which is kept as + // is); now that it decoded to a well-formed + // code point, undo that and \u-escape it + // like any other character instead + bytes = bytes_after_last_accept; + } if (codepoint <= 0xFFFF) { write_u_escape(bytes, static_cast(codepoint)); @@ -937,6 +940,44 @@ class serializer break; } + case error_handler_t::keep: + { + // the bytes of this (now abandoned) ill-formed + // sequence seen so far are already buffered below + // and are kept unchanged in the output + if (undumped_chars > 0) + { + // the byte that ended the sequence may be OK + // for itself (e.g., a quote that must still be + // escaped, or the lead byte of a well-formed + // code point), so read it again + --i; + } + else + { + // a byte that cannot start a sequence (e.g., + // 0xFF or a stray continuation byte) is kept + // as well + string_buffer[bytes++] = s[i]; + } + + // write buffer and reset index; there must be 13 bytes + // left, as this is the maximal number of bytes to be + // written ("\uxxxx\uxxxx\0") for one code point + if (string_buffer.size() - bytes < 13) + { + put_buffer(string_buffer, bytes); + bytes = 0; + } + + bytes_after_last_accept = bytes; + undumped_chars = 0; + + // continue processing the string + state = UTF8_ACCEPT; + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -945,9 +986,12 @@ class serializer default: // decode found yet incomplete multibyte code point { - if (!EnsureAscii) + if (!EnsureAscii || error_handler == error_handler_t::keep) { - // code point will not be escaped - copy byte to buffer + // code point will not be escaped (or will be kept as + // is if it turns out to be ill-formed) - copy byte to + // buffer; dropped again above if it decodes to a + // well-formed code point that needs \u-escaping string_buffer[bytes++] = s[i]; } ++undumped_chars; @@ -998,6 +1042,14 @@ class serializer break; } + case error_handler_t::keep: + { + // write the ill-formed trailing bytes as is; they were + // buffered above regardless of EnsureAscii + put_buffer(string_buffer, bytes); + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } diff --git a/include/nlohmann/detail/string_concat.hpp b/include/nlohmann/detail/string_concat.hpp index a39dd5e63..74002c878 100644 --- a/include/nlohmann/detail/string_concat.hpp +++ b/include/nlohmann/detail/string_concat.hpp @@ -39,7 +39,6 @@ inline std::size_t concat_length(const char /*c*/, const Args& ... rest) template inline std::size_t concat_length(const char* cstr, const Args& ... rest) { - // cppcheck-suppress ignoredReturnValue return ::strlen(cstr) + concat_length(rest...); } diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index 747202649..36d9ee8df 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -17,6 +17,7 @@ #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -118,13 +119,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally written by Björn Hoehrmann. See http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. -The library checks UTF-8 well-formedness (RFC 3629, section 4) in four +The library checks UTF-8 well-formedness (RFC 3629, section 4) in three places, which differ in speed, diagnostics, and how they read the input: -- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in - strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR, - MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in - text strings at decode time). +- decode() below: the serializer, to escape and, in strict mode, reject + ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON, + UBJSON and BJData readers do not use it: none of those specs requires a + decoder to reject ill-formed UTF-8 in text strings, so the readers keep + the bytes as is and leave the check to dump() and the binary writers. - the per-lead-byte switch in lexer::scan_string(): JSON text, with a diagnostic for each kind of error. - validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's @@ -180,19 +182,19 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std: } /*! -@brief check whether a string consists solely of valid UTF-8 +@brief check a string for well-formed UTF-8 (RFC 3629, section 4) -Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text -strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the -MessagePack/BSON specifications all require text strings to be UTF-8), so -that malformed input is caught immediately instead of only surfacing later -as a type_error.316 when the resulting value is dumped. +Used by the binary readers (CBOR, MessagePack, UBJSON, BJData, BSON) when an +@ref error_handler_t other than `keep` is requested for a text string value +or object key: none of those formats requires a decoder to reject ill-formed +UTF-8 on its own, so the check is opt-in there, unlike the JSON lexer and the +serializer's @ref decode -based escaping, which always run it. @param[in] s the string to check -@param[in] first index of the first byte to check; the bytes before it are - assumed to have been validated already and to end on a - code point boundary -@return whether @a s (from index @a first on) is valid UTF-8 +@param[in] first the index to start checking at +@return whether `s.substr(first)` is well-formed UTF-8 + +@sa @ref decode */ template inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept @@ -213,76 +215,99 @@ inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noex } /*! -@brief append U+FFFD REPLACEMENT CHARACTER, encoded in UTF-8 -@param[in,out] s the string to append to +@brief sanitize a string with ill-formed UTF-8 for @ref error_handler_t::replace or @ref error_handler_t::ignore + +Replaces every maximal ill-formed subsequence with U+FFFD (`replace`) or +drops it (`ignore`), using exactly the same boundaries @ref +serializer::dump_escaped_impl uses while escaping a string: a byte that does +not extend the sequence started by the previous byte(s) is reread as the +start of a new one, instead of being swallowed along with them. + +@pre @a error_handler is @ref error_handler_t::replace or @ref error_handler_t::ignore +@note Well-formed input is copied through unchanged, including bytes (e.g. + control characters or quotes) that @ref serializer::dump_escaped_impl + would itself escape; this function only concerns itself with + well-formedness, not with producing valid JSON text. + +@param[in] s the string to sanitize +@param[in] error_handler @ref error_handler_t::replace or @ref error_handler_t::ignore + +@return @a s with every ill-formed subsequence replaced or removed + +@sa @ref decode */ template -inline void append_replacement_character(StringType& s) +inline StringType sanitize_utf8(const StringType& s, const error_handler_t error_handler) { - s.push_back(static_cast(0xEFu)); - s.push_back(static_cast(0xBFu)); - s.push_back(static_cast(0xBDu)); -} + JSON_ASSERT(error_handler == error_handler_t::replace || error_handler == error_handler_t::ignore); -/*! -@brief replace ill-formed UTF-8 with U+FFFD REPLACEMENT CHARACTER + StringType result; + result.reserve(s.size()); -Each maximal subpart of an ill-formed sequence becomes one U+FFFD, as the -Unicode Standard recommends (Section 3.9, "U+FFFD Substitution of Maximal -Subparts"), and as the parser for JSON text does when it recovers from errors. - -@param[in,out] s the string to repair -@param[in] first index of the first byte to repair; the bytes before it are - assumed to be valid UTF-8 that ends on a code point boundary -*/ -template -inline void replace_invalid_utf8(StringType& s, const std::size_t first = 0) -{ - StringType result = s; - result.resize(first); - - std::uint8_t state = UTF8_ACCEPT; std::uint32_t codepoint = 0; - // the first byte of the sequence being decoded - std::size_t sequence_start = first; + std::uint8_t state = UTF8_ACCEPT; + // length of result after the last accepted code point + std::size_t result_len_after_last_accept = 0; + // whether bytes of an as yet unresolved sequence were already appended + bool pending = false; - std::size_t i = first; - while (i < s.size()) + for (std::size_t i = 0; i < s.size(); ++i) { switch (decode(state, codepoint, static_cast(s[i]))) { - case UTF8_ACCEPT: - for (++i; sequence_start < i; ++sequence_start) - { - result.push_back(s[sequence_start]); - } + case UTF8_ACCEPT: // decode found a well-formed code point + { + result.push_back(s[i]); + result_len_after_last_accept = result.size(); + pending = false; break; + } - case UTF8_REJECT: - append_replacement_character(result); - // the byte that made the sequence ill-formed begins the next - // one, unless it began this one - if (i == sequence_start) + case UTF8_REJECT: // decode found an ill-formed byte + { + // in case we saw this byte for the first time, read it again, + // because it may be fine for itself, just not for the + // sequence that came before it + if (pending) { - ++i; + --i; } + + // drop the bytes of the ill-formed sequence buffered below + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + result_len_after_last_accept = result.size(); + } + + pending = false; state = UTF8_ACCEPT; - sequence_start = i; break; + } - default: // in the middle of a sequence - ++i; + default: // decode found yet incomplete multibyte code point + { + result.push_back(s[i]); + pending = true; break; + } } } - // a sequence that the string ends in the middle of + // the string ended with an incomplete sequence if (state != UTF8_ACCEPT) { - append_replacement_character(result); + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + } } - s = std::move(result); + return result; } } // namespace detail diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 4ed4ef668..87d09e431 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -115,11 +115,12 @@ struct is_std_optional> : std::true_type {}; @brief a class to store JSON values @internal -@invariant The member variables @a m_value and @a m_type have the following -relationship: -- If `m_type == value_t::object`, then `m_value.object != nullptr`. -- If `m_type == value_t::array`, then `m_value.array != nullptr`. -- If `m_type == value_t::string`, then `m_value.string != nullptr`. +@invariant The member variables @a m_data.m_value and @a m_data.m_type have +the following relationship: +- If `m_data.m_type == value_t::object`, then `m_data.m_value.object != nullptr`. +- If `m_data.m_type == value_t::array`, then `m_data.m_value.array != nullptr`. +- If `m_data.m_type == value_t::string`, then `m_data.m_value.string != nullptr`. +- If `m_data.m_type == value_t::binary`, then `m_data.m_value.binary != nullptr`. The invariants are checked by member function assert_invariant(). @note ObjectType trick from https://stackoverflow.com/a/9860911 @@ -161,7 +162,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// workaround type for MSVC using basic_json_t = NLOHMANN_BASIC_JSON_TPL; - using json_base_class_t = ::nlohmann::detail::json_base_class; JSON_PRIVATE_UNLESS_TESTED: // convenience aliases for types residing in namespace detail; @@ -182,18 +182,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } private: - using primitive_iterator_t = ::nlohmann::detail::primitive_iterator_t; - template - using internal_iterator = ::nlohmann::detail::internal_iterator; template using iter_impl = ::nlohmann::detail::iter_impl; template using iteration_proxy = ::nlohmann::detail::iteration_proxy; template using json_reverse_iterator = ::nlohmann::detail::json_reverse_iterator; - template - using output_adapter_t = ::nlohmann::detail::output_adapter_t; - template using binary_reader = ::nlohmann::detail::binary_reader; template using binary_writer = ::nlohmann::detail::binary_writer; @@ -201,9 +195,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // used by the vector-returning to_* overloads template using vector_binary_writer = ::nlohmann::detail::binary_writer>; - template static vector_binary_writer vector_writer(std::vector& v) + template static vector_binary_writer vector_writer( + std::vector& v, const ::nlohmann::detail::error_handler_t error_handler = ::nlohmann::detail::binary_writer_default_error_handler()) { - return vector_binary_writer(::nlohmann::detail::output_vector_sink(v)); + return vector_binary_writer(::nlohmann::detail::output_vector_sink(v), error_handler); } JSON_PRIVATE_UNLESS_TESTED: @@ -221,6 +216,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec using cbor_tag_handler_t = detail::cbor_tag_handler_t; /// how to encode BJData using bjdata_version_t = detail::bjdata_version_t; + /// base class used to inject custom functionality into each instance of basic_json + /// @sa https://json.nlohmann.me/api/basic_json/json_base_class_t/ + using json_base_class_t = ::nlohmann::detail::json_base_class; /// helper type for initializer lists of basic_json values using initializer_list_t = std::initializer_list>; @@ -334,8 +332,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec std::to_string(__GNUC_PATCHLEVEL__)) } }; -#elif defined(__HP_cc) || defined(__HP_aCC) - result["compiler"] = "hp" +#elif defined(__HP_aCC) + result["compiler"] = {{"family", "hp"}, {"version", __HP_aCC}}; #elif defined(__IBMCPP__) result["compiler"] = {{"family", "ilecpp"}, {"version", __IBMCPP__}}; #elif defined(_MSC_VER) @@ -477,9 +475,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec binary | binary | pointer to @ref binary_t null | null | *no value is stored* - @note Variable-length types (objects, arrays, and strings) are stored as - pointers. The size of the union should not exceed 64 bits if the default - value types are used. + @note Variable-length types (objects, arrays, strings, and binary + values) are stored as pointers. The size of the union should not exceed + 64 bits if the default value types are used. @since version 1.0.0 */ @@ -731,7 +729,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec end of every constructor to make sure that created objects respect the invariant. Furthermore, it has to be called each time the type of a JSON value is changed, because the invariant expresses a relationship between - @a m_type and @a m_value. + @a m_data.m_type and @a m_data.m_value. Furthermore, the parent relation is checked for arrays and objects: If @a check_parents true and the value is an array or object, then the @@ -752,7 +750,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec #if JSON_DIAGNOSTICS JSON_TRY { - // cppcheck-suppress assertWithSideEffect JSON_ASSERT(!check_parents || !is_structured() || std::all_of(begin(), end(), [this](const basic_json & j) { return j.m_parent == this; @@ -1924,8 +1921,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief create a null object /// @sa https://json.nlohmann.me/api/basic_json/basic_json/ - basic_json(std::nullptr_t = nullptr) noexcept // NOLINT(bugprone-exception-escape) - : basic_json(value_t::null) + basic_json(std::nullptr_t = nullptr) noexcept { assert_invariant(); } @@ -2289,15 +2285,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief move constructor /// @sa https://json.nlohmann.me/api/basic_json/basic_json/ basic_json(basic_json&& other) noexcept - : json_base_class_t(std::forward(other)), - m_data(std::move(other.m_data)) // cppcheck-suppress[accessForwarded] TODO check + : json_base_class_t(std::move(static_cast(other))), + m_data(std::move(other.m_data)) #if JSON_DIAGNOSTIC_POSITIONS - , start_position(other.start_position) // cppcheck-suppress[accessForwarded] TODO check - , end_position(other.end_position) // cppcheck-suppress[accessForwarded] TODO check + , start_position(other.start_position) + , end_position(other.end_position) #endif { // check that the passed value is valid - other.assert_invariant(false); // cppcheck-suppress[accessForwarded] + other.assert_invariant(false); // invalidate payload other.m_data.m_type = value_t::null; @@ -2864,7 +2860,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec @tparam PointerType pointer type; must be a pointer to @ref array_t, @ref object_t, @ref string_t, @ref boolean_t, @ref number_integer_t, - @ref number_unsigned_t, or @ref number_float_t. + @ref number_unsigned_t, @ref number_float_t, or @ref binary_t. @return pointer to the internally stored JSON value if the requested pointer type @a PointerType fits to the JSON value; `nullptr` otherwise @@ -3030,8 +3026,106 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return *get_ptr(); } + /// @brief access the custom base class + /// @sa https://json.nlohmann.me/api/basic_json/as_base_class/ + json_base_class_t& as_base_class() noexcept + { + return static_cast(*this); + } + + /// @brief access the custom base class + /// @sa https://json.nlohmann.me/api/basic_json/as_base_class/ + const json_base_class_t& as_base_class() const noexcept + { + return static_cast(*this); + } + /// @} + private: + /// @brief look up @a key in the object held by @a j, for either constness of @a j + /// @note the single place that performs a (possibly transparent) object key lookup + template + static auto object_lookup(Self& j, KeyType&& key) + -> decltype(j.m_data.m_value.object->find(lookup_key(std::forward(key)))) + { + return j.m_data.m_value.object->find(lookup_key(std::forward(key))); + } + + /// @brief checked object element access used by the at() overloads taking a key + /// @throw type_error.304 if @a j is not an object + /// @throw out_of_range.403 if @a key is not found + template + static auto object_at(Self& j, KeyType&& key) + -> decltype((object_lookup(j, std::forward(key))->second)) + { + // at only works for objects + if (JSON_HEDLEY_UNLIKELY(!j.is_object())) + { + JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", j.type_name()), &j)); + } + + auto it = object_lookup(j, std::forward(key)); + if (it == j.m_data.m_value.object->end()) + { + // key is only forwarded into the lookup above: object_t::find() (a plain + // std::map or ordered_map) never moves from its argument, so key is still + // valid here regardless of whether KeyType was deduced as an rvalue reference + // NOLINTNEXTLINE(bugprone-use-after-move,hicpp-invalid-access-moved) + JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(key), "' not found"), &j)); + } + return it->second; + } + + /// @brief checked array element access used by the at() overloads taking an index + /// @throw type_error.304 if @a j is not an array + /// @throw out_of_range.401 if @a idx is out of range + template + static auto array_at(Self& j, size_type idx) + -> decltype((*j.m_data.m_value.array)[idx]) + { + // at only works for arrays + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", j.type_name()), &j)); + } + + if (JSON_HEDLEY_UNLIKELY(idx >= j.m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), &j)); + } + + return (*j.m_data.m_value.array)[idx]; + } + + /// @brief convert a null value to an empty container of type @a Container + /// @tparam Container array_t or object_t; any other type does not compile + template + void convert_null_to() + { + JSON_ASSERT(is_null()); + // create the container before touching the type, so a throwing + // allocation leaves this value as a valid null rather than a type + // tag with a dangling/null pointer behind it + set_container(create()); + assert_invariant(); + } + + /// @brief store a freshly created array and set the matching type + void set_container(array_t* array) noexcept + { + m_data.m_value.array = array; + m_data.m_type = value_t::array; + } + + /// @brief store a freshly created object and set the matching type + void set_container(object_t* object) noexcept + { + m_data.m_value.object = object; + m_data.m_type = value_t::object; + } + + public: //////////////////// // element access // //////////////////// @@ -3044,54 +3138,21 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/at/ reference at(size_type idx) { - // at only works for arrays - if (JSON_HEDLEY_UNLIKELY(!is_array())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) - { - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } - - return set_parent((*m_data.m_value.array)[idx]); + return set_parent(array_at(*this, idx)); } /// @brief access specified array element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ const_reference at(size_type idx) const { - // at only works for arrays - if (JSON_HEDLEY_UNLIKELY(!is_array())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) - { - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } - - return (*m_data.m_value.array)[idx]; + return array_at(*this, idx); } /// @brief access specified object element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ reference at(const typename object_t::key_type& key) { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(key); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", key, "' not found"), this)); - } - return set_parent(it->second); + return set_parent(object_at(*this, key)); } /// @brief access specified object element with bounds checking @@ -3100,36 +3161,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_usable_as_basic_json_key_type::value, int> = 0> reference at(KeyType && key) { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(std::forward(key)), "' not found"), this)); - } - return set_parent(it->second); + return set_parent(object_at(*this, std::forward(key))); } /// @brief access specified object element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ const_reference at(const typename object_t::key_type& key) const { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(key); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", key, "' not found"), this)); - } - return it->second; + return object_at(*this, key); } /// @brief access specified object element with bounds checking @@ -3138,18 +3177,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_usable_as_basic_json_key_type::value, int> = 0> const_reference at(KeyType && key) const { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(std::forward(key)), "' not found"), this)); - } - return it->second; + return object_at(*this, std::forward(key)); } /// @brief access specified array element @@ -3159,9 +3187,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value.array = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for arrays @@ -3227,9 +3253,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value.object = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for objects @@ -3249,7 +3273,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // const operator[] only works for objects if (JSON_HEDLEY_LIKELY(is_object())) { - auto it = m_data.m_value.object->find(key); + auto it = object_lookup(*this, key); JSON_ASSERT(it != m_data.m_value.object->end()); return it->second; } @@ -3280,9 +3304,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value.object = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for objects @@ -3304,7 +3326,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // const operator[] only works for objects if (JSON_HEDLEY_LIKELY(is_object())) { - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); + auto it = object_lookup(*this, std::forward(key)); JSON_ASSERT(it != m_data.m_value.object->end()); return it->second; } @@ -3324,6 +3346,36 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_c_string_uncvref::value, string_t, typename std::decay::type >; + /// @brief look up @a key for value(), the single place shared by all key-based overloads + /// @throw type_error.306 if this is not an object + /// @return a pointer to the found value, or `nullptr` if @a key was not found + template + const basic_json* value_member(KeyType&& key) const + { + // value only works for objects + if (JSON_HEDLEY_UNLIKELY(!is_object())) + { + JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + } + + const auto it = find(std::forward(key)); + return it != end() ? &*it : nullptr; + } + + /// @brief resolve @a ptr for value(), the single place shared by both json_pointer overloads + /// @throw type_error.306 if this is not an array or object + /// @return a pointer to the resolved value, or `nullptr` if @a ptr does not resolve + const basic_json* value_pointee(const json_pointer& ptr) const + { + // value only works for arrays and objects + if (JSON_HEDLEY_UNLIKELY(!is_structured())) + { + JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + } + + return ptr.get_checked_or_null(this); + } + public: // an integer literal 0 would otherwise convert to a null const char* and from there to key_type template::value, int> = 0> @@ -3337,20 +3389,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(const typename object_t::key_type& key, const ValueType& default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(key); - if (it != end()) - { - return it->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(key); + return found != nullptr ? found->template get() : default_value; } /// @brief access specified object element with default value @@ -3362,20 +3403,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(const typename object_t::key_type& key, ValueType && default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(key); - if (it != end()) - { - return it->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(key); + return found != nullptr ? found->template get() : std::forward(default_value); } /// @brief access specified object element with default value @@ -3388,23 +3418,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(KeyType && key, const ValueType& default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(std::forward(key)); - if (it != end()) - { - return it->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(std::forward(key)); + return found != nullptr ? found->template get() : default_value; } - /// @brief access specified object element via JSON Pointer with default value + /// @brief access specified object element with default value /// @sa https://json.nlohmann.me/api/basic_json/value/ template < class ValueType, class KeyType, class ReturnType = typename value_return_type::type, detail::enable_if_t < @@ -3415,20 +3434,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(KeyType && key, ValueType && default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(std::forward(key)); - if (it != end()) - { - return it->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(std::forward(key)); + return found != nullptr ? found->template get() : std::forward(default_value); } /// @brief access specified object element via JSON Pointer with default value @@ -3438,21 +3446,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(const json_pointer& ptr, const ValueType& default_value) const { - // value only works for arrays and objects - if (JSON_HEDLEY_LIKELY(is_structured())) - { - // If the pointer resolves to a value, return it. Otherwise, return - // 'default_value'. - const auto* res = ptr.get_checked_or_null(this); - if (JSON_HEDLEY_LIKELY(res != nullptr)) - { - return res->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If the pointer resolves to a value, return it. Otherwise, return + // 'default_value'. + const auto* found = value_pointee(ptr); + return found != nullptr ? found->template get() : default_value; } /// @brief access specified object element via JSON Pointer with default value @@ -3463,21 +3460,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(const json_pointer& ptr, ValueType && default_value) const { - // value only works for arrays and objects - if (JSON_HEDLEY_LIKELY(is_structured())) - { - // If the pointer resolves to a value, return it. Otherwise, return - // 'default_value'. - const auto* res = ptr.get_checked_or_null(this); - if (JSON_HEDLEY_LIKELY(res != nullptr)) - { - return res->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If the pointer resolves to a value, return it. Otherwise, return + // 'default_value'. + const auto* found = value_pointee(ptr); + return found != nullptr ? found->template get() : std::forward(default_value); } template < class ValueType, class BasicJsonType, detail::enable_if_t < @@ -3562,21 +3548,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(invalid_iterator::create(205, "iterator out of range", this)); } - if (is_string()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.string); - std::allocator_traits::deallocate(alloc, m_data.m_value.string, 1); - m_data.m_value.string = nullptr; - } - else if (is_binary()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.binary); - std::allocator_traits::deallocate(alloc, m_data.m_value.binary, 1); - m_data.m_value.binary = nullptr; - } - + m_data.m_value.destroy(m_data.m_type); + m_data.m_value = {}; m_data.m_type = value_t::null; assert_invariant(); break; @@ -3628,27 +3601,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::string: case value_t::binary: { - if (JSON_HEDLEY_LIKELY(!first.m_it.primitive_iterator.is_begin() - || !last.m_it.primitive_iterator.is_end())) + if (JSON_HEDLEY_UNLIKELY(!first.m_it.primitive_iterator.is_begin() + || !last.m_it.primitive_iterator.is_end())) { JSON_THROW(invalid_iterator::create(204, "iterators out of range", this)); } - if (is_string()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.string); - std::allocator_traits::deallocate(alloc, m_data.m_value.string, 1); - m_data.m_value.string = nullptr; - } - else if (is_binary()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.binary); - std::allocator_traits::deallocate(alloc, m_data.m_value.binary, 1); - m_data.m_value.binary = nullptr; - } - + m_data.m_value.destroy(m_data.m_type); + m_data.m_value = {}; m_data.m_type = value_t::null; assert_invariant(); break; @@ -3704,7 +3664,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - const auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); + const auto it = object_lookup(*this, std::forward(key)); if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); @@ -3781,7 +3741,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(key); + result.m_it.object_iterator = object_lookup(*this, key); } return result; @@ -3795,7 +3755,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(key); + result.m_it.object_iterator = object_lookup(*this, key); } return result; @@ -3811,7 +3771,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(lookup_key(std::forward(key))); + result.m_it.object_iterator = object_lookup(*this, std::forward(key)); } return result; @@ -3827,7 +3787,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(lookup_key(std::forward(key))); + result.m_it.object_iterator = object_lookup(*this, std::forward(key)); } return result; @@ -3858,7 +3818,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(const typename object_t::key_type& key) const { - return is_object() && m_data.m_value.object->find(key) != m_data.m_value.object->end(); + return is_object() && object_lookup(*this, key) != m_data.m_value.object->end(); } /// @brief check the existence of an element in a JSON object @@ -3868,7 +3828,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(KeyType && key) const { - return is_object() && m_data.m_value.object->find(lookup_key(std::forward(key))) != m_data.m_value.object->end(); + return is_object() && object_lookup(*this, std::forward(key)) != m_data.m_value.object->end(); } /// @brief check the existence of an element in a JSON object given a JSON pointer @@ -4233,9 +4193,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array (move semantics) @@ -4266,9 +4224,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array @@ -4298,9 +4254,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value = value_t::object; - assert_invariant(); + convert_null_to(); } // add the element to the object @@ -4354,9 +4308,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array (perfect forwarding) @@ -4379,9 +4331,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value = value_t::object; - assert_invariant(); + convert_null_to(); } // add the element to the array (perfect forwarding) @@ -4441,7 +4391,22 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/insert/ iterator insert(const_iterator pos, basic_json&& val) // NOLINT(performance-unnecessary-value-param) { - return insert(std::move(pos), val); + // insert only works for arrays + if (JSON_HEDLEY_LIKELY(is_array())) + { + // check if iterator pos fits to this JSON value + if (JSON_HEDLEY_UNLIKELY(pos.m_object != this)) + { + JSON_THROW(invalid_iterator::create(202, "iterator does not fit current value", this)); + } + + // moving into a local first keeps this safe even if val aliases + // an element of this array + basic_json tmp(std::move(val)); + return insert_iterator(pos, std::move(tmp)); + } + + JSON_THROW(type_error::create(309, detail::concat("cannot use insert() with ", type_name()), this)); } /// @brief inserts copies of element into array @@ -4617,14 +4582,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// value is an object; called first by both @ref update overloads void prepare_update() { - // implicitly convert a null value to an empty object; create the - // object before setting the type, so a throwing allocation leaves - // this value null + // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_value.object = create(); - m_data.m_type = value_t::object; - assert_invariant(); + convert_null_to(); } if (JSON_HEDLEY_UNLIKELY(!is_object())) @@ -4833,7 +4794,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(binary_t& other) // NOLINT(bugprone-exception-escape,cppcoreguidelines-noexcept-swap,performance-noexcept-swap) { - // swap only works for strings + // swap only works for binary values if (JSON_HEDLEY_LIKELY(is_binary())) { using std::swap; @@ -4849,7 +4810,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(typename binary_t::container_type& other) // NOLINT(bugprone-exception-escape) { - // swap only works for strings + // swap only works for binary values if (JSON_HEDLEY_LIKELY(is_binary())) { using std::swap; @@ -5350,7 +5311,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(detail::input_adapter(std::forward(i)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved,accessForwarded] + auto p = parser(detail::input_adapter(std::forward(i)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -5367,7 +5329,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(detail::input_adapter(std::move(first), std::move(last)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved] + auto p = parser(detail::input_adapter(std::move(first), std::move(last)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -5380,7 +5343,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(i.get(), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved] + auto p = parser(i.get(), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -5607,6 +5571,32 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #endif + private: + /*! + @brief shared implementation of the binary from_cbor/from_msgpack/ + from_ubjson/from_bjdata/from_bon8/from_bson overloads + + Building the @ref binary_reader as a named local, rather than as a + temporary that @ref detail::json_sax_dom_parser::parse is called on in + the same expression, avoids a false-positive cppcheck accessMoved + warning in each of the 16 callers. + */ + template + static basic_json from_binary_impl(InputAdapterType ia, const input_format_t format, + const bool strict, const bool allow_exceptions, + const error_handler_t error_handler = error_handler_t::keep, + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + { + basic_json result; + detail::json_sax_dom_parser sdp(result, allow_exceptions); + binary_reader reader(std::move(ia), format, error_handler); + if (!reader.sax_parse(&sdp, strict, tag_handler)) + { + result = value_t::discarded; + } + return result; + } + ////////////////////////////////////////// // binary serialization/deserialization // ////////////////////////////////////////// @@ -5617,78 +5607,87 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec public: /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static std::vector to_cbor(const basic_json& j) + static std::vector to_cbor(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_cbor(j); + vector_writer(result, error_handler).write_cbor(j); return result; } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static std::vector to_msgpack(const basic_json& j) + static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_msgpack(j); + vector_writer(result, error_handler).write_msgpack(j); return result; } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static std::vector to_ubjson(const basic_json& j, const bool use_size = false, - const bool use_type = false) + const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type); return result; } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a BJData serialization of a given JSON value @@ -5696,11 +5695,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_bjdata(const basic_json& j, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type, true, true, version); return result; } @@ -5708,42 +5708,47 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BJData serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static std::vector to_bson(const basic_json& j) + static std::vector to_bson(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_bson(j); + vector_writer(result, error_handler).write_bson(j); return result; } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BON8 serialization of a given JSON value @@ -5777,16 +5782,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(InputType&& i, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } /// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5797,16 +5796,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(IteratorType first, SentinelType last, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } template @@ -5827,15 +5820,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, error_handler_t::keep, tag_handler); } /// @brief create a JSON value from an input in MessagePack format @@ -5844,16 +5829,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5863,16 +5842,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } template @@ -5891,15 +5864,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::msgpack, strict, allow_exceptions); } /// @brief create a JSON value from an input in UBJSON format @@ -5908,16 +5873,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5927,16 +5886,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } template @@ -5955,15 +5908,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::ubjson, strict, allow_exceptions); } /// @brief create a JSON value from an input in BJData format @@ -5972,16 +5917,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5991,16 +5930,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BON8 format @@ -6011,14 +5944,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bon8, strict, allow_exceptions); } /// @brief create a JSON value from an input in BON8 format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -6030,14 +5956,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bon8, strict, allow_exceptions); } /// @brief create a JSON value from an input in BSON format @@ -6046,16 +5965,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -6065,16 +5978,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions, error_handler); } template @@ -6093,15 +6000,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::bson, strict, allow_exceptions); } /// @} diff --git a/include/nlohmann/ordered_map.hpp b/include/nlohmann/ordered_map.hpp index d1c247483..eb521114e 100644 --- a/include/nlohmann/ordered_map.hpp +++ b/include/nlohmann/ordered_map.hpp @@ -74,16 +74,43 @@ template , return *this; } +private: + /// @brief find the entry for @a key, for either constness of @a self + /// @note the single place that performs the linear key search + template + static auto find_impl(Self& self, KeyType&& key) -> decltype(self.begin()) + { + for (auto it = self.begin(); it != self.end(); ++it) + { + if (self.m_compare(it->first, key)) + { + return it; + } + } + return self.end(); + } + + /// @brief remove the entry @a it points to, preserving order + /// @note keys are not movable, so the tail is destroyed and re-constructed in place + void erase_at(iterator it) + { + for (auto next = it; ++next != this->end(); ++it) + { + it->~value_type(); // Destroy but keep allocation + new (&*it) value_type{std::move(*next)}; + } + Container::pop_back(); + } + +public: template::value, int> = 0> std::pair emplace(const key_type& key, V && t) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - return {it, false}; - } + return {it, false}; } append(key, std::forward(t)); return {std::prev(this->end()), true}; @@ -94,12 +121,10 @@ template , detail::is_constructible>::value, int> = 0> std::pair emplace(KeyType && key, V && t) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - return {it, false}; - } + return {it, false}; } append(std::forward(key), std::forward(t)); return {std::prev(this->end()), true}; @@ -131,75 +156,55 @@ template , T& at(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } template::value, int> = 0> T & at(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } const T& at(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } template::value, int> = 0> const T & at(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } size_type erase(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - // Since we cannot move const Keys, re-construct them in place - for (auto next = it; ++next != this->end(); ++it) - { - it->~value_type(); // Destroy but keep allocation - new (&*it) value_type{std::move(*next)}; - } - Container::pop_back(); - return 1; - } + erase_at(it); + return 1; } return 0; } @@ -208,19 +213,11 @@ template , detail::is_usable_as_key_type::value, int> = 0> size_type erase(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - // Since we cannot move const Keys, re-construct them in place - for (auto next = it; ++next != this->end(); ++it) - { - it->~value_type(); // Destroy but keep allocation - new (&*it) value_type{std::move(*next)}; - } - Container::pop_back(); - return 1; - } + erase_at(it); + return 1; } return 0; } @@ -285,80 +282,38 @@ template , size_type count(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return 1; - } - } - return 0; + return find_impl(*this, key) != this->end() ? 1 : 0; } template::value, int> = 0> size_type count(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return 1; - } - } - return 0; + return find_impl(*this, key) != this->end() ? 1 : 0; } iterator find(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } template::value, int> = 0> iterator find(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } const_iterator find(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } template::value, int> = 0> const_iterator find(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } std::pair insert( value_type&& value ) @@ -368,12 +323,10 @@ template , std::pair insert( const value_type& value ) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, value.first); + if (it != this->end()) { - if (m_compare(it->first, value.first)) - { - return {it, false}; - } + return {it, false}; } append(value); return {--this->end(), true}; diff --git a/nlohmann_json.natvis b/nlohmann_json.natvis index ed443145e..17029f1dd 100644 --- a/nlohmann_json.natvis +++ b/nlohmann_json.natvis @@ -455,6 +455,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -755,6 +815,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -995,6 +1115,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1175,6 +1355,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1295,6 +1535,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1355,6 +1655,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1595,6 +2015,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1775,6 +2255,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1895,6 +2435,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1955,6 +2555,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -2135,6 +2855,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -2255,6 +3035,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -2315,6 +3155,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -2435,6 +3395,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -2495,6 +3515,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -2555,6 +3695,186 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -2735,6 +4055,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -2855,6 +4235,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -2915,6 +4355,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3035,6 +4595,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3095,6 +4715,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3155,6 +4895,186 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3275,6 +5195,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3335,6 +5315,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3395,6 +5495,186 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3455,6 +5735,246 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3575,6 +6095,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3635,6 +6215,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3695,6 +6395,186 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3755,6 +6635,246 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3815,6 +6935,306 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -3875,4 +7295,424 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 47b085b32..beec7ec42 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -104,6 +104,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -140,14 +144,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -156,7 +166,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -276,104 +287,6 @@ #include // declval, pair -// #include -// __ _____ _____ _____ -// __| | __| | | | JSON for Modern C++ -// | | |__ | | | | | | version 3.12.0 -// |_____|_____|_____|_|___| https://github.com/nlohmann/json -// -// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann -// SPDX-License-Identifier: MIT - - - -#include - -// #include -// __ _____ _____ _____ -// __| | __| | | | JSON for Modern C++ -// | | |__ | | | | | | version 3.12.0 -// |_____|_____|_____|_|___| https://github.com/nlohmann/json -// -// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann -// SPDX-License-Identifier: MIT - - - -// #include - - -NLOHMANN_JSON_NAMESPACE_BEGIN -namespace detail -{ - -template struct make_void -{ - using type = void; -}; -template using void_t = typename make_void::type; - -} // namespace detail -NLOHMANN_JSON_NAMESPACE_END - - -NLOHMANN_JSON_NAMESPACE_BEGIN -namespace detail -{ - -// https://en.cppreference.com/w/cpp/experimental/is_detected -struct nonesuch -{ - nonesuch() = delete; - ~nonesuch() = delete; - nonesuch(nonesuch const&) = delete; - nonesuch(nonesuch const&&) = delete; - void operator=(nonesuch const&) = delete; - void operator=(nonesuch&&) = delete; -}; - -template class Op, - class... Args> -struct detector -{ - using value_t = std::false_type; - using type = Default; -}; - -template class Op, class... Args> -struct detector>, Op, Args...> -{ - using value_t = std::true_type; - using type = Op; -}; - -template class Op, class... Args> -using is_detected = typename detector::value_t; - -template class Op, class... Args> -struct is_detected_lazy : is_detected { }; - -template class Op, class... Args> -using detected_t = typename detector::type; - -template class Op, class... Args> -using detected_or = detector; - -template class Op, class... Args> -using detected_or_t = typename detected_or::type; - -template class Op, class... Args> -using is_detected_exact = std::is_same>; - -template class Op, class... Args> -using is_detected_convertible = - std::is_convertible, To>; - -} // namespace detail -NLOHMANN_JSON_NAMESPACE_END - // #include @@ -2552,10 +2465,12 @@ JSON_HEDLEY_DIAGNOSTIC_POP // libstdc++ < 11 has incomplete C++20 ranges (issue #4440) #elif defined(_GLIBCXX_RELEASE) && _GLIBCXX_RELEASE < 11 #define JSON_HAS_RANGES 0 - // libc++ < 16 has incomplete C++20 ranges (issue #4440) + // clang < 16 with libstdc++ does not implement the ranges customization + // points libstdc++ declares, so its C++20 ranges support is incomplete (issue #5161) #elif defined(__clang__) && !defined(__apple_build_version__) \ && __clang_major__ < 16 && defined(__GLIBCXX__) #define JSON_HAS_RANGES 0 + // libc++ < 16 has incomplete C++20 ranges (issue #4440) #elif defined(_LIBCPP_VERSION) && _LIBCPP_VERSION < 160000 #define JSON_HAS_RANGES 0 // nvcc CUDA 12.0/12.1 chokes on the enable_borrowed_range variable-template @@ -2570,6 +2485,18 @@ JSON_HEDLEY_DIAGNOSTIC_POP #endif #endif +// std::ranges view conversion (to_json/is_compatible_array_type_impl) additionally +// needs to be disabled on MinGW, whose std::ranges support is incomplete +// (issue #4916); this macro combines both conditions so the check and its +// reason are not duplicated at every use site. +#ifndef JSON_HAS_RANGE_VIEW_CONVERSION + #if JSON_HAS_RANGES && !defined(__MINGW32__) + #define JSON_HAS_RANGE_VIEW_CONVERSION 1 + #else + #define JSON_HAS_RANGE_VIEW_CONVERSION 0 + #endif +#endif + #ifndef JSON_HAS_STD_FORMAT #if defined(JSON_HAS_CPP_20) && defined(__cpp_lib_format) #define JSON_HAS_STD_FORMAT 1 @@ -2691,21 +2618,6 @@ JSON_HEDLEY_DIAGNOSTIC_POP -/*! -@brief function to wrap JSON_THROW_MACRO - there can be compilation errors about - there being no arguments to JSON_THROW that depend on template arguments - if this is not used to call JSON_THROW -*/ -template -void templated_json_throw(ExceptionType exception) -{ - JSON_THROW(exception); - - /* JSON_THROW(exception) discards exception and aborts - void cast needed to supress - compilation error if compiled with -Werror and Wunused-parameter */ - (void)exception; -} - /*! @brief macro to briefly define a mapping between an enum and JSON with exception on invalid input @@ -2726,7 +2638,7 @@ void templated_json_throw(ExceptionType exception) return ej_pair.first == e; \ }); \ if (it != std::end(m)) j = it->second; \ - else templated_json_throw(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE, nullptr)); \ + else ::nlohmann::detail::templated_json_throw(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE, nullptr)); \ } \ template \ inline void from_json(const BasicJsonType& j, ENUM_TYPE& e) \ @@ -2741,7 +2653,7 @@ void templated_json_throw(ExceptionType exception) return ej_pair.second == j; \ }); \ if (it != std::end(m)) e = it->first; \ - else templated_json_throw(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \ + else ::nlohmann::detail::templated_json_throw(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \ } // Ugly macros to avoid uglier copy-paste when specializing basic_json. They @@ -3286,30 +3198,6 @@ void templated_json_throw(ExceptionType exception) \ template \ using result_of_##std_name = decltype(std_name(std::declval()...)); \ - } \ - \ - namespace detail2 { \ - struct std_name##_tag \ - { \ - }; \ - \ - template \ - std_name##_tag std_name(T&&...); \ - \ - template \ - using result_of_##std_name = decltype(std_name(std::declval()...)); \ - \ - template \ - struct would_call_std_##std_name \ - { \ - static constexpr auto const value = ::nlohmann::detail:: \ - is_detected_exact::value; \ - }; \ - } /* namespace detail2 */ \ - \ - template \ - struct would_call_std_##std_name : detail2::would_call_std_##std_name \ - { \ } #ifndef JSON_USE_IMPLICIT_CONVERSIONS @@ -3851,6 +3739,31 @@ NLOHMANN_JSON_NAMESPACE_END // #include // #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +template struct make_void +{ + using type = void; +}; +template using void_t = typename make_void::type; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END // #include @@ -3922,7 +3835,7 @@ NLOHMANN_JSON_NAMESPACE_END NLOHMANN_JSON_NAMESPACE_BEGIN -NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin); +NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin) NLOHMANN_JSON_NAMESPACE_END @@ -3942,13 +3855,84 @@ NLOHMANN_JSON_NAMESPACE_END NLOHMANN_JSON_NAMESPACE_BEGIN -NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end); +NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end) NLOHMANN_JSON_NAMESPACE_END // #include // #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +// https://en.cppreference.com/w/cpp/experimental/is_detected +struct nonesuch +{ + nonesuch() = delete; + ~nonesuch() = delete; + nonesuch(nonesuch const&) = delete; + nonesuch(nonesuch const&&) = delete; + void operator=(nonesuch const&) = delete; + void operator=(nonesuch&&) = delete; +}; + +template class Op, + class... Args> +struct detector +{ + using value_t = std::false_type; + using type = Default; +}; + +template class Op, class... Args> +struct detector>, Op, Args...> +{ + using value_t = std::true_type; + using type = Op; +}; + +template class Op, class... Args> +using is_detected = typename detector::value_t; + +template class Op, class... Args> +struct is_detected_lazy : is_detected { }; + +template class Op, class... Args> +using detected_t = typename detector::type; + +template class Op, class... Args> +using detected_or = detector; + +template class Op, class... Args> +using detected_or_t = typename detected_or::type; + +template class Op, class... Args> +using is_detected_exact = std::is_same>; + +template class Op, class... Args> +using is_detected_convertible = + std::is_convertible, To>; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END // #include // __ _____ _____ _____ @@ -4314,6 +4298,13 @@ template struct conjunction : std::conditional(B::value), conjunction, B>::type {}; +// https://en.cppreference.com/w/cpp/types/disjunction +template struct disjunction : std::false_type { }; +template struct disjunction : B { }; +template +struct disjunction +: std::conditional(B::value), B, disjunction>::type {}; + // https://en.cppreference.com/w/cpp/types/negation template struct negation : std::integral_constant < bool, !B::value > { }; @@ -4508,9 +4499,7 @@ template struct is_range_view_optional_type> : std: template struct is_range_view_optional_type : std::false_type {}; #endif -// std::ranges does not work properly on MinGW due to incomplete C++20 support -// see https://github.com/nlohmann/json/issues/4916 -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION // SafeToCheck guards against types that trigger circular constraints when // std::ranges::view is evaluated on GCC 12 / libstdc++ 12: @@ -4549,7 +4538,7 @@ struct is_compatible_array_type_impl < // filter_view) can match BOTH this iterator-based specialization AND the view-based one // below, causing ambiguity. Exclude views here so the two specializations are mutually // exclusive: this one handles plain iterable containers, the other handles views. -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif >> @@ -4559,7 +4548,7 @@ struct is_compatible_array_type_impl < range_value_t>::value; }; -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template struct is_compatible_array_type_impl < BasicJsonType, CompatibleArrayType, @@ -4635,7 +4624,6 @@ struct is_compatible_integer_type_impl < std::is_integral::value&& !std::is_same::value >> { - // is there an assert somewhere on overflows? using RealLimits = std::numeric_limits; using CompatibleLimits = std::numeric_limits; @@ -4863,20 +4851,7 @@ struct has_capacity : std::integral_constant -struct is_ordered_map -{ - using one = char; - - struct two - { - char x[2]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - }; - - template static one test( decltype(&C::capacity) ) ; - template static two test(...); - - enum { value = sizeof(test(nullptr)) == sizeof(char) }; // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg,cppcoreguidelines-use-enum-class) -}; +struct is_ordered_map : has_capacity {}; // to avoid useless casts (see https://github.com/nlohmann/json/issues/2893#issuecomment-889152324) template < typename T, typename U, enable_if_t < !std::is_same::value, int > = 0 > @@ -4900,10 +4875,8 @@ using all_signed = conjunction...>; template using all_unsigned = conjunction...>; -// there's a disjunction trait in another PR; replace when merged template -using same_sign = std::integral_constant < bool, - all_signed::value || all_unsigned::value >; +using same_sign = disjunction, all_unsigned>; template using never_out_of_range = std::integral_constant < bool, @@ -5084,7 +5057,6 @@ inline std::size_t concat_length(const char /*c*/, const Args& ... rest) template inline std::size_t concat_length(const char* cstr, const Args& ... rest) { - // cppcheck-suppress ignoredReturnValue return ::strlen(cstr) + concat_length(rest...); } @@ -5452,6 +5424,27 @@ class other_error : public exception other_error(int id_, const char* what_arg) : exception(id_, what_arg) {} }; +/*! +@brief helper function to call JSON_THROW from a template +@note JSON_THROW is a macro that, depending on the JSON_THROW_USER / + JSON_TRY_USER / JSON_NOEXCEPTION configuration, may expand to code + that does not reference its argument (e.g. `std::abort()`), which + would trigger a compilation error if the argument's type depends on + a template parameter that is otherwise unused. Wrapping the call in + a templated function avoids this and gives the compiler a single + place to see the (possibly unused) parameter. +*/ +template +void templated_json_throw(ExceptionType exception) +{ + JSON_THROW(exception); + + // JSON_THROW may expand to code that discards its argument (e.g. when + // exceptions are disabled) - the cast below avoids an unused-parameter + // warning with -Werror in that case + (void)exception; +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -5521,63 +5514,6 @@ NLOHMANN_JSON_NAMESPACE_END // #include -// #include - - -// #include - - -NLOHMANN_JSON_NAMESPACE_BEGIN -namespace detail -{ -#ifdef JSON_HAS_CPP_17 - -template -struct cxpr_or_impl : std::integral_constant < bool, (Booleans || ...) > {}; - -template -struct cxpr_and_impl : std::integral_constant < bool, (Booleans &&...) > {}; - -#else - -template -struct cxpr_or_impl : std::false_type {}; - -template -struct cxpr_or_impl : std::true_type {}; - -template -struct cxpr_or_impl : cxpr_or_impl {}; - -template -struct cxpr_and_impl : std::true_type {}; - -template -struct cxpr_and_impl : cxpr_and_impl {}; - -template -struct cxpr_and_impl : std::false_type {}; - -#endif - -template -struct cxpr_not : std::integral_constant < bool, !Boolean::value > {}; - -template -struct cxpr_or : cxpr_or_impl {}; - -template -struct cxpr_or_c : cxpr_or_impl {}; - -template -struct cxpr_and : cxpr_and_impl {}; - -template -struct cxpr_and_c : cxpr_and_impl {}; - -} // namespace detail -NLOHMANN_JSON_NAMESPACE_END - // #include // #include @@ -5763,62 +5699,29 @@ inline void from_json(const BasicJsonType& j, std::valarray& l) }); } +// element is not itself a C array: read it directly +template +auto from_json_c_array_element(const BasicJsonType& j, T& e) +-> decltype(e = j.template get(), void()) +{ + e = j.template get(); +} + +// element is itself a C array: recurse one dimension at a time, so any rank is supported template -auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) +void from_json_c_array_element(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { for (std::size_t i = 0; i < N; ++i) { - arr[i] = j.at(i).template get(); + from_json_c_array_element(j.at(i), arr[i]); } } -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) +template +auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) +-> decltype(j.template get::type>(), void()) { - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - arr[i1][i2] = j.at(i1).at(i2).template get(); - } - } -} - -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) -{ - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - for (std::size_t i3 = 0; i3 < N3; ++i3) - { - arr[i1][i2][i3] = j.at(i1).at(i2).at(i3).template get(); - } - } - } -} - -template -auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3][N4]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) --> decltype(j.template get(), void()) -{ - for (std::size_t i1 = 0; i1 < N1; ++i1) - { - for (std::size_t i2 = 0; i2 < N2; ++i2) - { - for (std::size_t i3 = 0; i3 < N3; ++i3) - { - for (std::size_t i4 = 0; i4 < N4; ++i4) - { - arr[i1][i2][i3][i4] = j.at(i1).at(i2).at(i3).at(i4).template get(); - } - } - } - } + from_json_c_array_element(j, arr); } template @@ -5838,20 +5741,33 @@ auto from_json_array_impl(const BasicJsonType& j, std::array& arr, } } +// reserve() is called through this pair (modeled on from_json_object_reserve) +// so from_json_array_impl below has a single body for both ConstructibleArrayType +// that support reserve() and those that don't. +template +auto from_json_array_reserve(ConstructibleArrayType& arr, typename ConstructibleArrayType::size_type size, priority_tag<1> /*unused*/) +-> decltype(arr.reserve(size), void()) +{ + arr.reserve(size); +} + +template +void from_json_array_reserve(ConstructibleArrayType& /*arr*/, std::size_t /*size*/, priority_tag<0> /*unused*/) +{} + template::value, int> = 0> auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, priority_tag<1> /*unused*/) -> decltype( - arr.reserve(std::declval()), j.template get(), void()) { using std::end; ConstructibleArrayType ret; - ret.reserve(j.size()); + from_json_array_reserve(ret, j.size(), priority_tag<1> {}); std::transform(j.begin(), j.end(), std::inserter(ret, end(ret)), [](const BasicJsonType & i) { @@ -5862,27 +5778,6 @@ auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, p arr = std::move(ret); } -template::value, - int> = 0> -inline void from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, - priority_tag<0> /*unused*/) -{ - using std::end; - - ConstructibleArrayType ret; - std::transform( - j.begin(), j.end(), std::inserter(ret, end(ret)), - [](const BasicJsonType & i) - { - // get() returns *this, this won't call a from_json - // method when value_type is BasicJsonType - return i.template get(); - }); - arr = std::move(ret); -} - template < typename BasicJsonType, typename ConstructibleArrayType, enable_if_t < is_constructible_array_type::value&& @@ -5985,9 +5880,7 @@ inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj) } // overload for arithmetic types, not chosen for basic_json template arguments -// (BooleanType, etc.); note: Is it really necessary to provide explicit -// overloads for boolean_t etc. in case of a custom BooleanType which is not -// an arithmetic type? +// (BooleanType, etc.) template < typename BasicJsonType, typename ArithmeticType, enable_if_t < std::is_arithmetic::value&& @@ -6083,7 +5976,7 @@ inline void from_json_tuple_impl(const BasicJsonType& j, std::pair& p, p template std::tuple from_json_tuple_impl(const BasicJsonType& j, identity_tag> /*unused*/, priority_tag<2> /*unused*/) { - static_assert(cxpr_and>, is_compatible_reference_type>...>::value, + static_assert(conjunction>, is_compatible_reference_type>...>::value, "Can not return a tuple containing references to types not contained in a Json, try Json::get_to()"); return from_json_tuple_impl_base<1, Args...>(j, index_sequence_for {}); } @@ -6106,10 +5999,10 @@ auto from_json(const BasicJsonType& j, TupleRelated&& t) return from_json_tuple_impl(j, std::forward(t), priority_tag<3> {}); } -template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator, - typename = enable_if_t < !std::is_constructible < - typename BasicJsonType::string_t, Key >::value >> -inline void from_json(const BasicJsonType& j, std::map& m) +// shared body for std::map/std::unordered_map with a non-string Key: both +// containers are read from an array of [key, value] pairs the same way +template +void from_json_pair_array_to_map(const BasicJsonType& j, MapType& m) { if (JSON_HEDLEY_UNLIKELY(!j.is_array())) { @@ -6122,33 +6015,29 @@ inline void from_json(const BasicJsonType& j, std::map(), p.at(1).template get()); + m.emplace(p.at(0).template get(), p.at(1).template get()); } } +template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator, + typename = enable_if_t < !std::is_constructible < + typename BasicJsonType::string_t, Key >::value >> +void from_json(const BasicJsonType& j, std::map& m) +{ + from_json_pair_array_to_map(j, m); +} + template < typename BasicJsonType, typename Key, typename Value, typename Hash, typename KeyEqual, typename Allocator, typename = enable_if_t < !std::is_constructible < typename BasicJsonType::string_t, Key >::value >> -inline void from_json(const BasicJsonType& j, std::unordered_map& m) +void from_json(const BasicJsonType& j, std::unordered_map& m) { - if (JSON_HEDLEY_UNLIKELY(!j.is_array())) - { - JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); - } - m.clear(); - for (const auto& p : j) - { - if (JSON_HEDLEY_UNLIKELY(!p.is_array())) - { - JSON_THROW(type_error::create(302, concat("type must be array, but is ", p.type_name()), &p)); - } - m.emplace(p.at(0).template get(), p.at(1).template get()); - } + from_json_pair_array_to_map(j, m); } #if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM -// Workaround for MSVC 19.51 (and possibly later): in large in large cpp files, the compiler may fail to resolve with generic has_from_json (issue #4996) +// Workaround for MSVC 19.51 (and possibly later): in large cpp files, the compiler may fail to resolve with generic has_from_json (issue #4996) template struct has_from_json : std::true_type {}; @@ -6274,6 +6163,59 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// how to treat decoding errors +/// +/// @ref basic_json::dump uses this to decide what to do with ill-formed +/// UTF-8 while escaping a string, and the binary writers (@ref +/// basic_json::to_cbor, @ref basic_json::to_ubjson, @ref +/// basic_json::to_bjdata, @ref basic_json::to_bson) use it the same way for +/// string values and object keys. The binary readers (@ref +/// basic_json::from_cbor, @ref basic_json::from_msgpack, @ref +/// basic_json::from_ubjson, @ref basic_json::from_bjdata, @ref +/// basic_json::from_bson) use it to decide whether to check text strings +/// and object keys for well-formed UTF-8 at all, since none of those +/// formats requires a decoder to do so. +enum class error_handler_t +{ + strict, ///< throw a type_error/parse_error exception in case of invalid UTF-8 + replace, ///< replace invalid UTF-8 sequences with U+FFFD + ignore, ///< ignore invalid UTF-8 sequences + keep ///< keep invalid UTF-8 sequences unchanged +}; + +/// the default error handler of the CBOR, UBJSON, BJData, and BSON writers: +/// error_handler_t::strict if JSON_STRICT_BINARY_UTF8 is enabled, otherwise +/// error_handler_t::keep (the behavior before version 3.13.0) +constexpr error_handler_t binary_writer_default_error_handler() noexcept +{ +#if JSON_STRICT_BINARY_UTF8 + return error_handler_t::strict; +#else + return error_handler_t::keep; +#endif +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -6375,13 +6317,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally written by Björn Hoehrmann. See http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. -The library checks UTF-8 well-formedness (RFC 3629, section 4) in four +The library checks UTF-8 well-formedness (RFC 3629, section 4) in three places, which differ in speed, diagnostics, and how they read the input: -- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in - strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR, - MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in - text strings at decode time). +- decode() below: the serializer, to escape and, in strict mode, reject + ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON, + UBJSON and BJData readers do not use it: none of those specs requires a + decoder to reject ill-formed UTF-8 in text strings, so the readers keep + the bytes as is and leave the check to dump() and the binary writers. - the per-lead-byte switch in lexer::scan_string(): JSON text, with a diagnostic for each kind of error. - validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's @@ -6437,19 +6380,19 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std: } /*! -@brief check whether a string consists solely of valid UTF-8 +@brief check a string for well-formed UTF-8 (RFC 3629, section 4) -Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text -strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the -MessagePack/BSON specifications all require text strings to be UTF-8), so -that malformed input is caught immediately instead of only surfacing later -as a type_error.316 when the resulting value is dumped. +Used by the binary readers (CBOR, MessagePack, UBJSON, BJData, BSON) when an +@ref error_handler_t other than `keep` is requested for a text string value +or object key: none of those formats requires a decoder to reject ill-formed +UTF-8 on its own, so the check is opt-in there, unlike the JSON lexer and the +serializer's @ref decode -based escaping, which always run it. @param[in] s the string to check -@param[in] first index of the first byte to check; the bytes before it are - assumed to have been validated already and to end on a - code point boundary -@return whether @a s (from index @a first on) is valid UTF-8 +@param[in] first the index to start checking at +@return whether `s.substr(first)` is well-formed UTF-8 + +@sa @ref decode */ template inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept @@ -6470,76 +6413,99 @@ inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noex } /*! -@brief append U+FFFD REPLACEMENT CHARACTER, encoded in UTF-8 -@param[in,out] s the string to append to +@brief sanitize a string with ill-formed UTF-8 for @ref error_handler_t::replace or @ref error_handler_t::ignore + +Replaces every maximal ill-formed subsequence with U+FFFD (`replace`) or +drops it (`ignore`), using exactly the same boundaries @ref +serializer::dump_escaped_impl uses while escaping a string: a byte that does +not extend the sequence started by the previous byte(s) is reread as the +start of a new one, instead of being swallowed along with them. + +@pre @a error_handler is @ref error_handler_t::replace or @ref error_handler_t::ignore +@note Well-formed input is copied through unchanged, including bytes (e.g. + control characters or quotes) that @ref serializer::dump_escaped_impl + would itself escape; this function only concerns itself with + well-formedness, not with producing valid JSON text. + +@param[in] s the string to sanitize +@param[in] error_handler @ref error_handler_t::replace or @ref error_handler_t::ignore + +@return @a s with every ill-formed subsequence replaced or removed + +@sa @ref decode */ template -inline void append_replacement_character(StringType& s) +inline StringType sanitize_utf8(const StringType& s, const error_handler_t error_handler) { - s.push_back(static_cast(0xEFu)); - s.push_back(static_cast(0xBFu)); - s.push_back(static_cast(0xBDu)); -} + JSON_ASSERT(error_handler == error_handler_t::replace || error_handler == error_handler_t::ignore); -/*! -@brief replace ill-formed UTF-8 with U+FFFD REPLACEMENT CHARACTER + StringType result; + result.reserve(s.size()); -Each maximal subpart of an ill-formed sequence becomes one U+FFFD, as the -Unicode Standard recommends (Section 3.9, "U+FFFD Substitution of Maximal -Subparts"), and as the parser for JSON text does when it recovers from errors. - -@param[in,out] s the string to repair -@param[in] first index of the first byte to repair; the bytes before it are - assumed to be valid UTF-8 that ends on a code point boundary -*/ -template -inline void replace_invalid_utf8(StringType& s, const std::size_t first = 0) -{ - StringType result = s; - result.resize(first); - - std::uint8_t state = UTF8_ACCEPT; std::uint32_t codepoint = 0; - // the first byte of the sequence being decoded - std::size_t sequence_start = first; + std::uint8_t state = UTF8_ACCEPT; + // length of result after the last accepted code point + std::size_t result_len_after_last_accept = 0; + // whether bytes of an as yet unresolved sequence were already appended + bool pending = false; - std::size_t i = first; - while (i < s.size()) + for (std::size_t i = 0; i < s.size(); ++i) { switch (decode(state, codepoint, static_cast(s[i]))) { - case UTF8_ACCEPT: - for (++i; sequence_start < i; ++sequence_start) - { - result.push_back(s[sequence_start]); - } + case UTF8_ACCEPT: // decode found a well-formed code point + { + result.push_back(s[i]); + result_len_after_last_accept = result.size(); + pending = false; break; + } - case UTF8_REJECT: - append_replacement_character(result); - // the byte that made the sequence ill-formed begins the next - // one, unless it began this one - if (i == sequence_start) + case UTF8_REJECT: // decode found an ill-formed byte + { + // in case we saw this byte for the first time, read it again, + // because it may be fine for itself, just not for the + // sequence that came before it + if (pending) { - ++i; + --i; } + + // drop the bytes of the ill-formed sequence buffered below + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + result_len_after_last_accept = result.size(); + } + + pending = false; state = UTF8_ACCEPT; - sequence_start = i; break; + } - default: // in the middle of a sequence - ++i; + default: // decode found yet incomplete multibyte code point + { + result.push_back(s[i]); + pending = true; break; + } } } - // a sequence that the string ends in the middle of + // the string ended with an incomplete sequence if (state != UTF8_ACCEPT) { - append_replacement_character(result); + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + } } - s = std::move(result); + return result; } } // namespace detail @@ -6918,7 +6884,7 @@ struct external_constructor template < typename BasicJsonType, typename CompatibleArrayType, enable_if_t < !std::is_same::value -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif , int > = 0 > @@ -6962,9 +6928,7 @@ struct external_constructor j.assert_invariant(); } - // std::ranges does not work properly on MinGW due to incomplete C++20 support - // see https://github.com/nlohmann/json/issues/4916 -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template>::value, int> = 0> static void construct(BasicJsonType& j, CompatibleArrayType && arr) @@ -7124,7 +7088,7 @@ template < typename BasicJsonType, typename CompatibleArrayType, !std::is_same::value&& !is_compatible_binary_type::value&& !is_basic_json::value -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION && !is_compatible_range_view::value #endif , @@ -7134,7 +7098,7 @@ inline void to_json(BasicJsonType& j, const CompatibleArrayType& arr) external_constructor::construct(j, arr); } -#if JSON_HAS_RANGES && !defined(__MINGW32__) +#if JSON_HAS_RANGE_VIEW_CONVERSION template < typename BasicJsonType, typename T, enable_if_t < is_compatible_range_view>::value && !is_compatible_string_type>::value @@ -8222,8 +8186,10 @@ struct wide_string_input_helper } else { - // get the current character - const auto wc = input.get_character(); + // get the current character; converted to an unsigned type so that + // a negative unit (wint_t is signed on some platforms) is not + // mistaken for an ASCII character or for EOF + const auto wc = static_cast(input.get_character()); if (wc <= 0x10FFFF) { @@ -8291,9 +8257,11 @@ struct wide_string_input_helper bool valid_pair = false; if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty())) { - const auto wc2 = static_cast(input.get_character()); + // only consume the next unit if it completes the pair + const auto wc2 = static_cast(*input.current); if (0xDC00 <= wc2 && wc2 <= 0xDFFF) { + input.get_character(); const auto charcode = 0x10000u + (((static_cast(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); utf8_bytes_filled = 0; encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) @@ -8306,7 +8274,8 @@ struct wide_string_input_helper if (!valid_pair) { - utf8_bytes[0] = static_cast::int_type>(wc); + // emit a byte that is never valid UTF-8 (see the UTF-32 case) + utf8_bytes[0] = 0xFF; utf8_bytes_filled = 1; } } @@ -8515,7 +8484,7 @@ struct container_input_adapter_factory< ContainerType, { // container is forwarded twice on purpose: the resulting begin/end // iterator types must match adapter_type, computed the same way - // NOLINTNEXTLINE(bugprone-use-after-move) + // NOLINTNEXTLINE(bugprone-use-after-move,hicpp-invalid-access-moved) return input_adapter(begin(std::forward(container)), end(std::forward(container))); } }; @@ -14116,7 +14085,7 @@ NLOHMANN_JSON_NAMESPACE_END -#include // size_t +#include // size_t #include // declval #include // string @@ -14181,37 +14150,6 @@ using parse_error_function_t = decltype(std::declval().parse_error( std::declval(), std::declval(), std::declval())); -template -struct is_sax -{ - private: - static_assert(is_basic_json::value, - "BasicJsonType must be of type basic_json<...>"); - - using number_integer_t = typename BasicJsonType::number_integer_t; - using number_unsigned_t = typename BasicJsonType::number_unsigned_t; - using number_float_t = typename BasicJsonType::number_float_t; - using string_t = typename BasicJsonType::string_t; - using binary_t = typename BasicJsonType::binary_t; - using exception_t = typename BasicJsonType::exception; - - public: - static constexpr bool value = - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value && - is_detected_exact::value; -}; - template struct is_sax_static_asserts { @@ -14231,8 +14169,6 @@ struct is_sax_static_asserts "Missing/invalid function: bool null()"); static_assert(is_detected_exact::value, "Missing/invalid function: bool boolean(bool)"); - static_assert(is_detected_exact::value, - "Missing/invalid function: bool boolean(bool)"); static_assert( is_detected_exact::value, @@ -14271,6 +14207,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -14373,8 +14311,16 @@ class binary_reader @brief create a binary reader @param[in] adapter input adapter to read from + @param[in] format the binary format to parse + @param[in] error_handler how to treat text strings and object keys that + are not well-formed UTF-8; none of the supported formats + requires a decoder to reject those, so the default is to + @ref error_handler_t::keep them unchanged, as every binary + reader did before this parameter existed */ - explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json) noexcept : ia(std::move(adapter)), input_format(format) + explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json, + const error_handler_t error_handler = error_handler_t::keep) noexcept + : ia(std::move(adapter)), input_format(format), error_handler(error_handler) { (void)detail::is_sax_static_asserts {}; } @@ -14837,7 +14783,7 @@ class binary_reader { if (get_bson_cstr_bulk(result, std::integral_constant {})) { - return true; + return check_string_utf8(result, "key"); } auto out = std::back_inserter(result); @@ -14850,7 +14796,7 @@ class binary_reader } if (current == 0x00) { - return true; + return check_string_utf8(result, "key"); } *out++ = static_cast(current); } @@ -14936,7 +14882,7 @@ class binary_reader return current != char_traits::eof() && repair_requested(); } - return true; + return check_string_utf8(result, "string"); } /*! @@ -15596,7 +15542,7 @@ class binary_reader @return whether string creation completed */ - bool get_cbor_string(string_t& result) + bool get_cbor_string(string_t& result, const char* context = "string") { // number of indefinite-length strings that have been opened and not // closed yet. RFC 8949, Section 3.2.3 does not permit nesting them, @@ -15626,7 +15572,7 @@ class binary_reader { if (--open == 0) { - return true; + return check_string_utf8(result, context); } get(); continue; @@ -15639,7 +15585,7 @@ class binary_reader if (open == 0) { - return true; + return check_string_utf8(result, context); } get(); @@ -15664,7 +15610,7 @@ class binary_reader // EOF and major type 3 (text string) are left to get_cbor_string if (current == char_traits::eof() || (static_cast(current) & 0xE0u) == 0x60u) { - return get_cbor_string(result); + return get_cbor_string(result, "key"); } const char* found = nullptr; @@ -16577,7 +16523,7 @@ class binary_reader @return whether string creation completed */ - bool get_msgpack_string(string_t& result) + bool get_msgpack_string(string_t& result, const char* context = "string") { if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "string"))) { @@ -16620,25 +16566,25 @@ class binary_reader case 0xBE: case 0xBF: { - return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result); + return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result) && check_string_utf8(result, context); } case 0xD9: // str 8 { std::uint8_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDA: // str 16 { std::uint16_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDB: // str 32 { std::uint32_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } default: @@ -16717,7 +16663,7 @@ class binary_reader // byte 0xC1 are left to get_msgpack_string if (current == char_traits::eof()) { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } if (current <= 0x7F || current >= 0xE0) { @@ -16733,7 +16679,7 @@ class binary_reader } else { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } break; } @@ -17130,7 +17076,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, true, "key") || !sax->key(key))) { return false; } @@ -17152,7 +17098,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false, "key") || !sax->key(key))) { return false; } @@ -17220,7 +17166,7 @@ class binary_reader @return whether string creation completed */ - bool get_ubjson_string(string_t& result, const bool get_char = true) + bool get_ubjson_string(string_t& result, const bool get_char = true, const char* context = "string") { if (get_char) { @@ -17241,31 +17187,31 @@ class binary_reader case 'U': { std::uint8_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'i': { std::int8_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'I': { std::int16_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'l': { std::int32_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'L': { std::int64_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'u': @@ -17275,7 +17221,7 @@ class binary_reader break; } std::uint16_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'm': @@ -17285,7 +17231,7 @@ class binary_reader break; } std::uint32_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'M': @@ -17295,7 +17241,7 @@ class binary_reader break; } std::uint64_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } default: @@ -17800,10 +17746,9 @@ class binary_reader { return false; } - // when recovering, the character becomes U+FFFD, as an - // invalid byte in a string does - string_t replacement; - append_replacement_character(replacement); + // when recovering, the character becomes U+FFFD, as with + // error_handler_t::replace + string_t replacement = sanitize_utf8(string_t(1, static_cast(current)), error_handler_t::replace); return sax->string(replacement); } string_t s(1, static_cast(current)); @@ -18974,34 +18919,58 @@ class binary_reader const NumberType len, string_t& result) { - // get_bytes() appends to result, and CBOR indefinite-length strings - // collect all their chunks in the same result; validating only the - // newly read bytes keeps the check linear in the input size - const std::size_t old_size = result.size(); - if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) + // Strings are taken as is by default: none of CBOR (RFC 8949 §3.1 + // leaves the choice to the decoder), MessagePack (whose spec + // explicitly allows a str object to contain an invalid byte + // sequence), UBJSON, BJData, or BSON requires a decoder to reject + // ill-formed UTF-8. Checking (and, with @ref error_handler_t::strict, + // rejecting, or with `replace`/`ignore`, sanitizing) is opt-in via + // @ref error_handler, applied once the whole string (all chunks of + // an indefinite-length CBOR string included) has been assembled, by + // @ref check_string_utf8 at the call site. + return get_bytes(format, len, "string", result); + } + + /*! + @brief validate a decoded text string (value or object key) against @ref error_handler + + None of the binary formats requires a decoder to reject ill-formed UTF-8 + in a text string (see @ref get_string), so by default + (@ref error_handler_t::keep) this does nothing. A stricter + @ref error_handler opts into the same well-formedness check @ref + serializer::dump_escaped_impl applies when dumping a string: + @ref error_handler_t::strict rejects ill-formed input with + parse_error.113 (honoring `allow_exceptions` via @a sax), while + @ref error_handler_t::replace / @ref error_handler_t::ignore sanitize + @a result in place, using the exact same rules. + + @param[in,out] result the already assembled string to check + @param[in] context further context information (for diagnostics) + @return whether @a result is acceptable (always true for `keep`) + */ + bool check_string_utf8(string_t& result, const char* context) + { + if (error_handler == error_handler_t::keep || is_valid_utf8(result)) { - return false; + return true; } - // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications - // all require text strings to be valid UTF-8; reject anything else - // right here so malformed input is caught at decode time instead of - // only surfacing later as a type_error.316 when the value is dumped - // (which would defeat allow_exceptions=false / strict discarding). - if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) + if (error_handler == error_handler_t::strict) { - static_cast(report_error(chars_read, get_token_string(), - parse_error::create(113, chars_read, - exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr))); + auto last_token = get_token_string(); + static_cast(report_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format, "invalid string: ill-formed UTF-8 byte", context), nullptr))); + // when recovering, the string is repaired as with + // error_handler_t::replace if (!repair_requested()) { return false; } - // when recovering, each ill-formed sequence becomes U+FFFD, as it - // does in JSON text - replace_invalid_utf8(result, old_size); + result = sanitize_utf8(result, error_handler_t::replace); + return true; } + result = sanitize_utf8(result, error_handler); return true; } @@ -19465,6 +19434,9 @@ class binary_reader /// input format const input_format_t input_format = input_format_t::json; + /// how to treat text strings/object keys that are not well-formed UTF-8 + const error_handler_t error_handler = error_handler_t::keep; + /// the SAX parser json_sax_t* sax = nullptr; @@ -20999,9 +20971,11 @@ class iter_impl // NOLINT(cppcoreguidelines-special-member-functions,hicpp-speci static_assert(is_basic_json::type>::value, "iter_impl only accepts (const) basic_json"); // superficial check for the LegacyBidirectionalIterator named requirement - static_assert(std::is_base_of::value - && std::is_base_of::iterator_category>::value, - "basic_json iterator assumes array and object type iterators satisfy the LegacyBidirectionalIterator named requirement."); + // note: only array_t::iterator is checked here; object_t::iterator may be + // a forward-only iterator as long as reverse iteration and operator-- + // are never used on it + static_assert(std::is_base_of::iterator_category>::value, + "basic_json iterator assumes array type iterators satisfy the LegacyBidirectionalIterator named requirement."); public: /// The std::iterator class template (used as a base class to provide typedefs) is deprecated in C++17. @@ -22143,6 +22117,72 @@ class json_pointer } private: + /*! + @brief result of @ref parse_array_index + + @ref array_index maps each value to the corresponding parse_error/out_of_range + exception; @ref contains and @ref get_checked_or_null, which must not throw for + an out-of-range or unrepresentable index, switch on it directly instead. + */ + enum class array_index_status + { + ok, ///< @a s is a valid, representable array index + leading_zero, ///< @a s begins with '0' but has more than one character + not_a_number, ///< @a s does not begin with a digit + unresolved, ///< @a s could not be converted to an integer + exceeds_size_type ///< @a s converts to an integer that exceeds size_type + }; + + /*! + @param[in] s reference token to be converted into an array index + @param[out] idx the integer representation of @a s if @ref array_index_status::ok + is returned; left unchanged otherwise + + @return whether @a s is a valid array index, and if not, why + + @note this function never throws; @ref array_index and the callers that must not + throw (@ref contains, @ref get_checked_or_null) build on it instead of each + re-implementing the RFC 6901 digit rules and the @a size_type range check + */ + template + static array_index_status parse_array_index(const string_t& s, typename BasicJsonType::size_type& idx) noexcept + { + using size_type = typename BasicJsonType::size_type; + + // error condition (cf. RFC 6901, Sect. 4) + if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0')) + { + return array_index_status::leading_zero; + } + + // error condition (cf. RFC 6901, Sect. 4) + if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9'))) + { + return array_index_status::not_a_number; + } + + const char* p = s.data(); + char* p_end = nullptr; // NOLINT(misc-const-correctness) + errno = 0; // strtoull doesn't reset errno + const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) + if (p == p_end // invalid input or empty string + || errno == ERANGE // out of range + || JSON_HEDLEY_UNLIKELY(static_cast(p_end - p) != s.size())) // incomplete read + { + return array_index_status::unresolved; + } + + // the index does not fit into size_type; on 64-bit platforms this is + // only SIZE_MAX itself (see #2203 and #5395) + if (res >= static_cast((std::numeric_limits::max)())) // NOLINT(runtime/int) + { + return array_index_status::exceeds_size_type; + } + + idx = static_cast(res); + return array_index_status::ok; + } + /*! @param[in] s reference token to be converted into an array index @@ -22156,39 +22196,23 @@ class json_pointer template static typename BasicJsonType::size_type array_index(const string_t& s) { - using size_type = typename BasicJsonType::size_type; - - // error condition (cf. RFC 6901, Sect. 4) - if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0')) + typename BasicJsonType::size_type idx{}; + switch (parse_array_index(s, idx)) { - JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr)); + case array_index_status::leading_zero: + JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr)); + case array_index_status::not_a_number: + JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); + case array_index_status::unresolved: + JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr)); + case array_index_status::exceeds_size_type: + JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); + case array_index_status::ok: + default: + break; } - // error condition (cf. RFC 6901, Sect. 4) - if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9'))) - { - JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); - } - - const char* p = s.data(); - char* p_end = nullptr; // NOLINT(misc-const-correctness) - errno = 0; // strtoull doesn't reset errno - const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) - if (p == p_end // invalid input or empty string - || errno == ERANGE // out of range - || JSON_HEDLEY_UNLIKELY(static_cast(p_end - p) != s.size())) // incomplete read - { - JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr)); - } - - // the index does not fit into size_type; on 64-bit platforms this is - // only SIZE_MAX itself (see #2203 and #5395) - if (res >= static_cast((std::numeric_limits::max)())) // NOLINT(runtime/int) - { - JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); - } - - return static_cast(res); + return idx; } JSON_PRIVATE_UNLESS_TESTED: @@ -22493,63 +22517,6 @@ class json_pointer return *ptr; } - /*! - @throw parse_error.106 if an array index begins with '0' - @throw parse_error.109 if an array index was not a number - @throw out_of_range.402 if the array index '-' is used - @throw out_of_range.404 if the JSON pointer can not be resolved - */ - template - const BasicJsonType& get_checked(const BasicJsonType* ptr) const - { - for (const auto& reference_token : reference_tokens) - { - switch (ptr->type()) - { - case detail::value_t::object: - { - // note: at performs range check - ptr = &ptr->at(reference_token); - break; - } - - case detail::value_t::array: - { - if (JSON_HEDLEY_UNLIKELY(reference_token == "-")) - { - // "-" always fails the range check - JSON_THROW(detail::out_of_range::create(402, detail::concat( - "array index '-' (", std::to_string(ptr->m_data.m_value.array->size()), - ") is out of range"), ptr)); - } - - const auto idx = array_index(reference_token); - // Bounds check before access to avoid exception with JSON_NOEXCEPTION - if (JSON_HEDLEY_UNLIKELY(idx >= ptr->m_data.m_value.array->size())) - { - JSON_THROW(detail::out_of_range::create(401, detail::concat( - "array index ", std::to_string(idx), " is out of range"), ptr)); - } - ptr = &ptr->operator[](idx); - break; - } - - case detail::value_t::null: - case detail::value_t::string: - case detail::value_t::boolean: - case detail::value_t::number_integer: - case detail::value_t::number_unsigned: - case detail::value_t::number_float: - case detail::value_t::binary: - case detail::value_t::discarded: - default: - JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", reference_token, "'"), ptr)); - } - } - - return *ptr; - } - /*! @brief return a pointer to the pointed to value, or `nullptr` if the pointer cannot be resolved because a key is missing, an array @@ -22588,29 +22555,24 @@ class json_pointer return nullptr; } - // tokens that array_index() rejects with parse_error.106/109 - // are passed on to it; all other tokens that it would reject - // with out_of_range.404/410 are detected here, so that this - // also works without exceptions - if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1 && !(reference_token[0] >= '1' && reference_token[0] <= '9'))) + // a malformed index throws parse_error.106/109; an + // index that is syntactically valid but cannot be + // represented (out_of_range.404/410) is treated like an + // out-of-range index below + typename BasicJsonType::size_type idx{}; + switch (parse_array_index(reference_token, idx)) { - static_cast(array_index(reference_token)); // throws parse_error.106/109 + case array_index_status::leading_zero: + JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", reference_token, "' must not begin with '0'"), nullptr)); + case array_index_status::not_a_number: + JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", reference_token, "' is not a number"), nullptr)); + case array_index_status::unresolved: + case array_index_status::exceeds_size_type: + return nullptr; + case array_index_status::ok: + default: + break; } - if (JSON_HEDLEY_UNLIKELY(reference_token.empty() || !std::all_of(reference_token.begin(), reference_token.end(), [](const char c) - { - return c >= '0' && c <= '9'; - }))) - { - return nullptr; - } - errno = 0; // strtoull() does not reset errno on success - char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) - if (JSON_HEDLEY_UNLIKELY(errno == ERANGE || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) - { - return nullptr; - } - const auto idx = static_cast(magnitude); if (JSON_HEDLEY_UNLIKELY(idx >= ptr->m_data.m_value.array->size())) { @@ -22637,8 +22599,8 @@ class json_pointer } /*! - @throw parse_error.106 if an array index begins with '0' - @throw parse_error.109 if an array index was not a number + @note unlike array_index(), this never throws: a malformed or unrepresentable + array index reference token is treated like a missing key (see #5395) */ template bool contains(const BasicJsonType* ptr) const @@ -22666,49 +22628,17 @@ class json_pointer // "-" always fails the range check return false; } - if (JSON_HEDLEY_UNLIKELY(reference_token.empty())) - { - // an empty reference token is not an array index; array_index() - // would throw out_of_range.404 -- contains() must not throw (see #5395) - return false; - } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9'))) - { - // invalid char - return false; - } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1)) - { - if (JSON_HEDLEY_UNLIKELY(!('1' <= reference_token[0] && reference_token[0] <= '9'))) - { - // the first char should be between '1' and '9' - return false; - } - for (std::size_t i = 1; i < reference_token.size(); i++) - { - if (JSON_HEDLEY_UNLIKELY(!('0' <= reference_token[i] && reference_token[i] <= '9'))) - { - // other char should be between '0' and '9' - return false; - } - } - } - // the reference token consists only of digits at this point (cf. checks - // above); however, its numeric value might not be representable, in which - // case array_index() would throw out_of_range.404/410 -- contains() must - // not throw (see #5395), so such a reference token is treated as "not found" - errno = 0; // strtoull() does not reset errno on success - char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) - if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX - || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) + // any parse failure (malformed index, or one that is syntactically + // valid but not representable as size_type) means the reference + // token cannot denote an existing array element -- contains() must + // not throw (see #5395), so it is treated as "not found" + typename BasicJsonType::size_type idx{}; + if (JSON_HEDLEY_UNLIKELY(parse_array_index(reference_token, idx) != array_index_status::ok)) { - // the array index cannot be represented as size_type return false; } - const auto idx = array_index(reference_token); if (idx >= ptr->size()) { // index out of range @@ -23221,6 +23151,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // __ _____ _____ _____ // __| | __| | | | JSON for Modern C++ @@ -23588,8 +23520,12 @@ class binary_writer @param[in] sink output sink to write to (a value-type sink such as output_vector_sink, or output_adapter_sink wrapping a type-erased output adapter) + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ - explicit binary_writer(OutputSinkType sink) : oa(std::move(sink)) + explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(std::move(sink)), error_handler(error_handler_) {} /*! @@ -23602,14 +23538,20 @@ class binary_writer from one. @param[in] adapter output adapter to write to + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ template < typename SinkType = OutputSinkType, typename std::enable_if < std::is_constructible>::value, int >::type = 0 > - explicit binary_writer(output_adapter_t adapter) : oa(SinkType(std::move(adapter))) + explicit binary_writer(output_adapter_t adapter, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(SinkType(std::move(adapter))), error_handler(error_handler_) {} /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -23640,6 +23582,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -23706,13 +23650,16 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - write_cbor_head(0x60, j.m_data.m_value.string->size()); + write_cbor_head(0x60, value.size()); // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -23782,6 +23729,17 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // el.first is checked here, against the object as + // diagnostics context, because write_cbor(el.first) + // converts it to a temporary basic_json that would be + // used as the context instead; for error_handler_t::keep + // and ::replace/::ignore the recursive write_cbor(el.first) + // call below handles the key like any other string, so no + // separate check is needed here for those + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_cbor(el.first); write_cbor(el.second); } @@ -23929,8 +23887,11 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); + const auto N = to_msgpack_length(value.size(), j); if (N <= 31) { // fixstr @@ -23957,8 +23918,8 @@ class binary_writer // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -24105,6 +24066,13 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // as in write_cbor, el.first is checked here against the + // object as diagnostics context; the recursive call below + // handles keep/replace/ignore like any other string + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_msgpack(el.first); write_msgpack(el.second); } @@ -24124,6 +24092,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -24173,14 +24143,17 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + if (add_prefix) { oa.write_character(to_char_type('S')); } - write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); + write_number_with_ubjson_prefix(value.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -24335,10 +24308,12 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { - write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); + string_t storage; + const string_t& key = sanitize_utf8_for_write(el.first, j, storage); + write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(el.first.data()), - el.first.size()); + reinterpret_cast(key.data()), + key.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } @@ -24379,8 +24354,12 @@ class binary_writer /*! @return The size of a BSON document entry header, including the id marker and the entry name size (and its null-terminator). + @throw out_of_range.409 if @a name contains U+0000, before anything is + written + @throw type_error.316 if @a name is not valid UTF-8, before anything is + written */ - static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) + std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { const auto it = name.find(static_cast(0)); if (JSON_HEDLEY_UNLIKELY(it != BasicJsonType::string_t::npos)) @@ -24388,8 +24367,10 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - static_cast(j); - return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(name, j, storage); + + return /*id*/ 1ul + sanitized.size() + /*zero-terminator*/1u; } /*! @@ -24409,14 +24390,28 @@ class binary_writer /*! @brief Writes the given @a element_type and @a name to the output adapter + + @a name has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_entry_header_size during the earlier + size pass, so only @ref error_handler_t::replace / @ref + error_handler_t::ignore need to sanitize it again here, to actually write + the bytes that size was computed from. */ void write_bson_entry_header(const string_t& name, const std::uint8_t element_type) { oa.write_character(to_char_type(element_type)); - oa.write_characters( - reinterpret_cast(name.data()), - name.size()); + + if (error_handler == error_handler_t::keep || error_handler == error_handler_t::strict || is_valid_utf8(name)) + { + oa.write_characters(reinterpret_cast(name.data()), name.size()); + } + else + { + const string_t sanitized = sanitize_utf8(name, error_handler); + oa.write_characters(reinterpret_cast(sanitized.data()), sanitized.size()); + } + // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -24444,24 +24439,50 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value + @throw type_error.316 if @a value is not valid UTF-8, before anything is + written + + @note The UTF-8 check is skipped if @a value is already too long for the + 32-bit BSON length field (@ref to_bson_length rejects it later, once + the size of the whole document is known); this also keeps the check + from reading past a StringType that reports a size larger than what + it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value) + std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { + if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, j, storage); + return sizeof(std::int32_t) + sanitized.size() + 1ul; + } return sizeof(std::int32_t) + value.size() + 1ul; } /*! @brief Writes a BSON element with key @a name and string value @a value + + @a value has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_string_size during the earlier size + pass, so only @ref error_handler_t::replace / @ref error_handler_t::ignore + need to sanitize it again here, to actually write the bytes that size was + computed from. */ void write_bson_string(const string_t& name, const string_t& value) { write_bson_entry_header(name, 0x02); - write_number(to_bson_length(value.size() + 1ul), true); + const bool sanitize = error_handler != error_handler_t::keep + && error_handler != error_handler_t::strict + && !is_valid_utf8(value); + const string_t sanitized = sanitize ? sanitize_utf8(value, error_handler) : string_t{}; + const string_t& written = sanitize ? sanitized : value; + + write_number(to_bson_length(written.size() + 1ul), true); oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + reinterpret_cast(written.data()), + written.size()); // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -24575,8 +24596,10 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written + @throw type_error.316 if @a j is a string that is not valid UTF-8, before + anything is written */ - static std::size_t calc_bson_value_size(const BasicJsonType& j) + std::size_t calc_bson_value_size(const BasicJsonType& j) { switch (j.type()) { @@ -24596,7 +24619,7 @@ class binary_writer return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string, j); case value_t::null: return 0ul; @@ -24709,8 +24732,10 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written + @throw type_error.316 if a string value or a key is not valid UTF-8, + before anything is written */ - static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) + std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { // the object or array whose entries are being sized, and the ones it // is in; nothing is allocated unless the document nests @@ -25587,7 +25612,7 @@ class binary_writer */ void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) { - check_bon8_utf8(s, context); + check_utf8(s, context); // a string that follows another string terminates it if (string_open) @@ -25617,7 +25642,7 @@ class binary_writer @throw type_error.316 if @a s is not valid UTF-8; the message names the first byte of the first invalid or incomplete sequence */ - static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + static void check_utf8(const string_t& s, const BasicJsonType& context) { static_cast(context); // only used when exceptions are enabled const auto* data = reinterpret_cast(s.data()); @@ -25628,6 +25653,57 @@ class binary_writer } } + /*! + @brief return @a s as it should be written, honoring @ref error_handler + + Used by @ref write_cbor, @ref write_msgpack, @ref write_ubjson (and so + @ref write_bjdata), and the BSON writing functions for string values and + object keys; never by @ref write_bon8, which always validates, since UTF-8 + lead bytes are structural there. + + - @ref error_handler_t::keep: @a s is returned unchanged, without even + checking it (the behavior of release 3.12.0 and earlier). + - @ref error_handler_t::strict: @ref check_utf8 is called, which throws + type_error.316 if @a s is not valid UTF-8. + - @ref error_handler_t::replace / @ref error_handler_t::ignore: @a s is + sanitized into @a storage with exactly the rules @ref + serializer::dump_escaped_impl uses, so that parsing what @ref + basic_json::dump produces for the same string and the same handler + yields the same result. + + Well-formed input is never copied: this returns a reference to @a s + itself in every case but a sanitized `replace`/`ignore` one, so @a + storage must outlive the returned reference only then. + + @param[in] s the string (value or object key) to write + @param[in] context the value @a s belongs to (for diagnostics) + @param[out] storage backing storage for a sanitized copy + + @return a reference to @a s, or to @a storage once it holds a sanitized copy + */ + const string_t& sanitize_utf8_for_write(const string_t& s, const BasicJsonType& context, string_t& storage) const + { + switch (error_handler) + { + case error_handler_t::keep: + return s; + + case error_handler_t::strict: + check_utf8(s, context); + return s; + + case error_handler_t::replace: + case error_handler_t::ignore: + default: + if (is_valid_utf8(s)) + { + return s; + } + storage = sanitize_utf8(s, error_handler); + return storage; + } + } + /*! @brief write an integer in the shortest encoding @@ -25956,6 +26032,10 @@ class binary_writer /// the output OutputSinkType oa; + + /// how to treat a string value or object key that is not valid UTF-8 + /// (CBOR, MessagePack, UBJSON, BJData, and BSON; not BON8) + const error_handler_t error_handler = binary_writer_default_error_handler(); }; } // namespace detail @@ -27117,6 +27197,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -27136,14 +27218,6 @@ namespace detail // serialization // /////////////////// -/// how to treat decoding errors -enum class error_handler_t -{ - strict, ///< throw a type_error exception in case of invalid UTF-8 - replace, ///< replace invalid UTF-8 sequences with U+FFFD - ignore ///< ignore invalid UTF-8 sequences -}; - template class serializer { @@ -27934,6 +28008,16 @@ class serializer // EnsureAscii parameter is used, non-ASCII characters if ((codepoint <= 0x1F) || (EnsureAscii && (codepoint >= 0x7F))) { + if (EnsureAscii && error_handler == error_handler_t::keep) + { + // this character was buffered as raw bytes + // below in case it turned out to be part of + // an ill-formed sequence (which is kept as + // is); now that it decoded to a well-formed + // code point, undo that and \u-escape it + // like any other character instead + bytes = bytes_after_last_accept; + } if (codepoint <= 0xFFFF) { write_u_escape(bytes, static_cast(codepoint)); @@ -28032,6 +28116,44 @@ class serializer break; } + case error_handler_t::keep: + { + // the bytes of this (now abandoned) ill-formed + // sequence seen so far are already buffered below + // and are kept unchanged in the output + if (undumped_chars > 0) + { + // the byte that ended the sequence may be OK + // for itself (e.g., a quote that must still be + // escaped, or the lead byte of a well-formed + // code point), so read it again + --i; + } + else + { + // a byte that cannot start a sequence (e.g., + // 0xFF or a stray continuation byte) is kept + // as well + string_buffer[bytes++] = s[i]; + } + + // write buffer and reset index; there must be 13 bytes + // left, as this is the maximal number of bytes to be + // written ("\uxxxx\uxxxx\0") for one code point + if (string_buffer.size() - bytes < 13) + { + put_buffer(string_buffer, bytes); + bytes = 0; + } + + bytes_after_last_accept = bytes; + undumped_chars = 0; + + // continue processing the string + state = UTF8_ACCEPT; + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -28040,9 +28162,12 @@ class serializer default: // decode found yet incomplete multibyte code point { - if (!EnsureAscii) + if (!EnsureAscii || error_handler == error_handler_t::keep) { - // code point will not be escaped - copy byte to buffer + // code point will not be escaped (or will be kept as + // is if it turns out to be ill-formed) - copy byte to + // buffer; dropped again above if it decodes to a + // well-formed code point that needs \u-escaping string_buffer[bytes++] = s[i]; } ++undumped_chars; @@ -28093,6 +28218,14 @@ class serializer break; } + case error_handler_t::keep: + { + // write the ill-formed trailing bytes as is; they were + // buffered above regardless of EnsureAscii + put_buffer(string_buffer, bytes); + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -28745,16 +28878,43 @@ template , return *this; } +private: + /// @brief find the entry for @a key, for either constness of @a self + /// @note the single place that performs the linear key search + template + static auto find_impl(Self& self, KeyType&& key) -> decltype(self.begin()) + { + for (auto it = self.begin(); it != self.end(); ++it) + { + if (self.m_compare(it->first, key)) + { + return it; + } + } + return self.end(); + } + + /// @brief remove the entry @a it points to, preserving order + /// @note keys are not movable, so the tail is destroyed and re-constructed in place + void erase_at(iterator it) + { + for (auto next = it; ++next != this->end(); ++it) + { + it->~value_type(); // Destroy but keep allocation + new (&*it) value_type{std::move(*next)}; + } + Container::pop_back(); + } + +public: template::value, int> = 0> std::pair emplace(const key_type& key, V && t) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - return {it, false}; - } + return {it, false}; } append(key, std::forward(t)); return {std::prev(this->end()), true}; @@ -28765,12 +28925,10 @@ template , detail::is_constructible>::value, int> = 0> std::pair emplace(KeyType && key, V && t) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - return {it, false}; - } + return {it, false}; } append(std::forward(key), std::forward(t)); return {std::prev(this->end()), true}; @@ -28802,75 +28960,55 @@ template , T& at(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } template::value, int> = 0> T & at(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } const T& at(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } template::value, int> = 0> const T & at(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it == this->end()) { - if (m_compare(it->first, key)) - { - return it->second; - } + JSON_THROW(std::out_of_range("key not found")); } - - JSON_THROW(std::out_of_range("key not found")); + return it->second; } size_type erase(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - // Since we cannot move const Keys, re-construct them in place - for (auto next = it; ++next != this->end(); ++it) - { - it->~value_type(); // Destroy but keep allocation - new (&*it) value_type{std::move(*next)}; - } - Container::pop_back(); - return 1; - } + erase_at(it); + return 1; } return 0; } @@ -28879,19 +29017,11 @@ template , detail::is_usable_as_key_type::value, int> = 0> size_type erase(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, key); + if (it != this->end()) { - if (m_compare(it->first, key)) - { - // Since we cannot move const Keys, re-construct them in place - for (auto next = it; ++next != this->end(); ++it) - { - it->~value_type(); // Destroy but keep allocation - new (&*it) value_type{std::move(*next)}; - } - Container::pop_back(); - return 1; - } + erase_at(it); + return 1; } return 0; } @@ -28956,80 +29086,38 @@ template , size_type count(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return 1; - } - } - return 0; + return find_impl(*this, key) != this->end() ? 1 : 0; } template::value, int> = 0> size_type count(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return 1; - } - } - return 0; + return find_impl(*this, key) != this->end() ? 1 : 0; } iterator find(const key_type& key) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } template::value, int> = 0> iterator find(KeyType && key) // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } const_iterator find(const key_type& key) const { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } template::value, int> = 0> const_iterator find(KeyType && key) const // NOLINT(cppcoreguidelines-missing-std-forward) { - for (auto it = this->begin(); it != this->end(); ++it) - { - if (m_compare(it->first, key)) - { - return it; - } - } - return Container::end(); + return find_impl(*this, key); } std::pair insert( value_type&& value ) @@ -29039,12 +29127,10 @@ template , std::pair insert( const value_type& value ) { - for (auto it = this->begin(); it != this->end(); ++it) + const auto it = find_impl(*this, value.first); + if (it != this->end()) { - if (m_compare(it->first, value.first)) - { - return {it, false}; - } + return {it, false}; } append(value); return {--this->end(), true}; @@ -29167,11 +29253,12 @@ struct is_std_optional> : std::true_type {}; @brief a class to store JSON values @internal -@invariant The member variables @a m_value and @a m_type have the following -relationship: -- If `m_type == value_t::object`, then `m_value.object != nullptr`. -- If `m_type == value_t::array`, then `m_value.array != nullptr`. -- If `m_type == value_t::string`, then `m_value.string != nullptr`. +@invariant The member variables @a m_data.m_value and @a m_data.m_type have +the following relationship: +- If `m_data.m_type == value_t::object`, then `m_data.m_value.object != nullptr`. +- If `m_data.m_type == value_t::array`, then `m_data.m_value.array != nullptr`. +- If `m_data.m_type == value_t::string`, then `m_data.m_value.string != nullptr`. +- If `m_data.m_type == value_t::binary`, then `m_data.m_value.binary != nullptr`. The invariants are checked by member function assert_invariant(). @note ObjectType trick from https://stackoverflow.com/a/9860911 @@ -29213,7 +29300,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// workaround type for MSVC using basic_json_t = NLOHMANN_BASIC_JSON_TPL; - using json_base_class_t = ::nlohmann::detail::json_base_class; JSON_PRIVATE_UNLESS_TESTED: // convenience aliases for types residing in namespace detail; @@ -29234,18 +29320,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } private: - using primitive_iterator_t = ::nlohmann::detail::primitive_iterator_t; - template - using internal_iterator = ::nlohmann::detail::internal_iterator; template using iter_impl = ::nlohmann::detail::iter_impl; template using iteration_proxy = ::nlohmann::detail::iteration_proxy; template using json_reverse_iterator = ::nlohmann::detail::json_reverse_iterator; - template - using output_adapter_t = ::nlohmann::detail::output_adapter_t; - template using binary_reader = ::nlohmann::detail::binary_reader; template using binary_writer = ::nlohmann::detail::binary_writer; @@ -29253,9 +29333,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // used by the vector-returning to_* overloads template using vector_binary_writer = ::nlohmann::detail::binary_writer>; - template static vector_binary_writer vector_writer(std::vector& v) + template static vector_binary_writer vector_writer( + std::vector& v, const ::nlohmann::detail::error_handler_t error_handler = ::nlohmann::detail::binary_writer_default_error_handler()) { - return vector_binary_writer(::nlohmann::detail::output_vector_sink(v)); + return vector_binary_writer(::nlohmann::detail::output_vector_sink(v), error_handler); } JSON_PRIVATE_UNLESS_TESTED: @@ -29273,6 +29354,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec using cbor_tag_handler_t = detail::cbor_tag_handler_t; /// how to encode BJData using bjdata_version_t = detail::bjdata_version_t; + /// base class used to inject custom functionality into each instance of basic_json + /// @sa https://json.nlohmann.me/api/basic_json/json_base_class_t/ + using json_base_class_t = ::nlohmann::detail::json_base_class; /// helper type for initializer lists of basic_json values using initializer_list_t = std::initializer_list>; @@ -29386,8 +29470,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec std::to_string(__GNUC_PATCHLEVEL__)) } }; -#elif defined(__HP_cc) || defined(__HP_aCC) - result["compiler"] = "hp" +#elif defined(__HP_aCC) + result["compiler"] = {{"family", "hp"}, {"version", __HP_aCC}}; #elif defined(__IBMCPP__) result["compiler"] = {{"family", "ilecpp"}, {"version", __IBMCPP__}}; #elif defined(_MSC_VER) @@ -29529,9 +29613,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec binary | binary | pointer to @ref binary_t null | null | *no value is stored* - @note Variable-length types (objects, arrays, and strings) are stored as - pointers. The size of the union should not exceed 64 bits if the default - value types are used. + @note Variable-length types (objects, arrays, strings, and binary + values) are stored as pointers. The size of the union should not exceed + 64 bits if the default value types are used. @since version 1.0.0 */ @@ -29783,7 +29867,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec end of every constructor to make sure that created objects respect the invariant. Furthermore, it has to be called each time the type of a JSON value is changed, because the invariant expresses a relationship between - @a m_type and @a m_value. + @a m_data.m_type and @a m_data.m_value. Furthermore, the parent relation is checked for arrays and objects: If @a check_parents true and the value is an array or object, then the @@ -29804,7 +29888,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec #if JSON_DIAGNOSTICS JSON_TRY { - // cppcheck-suppress assertWithSideEffect JSON_ASSERT(!check_parents || !is_structured() || std::all_of(begin(), end(), [this](const basic_json & j) { return j.m_parent == this; @@ -30976,8 +31059,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief create a null object /// @sa https://json.nlohmann.me/api/basic_json/basic_json/ - basic_json(std::nullptr_t = nullptr) noexcept // NOLINT(bugprone-exception-escape) - : basic_json(value_t::null) + basic_json(std::nullptr_t = nullptr) noexcept { assert_invariant(); } @@ -31341,15 +31423,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief move constructor /// @sa https://json.nlohmann.me/api/basic_json/basic_json/ basic_json(basic_json&& other) noexcept - : json_base_class_t(std::forward(other)), - m_data(std::move(other.m_data)) // cppcheck-suppress[accessForwarded] TODO check + : json_base_class_t(std::move(static_cast(other))), + m_data(std::move(other.m_data)) #if JSON_DIAGNOSTIC_POSITIONS - , start_position(other.start_position) // cppcheck-suppress[accessForwarded] TODO check - , end_position(other.end_position) // cppcheck-suppress[accessForwarded] TODO check + , start_position(other.start_position) + , end_position(other.end_position) #endif { // check that the passed value is valid - other.assert_invariant(false); // cppcheck-suppress[accessForwarded] + other.assert_invariant(false); // invalidate payload other.m_data.m_type = value_t::null; @@ -31916,7 +31998,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec @tparam PointerType pointer type; must be a pointer to @ref array_t, @ref object_t, @ref string_t, @ref boolean_t, @ref number_integer_t, - @ref number_unsigned_t, or @ref number_float_t. + @ref number_unsigned_t, @ref number_float_t, or @ref binary_t. @return pointer to the internally stored JSON value if the requested pointer type @a PointerType fits to the JSON value; `nullptr` otherwise @@ -32082,8 +32164,106 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return *get_ptr(); } + /// @brief access the custom base class + /// @sa https://json.nlohmann.me/api/basic_json/as_base_class/ + json_base_class_t& as_base_class() noexcept + { + return static_cast(*this); + } + + /// @brief access the custom base class + /// @sa https://json.nlohmann.me/api/basic_json/as_base_class/ + const json_base_class_t& as_base_class() const noexcept + { + return static_cast(*this); + } + /// @} + private: + /// @brief look up @a key in the object held by @a j, for either constness of @a j + /// @note the single place that performs a (possibly transparent) object key lookup + template + static auto object_lookup(Self& j, KeyType&& key) + -> decltype(j.m_data.m_value.object->find(lookup_key(std::forward(key)))) + { + return j.m_data.m_value.object->find(lookup_key(std::forward(key))); + } + + /// @brief checked object element access used by the at() overloads taking a key + /// @throw type_error.304 if @a j is not an object + /// @throw out_of_range.403 if @a key is not found + template + static auto object_at(Self& j, KeyType&& key) + -> decltype((object_lookup(j, std::forward(key))->second)) + { + // at only works for objects + if (JSON_HEDLEY_UNLIKELY(!j.is_object())) + { + JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", j.type_name()), &j)); + } + + auto it = object_lookup(j, std::forward(key)); + if (it == j.m_data.m_value.object->end()) + { + // key is only forwarded into the lookup above: object_t::find() (a plain + // std::map or ordered_map) never moves from its argument, so key is still + // valid here regardless of whether KeyType was deduced as an rvalue reference + // NOLINTNEXTLINE(bugprone-use-after-move,hicpp-invalid-access-moved) + JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(key), "' not found"), &j)); + } + return it->second; + } + + /// @brief checked array element access used by the at() overloads taking an index + /// @throw type_error.304 if @a j is not an array + /// @throw out_of_range.401 if @a idx is out of range + template + static auto array_at(Self& j, size_type idx) + -> decltype((*j.m_data.m_value.array)[idx]) + { + // at only works for arrays + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", j.type_name()), &j)); + } + + if (JSON_HEDLEY_UNLIKELY(idx >= j.m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), &j)); + } + + return (*j.m_data.m_value.array)[idx]; + } + + /// @brief convert a null value to an empty container of type @a Container + /// @tparam Container array_t or object_t; any other type does not compile + template + void convert_null_to() + { + JSON_ASSERT(is_null()); + // create the container before touching the type, so a throwing + // allocation leaves this value as a valid null rather than a type + // tag with a dangling/null pointer behind it + set_container(create()); + assert_invariant(); + } + + /// @brief store a freshly created array and set the matching type + void set_container(array_t* array) noexcept + { + m_data.m_value.array = array; + m_data.m_type = value_t::array; + } + + /// @brief store a freshly created object and set the matching type + void set_container(object_t* object) noexcept + { + m_data.m_value.object = object; + m_data.m_type = value_t::object; + } + + public: //////////////////// // element access // //////////////////// @@ -32096,54 +32276,21 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/at/ reference at(size_type idx) { - // at only works for arrays - if (JSON_HEDLEY_UNLIKELY(!is_array())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) - { - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } - - return set_parent((*m_data.m_value.array)[idx]); + return set_parent(array_at(*this, idx)); } /// @brief access specified array element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ const_reference at(size_type idx) const { - // at only works for arrays - if (JSON_HEDLEY_UNLIKELY(!is_array())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) - { - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } - - return (*m_data.m_value.array)[idx]; + return array_at(*this, idx); } /// @brief access specified object element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ reference at(const typename object_t::key_type& key) { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(key); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", key, "' not found"), this)); - } - return set_parent(it->second); + return set_parent(object_at(*this, key)); } /// @brief access specified object element with bounds checking @@ -32152,36 +32299,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_usable_as_basic_json_key_type::value, int> = 0> reference at(KeyType && key) { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(std::forward(key)), "' not found"), this)); - } - return set_parent(it->second); + return set_parent(object_at(*this, std::forward(key))); } /// @brief access specified object element with bounds checking /// @sa https://json.nlohmann.me/api/basic_json/at/ const_reference at(const typename object_t::key_type& key) const { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(key); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", key, "' not found"), this)); - } - return it->second; + return object_at(*this, key); } /// @brief access specified object element with bounds checking @@ -32190,18 +32315,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_usable_as_basic_json_key_type::value, int> = 0> const_reference at(KeyType && key) const { - // at only works for objects - if (JSON_HEDLEY_UNLIKELY(!is_object())) - { - JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); - } - - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); - if (it == m_data.m_value.object->end()) - { - JSON_THROW(out_of_range::create(403, detail::concat("key '", string_t(std::forward(key)), "' not found"), this)); - } - return it->second; + return object_at(*this, std::forward(key)); } /// @brief access specified array element @@ -32211,9 +32325,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value.array = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for arrays @@ -32279,9 +32391,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value.object = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for objects @@ -32301,7 +32411,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // const operator[] only works for objects if (JSON_HEDLEY_LIKELY(is_object())) { - auto it = m_data.m_value.object->find(key); + auto it = object_lookup(*this, key); JSON_ASSERT(it != m_data.m_value.object->end()); return it->second; } @@ -32332,9 +32442,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value.object = create(); - assert_invariant(); + convert_null_to(); } // operator[] only works for objects @@ -32356,7 +32464,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // const operator[] only works for objects if (JSON_HEDLEY_LIKELY(is_object())) { - auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); + auto it = object_lookup(*this, std::forward(key)); JSON_ASSERT(it != m_data.m_value.object->end()); return it->second; } @@ -32376,6 +32484,36 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::is_c_string_uncvref::value, string_t, typename std::decay::type >; + /// @brief look up @a key for value(), the single place shared by all key-based overloads + /// @throw type_error.306 if this is not an object + /// @return a pointer to the found value, or `nullptr` if @a key was not found + template + const basic_json* value_member(KeyType&& key) const + { + // value only works for objects + if (JSON_HEDLEY_UNLIKELY(!is_object())) + { + JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + } + + const auto it = find(std::forward(key)); + return it != end() ? &*it : nullptr; + } + + /// @brief resolve @a ptr for value(), the single place shared by both json_pointer overloads + /// @throw type_error.306 if this is not an array or object + /// @return a pointer to the resolved value, or `nullptr` if @a ptr does not resolve + const basic_json* value_pointee(const json_pointer& ptr) const + { + // value only works for arrays and objects + if (JSON_HEDLEY_UNLIKELY(!is_structured())) + { + JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + } + + return ptr.get_checked_or_null(this); + } + public: // an integer literal 0 would otherwise convert to a null const char* and from there to key_type template::value, int> = 0> @@ -32389,20 +32527,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(const typename object_t::key_type& key, const ValueType& default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(key); - if (it != end()) - { - return it->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(key); + return found != nullptr ? found->template get() : default_value; } /// @brief access specified object element with default value @@ -32414,20 +32541,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(const typename object_t::key_type& key, ValueType && default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(key); - if (it != end()) - { - return it->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(key); + return found != nullptr ? found->template get() : std::forward(default_value); } /// @brief access specified object element with default value @@ -32440,23 +32556,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(KeyType && key, const ValueType& default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(std::forward(key)); - if (it != end()) - { - return it->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(std::forward(key)); + return found != nullptr ? found->template get() : default_value; } - /// @brief access specified object element via JSON Pointer with default value + /// @brief access specified object element with default value /// @sa https://json.nlohmann.me/api/basic_json/value/ template < class ValueType, class KeyType, class ReturnType = typename value_return_type::type, detail::enable_if_t < @@ -32467,20 +32572,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(KeyType && key, ValueType && default_value) const { - // value only works for objects - if (JSON_HEDLEY_LIKELY(is_object())) - { - // If 'key' is found, return its value. Otherwise, return `default_value'. - const auto it = find(std::forward(key)); - if (it != end()) - { - return it->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If 'key' is found, return its value. Otherwise, return `default_value'. + const auto* found = value_member(std::forward(key)); + return found != nullptr ? found->template get() : std::forward(default_value); } /// @brief access specified object element via JSON Pointer with default value @@ -32490,21 +32584,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ValueType value(const json_pointer& ptr, const ValueType& default_value) const { - // value only works for arrays and objects - if (JSON_HEDLEY_LIKELY(is_structured())) - { - // If the pointer resolves to a value, return it. Otherwise, return - // 'default_value'. - const auto* res = ptr.get_checked_or_null(this); - if (JSON_HEDLEY_LIKELY(res != nullptr)) - { - return res->template get(); - } - - return default_value; - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If the pointer resolves to a value, return it. Otherwise, return + // 'default_value'. + const auto* found = value_pointee(ptr); + return found != nullptr ? found->template get() : default_value; } /// @brief access specified object element via JSON Pointer with default value @@ -32515,21 +32598,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec && !std::is_same>::value, int > = 0 > ReturnType value(const json_pointer& ptr, ValueType && default_value) const { - // value only works for arrays and objects - if (JSON_HEDLEY_LIKELY(is_structured())) - { - // If the pointer resolves to a value, return it. Otherwise, return - // 'default_value'. - const auto* res = ptr.get_checked_or_null(this); - if (JSON_HEDLEY_LIKELY(res != nullptr)) - { - return res->template get(); - } - - return std::forward(default_value); - } - - JSON_THROW(type_error::create(306, detail::concat("cannot use value() with ", type_name()), this)); + // If the pointer resolves to a value, return it. Otherwise, return + // 'default_value'. + const auto* found = value_pointee(ptr); + return found != nullptr ? found->template get() : std::forward(default_value); } template < class ValueType, class BasicJsonType, detail::enable_if_t < @@ -32614,21 +32686,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(invalid_iterator::create(205, "iterator out of range", this)); } - if (is_string()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.string); - std::allocator_traits::deallocate(alloc, m_data.m_value.string, 1); - m_data.m_value.string = nullptr; - } - else if (is_binary()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.binary); - std::allocator_traits::deallocate(alloc, m_data.m_value.binary, 1); - m_data.m_value.binary = nullptr; - } - + m_data.m_value.destroy(m_data.m_type); + m_data.m_value = {}; m_data.m_type = value_t::null; assert_invariant(); break; @@ -32680,27 +32739,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::string: case value_t::binary: { - if (JSON_HEDLEY_LIKELY(!first.m_it.primitive_iterator.is_begin() - || !last.m_it.primitive_iterator.is_end())) + if (JSON_HEDLEY_UNLIKELY(!first.m_it.primitive_iterator.is_begin() + || !last.m_it.primitive_iterator.is_end())) { JSON_THROW(invalid_iterator::create(204, "iterators out of range", this)); } - if (is_string()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.string); - std::allocator_traits::deallocate(alloc, m_data.m_value.string, 1); - m_data.m_value.string = nullptr; - } - else if (is_binary()) - { - AllocatorType alloc; - std::allocator_traits::destroy(alloc, m_data.m_value.binary); - std::allocator_traits::deallocate(alloc, m_data.m_value.binary, 1); - m_data.m_value.binary = nullptr; - } - + m_data.m_value.destroy(m_data.m_type); + m_data.m_value = {}; m_data.m_type = value_t::null; assert_invariant(); break; @@ -32756,7 +32802,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - const auto it = m_data.m_value.object->find(lookup_key(std::forward(key))); + const auto it = object_lookup(*this, std::forward(key)); if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); @@ -32833,7 +32879,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(key); + result.m_it.object_iterator = object_lookup(*this, key); } return result; @@ -32847,7 +32893,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(key); + result.m_it.object_iterator = object_lookup(*this, key); } return result; @@ -32863,7 +32909,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(lookup_key(std::forward(key))); + result.m_it.object_iterator = object_lookup(*this, std::forward(key)); } return result; @@ -32879,7 +32925,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (is_object()) { - result.m_it.object_iterator = m_data.m_value.object->find(lookup_key(std::forward(key))); + result.m_it.object_iterator = object_lookup(*this, std::forward(key)); } return result; @@ -32910,7 +32956,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(const typename object_t::key_type& key) const { - return is_object() && m_data.m_value.object->find(key) != m_data.m_value.object->end(); + return is_object() && object_lookup(*this, key) != m_data.m_value.object->end(); } /// @brief check the existence of an element in a JSON object @@ -32920,7 +32966,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(KeyType && key) const { - return is_object() && m_data.m_value.object->find(lookup_key(std::forward(key))) != m_data.m_value.object->end(); + return is_object() && object_lookup(*this, std::forward(key)) != m_data.m_value.object->end(); } /// @brief check the existence of an element in a JSON object given a JSON pointer @@ -33285,9 +33331,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array (move semantics) @@ -33318,9 +33362,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array @@ -33350,9 +33392,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value = value_t::object; - assert_invariant(); + convert_null_to(); } // add the element to the object @@ -33406,9 +33446,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an array if (is_null()) { - m_data.m_type = value_t::array; - m_data.m_value = value_t::array; - assert_invariant(); + convert_null_to(); } // add the element to the array (perfect forwarding) @@ -33431,9 +33469,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // transform a null object into an object if (is_null()) { - m_data.m_type = value_t::object; - m_data.m_value = value_t::object; - assert_invariant(); + convert_null_to(); } // add the element to the array (perfect forwarding) @@ -33493,7 +33529,22 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/insert/ iterator insert(const_iterator pos, basic_json&& val) // NOLINT(performance-unnecessary-value-param) { - return insert(std::move(pos), val); + // insert only works for arrays + if (JSON_HEDLEY_LIKELY(is_array())) + { + // check if iterator pos fits to this JSON value + if (JSON_HEDLEY_UNLIKELY(pos.m_object != this)) + { + JSON_THROW(invalid_iterator::create(202, "iterator does not fit current value", this)); + } + + // moving into a local first keeps this safe even if val aliases + // an element of this array + basic_json tmp(std::move(val)); + return insert_iterator(pos, std::move(tmp)); + } + + JSON_THROW(type_error::create(309, detail::concat("cannot use insert() with ", type_name()), this)); } /// @brief inserts copies of element into array @@ -33669,14 +33720,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// value is an object; called first by both @ref update overloads void prepare_update() { - // implicitly convert a null value to an empty object; create the - // object before setting the type, so a throwing allocation leaves - // this value null + // implicitly convert a null value to an empty object if (is_null()) { - m_data.m_value.object = create(); - m_data.m_type = value_t::object; - assert_invariant(); + convert_null_to(); } if (JSON_HEDLEY_UNLIKELY(!is_object())) @@ -33885,7 +33932,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(binary_t& other) // NOLINT(bugprone-exception-escape,cppcoreguidelines-noexcept-swap,performance-noexcept-swap) { - // swap only works for strings + // swap only works for binary values if (JSON_HEDLEY_LIKELY(is_binary())) { using std::swap; @@ -33901,7 +33948,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(typename binary_t::container_type& other) // NOLINT(bugprone-exception-escape) { - // swap only works for strings + // swap only works for binary values if (JSON_HEDLEY_LIKELY(is_binary())) { using std::swap; @@ -34402,7 +34449,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(detail::input_adapter(std::forward(i)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved,accessForwarded] + auto p = parser(detail::input_adapter(std::forward(i)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -34419,7 +34467,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(detail::input_adapter(std::move(first), std::move(last)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved] + auto p = parser(detail::input_adapter(std::move(first), std::move(last)), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -34432,7 +34481,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_trailing_commas = false) { basic_json result; - parser(i.get(), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas).parse(true, result); // cppcheck-suppress[accessMoved] + auto p = parser(i.get(), std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + p.parse(true, result); return result; } @@ -34659,6 +34709,32 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #endif + private: + /*! + @brief shared implementation of the binary from_cbor/from_msgpack/ + from_ubjson/from_bjdata/from_bon8/from_bson overloads + + Building the @ref binary_reader as a named local, rather than as a + temporary that @ref detail::json_sax_dom_parser::parse is called on in + the same expression, avoids a false-positive cppcheck accessMoved + warning in each of the 16 callers. + */ + template + static basic_json from_binary_impl(InputAdapterType ia, const input_format_t format, + const bool strict, const bool allow_exceptions, + const error_handler_t error_handler = error_handler_t::keep, + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + { + basic_json result; + detail::json_sax_dom_parser sdp(result, allow_exceptions); + binary_reader reader(std::move(ia), format, error_handler); + if (!reader.sax_parse(&sdp, strict, tag_handler)) + { + result = value_t::discarded; + } + return result; + } + ////////////////////////////////////////// // binary serialization/deserialization // ////////////////////////////////////////// @@ -34669,78 +34745,87 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec public: /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static std::vector to_cbor(const basic_json& j) + static std::vector to_cbor(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_cbor(j); + vector_writer(result, error_handler).write_cbor(j); return result; } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static std::vector to_msgpack(const basic_json& j) + static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_msgpack(j); + vector_writer(result, error_handler).write_msgpack(j); return result; } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static std::vector to_ubjson(const basic_json& j, const bool use_size = false, - const bool use_type = false) + const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type); return result; } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a BJData serialization of a given JSON value @@ -34748,11 +34833,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_bjdata(const basic_json& j, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type, true, true, version); return result; } @@ -34760,42 +34846,47 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BJData serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static std::vector to_bson(const basic_json& j) + static std::vector to_bson(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_bson(j); + vector_writer(result, error_handler).write_bson(j); return result; } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BON8 serialization of a given JSON value @@ -34829,16 +34920,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(InputType&& i, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } /// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -34849,16 +34934,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(IteratorType first, SentinelType last, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } template @@ -34879,15 +34958,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, error_handler_t::keep, tag_handler); } /// @brief create a JSON value from an input in MessagePack format @@ -34896,16 +34967,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -34915,16 +34980,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } template @@ -34943,15 +35002,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::msgpack, strict, allow_exceptions); } /// @brief create a JSON value from an input in UBJSON format @@ -34960,16 +35011,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -34979,16 +35024,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } template @@ -35007,15 +35046,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::ubjson, strict, allow_exceptions); } /// @brief create a JSON value from an input in BJData format @@ -35024,16 +35055,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -35043,16 +35068,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BON8 format @@ -35063,14 +35082,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bon8, strict, allow_exceptions); } /// @brief create a JSON value from an input in BON8 format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -35082,14 +35094,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bon8, strict, allow_exceptions); } /// @brief create a JSON value from an input in BSON format @@ -35098,16 +35103,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::forward(i)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -35117,16 +35116,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - basic_json result; - auto ia = detail::input_adapter(std::move(first), std::move(last)); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions, error_handler); } template @@ -35145,15 +35138,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool strict = true, const bool allow_exceptions = true) { - basic_json result; - auto ia = i.get(); - detail::json_sax_dom_parser sdp(result, allow_exceptions); - // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] - { - result = value_t::discarded; - } - return result; + return from_binary_impl(i.get(), input_format_t::bson, strict, allow_exceptions); } /// @} @@ -36370,12 +36355,14 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM #undef JSON_HAS_THREE_WAY_COMPARISON #undef JSON_HAS_RANGES + #undef JSON_HAS_RANGE_VIEW_CONVERSION #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION #undef JSON_STRICT_NUL_HANDLING + #undef JSON_STRICT_BINARY_UTF8 #endif // #include diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index 3ee7afa73..7cadc9b1c 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -63,6 +63,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -99,14 +103,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -115,7 +125,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index 879322dd0..e4c627060 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -44,6 +44,10 @@ TEST_CASE("default namespace") expected += "_snul"; #endif +#if JSON_STRICT_BINARY_UTF8 + expected += "_sbu8"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 4b1eb6ee4..858964695 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -45,6 +45,10 @@ TEST_CASE("default namespace without version component") expected += "_snul"; #endif +#if JSON_STRICT_BINARY_UTF8 + expected += "_sbu8"; +#endif + expected += "::basic_json"; // fallback for Clang diff --git a/tests/src/unit-binary_utf8_error_handler.cpp b/tests/src/unit-binary_utf8_error_handler.cpp new file mode 100644 index 000000000..85505bffb --- /dev/null +++ b/tests/src/unit-binary_utf8_error_handler.cpp @@ -0,0 +1,362 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +using nlohmann::json; + +#include +#include + +namespace +{ + +struct ill_formed_case +{ + const char* name; + std::string bytes; +}; + +// RFC 3629 ill-formed sequences used throughout this file, plus one +// well-formed sequence for contrast +const std::vector ill_formed_cases = +{ + {"overlong", "\xC0\xAE"}, + {"lone_0xFF", "\xFF"}, + {"truncated", "\xE2\x82"}, + {"surrogate", "\xED\xA0\x80"}, +}; + +const std::string valid_sequence = "\xC3\xA9"; // U+00E9, "é" + +using eh = json::error_handler_t; +const std::vector all_handlers = {eh::strict, eh::replace, eh::ignore, eh::keep}; + +// what dump()+parse() produces for a sanitizing error_handler; this is the +// ground truth every binary writer/reader is checked against +std::string dump_and_parse(const std::string& raw, eh error_handler) +{ + return json::parse(json(raw).dump(-1, ' ', false, error_handler)).get(); +} + +} // namespace + +TEST_CASE("UTF-8 error_handler for the binary readers and writers") +{ + SECTION("writers: string value") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + const json jval = c.bytes; + + CHECK_THROWS_AS(json::to_cbor(jval, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_msgpack(jval, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_ubjson(jval, false, false, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&); + { + json jobj; + jobj["k"] = jval; + CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&); + } + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(json::to_cbor(jval, h)).get() == expected); + CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get() == expected); + CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get() == expected); + CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get() == expected); + { + json jobj; + jobj["k"] = jval; + const auto bytes = json::to_bson(jobj, h); + CHECK(json::from_bson(bytes)["k"].get() == expected); + } + } + + // keep: the writer passes the ill-formed bytes through unchanged, + // exactly as every binary writer did before this parameter existed + CHECK(json::from_cbor(json::to_cbor(jval, eh::keep)).get() == c.bytes); + CHECK(json::from_msgpack(json::to_msgpack(jval, eh::keep)).get() == c.bytes); + CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, eh::keep)).get() == c.bytes); + CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep)).get() == c.bytes); + { + json jobj; + jobj["k"] = jval; + const auto bytes = json::to_bson(jobj, eh::keep); + CHECK(json::from_bson(bytes)["k"].get() == c.bytes); + } + } + } + + SECTION("writers: object key") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + json jobj; + jobj[c.bytes] = 1; + + CHECK_THROWS_AS(json::to_cbor(jobj, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_msgpack(jobj, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_ubjson(jobj, false, false, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&); + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(json::to_cbor(jobj, h)).begin().key() == expected); + CHECK(json::from_msgpack(json::to_msgpack(jobj, h)).begin().key() == expected); + CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, h)).begin().key() == expected); + CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, h)).begin().key() == expected); + CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == expected); + } + + // keep: object keys round-trip unchanged too + CHECK(json::from_cbor(json::to_cbor(jobj, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_msgpack(json::to_msgpack(jobj, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_bson(json::to_bson(jobj, eh::keep)).begin().key() == c.bytes); + } + } + + SECTION("readers: string value") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + + // bytes produced the lenient (keep) way, as any binary reader + // accepted them before this parameter existed + const auto cbor_bytes = json::to_cbor(json(c.bytes), eh::keep); + const auto msgpack_bytes = json::to_msgpack(json(c.bytes)); // to_msgpack has no error_handler; always pass-through + const auto ubjson_bytes = json::to_ubjson(json(c.bytes), false, false, eh::keep); + const auto bjdata_bytes = json::to_bjdata(json(c.bytes), false, false, json::bjdata_version_t::draft2, eh::keep); + const auto bson_bytes = [&c] + { + json jobj; + jobj["k"] = c.bytes; + return json::to_bson(jobj, eh::keep); + }(); + + // keep (the default): bytes are kept unchanged + CHECK(json::from_cbor(cbor_bytes).get() == c.bytes); + CHECK(json::from_msgpack(msgpack_bytes).get() == c.bytes); + CHECK(json::from_ubjson(ubjson_bytes).get() == c.bytes); + CHECK(json::from_bjdata(bjdata_bytes).get() == c.bytes); + CHECK(json::from_bson(bson_bytes)["k"].get() == c.bytes); + + // strict: parse_error.113, discarded (not thrown) when allow_exceptions is false + CHECK_THROWS_AS(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, eh::strict), json::parse_error&); + CHECK(json::from_cbor(cbor_bytes, true, false, json::cbor_tag_handler_t::error, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_msgpack(msgpack_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_msgpack(msgpack_bytes, true, false, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_ubjson(ubjson_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_ubjson(ubjson_bytes, true, false, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_bjdata(bjdata_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_bjdata(bjdata_bytes, true, false, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_bson(bson_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_bson(bson_bytes, true, false, eh::strict).is_discarded()); + + // replace / ignore: match what dump() would have sanitized the same bytes to + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, h).get() == expected); + CHECK(json::from_msgpack(msgpack_bytes, true, true, h).get() == expected); + CHECK(json::from_ubjson(ubjson_bytes, true, true, h).get() == expected); + CHECK(json::from_bjdata(bjdata_bytes, true, true, h).get() == expected); + CHECK(json::from_bson(bson_bytes, true, true, h)["k"].get() == expected); + } + } + } + + SECTION("readers: object key") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + + json jobj; + jobj[c.bytes] = 1; + const auto cbor_bytes = json::to_cbor(jobj, eh::keep); + const auto msgpack_bytes = json::to_msgpack(jobj); + const auto ubjson_bytes = json::to_ubjson(jobj, false, false, eh::keep); + const auto bjdata_bytes = json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep); + const auto bson_bytes = json::to_bson(jobj, eh::keep); + + CHECK(json::from_cbor(cbor_bytes).begin().key() == c.bytes); + CHECK(json::from_msgpack(msgpack_bytes).begin().key() == c.bytes); + CHECK(json::from_ubjson(ubjson_bytes).begin().key() == c.bytes); + CHECK(json::from_bjdata(bjdata_bytes).begin().key() == c.bytes); + CHECK(json::from_bson(bson_bytes).begin().key() == c.bytes); + + CHECK_THROWS_AS(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_msgpack(msgpack_bytes, true, true, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_ubjson(ubjson_bytes, true, true, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_bjdata(bjdata_bytes, true, true, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_bson(bson_bytes, true, true, eh::strict), json::parse_error&); + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, h).begin().key() == expected); + CHECK(json::from_msgpack(msgpack_bytes, true, true, h).begin().key() == expected); + CHECK(json::from_ubjson(ubjson_bytes, true, true, h).begin().key() == expected); + CHECK(json::from_bjdata(bjdata_bytes, true, true, h).begin().key() == expected); + CHECK(json::from_bson(bson_bytes, true, true, h).begin().key() == expected); + } + } + } + + SECTION("well-formed UTF-8 is unaffected by error_handler") + { + const json jval = valid_sequence; + json jobj; + jobj[valid_sequence] = valid_sequence; + + for (const auto h : all_handlers) + { + CAPTURE(static_cast(h)); + + CHECK(json::from_cbor(json::to_cbor(jval, h)).get() == valid_sequence); + CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get() == valid_sequence); + CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get() == valid_sequence); + CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get() == valid_sequence); + CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == valid_sequence); + + CHECK(json::from_cbor(json::to_cbor(jval, eh::keep), true, true, json::cbor_tag_handler_t::error, h).get() == valid_sequence); + CHECK(json::from_msgpack(json::to_msgpack(jval), true, true, h).get() == valid_sequence); + } + } + + SECTION("dump() with error_handler_t::keep writes raw bytes as is") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + + const json jval = c.bytes; + const std::string dumped = jval.dump(-1, ' ', false, eh::keep); + CHECK(dumped.find(c.bytes) != std::string::npos); + + // even with ensure_ascii, the ill-formed bytes are written as is + const std::string dumped_ascii = jval.dump(-1, ' ', true, eh::keep); + CHECK(dumped_ascii.find(c.bytes) != std::string::npos); + } + + // well-formed characters around an ill-formed sequence are still + // escaped as usual under ensure_ascii + const json mixed = valid_sequence + ill_formed_cases[1].bytes; // "é" + lone 0xFF + const std::string dumped_mixed = mixed.dump(-1, ' ', true, eh::keep); + CHECK(dumped_mixed.find("\\u00e9") != std::string::npos); + CHECK(dumped_mixed.find(ill_formed_cases[1].bytes) != std::string::npos); + + // the byte that ends an ill-formed sequence is read again, so a quote, + // a backslash, or a control character after it is still escaped, and + // a well-formed code point after it is escaped under ensure_ascii + for (const bool ensure_ascii : + { + false, true + }) + { + CAPTURE(ensure_ascii); + CHECK(json("\xC3\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\\"\""); + CHECK(json("\xC3\\").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\\\\""); + CHECK(json("\xC3\n").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\n\""); + CHECK(json("\xE2\x82\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xE2\x82\\\"\""); + CHECK(json("\xFF\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xFF\\\"\""); + CHECK(json("a\xE2\x82").dump(-1, ' ', ensure_ascii, eh::keep) == "\"a\xE2\x82\""); + } + CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', false, eh::keep) == "\"\xC3\xC3\xA9\""); + CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', true, eh::keep) == "\"\xC3\\u00e9\""); + } + + SECTION("to_msgpack defaults to keep; to_bon8 is not affected by error_handler") + { + const json jval = ill_formed_cases[1].bytes; // lone 0xFF + + // to_msgpack's error_handler defaults to keep, as MessagePack's spec + // allows any bytes in a str, so the bytes are passed through + CHECK(json::to_msgpack(jval) == json::to_msgpack(jval, eh::keep)); + CHECK(json::from_msgpack(json::to_msgpack(jval)).get() == ill_formed_cases[1].bytes); + + // the diagnostics context of an ill-formed key is the object + json jobj; + jobj["\xFF"] = 1; + CHECK_THROWS_WITH_AS(json::to_msgpack(jobj, eh::strict), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + + // to_bon8 has no error_handler parameter; UTF-8 is structural for + // BON8, so it always rejects ill-formed input + CHECK_THROWS_AS(json::to_bon8(jval), json::type_error&); + } + + SECTION("allow_exceptions=false with error_handler_t::strict discards the value") + { + const auto bytes = json::to_cbor(json(ill_formed_cases[0].bytes), eh::keep); + const json result = json::from_cbor(bytes, true, false, json::cbor_tag_handler_t::error, eh::strict); + CHECK(result.is_discarded()); + } + + SECTION("default parameters are unchanged") + { + const json jval = ill_formed_cases[0].bytes; + + // to_*: the default error_handler is keep, so ill-formed bytes are + // written unchanged, exactly as in release 3.12.0 (it is strict only + // if JSON_STRICT_BINARY_UTF8 is enabled, see + // unit-binary_utf8_strict.cpp) + CHECK(json::to_cbor(jval) == json::to_cbor(jval, eh::keep)); + CHECK(json::to_ubjson(jval) == json::to_ubjson(jval, false, false, eh::keep)); + CHECK(json::to_bjdata(jval) == json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep)); + { + json jobj; + jobj["k"] = jval; + CHECK(json::to_bson(jobj) == json::to_bson(jobj, eh::keep)); + } + + // from_*: the default error_handler is keep, so ill-formed bytes are + // still accepted unchanged, exactly as in release 3.12.0 + const auto cbor_bytes = json::to_cbor(jval, eh::keep); + CHECK(json::from_cbor(cbor_bytes).get() == ill_formed_cases[0].bytes); + const auto ubjson_bytes = json::to_ubjson(jval, false, false, eh::keep); + CHECK(json::from_ubjson(ubjson_bytes).get() == ill_formed_cases[0].bytes); + const auto bjdata_bytes = json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep); + CHECK(json::from_bjdata(bjdata_bytes).get() == ill_formed_cases[0].bytes); + const auto msgpack_bytes = json::to_msgpack(jval); + CHECK(json::from_msgpack(msgpack_bytes).get() == ill_formed_cases[0].bytes); + json bson_obj; + bson_obj["k"] = jval; + const auto bson_bytes = json::to_bson(bson_obj, eh::keep); + CHECK(json::from_bson(bson_bytes)["k"].get() == ill_formed_cases[0].bytes); + } +} diff --git a/tests/src/unit-binary_utf8_strict.cpp b/tests/src/unit-binary_utf8_strict.cpp new file mode 100644 index 000000000..2ac8aabf4 --- /dev/null +++ b/tests/src/unit-binary_utf8_strict.cpp @@ -0,0 +1,122 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// The binary writers check strings and object keys for valid UTF-8 only if +// JSON_STRICT_BINARY_UTF8 is enabled (planned to be the default in 4.0.0). +// Without it, they write the bytes unchanged, as before version 3.13.0; the +// tests for that are next to the other tests of each format. +#ifdef JSON_STRICT_BINARY_UTF8 + #undef JSON_STRICT_BINARY_UTF8 +#endif + +#define JSON_STRICT_BINARY_UTF8 1 + +#include +using nlohmann::json; + +#include +#include + +TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)") +{ + SECTION("CBOR") + { + // a string value with ill-formed UTF-8 is rejected + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + + // binary values are not text and are unaffected + CHECK_NOTHROW(json::to_cbor(json::binary(std::vector({0xFF})))); + + // a value read back from CBOR with ill-formed bytes cannot be written + // back either (the reader is lenient regardless of the macro) + const json j = json::from_cbor(std::vector({0x62, 0xc0, 0xae})); + CHECK_THROWS_WITH_AS(json::to_cbor(j), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + } + + SECTION("UBJSON") + { + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("BJData") + { + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("BSON") + { + // to_bson() rejects the same kind of ill-formed string value, before + // any bytes reach the output adapter (the BSON document length + // prefix must be known up front, so nothing is written incrementally) + std::vector out{0x42}; // a sentinel byte the writer must not touch + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + CHECK(out == std::vector {0x42}); + + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected as well; unlike + // the reader (which never validates element names), the writer + // checks both string values and object keys + CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("an explicit error_handler overrides the default") + { + // the macro only changes the default of the error_handler parameter + CHECK(json::to_cbor(json("\xFF"), json::error_handler_t::keep) == std::vector({0x61, 0xff})); + CHECK(json::to_ubjson(json("\xFF"), false, false, json::error_handler_t::keep) == std::vector({'S', 'i', 1, 0xff})); + CHECK(json::to_bjdata(json("\xFF"), false, false, json::bjdata_version_t::draft2, json::error_handler_t::keep) == std::vector({'S', 'i', 1, 0xff})); + CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}}, json::error_handler_t::keep)) == json{{"s", "\xFF"}}); + CHECK(json::to_cbor(json("\xFF"), json::error_handler_t::replace) == std::vector({0x63, 0xef, 0xbf, 0xbd})); + } + + SECTION("MessagePack and BON8 are unaffected") + { + // MessagePack allows any bytes in a str, so to_msgpack() still + // defaults to keep (strict only if passed explicitly); BON8 always + // checks, because the lead bytes mark where strings end + CHECK(json::to_msgpack(json("\xFF")) == std::vector({0xa1, 0xff})); + CHECK_THROWS_AS(json::to_msgpack(json("\xFF"), json::error_handler_t::strict), json::type_error&); + CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&); + } +} diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index b6be66c9e..f34df1717 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -3906,6 +3906,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK(json::to_bjdata(j) == v); CHECK(json::from_bjdata(v) == j); } + + SECTION("ill-formed UTF-8 (see #5529, #5651)") + { + // none of the binary format specs requires a decoder to reject + // ill-formed UTF-8 in a text string, so a value whose bytes are + // not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') + // round-trips byte for byte as a string value; to_bjdata() writes + // the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp) + const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; + json j; + CHECK_NOTHROW(j = json::from_bjdata(v)); + REQUIRE(j.is_string()); + CHECK(j.get_ref() == std::string("\xc0\xae")); + CHECK_THROWS_AS(j.dump(), json::type_error&); + CHECK(json::from_bjdata(json::to_bjdata(j)) == j); + + // the same bytes as an object key round-trip as well + const std::vector v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'}; + json j_key; + CHECK_NOTHROW(j_key = json::from_bjdata(v_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_bjdata(json::to_bjdata(j_key)) == j_key); + + CHECK(json::from_bjdata(json::to_bjdata(json("\xFF"))) == json("\xFF")); + // a truncated multi-byte sequence + CHECK(json::from_bjdata(json::to_bjdata(json("\xC3"))) == json("\xC3")); + // an encoded surrogate half (U+D800) + CHECK(json::from_bjdata(json::to_bjdata(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); + // an overlong encoding of '.' + CHECK(json::from_bjdata(json::to_bjdata(json("\xC0\xAF"))) == json("\xC0\xAF")); + + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_bjdata(json::to_bjdata(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + } } SECTION("Array Type") diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 2f9a727ab..92a14e6fe 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -154,6 +154,43 @@ TEST_CASE("BSON") #endif } + SECTION("ill-formed UTF-8 (see #5529, #5651)") + { + // a BSON document {"s": "\xC0\xAE"} (0xC0 0xAE is an overlong + // encoding of '.'); the BSON spec does not require a decoder to + // reject ill-formed UTF-8 in a string value, so the reader hands the + // bytes back unchanged + const std::vector v = + { + 0x0F, 0x00, 0x00, 0x00, // document length + 0x02, 's', 0x00, // type 0x02 (string), key "s" + 0x03, 0x00, 0x00, 0x00, // string length (including null) + 0xc0, 0xae, 0x00, // string content and its null terminator + 0x00 // document terminator + }; + json j; + CHECK_NOTHROW(j = json::from_bson(v)); + REQUIRE(j.is_object()); + REQUIRE(j.contains("s")); + CHECK(j["s"].get_ref() == std::string("\xc0\xae")); + // dump() still requires valid UTF-8 and throws for such a value + CHECK_THROWS_AS(j.dump(), json::type_error&); + // to_bson() writes the bytes back unchanged, as before 3.13.0, + // unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp) + CHECK(json::from_bson(json::to_bson(j)) == j); + + CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}})) == json{{"s", "\xFF"}}); + // a truncated multi-byte sequence + CHECK(json::from_bson(json::to_bson(json{{"s", "\xC3"}})) == json{{"s", "\xC3"}}); + // an encoded surrogate half (U+D800) + CHECK(json::from_bson(json::to_bson(json{{"s", "\xED\xA0\x80"}})) == json{{"s", "\xED\xA0\x80"}}); + // an overlong encoding of '.' + CHECK(json::from_bson(json::to_bson(json{{"s", "\xC0\xAF"}})) == json{{"s", "\xC0\xAF"}}); + + // an object key with ill-formed UTF-8 is kept as well + CHECK(json::from_bson(json::to_bson(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + } + SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON") { // out_of_range.412 is thrown from a single shared helper diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index c3a425a16..633f50fea 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -1801,19 +1801,41 @@ TEST_CASE("CBOR") CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&); } - SECTION("invalid UTF-8 in string (see #5529)") + SECTION("ill-formed UTF-8 in string (see #5529, #5651)") { + // RFC 8949 §3.1 leaves it up to the decoder whether to reject + // ill-formed UTF-8 in a text string; this library does not, and + // hands the original bytes back unchanged, matching the + // MessagePack reader and the behavior before #5185/#5531 (not in + // any release) + // a two-character text string (major type 3) whose bytes are not - // valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be - // rejected at decode time, matching every other kind of - // malformed binary input, rather than only failing later when - // the resulting value is dumped - json _; - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_cbor(std::vector({0x62, 0xc0, 0xae}), true, false).is_discarded()); + // valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') round-trips + // byte for byte as a string value + const std::vector ill_formed_value = {0x62, 0xc0, 0xae}; + json j_value; + CHECK_NOTHROW(j_value = json::from_cbor(ill_formed_value)); + REQUIRE(j_value.is_string()); + CHECK(j_value.get_ref() == std::string("\xc0\xae")); + // dump() still requires valid UTF-8 and throws for such a value, + // unless an error handler that replaces or ignores the bytes is + // passed + CHECK_THROWS_AS(j_value.dump(), json::type_error&); + // to_cbor() writes the bytes back unchanged, as before 3.13.0, + // unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp) + CHECK(json::from_cbor(json::to_cbor(j_value)) == j_value); + + // the same bytes as an object key round-trip as well + const std::vector ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01}; + json j_key; + CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_cbor(json::to_cbor(j_key)) == j_key); // a CBOR byte string (major type 2) with the very same bytes is // NOT text and must still be accepted as-is + json _; CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x42, 0xc0, 0xae}))); CHECK(_ == json::binary(std::vector({0xc0, 0xae}))); @@ -1822,17 +1844,47 @@ TEST_CASE("CBOR") CHECK(json::from_cbor(json::to_cbor(j)) == j); } - SECTION("invalid UTF-8 in indefinite-length string") + SECTION("to_cbor keeps ill-formed UTF-8 (see #5651)") + { + // to_cbor() writes the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp); from_cbor() reads them back as is + CHECK(json::from_cbor(json::to_cbor(json("\xFF"))) == json("\xFF")); + // a truncated multi-byte sequence + CHECK(json::from_cbor(json::to_cbor(json("\xC3"))) == json("\xC3")); + // an encoded surrogate half (U+D800) + CHECK(json::from_cbor(json::to_cbor(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); + // an overlong encoding of '.' + CHECK(json::from_cbor(json::to_cbor(json("\xC0\xAF"))) == json("\xC0\xAF")); + + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_cbor(json::to_cbor(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + + // binary values are not text and are unaffected + CHECK_NOTHROW(json::to_cbor(json::binary(std::vector({0xFF})))); + } + + SECTION("ill-formed UTF-8 in indefinite-length string") { json _; - // every chunk must be valid UTF-8 on its own (RFC 8949, Section - // 3.2.3), so a code point split across two chunks is rejected - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded()); + // the chunks are concatenated as is, without checking that each + // chunk is valid UTF-8 on its own (RFC 8949, Section 3.2.3), so + // a code point split across two chunks yields a valid string + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}))); + CHECK(_ == "\xc3\xa9"); + CHECK(_.dump() == "\"\xc3\xa9\""); - // an ill-formed later chunk is rejected after valid ones - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + // a truncated code point is kept as is + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0xff}))); + CHECK(_ == "\xc3"); + CHECK_THROWS_AS(_.dump(), json::type_error&); + CHECK(json::from_cbor(json::to_cbor(_)) == _); + + // an ill-formed later chunk is kept after valid ones + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff}))); + CHECK(_ == "\xc3\xa9\xc0\xae"); + CHECK_THROWS_AS(_.dump(), json::type_error&); // valid multi-byte chunks are accepted CHECK(json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6"); @@ -1840,9 +1892,6 @@ TEST_CASE("CBOR") SECTION("many chunks in indefinite-length string") { - // only the newly read chunk is validated, not the whole string - // collected so far; validating the latter made this input take - // quadratic time (about ten seconds for 100000 chunks) constexpr std::size_t chunks = 100000; std::vector v{0x7f}; for (std::size_t i = 0; i < chunks; ++i) diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index 223891819..04be8620a 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -2686,12 +2686,10 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") SECTION("move constructor resets the moved-from value to npos") { - // basic_json(basic_json&&) (json.hpp, around line 1951) copies + // basic_json(basic_json&&) copies // other's start_position/end_position into *this and then resets - // other's to npos (see the cppcheck-suppress[accessForwarded] - // annotation there, which flags this reset as worth a second - // look). Only the top-level moved-from value is affected; its - // (moved-away) children are gone along with it. + // other's to npos. Only the top-level moved-from value is + // affected; its (moved-away) children are gone along with it. const std::string s = R"({"a":1,"b":[1,2,3]})"; json a = json::parse(s); const auto a_start = a.start_pos(); diff --git a/tests/src/unit-conversions.cpp b/tests/src/unit-conversions.cpp index df8685529..5f83a9253 100644 --- a/tests/src/unit-conversions.cpp +++ b/tests/src/unit-conversions.cpp @@ -430,6 +430,37 @@ TEST_CASE("value conversion") CHECK(std::equal(std::begin(nbs[0][0][0]), std::end(nbs[1][1][1]), std::begin(nbs2[0][0][0]))); } + SECTION("built-in arrays: 5D") + { + // NOLINTBEGIN(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + const int nbs[1][1][1][2][2] = {{{{{0, 1}, {2, 3}}}}}; + int nbs2[1][1][1][2][2] = {{{{{0, 0}, {0, 0}}}}}; + // NOLINTEND(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + + const json j2 = nbs; + j2.get_to(nbs2); + CHECK(std::equal(std::begin(nbs[0][0][0][0]), std::end(nbs[0][0][0][1]), std::begin(nbs2[0][0][0][0]))); + } + + SECTION("built-in arrays: mismatched shape") + { + // NOLINTBEGIN(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + int nbs2[2][3] = {{0, 0, 0}, {0, 0, 0}}; + // NOLINTEND(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + + SECTION("not an array") + { + const json j2 = 42; + CHECK_THROWS_WITH_AS(j2.get_to(nbs2), "[json.exception.type_error.304] cannot use at() with number", json::type_error&); + } + + SECTION("too few elements") + { + const json j2 = {{0, 1, 2}}; + CHECK_THROWS_WITH_AS(j2.get_to(nbs2), "[json.exception.out_of_range.401] array index 1 is out of range", json::out_of_range&); + } + } + SECTION("std::deque") { std::deque a{"previous", "value"}; @@ -1748,6 +1779,34 @@ NLOHMANN_JSON_SERIALIZE_ENUM_STRICT(StrictTaskState, {STRICT_TS_COMPLETED, "completed"}, }) +// regression test for #5708 item 2: NLOHMANN_JSON_SERIALIZE_ENUM_STRICT must not rely on +// unqualified lookup of a helper name that a user's own namespace may also declare +namespace ns_with_colliding_name +{ +// NOLINTNEXTLINE(misc-use-internal-linkage) - used to shadow the library's internal helper name +inline void templated_json_throw(int /*unused*/) {} + +enum class colliding_enum { a, b }; + +// NOLINTNEXTLINE(misc-use-internal-linkage,misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - false positive +NLOHMANN_JSON_SERIALIZE_ENUM_STRICT(colliding_enum, +{ + {colliding_enum::a, "a"}, + {colliding_enum::b, "b"} +}) +} // namespace ns_with_colliding_name + +TEST_CASE("NLOHMANN_JSON_SERIALIZE_ENUM_STRICT in a namespace with a colliding name") +{ + using ns_with_colliding_name::colliding_enum; + + CHECK(json(colliding_enum::a) == "a"); + CHECK(colliding_enum::b == json("b")); + + json _; + CHECK_THROWS_WITH_AS(_ = json("nope").get(), "[json.exception.out_of_range.410] enum value out of range for colliding_enum: \"nope\"", json::out_of_range&); +} + TEST_CASE("Strict JSON to enum mapping") { SECTION("enum class") diff --git a/tests/src/unit-custom-base-class.cpp b/tests/src/unit-custom-base-class.cpp index a6b9b9ea4..38b665793 100644 --- a/tests/src/unit-custom-base-class.cpp +++ b/tests/src/unit-custom-base-class.cpp @@ -10,6 +10,8 @@ #include #include #include +#include +#include #include #include "doctest_compatibility.h" @@ -406,6 +408,74 @@ TEST_CASE("JSON Visit Node") CHECK(expected.empty()); } +// Test accessing members of a custom base class that are hidden by members of nlohmann::basic_json +class base_class_with_hidden_members +{ + public: + const char* type_name() const noexcept // NOLINT(readability-convert-member-functions-to-static) + { + return "custom type_name"; + } + + std::size_t size() const noexcept + { + return m_size; + } + + std::size_t m_size = 42; +}; + +using json_with_hidden_base_members = + nlohmann::basic_json < + std::map, + std::vector, + std::string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer, + std::vector, + base_class_with_hidden_members + >; + +TEST_CASE("JSON Node as_base_class") +{ + using json = json_with_hidden_base_members; + + static_assert(std::is_same().as_base_class()), json::json_base_class_t&>::value, ""); + static_assert(std::is_same().as_base_class()), const json::json_base_class_t&>::value, ""); + static_assert(noexcept(std::declval().as_base_class()), ""); + static_assert(noexcept(std::declval().as_base_class()), ""); + + SECTION("non-const") + { + json j = {1, 2, 3}; + + CHECK(std::string(j.type_name()) == "array"); + CHECK(j.size() == 3); + CHECK(std::string(j.as_base_class().type_name()) == "custom type_name"); + CHECK(j.as_base_class().size() == 42); + CHECK(&j.as_base_class() == &static_cast(j)); + + j.as_base_class().m_size = 7; + CHECK(j.as_base_class().size() == 7); + CHECK(j.size() == 3); + } + + SECTION("const") + { + const json j = {1, 2, 3}; + + CHECK(std::string(j.type_name()) == "array"); + CHECK(j.size() == 3); + CHECK(std::string(j.as_base_class().type_name()) == "custom type_name"); + CHECK(j.as_base_class().size() == 42); + CHECK(&j.as_base_class() == &static_cast(j)); + } +} + // A custom base class with a const member: copy-constructible (initializing a // const member works fine), but not copy-/move-assignable (assigning one does // not). Used to check that copy construction never requires more than that. diff --git a/tests/src/unit-modifiers.cpp b/tests/src/unit-modifiers.cpp index ee81120cf..532283f95 100644 --- a/tests/src/unit-modifiers.cpp +++ b/tests/src/unit-modifiers.cpp @@ -630,6 +630,49 @@ TEST_CASE("modifiers") } } + SECTION("rvalue at position moves rather than copies") + { + // regression test: insert(pos, basic_json&&) used to forward to + // insert(pos, const basic_json&) because the named rvalue + // reference parameter is itself an lvalue, so it always + // deep-copied its argument instead of moving it + json j_big = std::string(1000, 'x'); + const auto* const original_buffer = j_big.get_ref().data(); + + auto it = j_array.insert(j_array.begin(), std::move(j_big)); + CHECK(j_array.size() == 5); + CHECK(*it == json(std::string(1000, 'x'))); + CHECK((*it).get_ref().data() == original_buffer); + + // the moved-from value is null, the same as after push_back(&&) + CHECK(j_big.is_null()); // NOLINT(bugprone-use-after-move,hicpp-invalid-access-moved) + } + + SECTION("self-aliasing insertion") + { + SECTION("without reallocation") + { + json j_self = {1, 2, 3, 4}; + j_self.get_ref().reserve(j_self.size() + 1); + + auto it = j_self.insert(j_self.begin(), std::move(j_self[1])); + CHECK(j_self.size() == 5); + CHECK(*it == json(2)); + CHECK(j_self == json({2, 1, nullptr, 3, 4})); + } + + SECTION("with reallocation") + { + json j_self = {1, 2, 3, 4}; + j_self.get_ref().shrink_to_fit(); + + auto it = j_self.insert(j_self.begin(), std::move(j_self[1])); + CHECK(j_self.size() == 5); + CHECK(*it == json(2)); + CHECK(j_self == json({2, 1, nullptr, 3, 4})); + } + } + SECTION("copies at position") { SECTION("insert before begin()") diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index afabb85c6..a92496144 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -1540,19 +1540,39 @@ TEST_CASE("MessagePack") CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&); } - SECTION("invalid UTF-8 in string (see #5529)") + SECTION("ill-formed UTF-8 in string (see #5529, #5651)") { + // the MessagePack specification explicitly allows a str object to + // contain a byte sequence that is not valid UTF-8 and expects a + // deserializer to hand the original bytes back unchanged; this + // library follows that, unlike CBOR/UBJSON/BJData/BSON, whose + // specifications require text strings to be valid UTF-8 + // a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8 - // (0xC0 0xAE is an overlong encoding of '.') must be rejected at - // decode time, matching every other kind of malformed binary - // input, rather than only failing later when the resulting - // value is dumped - json _; - CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0xa2, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing MessagePack string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_msgpack(std::vector({0xa2, 0xc0, 0xae}), true, false).is_discarded()); + // (0xC0 0xAE is an overlong encoding of '.') round-trips byte for + // byte as a string value + const std::vector ill_formed_value = {0xa2, 0xc0, 0xae}; + json j_value; + CHECK_NOTHROW(j_value = json::from_msgpack(ill_formed_value)); + REQUIRE(j_value.is_string()); + CHECK(j_value.get_ref() == std::string("\xc0\xae")); + CHECK(json::from_msgpack(json::to_msgpack(j_value)) == j_value); + // dump() still requires valid UTF-8 and throws for such a value, + // unless an error handler that replaces or ignores the bytes is + // passed + CHECK_THROWS_AS(j_value.dump(), json::type_error&); + + // the same bytes as an object key round-trip as well + const std::vector ill_formed_key = {0x81, 0xa2, 0xc0, 0xae, 0x01}; + json j_key; + CHECK_NOTHROW(j_key = json::from_msgpack(ill_formed_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_msgpack(json::to_msgpack(j_key)) == j_key); // a MessagePack bin8 blob with the very same bytes is NOT text // and must still be accepted as-is + json _; CHECK_NOTHROW(_ = json::from_msgpack(std::vector({0xc4, 0x02, 0xc0, 0xae}))); CHECK(_ == json::binary(std::vector({0xc0, 0xae}))); diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index 2ab9b86cd..954331cbc 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -808,6 +808,26 @@ TEST_CASE("regression tests 2") CHECK(j == k); } +#ifdef JSON_HAS_CPP_17 + SECTION("issue #5066 - MSVC converts json to std::variant via the conversion operator") + { + // std::variant must not be retrievable via get<>(), because otherwise the + // implicit conversion operator becomes a candidate that MSVC picks over the variant's + // converting constructor, routing a number through the string from_json overload + static_assert(!nlohmann::detail::is_detected>::value, + "std::variant must not be retrievable via get<>()"); + + // clang before 7 cannot instantiate libstdc++'s std::variant +#if !(defined(__clang__) && __clang_major__ < 7) + // push_back, not emplace_back: #5066 needs the implicit conversion + // from json to the vector's value type + std::vector> v; + v.push_back(json(1)); // NOLINT(hicpp-use-emplace,modernize-use-emplace) + CHECK(std::get<0>(v[0]) == 1); +#endif + } +#endif + SECTION("issue #3669 - invalid use of incomplete type with optional member and to_json") { const Issue3669Holder h{}; @@ -1286,15 +1306,11 @@ TEST_CASE("regression test - #3989 SAX parse_error() returning true") {json::input_format_t::cbor, {0x82, 0xC1, 0x05, 0xD9, 0xD9, 0xF7, 0x06}, {5, 6}, 2}, // CBOR: undefined and other simple values become null {json::input_format_t::cbor, {0x84, 0xF7, 0xE0, 0xF8, 0x20, 0x01}, {nullptr, nullptr, nullptr, 1}, 3}, - // CBOR: ill-formed UTF-8 becomes U+FFFD, also in keys - {json::input_format_t::cbor, {0xA1, 0x61, 0xFF, 0x62, 0xC3, 0x28}, {{replacement_character(), replacement_character() + "("}}, 2}, // CBOR: members whose key is not a string are skipped, whatever their key and value {json::input_format_t::cbor, {0xA4, 0x01, 0x02, 0x82, 0x01, 0x02, 0xA1, 0x61, 'x', 0x9F, 0xFF, 0xC1, 0x01, 0x5F, 0x41, 0x00, 0xFF, 0x61, 'a', 0x03}, {{"a", 3}}, 3}, {json::input_format_t::cbor, {0xBF, 0xF5, 0xBF, 0x61, 'x', 0x7F, 0x61, 'y', 0xFF, 0xFF, 0x61, 'a', 0x03, 0xFF}, {{"a", 3}}, 1}, // MessagePack: members whose key is not a string are skipped {json::input_format_t::msgpack, {0x84, 0x01, 0x02, 0x81, 0xA1, 'x', 0x01, 0x92, 0x01, 0x02, 0xD4, 0x01, 0x02, 0xC0, 0xA1, 'a', 0x04}, {{"a", 4}}, 3}, - // MessagePack: ill-formed UTF-8 becomes U+FFFD - {json::input_format_t::msgpack, {0x92, 0xA2, 0xC3, 0x28, 0xA3, 0xE2, 0x82, 'x'}, {replacement_character() + "(", replacement_character() + "x"}, 2}, // UBJSON: a char that is not ASCII becomes U+FFFD {json::input_format_t::ubjson, {'[', 'C', 0x80, 'C', 'A', ']'}, {replacement_character(), "A"}, 1}, // UBJSON: the longest beginning of a high-precision number is kept diff --git a/tests/src/unit-type_traits.cpp b/tests/src/unit-type_traits.cpp index 6dc166d03..a3937bf2e 100644 --- a/tests/src/unit-type_traits.cpp +++ b/tests/src/unit-type_traits.cpp @@ -10,10 +10,14 @@ #if JSON_TEST_USING_MULTIPLE_HEADERS #include + #include #else #include #endif +#include +#include + TEST_CASE("type traits") { SECTION("is_c_string") @@ -83,4 +87,12 @@ TEST_CASE("type traits") } } } + + SECTION("is_ordered_map") + { + using nlohmann::detail::is_ordered_map; + + CHECK(is_ordered_map>::value); + CHECK_FALSE(is_ordered_map>::value); + } } diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index 9a9710b48..450882a12 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -2505,6 +2505,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK(json::to_ubjson(j) == v); CHECK(json::from_ubjson(v) == j); } + + SECTION("ill-formed UTF-8 (see #5529, #5651)") + { + // none of the binary format specs requires a decoder to reject + // ill-formed UTF-8 in a text string, so a value whose bytes are + // not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') + // round-trips byte for byte as a string value; to_ubjson() writes + // the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp) + const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; + json j; + CHECK_NOTHROW(j = json::from_ubjson(v)); + REQUIRE(j.is_string()); + CHECK(j.get_ref() == std::string("\xc0\xae")); + CHECK_THROWS_AS(j.dump(), json::type_error&); + CHECK(json::from_ubjson(json::to_ubjson(j)) == j); + + // the same bytes as an object key round-trip as well + const std::vector v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'}; + json j_key; + CHECK_NOTHROW(j_key = json::from_ubjson(v_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_ubjson(json::to_ubjson(j_key)) == j_key); + + CHECK(json::from_ubjson(json::to_ubjson(json("\xFF"))) == json("\xFF")); + // a truncated multi-byte sequence + CHECK(json::from_ubjson(json::to_ubjson(json("\xC3"))) == json("\xC3")); + // an encoded surrogate half (U+D800) + CHECK(json::from_ubjson(json::to_ubjson(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); + // an overlong encoding of '.' + CHECK(json::from_ubjson(json::to_ubjson(json("\xC0\xAF"))) == json("\xC0\xAF")); + + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_ubjson(json::to_ubjson(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + } } SECTION("Array Type") diff --git a/tests/src/unit-wstring.cpp b/tests/src/unit-wstring.cpp index b553ce246..c71684002 100644 --- a/tests/src/unit-wstring.cpp +++ b/tests/src/unit-wstring.cpp @@ -8,6 +8,7 @@ #include "doctest_compatibility.h" +#include #include using nlohmann::json; @@ -68,15 +69,15 @@ TEST_CASE("wide strings") CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&); // a lone low surrogate cannot start a pair - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // a high surrogate followed by a non-low-surrogate unit is invalid - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // ... also when the unit is above the low surrogates - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // a lone low surrogate must not swallow the following unit: pairing // it with any second unit would produce valid UTF-8, so the error // has to report an ill-formed byte at the surrogate's own position - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // a valid surrogate pair is still decoded (U+1F600) CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get() == "\xF0\x9F\x98\x80"); } @@ -104,4 +105,55 @@ TEST_CASE("wide strings") // the same unit inside a string is reported as an ill-formed byte CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); } + + SECTION("malformed wide-string input outside strings (#5645)") + { + json _; + + // a lone low surrogate inside a literal must not be truncated to its + // low byte and mistaken for the letter the literal expects next + // (0xDC72 truncates to 'r', which is what "true" expects after 't') + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast(0xDC72), u'u', u'e'}), + "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&); + + // ... also when the lone surrogate is the last unit of the input + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'f', u'a', u'l', u's', static_cast(0xDD65)}), + "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'fals\xFF'", json::parse_error&); + + // a high surrogate followed by a unit that is not its low surrogate + // must not silently swallow that unit + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast(0xD872), u'X', u'u', u'e'}), + "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&); + + // ... in particular, if the swallowed unit is the newline that ends a + // // comment, the comment must not extend over the following line + CHECK(json::parse(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast(0xD800), u'\n', + u',', u'2', u' ', u'/', u'/', u'\n', u']'}, + nullptr, true, /*ignore_comments*/true) == json::parse("[1,2]")); + CHECK(json::accept(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast(0xD800), u'\n', + u',', u'2', u' ', u'/', u'/', u'\n', u']'}, /*ignore_comments*/true)); + + // cases 5 and 6 use a 32-bit wchar_t (Linux, macOS, the BSDs) to reach + // the UTF-32 helper tested above via u32string; the 16-bit wchar_t of + // Windows goes through the UTF-16 helper instead, already covered by + // the u16string cases above +#if WCHAR_MAX > 0xFFFFu + // a negative wchar_t must not be mistaken for + // char_traits::eof() and silently end the input, letting + // trailing garbage pass the strict end-of-input check (only observable + // where wint_t is signed, e.g. macOS/the BSDs; on Linux wint_t is + // unsigned and this was already handled by #5348) + std::wstring w = L"[1]"; + w.push_back(static_cast(-1)); + w += L"garbage"; + CHECK(!json::accept(w)); + CHECK_THROWS_WITH_AS(_ = json::parse(w), + "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&); + + // other negative wchar_t units must not be truncated to their low + // byte (0xFFFFFF72 truncates to 'r', as in the u16string case above) + CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L't', static_cast(0xFFFFFF72), L'u', L'e'}), + "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&); +#endif + } }