From 73e9eae3c135e262dac3fe3e7978b46694e531d3 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 4 Oct 2026 12:13:49 +0200 Subject: [PATCH] Add an error_handler parameter for UTF-8 to the binary readers and writers (#5746) Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/dump.md | 9 +- .../docs/api/basic_json/error_handler_t.md | 32 +- .../mkdocs/docs/api/basic_json/from_bjdata.md | 15 +- docs/mkdocs/docs/api/basic_json/from_bson.md | 15 +- docs/mkdocs/docs/api/basic_json/from_cbor.md | 18 +- .../docs/api/basic_json/from_msgpack.md | 18 +- .../mkdocs/docs/api/basic_json/from_ubjson.md | 15 +- docs/mkdocs/docs/api/basic_json/to_bjdata.md | 26 +- docs/mkdocs/docs/api/basic_json/to_bson.md | 28 +- docs/mkdocs/docs/api/basic_json/to_cbor.md | 26 +- docs/mkdocs/docs/api/basic_json/to_msgpack.md | 20 +- docs/mkdocs/docs/api/basic_json/to_ubjson.md | 26 +- .../api/macros/json_strict_binary_utf8.md | 18 +- docs/mkdocs/docs/examples/error_handler_t.cpp | 3 +- .../docs/examples/error_handler_t.output | 1 + .../docs/features/binary_formats/bjdata.md | 24 +- .../docs/features/binary_formats/bson.md | 19 +- .../docs/features/binary_formats/cbor.md | 17 +- .../features/binary_formats/messagepack.md | 18 +- .../docs/features/binary_formats/ubjson.md | 24 +- docs/mkdocs/docs/home/exceptions.md | 18 +- .../nlohmann/detail/input/binary_reader.hpp | 114 ++- .../nlohmann/detail/output/binary_writer.hpp | 213 +++-- .../nlohmann/detail/output/error_handler.hpp | 50 ++ include/nlohmann/detail/output/serializer.hpp | 72 +- include/nlohmann/detail/string_utils.hpp | 130 ++++ include/nlohmann/json.hpp | 135 ++-- single_include/nlohmann/json.hpp | 734 ++++++++++++++---- tests/src/unit-binary_utf8_error_handler.cpp | 362 +++++++++ tests/src/unit-binary_utf8_strict.cpp | 16 +- 30 files changed, 1794 insertions(+), 422 deletions(-) create mode 100644 include/nlohmann/detail/output/error_handler.hpp create mode 100644 tests/src/unit-binary_utf8_error_handler.cpp diff --git a/docs/mkdocs/docs/api/basic_json/dump.md b/docs/mkdocs/docs/api/basic_json/dump.md index a3c9db3a8..d805f3db1 100644 --- a/docs/mkdocs/docs/api/basic_json/dump.md +++ b/docs/mkdocs/docs/api/basic_json/dump.md @@ -25,10 +25,12 @@ and `ensure_ascii` parameters. result consists of ASCII characters only. `error_handler` (in) -: how to react on decoding errors; there are three possible values (see [`error_handler_t`](error_handler_t.md): +: how to react on decoding errors; there are four possible values (see [`error_handler_t`](error_handler_t.md): `strict` (throws an exception in case a decoding error occurs; default), `replace` (replace invalid UTF-8 sequences - with U+FFFD), and `ignore` (ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the - output unchanged, and invalid bytes are dropped)). + with U+FFFD), `ignore` (ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the + output unchanged, and invalid bytes are dropped), and `keep` (write the ill-formed bytes to the output as is, + without escaping them, even if `ensure_ascii` is `#!cpp true`; the result is then not valid UTF-8, but equals the + input bytes exactly, and well-formed characters around the ill-formed bytes are still escaped as usual)). ## Return value @@ -94,3 +96,4 @@ Binary values are serialized as an object containing two keys: - Indentation character `indent_char`, option `ensure_ascii` and exceptions added in version 3.0.0. - Error handlers added in version 3.4.0. - Serialization of binary values added in version 3.8.0. +- Error handler `keep` added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/error_handler_t.md b/docs/mkdocs/docs/api/basic_json/error_handler_t.md index 51dc6510f..17327fe26 100644 --- a/docs/mkdocs/docs/api/basic_json/error_handler_t.md +++ b/docs/mkdocs/docs/api/basic_json/error_handler_t.md @@ -4,15 +4,31 @@ enum class error_handler_t { strict, replace, - ignore + ignore, + keep }; ``` -This enumeration is used in the [`dump`](dump.md) function to choose how to treat decoding errors while serializing a -`basic_json` value. Three values are differentiated: +This enumeration is used to choose how to treat ill-formed UTF-8 in a string value or object key: + +- [`dump`](dump.md) uses it while serializing a `basic_json` value to text. +- [`to_cbor`](to_cbor.md), [`to_msgpack`](to_msgpack.md), [`to_ubjson`](to_ubjson.md), [`to_bjdata`](to_bjdata.md), + and [`to_bson`](to_bson.md) use it while serializing a `basic_json` value to that binary format. Their default is + `keep`, as no binary writer checked before this parameter was added. CBOR, UBJSON, BJData, and BSON require valid + UTF-8, so for these four the default is `strict` if [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) + is enabled; MessagePack's specification explicitly allows a string to contain ill-formed UTF-8, so `to_msgpack` + stays at `keep`. `to_bon8` does not take this parameter: BON8 always validates, since UTF-8 lead bytes are + structural to that format. +- [`from_cbor`](from_cbor.md), [`from_msgpack`](from_msgpack.md), [`from_ubjson`](from_ubjson.md), + [`from_bjdata`](from_bjdata.md), and [`from_bson`](from_bson.md) use it while parsing that binary format, to decide + whether to check a string value or object key for well-formed UTF-8 at all; by default (`keep`) they do not, as no + binary reader did before this parameter was added. `from_bon8` does not take this parameter, for the same reason + `to_bon8` does not. + +Four values are differentiated: strict -: throw a `type_error` exception in case of invalid UTF-8 +: throw a `type_error`/`parse_error` exception in case of invalid UTF-8 replace : replace invalid UTF-8 sequences with U+FFFD (� REPLACEMENT CHARACTER) @@ -20,6 +36,12 @@ replace ignore : ignore invalid UTF-8 sequences; all valid bytes are copied to the output unchanged, and invalid bytes are dropped +keep +: keep invalid UTF-8 sequences unchanged; only meaningful for the binary formats mentioned above, since [`dump`] + (dump.md) itself must produce text, and `keep` there writes the ill-formed bytes to the output as is, so the + result is then not valid UTF-8 (but still equals the input bytes exactly, including around any well-formed + characters, which are still escaped as usual) + ## Examples ??? example @@ -45,3 +67,5 @@ ignore ## Version history - Added in version 3.4.0. +- Added `keep`, and made this enumeration apply to the binary readers and writers in addition to `dump`, in version + 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/from_bjdata.md b/docs/mkdocs/docs/api/basic_json/from_bjdata.md index 9df21ab30..a45e00ad5 100644 --- a/docs/mkdocs/docs/api/basic_json/from_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/from_bjdata.md @@ -5,12 +5,14 @@ template static basic_json from_bjdata(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_bjdata(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the BJData (Binary JData) serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + BJData does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -73,7 +81,7 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if a parse error occurs - Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string could not be parsed - successfully + successfully, or if a string value or object key is not valid UTF-8 and `error_handler` is `strict` - Throws [out_of_range.408](../../home/exceptions.md#jsonexceptionout_of_range408) if the size of an optimized container or n-dimensional array cannot be represented by `std::size_t` @@ -111,3 +119,4 @@ Linear in the size of the input. - Added in version 3.11.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/from_bson.md b/docs/mkdocs/docs/api/basic_json/from_bson.md index 9dfd9dc18..b037e8e07 100644 --- a/docs/mkdocs/docs/api/basic_json/from_bson.md +++ b/docs/mkdocs/docs/api/basic_json/from_bson.md @@ -5,12 +5,14 @@ template static basic_json from_bson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_bson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the BSON (Binary JSON) serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + BSON does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -75,6 +83,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va invalid string or byte array length) - Throws [`parse_error.114`](../../home/exceptions.md#jsonexceptionparse_error114) if an unsupported BSON record type is encountered +- Throws [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) if a string value or object key is + not valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -111,6 +121,7 @@ Linear in the size of the input. - Added in version 3.4.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/from_cbor.md b/docs/mkdocs/docs/api/basic_json/from_cbor.md index 791c183bb..7d7df08a5 100644 --- a/docs/mkdocs/docs/api/basic_json/from_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/from_cbor.md @@ -6,14 +6,16 @@ template static basic_json from_cbor(InputType&& i, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error); + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_cbor(IteratorType first, SentinelType last, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error); + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the CBOR (Concise Binary Object Representation) serialization format. @@ -65,6 +67,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ : how to treat CBOR tags (optional, `error` by default); see [`cbor_tag_handler_t`](cbor_tag_handler_t.md) for more information +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + CBOR does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -80,8 +88,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from CBOR were used in the given input or if the input is not valid CBOR -- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other - types are not supported, as JSON object keys are always strings) or a string is malformed +- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of + other types are not supported, as JSON object keys are always strings), or if a string value or object key is not + valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -121,6 +130,7 @@ Linear in the size of the input. - Added `tag_handler` parameter in version 3.9.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/from_msgpack.md b/docs/mkdocs/docs/api/basic_json/from_msgpack.md index 395512acb..2e7fa5062 100644 --- a/docs/mkdocs/docs/api/basic_json/from_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/from_msgpack.md @@ -5,12 +5,14 @@ template static basic_json from_msgpack(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_msgpack(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the MessagePack serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + MessagePack's specification explicitly allows ill-formed UTF-8, so checking is opt-in: the default, `keep`, does + not check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -73,8 +81,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from MessagePack were used in the given input or if the input is not valid MessagePack -- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other - types are not supported, as JSON object keys are always strings) or a string is malformed +- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of + other types are not supported, as JSON object keys are always strings), or if a string value or object key is not + valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -113,6 +122,7 @@ Linear in the size of the input. - Added `allow_exceptions` parameter in version 3.2.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/from_ubjson.md b/docs/mkdocs/docs/api/basic_json/from_ubjson.md index 1ad076588..ac9404d4c 100644 --- a/docs/mkdocs/docs/api/basic_json/from_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/from_ubjson.md @@ -5,12 +5,14 @@ template static basic_json from_ubjson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); // (2) template static basic_json from_ubjson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true); + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep); ``` Deserializes a given input to a JSON value using the UBJSON (Universal Binary JSON) serialization format. @@ -58,6 +60,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `allow_exceptions` (in) : whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) +`error_handler` (in) +: how to treat a string value or object key that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + UBJSON does not require a decoder to reject ill-formed UTF-8, so checking is opt-in: the default, `keep`, does not + check at all, as every binary reader did before this parameter was added; `strict` checks and throws; + `replace`/`ignore` sanitize the string the same way [`dump`](dump.md) would + ## Return value deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be @@ -73,7 +81,7 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if a parse error occurs - Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string could not be parsed - successfully + successfully, or if a string value or object key is not valid UTF-8 and `error_handler` is `strict` - Throws [out_of_range.408](../../home/exceptions.md#jsonexceptionout_of_range408) if the size of an optimized container or n-dimensional array cannot be represented by `std::size_t` @@ -112,6 +120,7 @@ Linear in the size of the input. - Added `allow_exceptions` parameter in version 3.2.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/to_bjdata.md b/docs/mkdocs/docs/api/basic_json/to_bjdata.md index 63dc5379e..df3b69004 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/to_bjdata.md @@ -5,15 +5,18 @@ static std::vector to_bjdata(const basic_json& j, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2); + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = error_handler_t::keep); // (2) static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2); + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = error_handler_t::keep); static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2); + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the BJData (Binary JData) serialization format. BJData aims to @@ -43,6 +46,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ : which version of BJData to use (see note on "Binary values" on [BJData](../../features/binary_formats/bjdata.md)); optional, `#!cpp bjdata_version_t::draft2` by default. +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_bjdata` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. BJData serialization as byte vector @@ -56,9 +65,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false, and `j` contains a non-empty array, object, or binary value. -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not - valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are - written unchanged +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -108,5 +117,6 @@ Linear in the size of the JSON value `j`. - Added in version 3.11.0. - BJData version parameter (for draft3 binary encoding) added in version 3.12.0. -- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if - [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. \ No newline at end of file +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`. \ No newline at end of file diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index e3104b13c..ad974e048 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -2,11 +2,14 @@ ```cpp // (1) -static std::vector to_bson(const basic_json& j); +static std::vector to_bson(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep); // (2) -static void to_bson(const basic_json& j, detail::output_adapter o); -static void to_bson(const basic_json& j, detail::output_adapter o); +static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); +static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); ``` BSON (Binary JSON) is a binary format in which zero or more ordered key/value pairs are stored as a single entity (a @@ -25,6 +28,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `o` (in) : output adapter to write serialization to +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_bson` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. BSON serialization as a byte vector @@ -46,9 +55,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the BSON binary subtype; example: `"subtype 70000 is too large for the BSON binary subtype (max 255)"` -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is not valid - UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are - written unchanged +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -101,6 +110,7 @@ pass before anything is written. - Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. - Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0. - `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0. -- Throwing `type_error.316` for a string value or object key that is not valid UTF-8 if - [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, detected before anything is written, - added in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316` before anything + is written. diff --git a/docs/mkdocs/docs/api/basic_json/to_cbor.md b/docs/mkdocs/docs/api/basic_json/to_cbor.md index cac8a0917..c72f310bb 100644 --- a/docs/mkdocs/docs/api/basic_json/to_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/to_cbor.md @@ -2,11 +2,14 @@ ```cpp // (1) -static std::vector to_cbor(const basic_json& j); +static std::vector to_cbor(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep); // (2) -static void to_cbor(const basic_json& j, detail::output_adapter o); -static void to_cbor(const basic_json& j, detail::output_adapter o); +static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); +static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the CBOR (Concise Binary Object Representation) serialization @@ -26,6 +29,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `o` (in) : output adapter to write serialization to +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_cbor` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. CBOR serialization as a byte vector @@ -37,9 +46,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va ## Exceptions -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not - valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are - written unchanged +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -74,5 +83,6 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Compact representation of floating-point numbers added in version 3.8.0. -- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if - [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`. diff --git a/docs/mkdocs/docs/api/basic_json/to_msgpack.md b/docs/mkdocs/docs/api/basic_json/to_msgpack.md index 707de9514..17b9fe29c 100644 --- a/docs/mkdocs/docs/api/basic_json/to_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/to_msgpack.md @@ -2,11 +2,14 @@ ```cpp // (1) -static std::vector to_msgpack(const basic_json& j); +static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep); // (2) -static void to_msgpack(const basic_json& j, detail::output_adapter o); -static void to_msgpack(const basic_json& j, detail::output_adapter o); +static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); +static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the MessagePack serialization format. MessagePack is a binary @@ -25,6 +28,13 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `o` (in) : output adapter to write serialization to +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_msgpack` did before + this parameter was added and as the MessagePack specification allows; `strict` throws; `replace`/`ignore` sanitize + it the same way [`dump`](dump.md) would. Unlike the other binary writers, the default stays `keep` even if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled. + ## Return value 1. MessagePack serialization as a byte vector @@ -42,6 +52,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the MessagePack ext type; example: `"subtype 70000 is too large for the MessagePack ext type (max 255)"` +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -91,6 +103,8 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before. - Fixed in version 3.13.0 to serialize `number_integer_t`/`number_unsigned_t` pairs of different width correctly; before, integers could be serialized with the wrong value if `number_integer_t` was narrower than `number_unsigned_t`. diff --git a/docs/mkdocs/docs/api/basic_json/to_ubjson.md b/docs/mkdocs/docs/api/basic_json/to_ubjson.md index 8f2ea5eb9..6860c01c6 100644 --- a/docs/mkdocs/docs/api/basic_json/to_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/to_ubjson.md @@ -4,13 +4,16 @@ // (1) static std::vector to_ubjson(const basic_json& j, const bool use_size = false, - const bool use_type = false); + const bool use_type = false, + const error_handler_t error_handler = error_handler_t::keep); // (2) static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false); + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = error_handler_t::keep); static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false); + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the UBJSON (Universal Binary JSON) serialization format. UBJSON @@ -36,6 +39,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ : whether to add type annotations to container types (must be combined with `#!cpp use_size = true`); optional, `#!cpp false` by default. +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_ubjson` did before + this parameter was added; `strict` throws; `replace`/`ignore` sanitize it the same way [`dump`](dump.md) would. + If [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, the default is `strict` instead. + ## Return value 1. UBJSON serialization as a byte vector @@ -49,9 +58,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false, and `j` contains a non-empty array, object, or binary value. -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not - valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are - written unchanged +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` (the default only if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) ## Complexity @@ -100,5 +109,6 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.1.0. -- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if - [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before; `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`. diff --git a/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md index 7fe6d9a58..ed7bbe6b3 100644 --- a/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md +++ b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md @@ -4,15 +4,18 @@ #define JSON_STRICT_BINARY_UTF8 /* value */ ``` -When defined to `1`, the binary writers [`to_cbor`](../basic_json/to_cbor.md), [`to_ubjson`](../basic_json/to_ubjson.md), -[`to_bjdata`](../basic_json/to_bjdata.md), and [`to_bson`](../basic_json/to_bson.md) check every string value and -object key for valid UTF-8 and throw [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for -ill-formed UTF-8, like [`dump`](../basic_json/dump.md) does. Without it, they write the bytes unchanged. +When defined to `1`, the `error_handler` parameter of the binary writers [`to_cbor`](../basic_json/to_cbor.md), +[`to_ubjson`](../basic_json/to_ubjson.md), [`to_bjdata`](../basic_json/to_bjdata.md), and +[`to_bson`](../basic_json/to_bson.md) defaults to [`error_handler_t::strict`](../basic_json/error_handler_t.md) instead +of `error_handler_t::keep`. These writers then check every string value and object key for valid UTF-8 and throw +[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, like +[`dump`](../basic_json/dump.md) does. Without it, they write the bytes unchanged. An `error_handler` passed explicitly +always takes precedence. The macro does not affect: - [`to_msgpack`](../basic_json/to_msgpack.md): the MessagePack specification allows a `str` value to contain bytes that - are not valid UTF-8, so it always writes them unchanged. + are not valid UTF-8, so its `error_handler` always defaults to `keep`. - [`to_bon8`](../basic_json/to_bon8.md): BON8 always checks, because the UTF-8 lead bytes mark where a string ends. - The binary readers ([`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md), [`from_ubjson`](../basic_json/from_ubjson.md), [`from_bjdata`](../basic_json/from_bjdata.md), @@ -33,8 +36,9 @@ The default value is `0` (disabled, the behavior of version 3.12.0 and earlier i CBOR, UBJSON, BJData, and BSON all require strings to be UTF-8. Up to version 3.12.0, the writers did not check this, so they could produce output that other decoders reject. Checking by default would break code that stores - other encodings (for instance ISO 8859-1) in a string and only ever writes it to a binary format, so this macro - offers the check as an opt-in ahead of version 4.0.0, where it is planned to become the default (see + other encodings (for instance ISO 8859-1) in a string and only ever writes it to a binary format. You can pass + `error_handler_t::strict` to each call, or use this macro to check by default ahead of version 4.0.0, where + `strict` is planned to become the default (see [#5529](https://github.com/nlohmann/json/issues/5529) and [#5651](https://github.com/nlohmann/json/issues/5651)). !!! warning "Opt-in only" diff --git a/docs/mkdocs/docs/examples/error_handler_t.cpp b/docs/mkdocs/docs/examples/error_handler_t.cpp index b4718d7e6..bf035cea3 100644 --- a/docs/mkdocs/docs/examples/error_handler_t.cpp +++ b/docs/mkdocs/docs/examples/error_handler_t.cpp @@ -20,5 +20,6 @@ int main() << j_invalid.dump(-1, ' ', false, json::error_handler_t::replace) << "\nstring with ignored invalid characters: " << j_invalid.dump(-1, ' ', false, json::error_handler_t::ignore) - << '\n'; + << "\nstring with the invalid byte kept as is (" << j_invalid.dump(-1, ' ', false, json::error_handler_t::keep).size() + << " bytes, not valid UTF-8 itself)\n"; } diff --git a/docs/mkdocs/docs/examples/error_handler_t.output b/docs/mkdocs/docs/examples/error_handler_t.output index 718d62bee..37cae62a2 100644 --- a/docs/mkdocs/docs/examples/error_handler_t.output +++ b/docs/mkdocs/docs/examples/error_handler_t.output @@ -1,3 +1,4 @@ [json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9 string with replaced invalid characters: "ä�ü" string with ignored invalid characters: "äü" +string with the invalid byte kept as is (7 bytes, not valid UTF-8 itself) diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index 12d1af51b..3e3d8e885 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -65,10 +65,12 @@ The library uses the following mapping from JSON values types to BJData types ac !!! warning "UTF-8 validation of string values and object keys" - BJData strings must use UTF-8 encoding. By default, `to_bjdata()` writes the bytes of string values and object keys - unchanged, even if they are not valid UTF-8. If - [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead. + BJData strings must use UTF-8 encoding. By default (the [`error_handler`](../../api/basic_json/to_bjdata.md) + parameter left at `keep`), `to_bjdata()` writes the bytes of string values and object keys unchanged, even if they + are not valid UTF-8. With `error_handler_t::strict`, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead; + `replace`/`ignore` sanitize the string. [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) + makes `strict` the default. !!! info "Unused BJData markers" @@ -217,12 +219,16 @@ The library maps BJData types to JSON value types as follows: !!! warning "Ill-formed UTF-8 in string values and object keys" - BJData strings must use UTF-8 encoding, but this is not enforced on read: `from_bjdata()` accepts a string - value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, + BJData strings must use UTF-8 encoding, but checking it on read is opt-in: with the + [`error_handler`](../../api/basic_json/from_bjdata.md) parameter left at `keep` (the default), `from_bjdata()` + accepts a string value or object key whose bytes are not valid UTF-8 and hands them back unchanged. Passing + `error_handler_t::strict` makes `from_bjdata()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error - handler is passed that replaces or ignores the ill-formed bytes. By default, `to_bjdata()` writes such a value - back unchanged (see above). + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read with the default + `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. `to_bjdata()`'s + own `error_handler` parameter defaults to `keep` (see above), so such a value is written back unchanged. !!! info "Round trips" diff --git a/docs/mkdocs/docs/features/binary_formats/bson.md b/docs/mkdocs/docs/features/binary_formats/bson.md index 245ed3ebf..f205cba17 100644 --- a/docs/mkdocs/docs/features/binary_formats/bson.md +++ b/docs/mkdocs/docs/features/binary_formats/bson.md @@ -112,13 +112,18 @@ The library maps BSON record types to JSON value types as follows: !!! warning "Ill-formed UTF-8 in string values" The BSON specification requires `string` values (type `0x02`) to be valid UTF-8, but this is not required of a - decoder. `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back unchanged. - However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error handler is - passed that replaces or ignores the ill-formed bytes. By default, `to_bson()` writes such a string value or element - (key) name unchanged; if [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it - throws the same exception instead. Element (key) names are never validated on read, since they are read byte-by-byte - as a C string. `binary` values (type `0x05`) are unaffected, since they are not required to hold text. + decoder, so checking is opt-in: with the [`error_handler`](../../api/basic_json/from_bson.md) parameter left at + `keep` (the default), `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back + unchanged. Passing `error_handler_t::strict` makes `from_bson()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md) + still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a + value read with the default `keep` handler, unless an error handler is passed that replaces or ignores the + ill-formed bytes. `to_bson()`'s own `error_handler` parameter defaults to `keep`, so such a string value or element + (key) name is written unchanged; with `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled), it throws the same exception + instead. Element (key) names are never validated on read, since they are read byte-by-byte as a C string. `binary` + values (type `0x05`) are unaffected, since they are not required to hold text. ??? example "Example: deserialize a JSON value from BSON" diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index 66300c5fa..7b5be4631 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -192,12 +192,17 @@ The library maps CBOR types to JSON value types as follows: !!! warning "Ill-formed UTF-8 in text strings" [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings (major - type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this. This library does not: - `from_cbor()` accepts a text string (object keys included) whose bytes are not valid UTF-8 and hands them back - unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error handler is - passed that replaces or ignores the ill-formed bytes. By default, `to_cbor()` writes such a value back unchanged; if - [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws the same exception + type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this, so checking is opt-in: with the + [`error_handler`](../../api/basic_json/from_cbor.md) parameter left at `keep` (the default), `from_cbor()` accepts a + text string (object keys included) whose bytes are not valid UTF-8 and hands them back unchanged. Passing + `error_handler_t::strict` makes `from_cbor()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md) + still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a + value read with the default `keep` handler, unless an error handler is passed that replaces or ignores the + ill-formed bytes. `to_cbor()`'s own [`error_handler`](../../api/basic_json/to_cbor.md) parameter defaults to `keep`, + so such a value is written back unchanged; with `strict` (the default if + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled), it throws the same exception instead. Byte strings (major type 2) are unaffected, since they are not required to hold text. !!! warning "Tagged items" diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index 047944852..24eeab139 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -157,11 +157,19 @@ The library maps MessagePack types to JSON value types as follows: The MessagePack specification explicitly allows a `str` value (`fixstr`, `str 8`, `str 16`, `str 32`) to contain a byte sequence that is not valid UTF-8, and expects a deserializer to hand the original bytes back unchanged. - This library follows that: `from_msgpack()` reads `str` bytes (object keys included) as-is, without validating - them, and `to_msgpack()` writes them back as-is, so such a value round-trips through `from_msgpack(to_msgpack(j))` - byte for byte. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way, unless an - error handler is passed that replaces or ignores the ill-formed bytes. + This library follows that by default: with its + [`error_handler`](../../api/basic_json/from_msgpack.md) parameter left at `keep` (the default), + `from_msgpack()` reads `str` bytes (object keys included) as-is, without validating them, so such a value + round-trips through `from_msgpack(to_msgpack(j))` byte for byte. Passing `error_handler_t::strict` makes + `from_msgpack()` check anyway and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. `to_msgpack()` also writes `str` bytes as-is by + default, since the specification permits it; its [`error_handler`](../../api/basic_json/to_msgpack.md) parameter + can be set to `strict` to throw [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) instead, or + to `replace`/`ignore` to sanitize the string, for instance for a decoder that rejects ill-formed UTF-8. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way with the + default `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. ??? example "Example: deserialize a JSON value from MessagePack" diff --git a/docs/mkdocs/docs/features/binary_formats/ubjson.md b/docs/mkdocs/docs/features/binary_formats/ubjson.md index dbdff6e6c..f7a14d855 100644 --- a/docs/mkdocs/docs/features/binary_formats/ubjson.md +++ b/docs/mkdocs/docs/features/binary_formats/ubjson.md @@ -49,10 +49,12 @@ The library uses the following mapping from JSON values types to UBJSON types ac !!! warning "UTF-8 validation of string values and object keys" - UBJSON's required string encoding is UTF-8. By default, `to_ubjson()` writes the bytes of string values and object - keys unchanged, even if they are not valid UTF-8. If - [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead. + UBJSON's required string encoding is UTF-8. By default (the [`error_handler`](../../api/basic_json/to_ubjson.md) + parameter left at `keep`), `to_ubjson()` writes the bytes of string values and object keys unchanged, even if they + are not valid UTF-8. With `error_handler_t::strict`, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead; + `replace`/`ignore` sanitize the string. [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) + makes `strict` the default. !!! info "Unused UBJSON markers" @@ -129,12 +131,16 @@ The library maps UBJSON types to JSON value types as follows: !!! warning "Ill-formed UTF-8 in string values and object keys" - UBJSON's required string encoding is UTF-8, but this is not enforced on read: `from_ubjson()` accepts a string - value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, + UBJSON's required string encoding is UTF-8, but checking it on read is opt-in: with the + [`error_handler`](../../api/basic_json/from_ubjson.md) parameter left at `keep` (the default), `from_ubjson()` + accepts a string value or object key whose bytes are not valid UTF-8 and hands them back unchanged. Passing + `error_handler_t::strict` makes `from_ubjson()` check and throw + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and + `replace`/`ignore` sanitize the string instead of keeping it. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error - handler is passed that replaces or ignores the ill-formed bytes. By default, `to_ubjson()` writes such a value - back unchanged (see above). + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read with the default + `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. `to_ubjson()`'s + own `error_handler` parameter defaults to `keep` (see above), so such a value is written back unchanged. ??? example "Example: deserialize a JSON value from UBJSON" diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index a3797cf04..e7eb8fd03 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -340,9 +340,11 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde ### json.exception.parse_error.113 A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a -string was read where one was required (for instance as a map key), or the string's length specification is invalid. -The bytes of a string itself are not checked for valid UTF-8 on read; see the ill-formed UTF-8 notes on the -individual [binary format](../features/binary_formats/index.md) pages for how such a string is handled afterward. +string was read where one was required (for instance as a map key), the string's length specification is invalid, or +the string's bytes are not valid UTF-8 and the `error_handler` parameter of the corresponding `from_*` function is +set to `strict`. By default (`error_handler_t::keep`), the bytes of a string are not checked for valid UTF-8 on read; +see the ill-formed UTF-8 notes on the individual [binary format](../features/binary_formats/index.md) pages for how +such a string is handled depending on `error_handler`. CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other type (for instance integers or `null`) are therefore not supported; see the notes on @@ -365,6 +367,9 @@ type (for instance integers or `null`) are therefore not supported; see the note ``` [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData string: string length must not be negative ``` + ``` + [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte + ``` ### json.exception.parse_error.114 @@ -747,10 +752,11 @@ The [`unflatten()`](../api/basic_json/unflatten.md) function only works for an o The [`dump()`](../api/basic_json/dump.md) function only works with UTF-8 encoded strings; that is, if you assign a `std::string` to a JSON value, make sure it is UTF-8 encoded. See the FAQ entry on [serializing untrusted or invalid UTF-8](faq.md#serializing-untrusted-or-invalid-utf-8) for background and the recommended fix. -If [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled, the binary writers -[`to_cbor()`](../api/basic_json/to_cbor.md), [`to_ubjson()`](../api/basic_json/to_ubjson.md), +The binary writers [`to_cbor()`](../api/basic_json/to_cbor.md), [`to_ubjson()`](../api/basic_json/to_ubjson.md), [`to_bjdata()`](../api/basic_json/to_bjdata.md), and [`to_bson()`](../api/basic_json/to_bson.md) throw this exception -for a string value or object key that is not valid UTF-8 as well. +as well for a string value or object key that is not valid UTF-8 if their `error_handler` is `strict` (the default if +[`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled). So does +[`to_msgpack()`](../api/basic_json/to_msgpack.md) if `error_handler_t::strict` is passed. !!! failure "Example message" diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 3d79ad41b..1a78a6502 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -31,6 +31,7 @@ #include #include #include +#include #include #include #include @@ -108,8 +109,16 @@ class binary_reader @brief create a binary reader @param[in] adapter input adapter to read from + @param[in] format the binary format to parse + @param[in] error_handler how to treat text strings and object keys that + are not well-formed UTF-8; none of the supported formats + requires a decoder to reject those, so the default is to + @ref error_handler_t::keep them unchanged, as every binary + reader did before this parameter existed */ - explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json) noexcept : ia(std::move(adapter)), input_format(format) + explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json, + const error_handler_t error_handler = error_handler_t::keep) noexcept + : ia(std::move(adapter)), input_format(format), error_handler(error_handler) { (void)detail::is_sax_static_asserts {}; } @@ -428,7 +437,7 @@ class binary_reader { if (get_bson_cstr_bulk(result, std::integral_constant {})) { - return true; + return check_string_utf8(result, "key"); } auto out = std::back_inserter(result); @@ -441,7 +450,7 @@ class binary_reader } if (current == 0x00) { - return true; + return check_string_utf8(result, "key"); } *out++ = static_cast(current); } @@ -522,7 +531,7 @@ class binary_reader "string"), nullptr)); } - return true; + return check_string_utf8(result, "string"); } /*! @@ -1149,7 +1158,7 @@ class binary_reader @return whether string creation completed */ - bool get_cbor_string(string_t& result) + bool get_cbor_string(string_t& result, const char* context = "string") { // number of indefinite-length strings that have been opened and not // closed yet. RFC 8949, Section 3.2.3 does not permit nesting them, @@ -1179,7 +1188,7 @@ class binary_reader { if (--open == 0) { - return true; + return check_string_utf8(result, context); } get(); continue; @@ -1192,7 +1201,7 @@ class binary_reader if (open == 0) { - return true; + return check_string_utf8(result, context); } get(); @@ -1216,7 +1225,7 @@ class binary_reader // EOF and major type 3 (text string) are left to get_cbor_string if (current == char_traits::eof() || (static_cast(current) & 0xE0u) == 0x60u) { - return get_cbor_string(result); + return get_cbor_string(result, "key"); } const char* found = nullptr; @@ -2004,7 +2013,7 @@ class binary_reader @return whether string creation completed */ - bool get_msgpack_string(string_t& result) + bool get_msgpack_string(string_t& result, const char* context = "string") { if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "string"))) { @@ -2047,25 +2056,25 @@ class binary_reader case 0xBE: case 0xBF: { - return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result); + return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result) && check_string_utf8(result, context); } case 0xD9: // str 8 { std::uint8_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDA: // str 16 { std::uint16_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDB: // str 32 { std::uint32_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } default: @@ -2143,7 +2152,7 @@ class binary_reader // byte 0xC1 are left to get_msgpack_string if (current == char_traits::eof()) { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } if (current <= 0x7F || current >= 0xE0) { @@ -2159,7 +2168,7 @@ class binary_reader } else { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } break; } @@ -2405,7 +2414,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, true, "key") || !sax->key(key))) { return false; } @@ -2427,7 +2436,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false, "key") || !sax->key(key))) { return false; } @@ -2495,7 +2504,7 @@ class binary_reader @return whether string creation completed */ - bool get_ubjson_string(string_t& result, const bool get_char = true) + bool get_ubjson_string(string_t& result, const bool get_char = true, const char* context = "string") { if (get_char) { @@ -2516,31 +2525,31 @@ class binary_reader case 'U': { std::uint8_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'i': { std::int8_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'I': { std::int16_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'l': { std::int32_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'L': { std::int64_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'u': @@ -2550,7 +2559,7 @@ class binary_reader break; } std::uint16_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'm': @@ -2560,7 +2569,7 @@ class binary_reader break; } std::uint32_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'M': @@ -2570,7 +2579,7 @@ class binary_reader break; } std::uint64_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } default: @@ -4044,15 +4053,53 @@ class binary_reader const NumberType len, string_t& result) { - // Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the - // choice to the decoder), MessagePack (whose spec explicitly allows - // a str object to contain an invalid byte sequence), UBJSON, BJData, - // or BSON requires a decoder to reject ill-formed UTF-8. The bytes - // are kept unchanged; dump() and the binary writers are the ones - // that check them and report type_error.316 if they are not valid. + // Strings are taken as is by default: none of CBOR (RFC 8949 §3.1 + // leaves the choice to the decoder), MessagePack (whose spec + // explicitly allows a str object to contain an invalid byte + // sequence), UBJSON, BJData, or BSON requires a decoder to reject + // ill-formed UTF-8. Checking (and, with @ref error_handler_t::strict, + // rejecting, or with `replace`/`ignore`, sanitizing) is opt-in via + // @ref error_handler, applied once the whole string (all chunks of + // an indefinite-length CBOR string included) has been assembled, by + // @ref check_string_utf8 at the call site. return get_bytes(format, len, "string", result); } + /*! + @brief validate a decoded text string (value or object key) against @ref error_handler + + None of the binary formats requires a decoder to reject ill-formed UTF-8 + in a text string (see @ref get_string), so by default + (@ref error_handler_t::keep) this does nothing. A stricter + @ref error_handler opts into the same well-formedness check @ref + serializer::dump_escaped_impl applies when dumping a string: + @ref error_handler_t::strict rejects ill-formed input with + parse_error.113 (honoring `allow_exceptions` via @a sax), while + @ref error_handler_t::replace / @ref error_handler_t::ignore sanitize + @a result in place, using the exact same rules. + + @param[in,out] result the already assembled string to check + @param[in] context further context information (for diagnostics) + @return whether @a result is acceptable (always true for `keep`) + */ + bool check_string_utf8(string_t& result, const char* context) + { + if (error_handler == error_handler_t::keep || is_valid_utf8(result)) + { + return true; + } + + if (error_handler == error_handler_t::strict) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format, "invalid string: ill-formed UTF-8 byte", context), nullptr)); + } + + result = sanitize_utf8(result, error_handler); + return true; + } + /*! @brief create a byte array by reading bytes from the input @@ -4226,6 +4273,9 @@ class binary_reader /// input format const input_format_t input_format = input_format_t::json; + /// how to treat text strings/object keys that are not well-formed UTF-8 + const error_handler_t error_handler = error_handler_t::keep; + /// the SAX parser json_sax_t* sax = nullptr; diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index da5aa0cf4..e74671db2 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -26,6 +26,7 @@ #include #include #include +#include #include #include #include @@ -93,8 +94,12 @@ class binary_writer @param[in] sink output sink to write to (a value-type sink such as output_vector_sink, or output_adapter_sink wrapping a type-erased output adapter) + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ - explicit binary_writer(OutputSinkType sink) : oa(std::move(sink)) + explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(std::move(sink)), error_handler(error_handler_) {} /*! @@ -107,16 +112,20 @@ class binary_writer from one. @param[in] adapter output adapter to write to + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ template < typename SinkType = OutputSinkType, typename std::enable_if < std::is_constructible>::value, int >::type = 0 > - explicit binary_writer(output_adapter_t adapter) : oa(SinkType(std::move(adapter))) + explicit binary_writer(output_adapter_t adapter, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(SinkType(std::move(adapter))), error_handler(error_handler_) {} /*! @param[in] j JSON value to serialize - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string - value or an object key is not valid UTF-8 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -147,8 +156,8 @@ class binary_writer /*! @param[in] j JSON value to serialize - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string - value or an object key is not valid UTF-8 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -215,15 +224,16 @@ class binary_writer case value_t::string: { - check_text_utf8(*j.m_data.m_value.string, j); + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); // step 1: write control byte and the string length - write_cbor_head(0x60, j.m_data.m_value.string->size()); + write_cbor_head(0x60, value.size()); // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -296,8 +306,14 @@ class binary_writer // el.first is checked here, against the object as // diagnostics context, because write_cbor(el.first) // converts it to a temporary basic_json that would be - // used as the context instead - check_text_utf8(el.first, j); + // used as the context instead; for error_handler_t::keep + // and ::replace/::ignore the recursive write_cbor(el.first) + // call below handles the key like any other string, so no + // separate check is needed here for those + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_cbor(el.first); write_cbor(el.second); } @@ -445,8 +461,11 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); + const auto N = to_msgpack_length(value.size(), j); if (N <= 31) { // fixstr @@ -473,8 +492,8 @@ class binary_writer // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -621,6 +640,13 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // as in write_cbor, el.first is checked here against the + // object as diagnostics context; the recursive call below + // handles keep/replace/ignore like any other string + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_msgpack(el.first); write_msgpack(el.second); } @@ -640,8 +666,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string - value or an object key is not valid UTF-8 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -691,16 +717,17 @@ class binary_writer case value_t::string: { - check_text_utf8(*j.m_data.m_value.string, j); + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); if (add_prefix) { oa.write_character(to_char_type('S')); } - write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); + write_number_with_ubjson_prefix(value.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -855,11 +882,12 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { - check_text_utf8(el.first, j); - write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); + string_t storage; + const string_t& key = sanitize_utf8_for_write(el.first, j, storage); + write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(el.first.data()), - el.first.size()); + reinterpret_cast(key.data()), + key.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } @@ -902,10 +930,10 @@ class binary_writer and the entry name size (and its null-terminator). @throw out_of_range.409 if @a name contains U+0000, before anything is written - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a name is - not valid UTF-8, before anything is written + @throw type_error.316 if @a name is not valid UTF-8, before anything is + written */ - static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) + std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { const auto it = name.find(static_cast(0)); if (JSON_HEDLEY_UNLIKELY(it != BasicJsonType::string_t::npos)) @@ -913,9 +941,10 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - check_text_utf8(name, j); + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(name, j, storage); - return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; + return /*id*/ 1ul + sanitized.size() + /*zero-terminator*/1u; } /*! @@ -935,14 +964,28 @@ class binary_writer /*! @brief Writes the given @a element_type and @a name to the output adapter + + @a name has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_entry_header_size during the earlier + size pass, so only @ref error_handler_t::replace / @ref + error_handler_t::ignore need to sanitize it again here, to actually write + the bytes that size was computed from. */ void write_bson_entry_header(const string_t& name, const std::uint8_t element_type) { oa.write_character(to_char_type(element_type)); - oa.write_characters( - reinterpret_cast(name.data()), - name.size()); + + if (error_handler == error_handler_t::keep || error_handler == error_handler_t::strict || is_valid_utf8(name)) + { + oa.write_characters(reinterpret_cast(name.data()), name.size()); + } + else + { + const string_t sanitized = sanitize_utf8(name, error_handler); + oa.write_characters(reinterpret_cast(sanitized.data()), sanitized.size()); + } + // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -970,8 +1013,8 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a value - is not valid UTF-8, before anything is written + @throw type_error.316 if @a value is not valid UTF-8, before anything is + written @note The UTF-8 check is skipped if @a value is already too long for the 32-bit BSON length field (@ref to_bson_length rejects it later, once @@ -979,27 +1022,41 @@ class binary_writer from reading past a StringType that reports a size larger than what it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) + std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) { - check_text_utf8(value, j); + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, j, storage); + return sizeof(std::int32_t) + sanitized.size() + 1ul; } return sizeof(std::int32_t) + value.size() + 1ul; } /*! @brief Writes a BSON element with key @a name and string value @a value + + @a value has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_string_size during the earlier size + pass, so only @ref error_handler_t::replace / @ref error_handler_t::ignore + need to sanitize it again here, to actually write the bytes that size was + computed from. */ void write_bson_string(const string_t& name, const string_t& value) { write_bson_entry_header(name, 0x02); - write_number(to_bson_length(value.size() + 1ul), true); + const bool sanitize = error_handler != error_handler_t::keep + && error_handler != error_handler_t::strict + && !is_valid_utf8(value); + const string_t sanitized = sanitize ? sanitize_utf8(value, error_handler) : string_t{}; + const string_t& written = sanitize ? sanitized : value; + + write_number(to_bson_length(written.size() + 1ul), true); oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + reinterpret_cast(written.data()), + written.size()); // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -1113,10 +1170,10 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a j is a - string that is not valid UTF-8, before anything is written + @throw type_error.316 if @a j is a string that is not valid UTF-8, before + anything is written */ - static std::size_t calc_bson_value_size(const BasicJsonType& j) + std::size_t calc_bson_value_size(const BasicJsonType& j) { switch (j.type()) { @@ -1249,10 +1306,10 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string - value or a key is not valid UTF-8, before anything is written + @throw type_error.316 if a string value or a key is not valid UTF-8, + before anything is written */ - static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) + std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { // the object or array whose entries are being sized, and the ones it // is in; nothing is allocated unless the document nests @@ -2171,26 +2228,54 @@ class binary_writer } /*! - @brief check a CBOR, UBJSON, BJData, or BSON text string for valid UTF-8 + @brief return @a s as it should be written, honoring @ref error_handler - The check only happens if JSON_STRICT_BINARY_UTF8 is enabled. Otherwise, - the bytes are written unchanged, as before version 3.13.0. MessagePack - always writes the bytes as is, and BON8 always checks them (see - @ref check_utf8). + Used by @ref write_cbor, @ref write_msgpack, @ref write_ubjson (and so + @ref write_bjdata), and the BSON writing functions for string values and + object keys; never by @ref write_bon8, which always validates, since UTF-8 + lead bytes are structural there. - @param[in] s the string to check - @param[in] context the value that holds @a s (for diagnostics) - @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a s is - not valid UTF-8 + - @ref error_handler_t::keep: @a s is returned unchanged, without even + checking it (the behavior of release 3.12.0 and earlier). + - @ref error_handler_t::strict: @ref check_utf8 is called, which throws + type_error.316 if @a s is not valid UTF-8. + - @ref error_handler_t::replace / @ref error_handler_t::ignore: @a s is + sanitized into @a storage with exactly the rules @ref + serializer::dump_escaped_impl uses, so that parsing what @ref + basic_json::dump produces for the same string and the same handler + yields the same result. + + Well-formed input is never copied: this returns a reference to @a s + itself in every case but a sanitized `replace`/`ignore` one, so @a + storage must outlive the returned reference only then. + + @param[in] s the string (value or object key) to write + @param[in] context the value @a s belongs to (for diagnostics) + @param[out] storage backing storage for a sanitized copy + + @return a reference to @a s, or to @a storage once it holds a sanitized copy */ - static void check_text_utf8(const string_t& s, const BasicJsonType& context) + const string_t& sanitize_utf8_for_write(const string_t& s, const BasicJsonType& context, string_t& storage) const { -#if JSON_STRICT_BINARY_UTF8 - check_utf8(s, context); -#else - static_cast(s); - static_cast(context); -#endif + switch (error_handler) + { + case error_handler_t::keep: + return s; + + case error_handler_t::strict: + check_utf8(s, context); + return s; + + case error_handler_t::replace: + case error_handler_t::ignore: + default: + if (is_valid_utf8(s)) + { + return s; + } + storage = sanitize_utf8(s, error_handler); + return storage; + } } /*! @@ -2521,6 +2606,10 @@ class binary_writer /// the output OutputSinkType oa; + + /// how to treat a string value or object key that is not valid UTF-8 + /// (CBOR, MessagePack, UBJSON, BJData, and BSON; not BON8) + const error_handler_t error_handler = binary_writer_default_error_handler(); }; } // namespace detail diff --git a/include/nlohmann/detail/output/error_handler.hpp b/include/nlohmann/detail/output/error_handler.hpp new file mode 100644 index 000000000..b95d70fce --- /dev/null +++ b/include/nlohmann/detail/output/error_handler.hpp @@ -0,0 +1,50 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// how to treat decoding errors +/// +/// @ref basic_json::dump uses this to decide what to do with ill-formed +/// UTF-8 while escaping a string, and the binary writers (@ref +/// basic_json::to_cbor, @ref basic_json::to_ubjson, @ref +/// basic_json::to_bjdata, @ref basic_json::to_bson) use it the same way for +/// string values and object keys. The binary readers (@ref +/// basic_json::from_cbor, @ref basic_json::from_msgpack, @ref +/// basic_json::from_ubjson, @ref basic_json::from_bjdata, @ref +/// basic_json::from_bson) use it to decide whether to check text strings +/// and object keys for well-formed UTF-8 at all, since none of those +/// formats requires a decoder to do so. +enum class error_handler_t +{ + strict, ///< throw a type_error/parse_error exception in case of invalid UTF-8 + replace, ///< replace invalid UTF-8 sequences with U+FFFD + ignore, ///< ignore invalid UTF-8 sequences + keep ///< keep invalid UTF-8 sequences unchanged +}; + +/// the default error handler of the CBOR, UBJSON, BJData, and BSON writers: +/// error_handler_t::strict if JSON_STRICT_BINARY_UTF8 is enabled, otherwise +/// error_handler_t::keep (the behavior before version 3.13.0) +constexpr error_handler_t binary_writer_default_error_handler() noexcept +{ +#if JSON_STRICT_BINARY_UTF8 + return error_handler_t::strict; +#else + return error_handler_t::keep; +#endif +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index 1c519d4c7..dcaae8055 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -27,6 +27,7 @@ #include #include #include +#include #include #include #include @@ -41,14 +42,6 @@ namespace detail // serialization // /////////////////// -/// how to treat decoding errors -enum class error_handler_t -{ - strict, ///< throw a type_error exception in case of invalid UTF-8 - replace, ///< replace invalid UTF-8 sequences with U+FFFD - ignore ///< ignore invalid UTF-8 sequences -}; - template class serializer { @@ -839,6 +832,16 @@ class serializer // EnsureAscii parameter is used, non-ASCII characters if ((codepoint <= 0x1F) || (EnsureAscii && (codepoint >= 0x7F))) { + if (EnsureAscii && error_handler == error_handler_t::keep) + { + // this character was buffered as raw bytes + // below in case it turned out to be part of + // an ill-formed sequence (which is kept as + // is); now that it decoded to a well-formed + // code point, undo that and \u-escape it + // like any other character instead + bytes = bytes_after_last_accept; + } if (codepoint <= 0xFFFF) { write_u_escape(bytes, static_cast(codepoint)); @@ -937,6 +940,44 @@ class serializer break; } + case error_handler_t::keep: + { + // the bytes of this (now abandoned) ill-formed + // sequence seen so far are already buffered below + // and are kept unchanged in the output + if (undumped_chars > 0) + { + // the byte that ended the sequence may be OK + // for itself (e.g., a quote that must still be + // escaped, or the lead byte of a well-formed + // code point), so read it again + --i; + } + else + { + // a byte that cannot start a sequence (e.g., + // 0xFF or a stray continuation byte) is kept + // as well + string_buffer[bytes++] = s[i]; + } + + // write buffer and reset index; there must be 13 bytes + // left, as this is the maximal number of bytes to be + // written ("\uxxxx\uxxxx\0") for one code point + if (string_buffer.size() - bytes < 13) + { + put_buffer(string_buffer, bytes); + bytes = 0; + } + + bytes_after_last_accept = bytes; + undumped_chars = 0; + + // continue processing the string + state = UTF8_ACCEPT; + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -945,9 +986,12 @@ class serializer default: // decode found yet incomplete multibyte code point { - if (!EnsureAscii) + if (!EnsureAscii || error_handler == error_handler_t::keep) { - // code point will not be escaped - copy byte to buffer + // code point will not be escaped (or will be kept as + // is if it turns out to be ill-formed) - copy byte to + // buffer; dropped again above if it decodes to a + // well-formed code point that needs \u-escaping string_buffer[bytes++] = s[i]; } ++undumped_chars; @@ -998,6 +1042,14 @@ class serializer break; } + case error_handler_t::keep: + { + // write the ill-formed trailing bytes as is; they were + // buffered above regardless of EnsureAscii + put_buffer(string_buffer, bytes); + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index 2b6864d0d..4ef748f13 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -16,6 +16,7 @@ #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -179,5 +180,134 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std: return state; } +/*! +@brief check a string for well-formed UTF-8 (RFC 3629, section 4) + +Used by the binary readers (CBOR, MessagePack, UBJSON, BJData, BSON) when an +@ref error_handler_t other than `keep` is requested for a text string value +or object key: none of those formats requires a decoder to reject ill-formed +UTF-8 on its own, so the check is opt-in there, unlike the JSON lexer and the +serializer's @ref decode -based escaping, which always run it. + +@param[in] s the string to check +@param[in] first the index to start checking at +@return whether `s.substr(first)` is well-formed UTF-8 + +@sa @ref decode +*/ +template +inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept +{ + std::uint8_t state = UTF8_ACCEPT; + std::uint32_t codepoint = 0; + + for (std::size_t i = first; i < s.size(); ++i) + { + decode(state, codepoint, static_cast(s[i])); + if (state == UTF8_REJECT) + { + return false; + } + } + + return state == UTF8_ACCEPT; +} + +/*! +@brief sanitize a string with ill-formed UTF-8 for @ref error_handler_t::replace or @ref error_handler_t::ignore + +Replaces every maximal ill-formed subsequence with U+FFFD (`replace`) or +drops it (`ignore`), using exactly the same boundaries @ref +serializer::dump_escaped_impl uses while escaping a string: a byte that does +not extend the sequence started by the previous byte(s) is reread as the +start of a new one, instead of being swallowed along with them. + +@pre @a error_handler is @ref error_handler_t::replace or @ref error_handler_t::ignore +@note Well-formed input is copied through unchanged, including bytes (e.g. + control characters or quotes) that @ref serializer::dump_escaped_impl + would itself escape; this function only concerns itself with + well-formedness, not with producing valid JSON text. + +@param[in] s the string to sanitize +@param[in] error_handler @ref error_handler_t::replace or @ref error_handler_t::ignore + +@return @a s with every ill-formed subsequence replaced or removed + +@sa @ref decode +*/ +template +inline StringType sanitize_utf8(const StringType& s, const error_handler_t error_handler) +{ + JSON_ASSERT(error_handler == error_handler_t::replace || error_handler == error_handler_t::ignore); + + StringType result; + result.reserve(s.size()); + + std::uint32_t codepoint = 0; + std::uint8_t state = UTF8_ACCEPT; + // length of result after the last accepted code point + std::size_t result_len_after_last_accept = 0; + // whether bytes of an as yet unresolved sequence were already appended + bool pending = false; + + for (std::size_t i = 0; i < s.size(); ++i) + { + switch (decode(state, codepoint, static_cast(s[i]))) + { + case UTF8_ACCEPT: // decode found a well-formed code point + { + result.push_back(s[i]); + result_len_after_last_accept = result.size(); + pending = false; + break; + } + + case UTF8_REJECT: // decode found an ill-formed byte + { + // in case we saw this byte for the first time, read it again, + // because it may be fine for itself, just not for the + // sequence that came before it + if (pending) + { + --i; + } + + // drop the bytes of the ill-formed sequence buffered below + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + result_len_after_last_accept = result.size(); + } + + pending = false; + state = UTF8_ACCEPT; + break; + } + + default: // decode found yet incomplete multibyte code point + { + result.push_back(s[i]); + pending = true; + break; + } + } + } + + // the string ended with an incomplete sequence + if (state != UTF8_ACCEPT) + { + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + } + } + + return result; +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 07b625a89..421feda50 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -195,9 +195,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // used by the vector-returning to_* overloads template using vector_binary_writer = ::nlohmann::detail::binary_writer>; - template static vector_binary_writer vector_writer(std::vector& v) + template static vector_binary_writer vector_writer( + std::vector& v, const ::nlohmann::detail::error_handler_t error_handler = ::nlohmann::detail::binary_writer_default_error_handler()) { - return vector_binary_writer(::nlohmann::detail::output_vector_sink(v)); + return vector_binary_writer(::nlohmann::detail::output_vector_sink(v), error_handler); } JSON_PRIVATE_UNLESS_TESTED: @@ -5583,11 +5584,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template static basic_json from_binary_impl(InputAdapterType ia, const input_format_t format, const bool strict, const bool allow_exceptions, + const error_handler_t error_handler = error_handler_t::keep, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { basic_json result; detail::json_sax_dom_parser sdp(result, allow_exceptions); - binary_reader reader(std::move(ia), format); + binary_reader reader(std::move(ia), format, error_handler); if (!reader.sax_parse(&sdp, strict, tag_handler)) { result = value_t::discarded; @@ -5605,78 +5607,87 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec public: /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static std::vector to_cbor(const basic_json& j) + static std::vector to_cbor(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_cbor(j); + vector_writer(result, error_handler).write_cbor(j); return result; } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static std::vector to_msgpack(const basic_json& j) + static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_msgpack(j); + vector_writer(result, error_handler).write_msgpack(j); return result; } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static std::vector to_ubjson(const basic_json& j, const bool use_size = false, - const bool use_type = false) + const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type); return result; } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a BJData serialization of a given JSON value @@ -5684,11 +5695,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_bjdata(const basic_json& j, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type, true, true, version); return result; } @@ -5696,42 +5708,47 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BJData serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static std::vector to_bson(const basic_json& j) + static std::vector to_bson(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_bson(j); + vector_writer(result, error_handler).write_bson(j); return result; } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BON8 serialization of a given JSON value @@ -5765,9 +5782,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(InputType&& i, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } /// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5778,9 +5796,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(IteratorType first, SentinelType last, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } template @@ -5801,7 +5820,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, error_handler_t::keep, tag_handler); } /// @brief create a JSON value from an input in MessagePack format @@ -5810,9 +5829,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5822,9 +5842,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } template @@ -5852,9 +5873,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5864,9 +5886,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } template @@ -5894,9 +5917,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5906,9 +5930,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BON8 format @@ -5940,9 +5965,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5952,9 +5978,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions, error_handler); } template diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 16e87cc86..511401699 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -104,6 +104,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -140,14 +144,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -156,7 +166,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -6151,6 +6162,59 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// how to treat decoding errors +/// +/// @ref basic_json::dump uses this to decide what to do with ill-formed +/// UTF-8 while escaping a string, and the binary writers (@ref +/// basic_json::to_cbor, @ref basic_json::to_ubjson, @ref +/// basic_json::to_bjdata, @ref basic_json::to_bson) use it the same way for +/// string values and object keys. The binary readers (@ref +/// basic_json::from_cbor, @ref basic_json::from_msgpack, @ref +/// basic_json::from_ubjson, @ref basic_json::from_bjdata, @ref +/// basic_json::from_bson) use it to decide whether to check text strings +/// and object keys for well-formed UTF-8 at all, since none of those +/// formats requires a decoder to do so. +enum class error_handler_t +{ + strict, ///< throw a type_error/parse_error exception in case of invalid UTF-8 + replace, ///< replace invalid UTF-8 sequences with U+FFFD + ignore, ///< ignore invalid UTF-8 sequences + keep ///< keep invalid UTF-8 sequences unchanged +}; + +/// the default error handler of the CBOR, UBJSON, BJData, and BSON writers: +/// error_handler_t::strict if JSON_STRICT_BINARY_UTF8 is enabled, otherwise +/// error_handler_t::keep (the behavior before version 3.13.0) +constexpr error_handler_t binary_writer_default_error_handler() noexcept +{ +#if JSON_STRICT_BINARY_UTF8 + return error_handler_t::strict; +#else + return error_handler_t::keep; +#endif +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -6252,13 +6316,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally written by Björn Hoehrmann. See http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. -The library checks UTF-8 well-formedness (RFC 3629, section 4) in four +The library checks UTF-8 well-formedness (RFC 3629, section 4) in three places, which differ in speed, diagnostics, and how they read the input: -- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in - strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR, - MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in - text strings at decode time). +- decode() below: the serializer, to escape and, in strict mode, reject + ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON, + UBJSON and BJData readers do not use it: none of those specs requires a + decoder to reject ill-formed UTF-8 in text strings, so the readers keep + the bytes as is and leave the check to dump() and the binary writers. - the per-lead-byte switch in lexer::scan_string(): JSON text, with a diagnostic for each kind of error. - validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's @@ -6314,19 +6379,19 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std: } /*! -@brief check whether a string consists solely of valid UTF-8 +@brief check a string for well-formed UTF-8 (RFC 3629, section 4) -Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text -strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the -MessagePack/BSON specifications all require text strings to be UTF-8), so -that malformed input is caught immediately instead of only surfacing later -as a type_error.316 when the resulting value is dumped. +Used by the binary readers (CBOR, MessagePack, UBJSON, BJData, BSON) when an +@ref error_handler_t other than `keep` is requested for a text string value +or object key: none of those formats requires a decoder to reject ill-formed +UTF-8 on its own, so the check is opt-in there, unlike the JSON lexer and the +serializer's @ref decode -based escaping, which always run it. @param[in] s the string to check -@param[in] first index of the first byte to check; the bytes before it are - assumed to have been validated already and to end on a - code point boundary -@return whether @a s (from index @a first on) is valid UTF-8 +@param[in] first the index to start checking at +@return whether `s.substr(first)` is well-formed UTF-8 + +@sa @ref decode */ template inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept @@ -6346,6 +6411,102 @@ inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noex return state == UTF8_ACCEPT; } +/*! +@brief sanitize a string with ill-formed UTF-8 for @ref error_handler_t::replace or @ref error_handler_t::ignore + +Replaces every maximal ill-formed subsequence with U+FFFD (`replace`) or +drops it (`ignore`), using exactly the same boundaries @ref +serializer::dump_escaped_impl uses while escaping a string: a byte that does +not extend the sequence started by the previous byte(s) is reread as the +start of a new one, instead of being swallowed along with them. + +@pre @a error_handler is @ref error_handler_t::replace or @ref error_handler_t::ignore +@note Well-formed input is copied through unchanged, including bytes (e.g. + control characters or quotes) that @ref serializer::dump_escaped_impl + would itself escape; this function only concerns itself with + well-formedness, not with producing valid JSON text. + +@param[in] s the string to sanitize +@param[in] error_handler @ref error_handler_t::replace or @ref error_handler_t::ignore + +@return @a s with every ill-formed subsequence replaced or removed + +@sa @ref decode +*/ +template +inline StringType sanitize_utf8(const StringType& s, const error_handler_t error_handler) +{ + JSON_ASSERT(error_handler == error_handler_t::replace || error_handler == error_handler_t::ignore); + + StringType result; + result.reserve(s.size()); + + std::uint32_t codepoint = 0; + std::uint8_t state = UTF8_ACCEPT; + // length of result after the last accepted code point + std::size_t result_len_after_last_accept = 0; + // whether bytes of an as yet unresolved sequence were already appended + bool pending = false; + + for (std::size_t i = 0; i < s.size(); ++i) + { + switch (decode(state, codepoint, static_cast(s[i]))) + { + case UTF8_ACCEPT: // decode found a well-formed code point + { + result.push_back(s[i]); + result_len_after_last_accept = result.size(); + pending = false; + break; + } + + case UTF8_REJECT: // decode found an ill-formed byte + { + // in case we saw this byte for the first time, read it again, + // because it may be fine for itself, just not for the + // sequence that came before it + if (pending) + { + --i; + } + + // drop the bytes of the ill-formed sequence buffered below + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + result_len_after_last_accept = result.size(); + } + + pending = false; + state = UTF8_ACCEPT; + break; + } + + default: // decode found yet incomplete multibyte code point + { + result.push_back(s[i]); + pending = true; + break; + } + } + } + + // the string ended with an incomplete sequence + if (state != UTF8_ACCEPT) + { + result.resize(result_len_after_last_accept); + + if (error_handler == error_handler_t::replace) + { + result.append("\xEF\xBF\xBD"); + } + } + + return result; +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -13452,6 +13613,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -13532,8 +13695,16 @@ class binary_reader @brief create a binary reader @param[in] adapter input adapter to read from + @param[in] format the binary format to parse + @param[in] error_handler how to treat text strings and object keys that + are not well-formed UTF-8; none of the supported formats + requires a decoder to reject those, so the default is to + @ref error_handler_t::keep them unchanged, as every binary + reader did before this parameter existed */ - explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json) noexcept : ia(std::move(adapter)), input_format(format) + explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json, + const error_handler_t error_handler = error_handler_t::keep) noexcept + : ia(std::move(adapter)), input_format(format), error_handler(error_handler) { (void)detail::is_sax_static_asserts {}; } @@ -13852,7 +14023,7 @@ class binary_reader { if (get_bson_cstr_bulk(result, std::integral_constant {})) { - return true; + return check_string_utf8(result, "key"); } auto out = std::back_inserter(result); @@ -13865,7 +14036,7 @@ class binary_reader } if (current == 0x00) { - return true; + return check_string_utf8(result, "key"); } *out++ = static_cast(current); } @@ -13946,7 +14117,7 @@ class binary_reader "string"), nullptr)); } - return true; + return check_string_utf8(result, "string"); } /*! @@ -14573,7 +14744,7 @@ class binary_reader @return whether string creation completed */ - bool get_cbor_string(string_t& result) + bool get_cbor_string(string_t& result, const char* context = "string") { // number of indefinite-length strings that have been opened and not // closed yet. RFC 8949, Section 3.2.3 does not permit nesting them, @@ -14603,7 +14774,7 @@ class binary_reader { if (--open == 0) { - return true; + return check_string_utf8(result, context); } get(); continue; @@ -14616,7 +14787,7 @@ class binary_reader if (open == 0) { - return true; + return check_string_utf8(result, context); } get(); @@ -14640,7 +14811,7 @@ class binary_reader // EOF and major type 3 (text string) are left to get_cbor_string if (current == char_traits::eof() || (static_cast(current) & 0xE0u) == 0x60u) { - return get_cbor_string(result); + return get_cbor_string(result, "key"); } const char* found = nullptr; @@ -15428,7 +15599,7 @@ class binary_reader @return whether string creation completed */ - bool get_msgpack_string(string_t& result) + bool get_msgpack_string(string_t& result, const char* context = "string") { if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "string"))) { @@ -15471,25 +15642,25 @@ class binary_reader case 0xBE: case 0xBF: { - return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result); + return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result) && check_string_utf8(result, context); } case 0xD9: // str 8 { std::uint8_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDA: // str 16 { std::uint16_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } case 0xDB: // str 32 { std::uint32_t len{}; - return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context); } default: @@ -15567,7 +15738,7 @@ class binary_reader // byte 0xC1 are left to get_msgpack_string if (current == char_traits::eof()) { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } if (current <= 0x7F || current >= 0xE0) { @@ -15583,7 +15754,7 @@ class binary_reader } else { - return get_msgpack_string(result); + return get_msgpack_string(result, "key"); } break; } @@ -15829,7 +16000,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, true, "key") || !sax->key(key))) { return false; } @@ -15851,7 +16022,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false, "key") || !sax->key(key))) { return false; } @@ -15919,7 +16090,7 @@ class binary_reader @return whether string creation completed */ - bool get_ubjson_string(string_t& result, const bool get_char = true) + bool get_ubjson_string(string_t& result, const bool get_char = true, const char* context = "string") { if (get_char) { @@ -15940,31 +16111,31 @@ class binary_reader case 'U': { std::uint8_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'i': { std::int8_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'I': { std::int16_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'l': { std::int32_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'L': { std::int64_t len{}; - return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result); + return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'u': @@ -15974,7 +16145,7 @@ class binary_reader break; } std::uint16_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'm': @@ -15984,7 +16155,7 @@ class binary_reader break; } std::uint32_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } case 'M': @@ -15994,7 +16165,7 @@ class binary_reader break; } std::uint64_t len{}; - return get_number(input_format, len) && get_string(input_format, len, result); + return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context); } default: @@ -17468,27 +17639,50 @@ class binary_reader const NumberType len, string_t& result) { - // get_bytes() appends to result, and CBOR indefinite-length strings - // collect all their chunks in the same result; validating only the - // newly read bytes keeps the check linear in the input size - const std::size_t old_size = result.size(); - if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) + // Strings are taken as is by default: none of CBOR (RFC 8949 §3.1 + // leaves the choice to the decoder), MessagePack (whose spec + // explicitly allows a str object to contain an invalid byte + // sequence), UBJSON, BJData, or BSON requires a decoder to reject + // ill-formed UTF-8. Checking (and, with @ref error_handler_t::strict, + // rejecting, or with `replace`/`ignore`, sanitizing) is opt-in via + // @ref error_handler, applied once the whole string (all chunks of + // an indefinite-length CBOR string included) has been assembled, by + // @ref check_string_utf8 at the call site. + return get_bytes(format, len, "string", result); + } + + /*! + @brief validate a decoded text string (value or object key) against @ref error_handler + + None of the binary formats requires a decoder to reject ill-formed UTF-8 + in a text string (see @ref get_string), so by default + (@ref error_handler_t::keep) this does nothing. A stricter + @ref error_handler opts into the same well-formedness check @ref + serializer::dump_escaped_impl applies when dumping a string: + @ref error_handler_t::strict rejects ill-formed input with + parse_error.113 (honoring `allow_exceptions` via @a sax), while + @ref error_handler_t::replace / @ref error_handler_t::ignore sanitize + @a result in place, using the exact same rules. + + @param[in,out] result the already assembled string to check + @param[in] context further context information (for diagnostics) + @return whether @a result is acceptable (always true for `keep`) + */ + bool check_string_utf8(string_t& result, const char* context) + { + if (error_handler == error_handler_t::keep || is_valid_utf8(result)) { - return false; + return true; } - // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications - // all require text strings to be valid UTF-8; reject anything else - // right here so malformed input is caught at decode time instead of - // only surfacing later as a type_error.316 when the value is dumped - // (which would defeat allow_exceptions=false / strict discarding). - if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) + if (error_handler == error_handler_t::strict) { - return sax->parse_error(chars_read, get_token_string(), - parse_error::create(113, chars_read, - exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)); + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format, "invalid string: ill-formed UTF-8 byte", context), nullptr)); } + result = sanitize_utf8(result, error_handler); return true; } @@ -17665,6 +17859,9 @@ class binary_reader /// input format const input_format_t input_format = input_format_t::json; + /// how to treat text strings/object keys that are not well-formed UTF-8 + const error_handler_t error_handler = error_handler_t::keep; + /// the SAX parser json_sax_t* sax = nullptr; @@ -20753,6 +20950,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // __ _____ _____ _____ // __| | __| | | | JSON for Modern C++ @@ -21120,8 +21319,12 @@ class binary_writer @param[in] sink output sink to write to (a value-type sink such as output_vector_sink, or output_adapter_sink wrapping a type-erased output adapter) + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ - explicit binary_writer(OutputSinkType sink) : oa(std::move(sink)) + explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(std::move(sink)), error_handler(error_handler_) {} /*! @@ -21134,14 +21337,20 @@ class binary_writer from one. @param[in] adapter output adapter to write to + @param[in] error_handler_ how to treat a string value or object key that + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ template < typename SinkType = OutputSinkType, typename std::enable_if < std::is_constructible>::value, int >::type = 0 > - explicit binary_writer(output_adapter_t adapter) : oa(SinkType(std::move(adapter))) + explicit binary_writer(output_adapter_t adapter, const error_handler_t error_handler_ = binary_writer_default_error_handler()) + : oa(SinkType(std::move(adapter))), error_handler(error_handler_) {} /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -21172,6 +21381,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -21238,13 +21449,16 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - write_cbor_head(0x60, j.m_data.m_value.string->size()); + write_cbor_head(0x60, value.size()); // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -21314,6 +21528,17 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // el.first is checked here, against the object as + // diagnostics context, because write_cbor(el.first) + // converts it to a temporary basic_json that would be + // used as the context instead; for error_handler_t::keep + // and ::replace/::ignore the recursive write_cbor(el.first) + // call below handles the key like any other string, so no + // separate check is needed here for those + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_cbor(el.first); write_cbor(el.second); } @@ -21461,8 +21686,11 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); + const auto N = to_msgpack_length(value.size(), j); if (N <= 31) { // fixstr @@ -21489,8 +21717,8 @@ class binary_writer // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -21637,6 +21865,13 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // as in write_cbor, el.first is checked here against the + // object as diagnostics context; the recursive call below + // handles keep/replace/ignore like any other string + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_msgpack(el.first); write_msgpack(el.second); } @@ -21656,6 +21891,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @throw type_error.316 if a string value or an object key is not valid + UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -21705,14 +21942,17 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + if (add_prefix) { oa.write_character(to_char_type('S')); } - write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); + write_number_with_ubjson_prefix(value.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -21867,10 +22107,12 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { - write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); + string_t storage; + const string_t& key = sanitize_utf8_for_write(el.first, j, storage); + write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( - reinterpret_cast(el.first.data()), - el.first.size()); + reinterpret_cast(key.data()), + key.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } @@ -21911,8 +22153,12 @@ class binary_writer /*! @return The size of a BSON document entry header, including the id marker and the entry name size (and its null-terminator). + @throw out_of_range.409 if @a name contains U+0000, before anything is + written + @throw type_error.316 if @a name is not valid UTF-8, before anything is + written */ - static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) + std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { const auto it = name.find(static_cast(0)); if (JSON_HEDLEY_UNLIKELY(it != BasicJsonType::string_t::npos)) @@ -21920,8 +22166,10 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - static_cast(j); - return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(name, j, storage); + + return /*id*/ 1ul + sanitized.size() + /*zero-terminator*/1u; } /*! @@ -21941,14 +22189,28 @@ class binary_writer /*! @brief Writes the given @a element_type and @a name to the output adapter + + @a name has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_entry_header_size during the earlier + size pass, so only @ref error_handler_t::replace / @ref + error_handler_t::ignore need to sanitize it again here, to actually write + the bytes that size was computed from. */ void write_bson_entry_header(const string_t& name, const std::uint8_t element_type) { oa.write_character(to_char_type(element_type)); - oa.write_characters( - reinterpret_cast(name.data()), - name.size()); + + if (error_handler == error_handler_t::keep || error_handler == error_handler_t::strict || is_valid_utf8(name)) + { + oa.write_characters(reinterpret_cast(name.data()), name.size()); + } + else + { + const string_t sanitized = sanitize_utf8(name, error_handler); + oa.write_characters(reinterpret_cast(sanitized.data()), sanitized.size()); + } + // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -21976,24 +22238,50 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value + @throw type_error.316 if @a value is not valid UTF-8, before anything is + written + + @note The UTF-8 check is skipped if @a value is already too long for the + 32-bit BSON length field (@ref to_bson_length rejects it later, once + the size of the whole document is known); this also keeps the check + from reading past a StringType that reports a size larger than what + it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value) + std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { + if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, j, storage); + return sizeof(std::int32_t) + sanitized.size() + 1ul; + } return sizeof(std::int32_t) + value.size() + 1ul; } /*! @brief Writes a BSON element with key @a name and string value @a value + + @a value has already been validated (and, for @ref error_handler_t::strict, + found well-formed) by @ref calc_bson_string_size during the earlier size + pass, so only @ref error_handler_t::replace / @ref error_handler_t::ignore + need to sanitize it again here, to actually write the bytes that size was + computed from. */ void write_bson_string(const string_t& name, const string_t& value) { write_bson_entry_header(name, 0x02); - write_number(to_bson_length(value.size() + 1ul), true); + const bool sanitize = error_handler != error_handler_t::keep + && error_handler != error_handler_t::strict + && !is_valid_utf8(value); + const string_t sanitized = sanitize ? sanitize_utf8(value, error_handler) : string_t{}; + const string_t& written = sanitize ? sanitized : value; + + write_number(to_bson_length(written.size() + 1ul), true); oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + reinterpret_cast(written.data()), + written.size()); // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated oa.write_character(to_char_type(0x00)); @@ -22107,8 +22395,10 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written + @throw type_error.316 if @a j is a string that is not valid UTF-8, before + anything is written */ - static std::size_t calc_bson_value_size(const BasicJsonType& j) + std::size_t calc_bson_value_size(const BasicJsonType& j) { switch (j.type()) { @@ -22128,7 +22418,7 @@ class binary_writer return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string, j); case value_t::null: return 0ul; @@ -22241,8 +22531,10 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written + @throw type_error.316 if a string value or a key is not valid UTF-8, + before anything is written */ - static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) + std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { // the object or array whose entries are being sized, and the ones it // is in; nothing is allocated unless the document nests @@ -23119,7 +23411,7 @@ class binary_writer */ void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) { - check_bon8_utf8(s, context); + check_utf8(s, context); // a string that follows another string terminates it if (string_open) @@ -23149,7 +23441,7 @@ class binary_writer @throw type_error.316 if @a s is not valid UTF-8; the message names the first byte of the first invalid or incomplete sequence */ - static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + static void check_utf8(const string_t& s, const BasicJsonType& context) { static_cast(context); // only used when exceptions are enabled const auto* data = reinterpret_cast(s.data()); @@ -23160,6 +23452,57 @@ class binary_writer } } + /*! + @brief return @a s as it should be written, honoring @ref error_handler + + Used by @ref write_cbor, @ref write_msgpack, @ref write_ubjson (and so + @ref write_bjdata), and the BSON writing functions for string values and + object keys; never by @ref write_bon8, which always validates, since UTF-8 + lead bytes are structural there. + + - @ref error_handler_t::keep: @a s is returned unchanged, without even + checking it (the behavior of release 3.12.0 and earlier). + - @ref error_handler_t::strict: @ref check_utf8 is called, which throws + type_error.316 if @a s is not valid UTF-8. + - @ref error_handler_t::replace / @ref error_handler_t::ignore: @a s is + sanitized into @a storage with exactly the rules @ref + serializer::dump_escaped_impl uses, so that parsing what @ref + basic_json::dump produces for the same string and the same handler + yields the same result. + + Well-formed input is never copied: this returns a reference to @a s + itself in every case but a sanitized `replace`/`ignore` one, so @a + storage must outlive the returned reference only then. + + @param[in] s the string (value or object key) to write + @param[in] context the value @a s belongs to (for diagnostics) + @param[out] storage backing storage for a sanitized copy + + @return a reference to @a s, or to @a storage once it holds a sanitized copy + */ + const string_t& sanitize_utf8_for_write(const string_t& s, const BasicJsonType& context, string_t& storage) const + { + switch (error_handler) + { + case error_handler_t::keep: + return s; + + case error_handler_t::strict: + check_utf8(s, context); + return s; + + case error_handler_t::replace: + case error_handler_t::ignore: + default: + if (is_valid_utf8(s)) + { + return s; + } + storage = sanitize_utf8(s, error_handler); + return storage; + } + } + /*! @brief write an integer in the shortest encoding @@ -23488,6 +23831,10 @@ class binary_writer /// the output OutputSinkType oa; + + /// how to treat a string value or object key that is not valid UTF-8 + /// (CBOR, MessagePack, UBJSON, BJData, and BSON; not BON8) + const error_handler_t error_handler = binary_writer_default_error_handler(); }; } // namespace detail @@ -24649,6 +24996,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -24668,14 +25017,6 @@ namespace detail // serialization // /////////////////// -/// how to treat decoding errors -enum class error_handler_t -{ - strict, ///< throw a type_error exception in case of invalid UTF-8 - replace, ///< replace invalid UTF-8 sequences with U+FFFD - ignore ///< ignore invalid UTF-8 sequences -}; - template class serializer { @@ -25466,6 +25807,16 @@ class serializer // EnsureAscii parameter is used, non-ASCII characters if ((codepoint <= 0x1F) || (EnsureAscii && (codepoint >= 0x7F))) { + if (EnsureAscii && error_handler == error_handler_t::keep) + { + // this character was buffered as raw bytes + // below in case it turned out to be part of + // an ill-formed sequence (which is kept as + // is); now that it decoded to a well-formed + // code point, undo that and \u-escape it + // like any other character instead + bytes = bytes_after_last_accept; + } if (codepoint <= 0xFFFF) { write_u_escape(bytes, static_cast(codepoint)); @@ -25564,6 +25915,44 @@ class serializer break; } + case error_handler_t::keep: + { + // the bytes of this (now abandoned) ill-formed + // sequence seen so far are already buffered below + // and are kept unchanged in the output + if (undumped_chars > 0) + { + // the byte that ended the sequence may be OK + // for itself (e.g., a quote that must still be + // escaped, or the lead byte of a well-formed + // code point), so read it again + --i; + } + else + { + // a byte that cannot start a sequence (e.g., + // 0xFF or a stray continuation byte) is kept + // as well + string_buffer[bytes++] = s[i]; + } + + // write buffer and reset index; there must be 13 bytes + // left, as this is the maximal number of bytes to be + // written ("\uxxxx\uxxxx\0") for one code point + if (string_buffer.size() - bytes < 13) + { + put_buffer(string_buffer, bytes); + bytes = 0; + } + + bytes_after_last_accept = bytes; + undumped_chars = 0; + + // continue processing the string + state = UTF8_ACCEPT; + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -25572,9 +25961,12 @@ class serializer default: // decode found yet incomplete multibyte code point { - if (!EnsureAscii) + if (!EnsureAscii || error_handler == error_handler_t::keep) { - // code point will not be escaped - copy byte to buffer + // code point will not be escaped (or will be kept as + // is if it turns out to be ill-formed) - copy byte to + // buffer; dropped again above if it decodes to a + // well-formed code point that needs \u-escaping string_buffer[bytes++] = s[i]; } ++undumped_chars; @@ -25625,6 +26017,14 @@ class serializer break; } + case error_handler_t::keep: + { + // write the ill-formed trailing bytes as is; they were + // buffered above regardless of EnsureAscii + put_buffer(string_buffer, bytes); + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -26732,9 +27132,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // used by the vector-returning to_* overloads template using vector_binary_writer = ::nlohmann::detail::binary_writer>; - template static vector_binary_writer vector_writer(std::vector& v) + template static vector_binary_writer vector_writer( + std::vector& v, const ::nlohmann::detail::error_handler_t error_handler = ::nlohmann::detail::binary_writer_default_error_handler()) { - return vector_binary_writer(::nlohmann::detail::output_vector_sink(v)); + return vector_binary_writer(::nlohmann::detail::output_vector_sink(v), error_handler); } JSON_PRIVATE_UNLESS_TESTED: @@ -32120,11 +32521,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template static basic_json from_binary_impl(InputAdapterType ia, const input_format_t format, const bool strict, const bool allow_exceptions, + const error_handler_t error_handler = error_handler_t::keep, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { basic_json result; detail::json_sax_dom_parser sdp(result, allow_exceptions); - binary_reader reader(std::move(ia), format); + binary_reader reader(std::move(ia), format, error_handler); if (!reader.sax_parse(&sdp, strict, tag_handler)) { result = value_t::discarded; @@ -32142,78 +32544,87 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec public: /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static std::vector to_cbor(const basic_json& j) + static std::vector to_cbor(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_cbor(j); + vector_writer(result, error_handler).write_cbor(j); return result; } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a CBOR serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_cbor/ - static void to_cbor(const basic_json& j, detail::output_adapter o) + static void to_cbor(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_cbor(j); + binary_writer(o, error_handler).write_cbor(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static std::vector to_msgpack(const basic_json& j) + static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_msgpack(j); + vector_writer(result, error_handler).write_msgpack(j); return result; } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static std::vector to_ubjson(const basic_json& j, const bool use_size = false, - const bool use_type = false) + const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type); return result; } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a UBJSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/ static void to_ubjson(const basic_json& j, detail::output_adapter o, - const bool use_size = false, const bool use_type = false) + const bool use_size = false, const bool use_type = false, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type); } /// @brief create a BJData serialization of a given JSON value @@ -32221,11 +32632,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_bjdata(const basic_json& j, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version); + vector_writer(result, error_handler).write_ubjson(j, use_size, use_type, true, true, version); return result; } @@ -32233,42 +32645,47 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BJData serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/ static void to_bjdata(const basic_json& j, detail::output_adapter o, const bool use_size = false, const bool use_type = false, - const bjdata_version_t version = bjdata_version_t::draft2) + const bjdata_version_t version = bjdata_version_t::draft2, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_ubjson(j, use_size, use_type, true, true, version); + binary_writer(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static std::vector to_bson(const basic_json& j) + static std::vector to_bson(const basic_json& j, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_bson(j); + vector_writer(result, error_handler).write_bson(j); return result; } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BSON serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_bson/ - static void to_bson(const basic_json& j, detail::output_adapter o) + static void to_bson(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = detail::binary_writer_default_error_handler()) { - binary_writer(o).write_bson(j); + binary_writer(o, error_handler).write_bson(j); } /// @brief create a BON8 serialization of a given JSON value @@ -32302,9 +32719,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(InputType&& i, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } /// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32315,9 +32733,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static basic_json from_cbor(IteratorType first, SentinelType last, const bool strict = true, const bool allow_exceptions = true, - const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::cbor, strict, allow_exceptions, error_handler, tag_handler); } template @@ -32338,7 +32757,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool allow_exceptions = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { - return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, tag_handler); + return from_binary_impl(i.get(), input_format_t::cbor, strict, allow_exceptions, error_handler_t::keep, tag_handler); } /// @brief create a JSON value from an input in MessagePack format @@ -32347,9 +32766,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32359,9 +32779,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_msgpack(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::msgpack, strict, allow_exceptions, error_handler); } template @@ -32389,9 +32810,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32401,9 +32823,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_ubjson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::ubjson, strict, allow_exceptions, error_handler); } template @@ -32431,9 +32854,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32443,9 +32867,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bjdata(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bjdata, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BON8 format @@ -32477,9 +32902,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(InputType&& i, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::forward(i)), input_format_t::bson, strict, allow_exceptions, error_handler); } /// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32489,9 +32915,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_WARN_UNUSED_RESULT static basic_json from_bson(IteratorType first, SentinelType last, const bool strict = true, - const bool allow_exceptions = true) + const bool allow_exceptions = true, + const error_handler_t error_handler = error_handler_t::keep) { - return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions); + return from_binary_impl(detail::input_adapter(std::move(first), std::move(last)), input_format_t::bson, strict, allow_exceptions, error_handler); } template @@ -33734,6 +34161,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION #undef JSON_STRICT_NUL_HANDLING + #undef JSON_STRICT_BINARY_UTF8 #endif // #include diff --git a/tests/src/unit-binary_utf8_error_handler.cpp b/tests/src/unit-binary_utf8_error_handler.cpp new file mode 100644 index 000000000..85505bffb --- /dev/null +++ b/tests/src/unit-binary_utf8_error_handler.cpp @@ -0,0 +1,362 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +using nlohmann::json; + +#include +#include + +namespace +{ + +struct ill_formed_case +{ + const char* name; + std::string bytes; +}; + +// RFC 3629 ill-formed sequences used throughout this file, plus one +// well-formed sequence for contrast +const std::vector ill_formed_cases = +{ + {"overlong", "\xC0\xAE"}, + {"lone_0xFF", "\xFF"}, + {"truncated", "\xE2\x82"}, + {"surrogate", "\xED\xA0\x80"}, +}; + +const std::string valid_sequence = "\xC3\xA9"; // U+00E9, "é" + +using eh = json::error_handler_t; +const std::vector all_handlers = {eh::strict, eh::replace, eh::ignore, eh::keep}; + +// what dump()+parse() produces for a sanitizing error_handler; this is the +// ground truth every binary writer/reader is checked against +std::string dump_and_parse(const std::string& raw, eh error_handler) +{ + return json::parse(json(raw).dump(-1, ' ', false, error_handler)).get(); +} + +} // namespace + +TEST_CASE("UTF-8 error_handler for the binary readers and writers") +{ + SECTION("writers: string value") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + const json jval = c.bytes; + + CHECK_THROWS_AS(json::to_cbor(jval, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_msgpack(jval, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_ubjson(jval, false, false, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&); + { + json jobj; + jobj["k"] = jval; + CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&); + } + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(json::to_cbor(jval, h)).get() == expected); + CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get() == expected); + CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get() == expected); + CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get() == expected); + { + json jobj; + jobj["k"] = jval; + const auto bytes = json::to_bson(jobj, h); + CHECK(json::from_bson(bytes)["k"].get() == expected); + } + } + + // keep: the writer passes the ill-formed bytes through unchanged, + // exactly as every binary writer did before this parameter existed + CHECK(json::from_cbor(json::to_cbor(jval, eh::keep)).get() == c.bytes); + CHECK(json::from_msgpack(json::to_msgpack(jval, eh::keep)).get() == c.bytes); + CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, eh::keep)).get() == c.bytes); + CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep)).get() == c.bytes); + { + json jobj; + jobj["k"] = jval; + const auto bytes = json::to_bson(jobj, eh::keep); + CHECK(json::from_bson(bytes)["k"].get() == c.bytes); + } + } + } + + SECTION("writers: object key") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + json jobj; + jobj[c.bytes] = 1; + + CHECK_THROWS_AS(json::to_cbor(jobj, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_msgpack(jobj, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_ubjson(jobj, false, false, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&); + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(json::to_cbor(jobj, h)).begin().key() == expected); + CHECK(json::from_msgpack(json::to_msgpack(jobj, h)).begin().key() == expected); + CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, h)).begin().key() == expected); + CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, h)).begin().key() == expected); + CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == expected); + } + + // keep: object keys round-trip unchanged too + CHECK(json::from_cbor(json::to_cbor(jobj, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_msgpack(json::to_msgpack(jobj, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_bson(json::to_bson(jobj, eh::keep)).begin().key() == c.bytes); + } + } + + SECTION("readers: string value") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + + // bytes produced the lenient (keep) way, as any binary reader + // accepted them before this parameter existed + const auto cbor_bytes = json::to_cbor(json(c.bytes), eh::keep); + const auto msgpack_bytes = json::to_msgpack(json(c.bytes)); // to_msgpack has no error_handler; always pass-through + const auto ubjson_bytes = json::to_ubjson(json(c.bytes), false, false, eh::keep); + const auto bjdata_bytes = json::to_bjdata(json(c.bytes), false, false, json::bjdata_version_t::draft2, eh::keep); + const auto bson_bytes = [&c] + { + json jobj; + jobj["k"] = c.bytes; + return json::to_bson(jobj, eh::keep); + }(); + + // keep (the default): bytes are kept unchanged + CHECK(json::from_cbor(cbor_bytes).get() == c.bytes); + CHECK(json::from_msgpack(msgpack_bytes).get() == c.bytes); + CHECK(json::from_ubjson(ubjson_bytes).get() == c.bytes); + CHECK(json::from_bjdata(bjdata_bytes).get() == c.bytes); + CHECK(json::from_bson(bson_bytes)["k"].get() == c.bytes); + + // strict: parse_error.113, discarded (not thrown) when allow_exceptions is false + CHECK_THROWS_AS(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, eh::strict), json::parse_error&); + CHECK(json::from_cbor(cbor_bytes, true, false, json::cbor_tag_handler_t::error, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_msgpack(msgpack_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_msgpack(msgpack_bytes, true, false, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_ubjson(ubjson_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_ubjson(ubjson_bytes, true, false, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_bjdata(bjdata_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_bjdata(bjdata_bytes, true, false, eh::strict).is_discarded()); + CHECK_THROWS_AS(json::from_bson(bson_bytes, true, true, eh::strict), json::parse_error&); + CHECK(json::from_bson(bson_bytes, true, false, eh::strict).is_discarded()); + + // replace / ignore: match what dump() would have sanitized the same bytes to + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, h).get() == expected); + CHECK(json::from_msgpack(msgpack_bytes, true, true, h).get() == expected); + CHECK(json::from_ubjson(ubjson_bytes, true, true, h).get() == expected); + CHECK(json::from_bjdata(bjdata_bytes, true, true, h).get() == expected); + CHECK(json::from_bson(bson_bytes, true, true, h)["k"].get() == expected); + } + } + } + + SECTION("readers: object key") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + + json jobj; + jobj[c.bytes] = 1; + const auto cbor_bytes = json::to_cbor(jobj, eh::keep); + const auto msgpack_bytes = json::to_msgpack(jobj); + const auto ubjson_bytes = json::to_ubjson(jobj, false, false, eh::keep); + const auto bjdata_bytes = json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep); + const auto bson_bytes = json::to_bson(jobj, eh::keep); + + CHECK(json::from_cbor(cbor_bytes).begin().key() == c.bytes); + CHECK(json::from_msgpack(msgpack_bytes).begin().key() == c.bytes); + CHECK(json::from_ubjson(ubjson_bytes).begin().key() == c.bytes); + CHECK(json::from_bjdata(bjdata_bytes).begin().key() == c.bytes); + CHECK(json::from_bson(bson_bytes).begin().key() == c.bytes); + + CHECK_THROWS_AS(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_msgpack(msgpack_bytes, true, true, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_ubjson(ubjson_bytes, true, true, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_bjdata(bjdata_bytes, true, true, eh::strict), json::parse_error&); + CHECK_THROWS_AS(json::from_bson(bson_bytes, true, true, eh::strict), json::parse_error&); + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)); + const std::string expected = dump_and_parse(c.bytes, h); + + CHECK(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, h).begin().key() == expected); + CHECK(json::from_msgpack(msgpack_bytes, true, true, h).begin().key() == expected); + CHECK(json::from_ubjson(ubjson_bytes, true, true, h).begin().key() == expected); + CHECK(json::from_bjdata(bjdata_bytes, true, true, h).begin().key() == expected); + CHECK(json::from_bson(bson_bytes, true, true, h).begin().key() == expected); + } + } + } + + SECTION("well-formed UTF-8 is unaffected by error_handler") + { + const json jval = valid_sequence; + json jobj; + jobj[valid_sequence] = valid_sequence; + + for (const auto h : all_handlers) + { + CAPTURE(static_cast(h)); + + CHECK(json::from_cbor(json::to_cbor(jval, h)).get() == valid_sequence); + CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get() == valid_sequence); + CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get() == valid_sequence); + CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get() == valid_sequence); + CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == valid_sequence); + + CHECK(json::from_cbor(json::to_cbor(jval, eh::keep), true, true, json::cbor_tag_handler_t::error, h).get() == valid_sequence); + CHECK(json::from_msgpack(json::to_msgpack(jval), true, true, h).get() == valid_sequence); + } + } + + SECTION("dump() with error_handler_t::keep writes raw bytes as is") + { + for (const auto& c : ill_formed_cases) + { + CAPTURE(c.name); + + const json jval = c.bytes; + const std::string dumped = jval.dump(-1, ' ', false, eh::keep); + CHECK(dumped.find(c.bytes) != std::string::npos); + + // even with ensure_ascii, the ill-formed bytes are written as is + const std::string dumped_ascii = jval.dump(-1, ' ', true, eh::keep); + CHECK(dumped_ascii.find(c.bytes) != std::string::npos); + } + + // well-formed characters around an ill-formed sequence are still + // escaped as usual under ensure_ascii + const json mixed = valid_sequence + ill_formed_cases[1].bytes; // "é" + lone 0xFF + const std::string dumped_mixed = mixed.dump(-1, ' ', true, eh::keep); + CHECK(dumped_mixed.find("\\u00e9") != std::string::npos); + CHECK(dumped_mixed.find(ill_formed_cases[1].bytes) != std::string::npos); + + // the byte that ends an ill-formed sequence is read again, so a quote, + // a backslash, or a control character after it is still escaped, and + // a well-formed code point after it is escaped under ensure_ascii + for (const bool ensure_ascii : + { + false, true + }) + { + CAPTURE(ensure_ascii); + CHECK(json("\xC3\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\\"\""); + CHECK(json("\xC3\\").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\\\\""); + CHECK(json("\xC3\n").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\n\""); + CHECK(json("\xE2\x82\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xE2\x82\\\"\""); + CHECK(json("\xFF\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xFF\\\"\""); + CHECK(json("a\xE2\x82").dump(-1, ' ', ensure_ascii, eh::keep) == "\"a\xE2\x82\""); + } + CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', false, eh::keep) == "\"\xC3\xC3\xA9\""); + CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', true, eh::keep) == "\"\xC3\\u00e9\""); + } + + SECTION("to_msgpack defaults to keep; to_bon8 is not affected by error_handler") + { + const json jval = ill_formed_cases[1].bytes; // lone 0xFF + + // to_msgpack's error_handler defaults to keep, as MessagePack's spec + // allows any bytes in a str, so the bytes are passed through + CHECK(json::to_msgpack(jval) == json::to_msgpack(jval, eh::keep)); + CHECK(json::from_msgpack(json::to_msgpack(jval)).get() == ill_formed_cases[1].bytes); + + // the diagnostics context of an ill-formed key is the object + json jobj; + jobj["\xFF"] = 1; + CHECK_THROWS_WITH_AS(json::to_msgpack(jobj, eh::strict), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + + // to_bon8 has no error_handler parameter; UTF-8 is structural for + // BON8, so it always rejects ill-formed input + CHECK_THROWS_AS(json::to_bon8(jval), json::type_error&); + } + + SECTION("allow_exceptions=false with error_handler_t::strict discards the value") + { + const auto bytes = json::to_cbor(json(ill_formed_cases[0].bytes), eh::keep); + const json result = json::from_cbor(bytes, true, false, json::cbor_tag_handler_t::error, eh::strict); + CHECK(result.is_discarded()); + } + + SECTION("default parameters are unchanged") + { + const json jval = ill_formed_cases[0].bytes; + + // to_*: the default error_handler is keep, so ill-formed bytes are + // written unchanged, exactly as in release 3.12.0 (it is strict only + // if JSON_STRICT_BINARY_UTF8 is enabled, see + // unit-binary_utf8_strict.cpp) + CHECK(json::to_cbor(jval) == json::to_cbor(jval, eh::keep)); + CHECK(json::to_ubjson(jval) == json::to_ubjson(jval, false, false, eh::keep)); + CHECK(json::to_bjdata(jval) == json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep)); + { + json jobj; + jobj["k"] = jval; + CHECK(json::to_bson(jobj) == json::to_bson(jobj, eh::keep)); + } + + // from_*: the default error_handler is keep, so ill-formed bytes are + // still accepted unchanged, exactly as in release 3.12.0 + const auto cbor_bytes = json::to_cbor(jval, eh::keep); + CHECK(json::from_cbor(cbor_bytes).get() == ill_formed_cases[0].bytes); + const auto ubjson_bytes = json::to_ubjson(jval, false, false, eh::keep); + CHECK(json::from_ubjson(ubjson_bytes).get() == ill_formed_cases[0].bytes); + const auto bjdata_bytes = json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep); + CHECK(json::from_bjdata(bjdata_bytes).get() == ill_formed_cases[0].bytes); + const auto msgpack_bytes = json::to_msgpack(jval); + CHECK(json::from_msgpack(msgpack_bytes).get() == ill_formed_cases[0].bytes); + json bson_obj; + bson_obj["k"] = jval; + const auto bson_bytes = json::to_bson(bson_obj, eh::keep); + CHECK(json::from_bson(bson_bytes)["k"].get() == ill_formed_cases[0].bytes); + } +} diff --git a/tests/src/unit-binary_utf8_strict.cpp b/tests/src/unit-binary_utf8_strict.cpp index c83b5938c..2ac8aabf4 100644 --- a/tests/src/unit-binary_utf8_strict.cpp +++ b/tests/src/unit-binary_utf8_strict.cpp @@ -100,11 +100,23 @@ TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)") CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); } + SECTION("an explicit error_handler overrides the default") + { + // the macro only changes the default of the error_handler parameter + CHECK(json::to_cbor(json("\xFF"), json::error_handler_t::keep) == std::vector({0x61, 0xff})); + CHECK(json::to_ubjson(json("\xFF"), false, false, json::error_handler_t::keep) == std::vector({'S', 'i', 1, 0xff})); + CHECK(json::to_bjdata(json("\xFF"), false, false, json::bjdata_version_t::draft2, json::error_handler_t::keep) == std::vector({'S', 'i', 1, 0xff})); + CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}}, json::error_handler_t::keep)) == json{{"s", "\xFF"}}); + CHECK(json::to_cbor(json("\xFF"), json::error_handler_t::replace) == std::vector({0x63, 0xef, 0xbf, 0xbd})); + } + SECTION("MessagePack and BON8 are unaffected") { - // MessagePack allows any bytes in a str, so to_msgpack() writes them as - // is; BON8 always checks, because the lead bytes mark where strings end + // MessagePack allows any bytes in a str, so to_msgpack() still + // defaults to keep (strict only if passed explicitly); BON8 always + // checks, because the lead bytes mark where strings end CHECK(json::to_msgpack(json("\xFF")) == std::vector({0xa1, 0xff})); + CHECK_THROWS_AS(json::to_msgpack(json("\xFF"), json::error_handler_t::strict), json::type_error&); CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&); } }