From bbf9b69d38bc66e8cbf82472ed01a7815f3f427b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 2 Oct 2026 09:02:34 +0200 Subject: [PATCH] Add an error_handler parameter to to_msgpack The MessagePack spec allows ill-formed UTF-8 in a str, so the default stays keep (also if JSON_STRICT_BINARY_UTF8 is enabled), but decoders such as msgpack-python reject it by default. With the parameter, strict catches such strings and replace/ignore sanitize them, the same way as for the other binary writers and dump(). Signed-off-by: Niels Lohmann --- .../docs/api/basic_json/error_handler_t.md | 15 +++--- docs/mkdocs/docs/api/basic_json/to_msgpack.md | 20 ++++++-- .../api/macros/json_strict_binary_utf8.md | 2 +- .../features/binary_formats/messagepack.md | 6 ++- docs/mkdocs/docs/home/exceptions.md | 3 +- .../nlohmann/detail/output/binary_writer.hpp | 36 ++++++++----- include/nlohmann/json.hpp | 15 +++--- single_include/nlohmann/json.hpp | 51 +++++++++++-------- tests/src/unit-binary_utf8_error_handler.cpp | 23 +++++++-- tests/src/unit-binary_utf8_strict.cpp | 6 ++- 10 files changed, 115 insertions(+), 62 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/error_handler_t.md b/docs/mkdocs/docs/api/basic_json/error_handler_t.md index 65c9c4644..8937dddee 100644 --- a/docs/mkdocs/docs/api/basic_json/error_handler_t.md +++ b/docs/mkdocs/docs/api/basic_json/error_handler_t.md @@ -12,14 +12,13 @@ enum class error_handler_t { This enumeration is used to choose how to treat ill-formed UTF-8 in a string value or object key: - [`dump`](dump.md) uses it while serializing a `basic_json` value to text. -- [`to_cbor`](to_cbor.md), [`to_ubjson`](to_ubjson.md), [`to_bjdata`](to_bjdata.md), and [`to_bson`](to_bson.md) use it - while serializing a `basic_json` value to that binary format; none of CBOR, UBJSON, BJData, or BSON requires a - decoder to reject ill-formed UTF-8, so the library can check on write instead. Their default is `keep`, as no - binary writer checked before this parameter was added, or `strict` if - [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled. `to_msgpack` and - `to_bon8` do not take this parameter: MessagePack's specification explicitly allows a string to contain ill-formed - UTF-8, so `to_msgpack` always passes it through, while BON8 always validates, since UTF-8 lead bytes are structural - to that format. +- [`to_cbor`](to_cbor.md), [`to_msgpack`](to_msgpack.md), [`to_ubjson`](to_ubjson.md), [`to_bjdata`](to_bjdata.md), + and [`to_bson`](to_bson.md) use it while serializing a `basic_json` value to that binary format. Their default is + `keep`, as no binary writer checked before this parameter was added. CBOR, UBJSON, BJData, and BSON require valid + UTF-8, so for these four the default is `strict` if [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) + is enabled; MessagePack's specification explicitly allows a string to contain ill-formed UTF-8, so `to_msgpack` + stays at `keep`. `to_bon8` does not take this parameter: BON8 always validates, since UTF-8 lead bytes are + structural to that format. - [`from_cbor`](from_cbor.md), [`from_msgpack`](from_msgpack.md), [`from_ubjson`](from_ubjson.md), [`from_bjdata`](from_bjdata.md), and [`from_bson`](from_bson.md) use it while parsing that binary format, to decide whether to check a string value or object key for well-formed UTF-8 at all; by default (`keep`) they do not, as no diff --git a/docs/mkdocs/docs/api/basic_json/to_msgpack.md b/docs/mkdocs/docs/api/basic_json/to_msgpack.md index 19f86f9c2..2b43903ca 100644 --- a/docs/mkdocs/docs/api/basic_json/to_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/to_msgpack.md @@ -2,11 +2,14 @@ ```cpp // (1) -static std::vector to_msgpack(const basic_json& j); +static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep); // (2) -static void to_msgpack(const basic_json& j, detail::output_adapter o); -static void to_msgpack(const basic_json& j, detail::output_adapter o); +static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); +static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep); ``` Serializes a given JSON value `j` to a byte vector using the MessagePack serialization format. MessagePack is a binary @@ -25,6 +28,13 @@ The exact mapping and its limitations are described on a [dedicated page](../../ `o` (in) : output adapter to write serialization to +`error_handler` (in) +: how to treat a string or object key in `j` that is not valid UTF-8; see [`error_handler_t`](error_handler_t.md). + The default, `keep`, writes the ill-formed bytes to the output as is, as every version of `to_msgpack` did before + this parameter was added and as the MessagePack specification allows; `strict` throws; `replace`/`ignore` sanitize + it the same way [`dump`](dump.md) would. Unlike the other binary writers, the default stays `keep` even if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled. + ## Return value 1. MessagePack serialization as a byte vector @@ -42,6 +52,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the MessagePack ext type; example: `"subtype 70000 is too large for the MessagePack ext type (max 255)"` +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is + not valid UTF-8 and `error_handler` is `strict` ## Complexity @@ -76,6 +88,8 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. +- Added `error_handler` parameter in version 3.13.0. Its default, `keep`, writes the bytes of a string or object key + that is not valid UTF-8 unchanged, as before. - Fixed in version 3.13.0 to serialize `number_integer_t`/`number_unsigned_t` pairs of different width correctly; before, integers could be serialized with the wrong value if `number_integer_t` was narrower than `number_unsigned_t`. diff --git a/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md index eb457249c..ed7bbe6b3 100644 --- a/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md +++ b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md @@ -15,7 +15,7 @@ always takes precedence. The macro does not affect: - [`to_msgpack`](../basic_json/to_msgpack.md): the MessagePack specification allows a `str` value to contain bytes that - are not valid UTF-8, so it always writes them unchanged. + are not valid UTF-8, so its `error_handler` always defaults to `keep`. - [`to_bon8`](../basic_json/to_bon8.md): BON8 always checks, because the UTF-8 lead bytes mark where a string ends. - The binary readers ([`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md), [`from_ubjson`](../basic_json/from_ubjson.md), [`from_bjdata`](../basic_json/from_bjdata.md), diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index 3991d16d0..9c2130693 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -163,8 +163,10 @@ The library maps MessagePack types to JSON value types as follows: round-trips through `from_msgpack(to_msgpack(j))` byte for byte. Passing `error_handler_t::strict` makes `from_msgpack()` check anyway and throw [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) for ill-formed UTF-8, and - `replace`/`ignore` sanitize the string instead of keeping it. `to_msgpack()` itself has no `error_handler` - parameter and always writes `str` bytes as-is, since the specification permits it. However, + `replace`/`ignore` sanitize the string instead of keeping it. `to_msgpack()` also writes `str` bytes as-is by + default, since the specification permits it; its [`error_handler`](../../api/basic_json/to_msgpack.md) parameter + can be set to `strict` to throw [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) instead, or + to `replace`/`ignore` to sanitize the string, for instance for a decoder that rejects ill-formed UTF-8. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way with the default `keep` handler, unless an error handler is passed that replaces or ignores the ill-formed bytes. diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 407a0fde7..b909fdd9f 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -755,7 +755,8 @@ The `dump()` function only works with UTF-8 encoded strings; that is, if you ass The binary writers [`to_cbor()`](../api/basic_json/to_cbor.md), [`to_ubjson()`](../api/basic_json/to_ubjson.md), [`to_bjdata()`](../api/basic_json/to_bjdata.md), and [`to_bson()`](../api/basic_json/to_bson.md) throw this exception as well for a string value or object key that is not valid UTF-8 if their `error_handler` is `strict` (the default if -[`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled). +[`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled). So does +[`to_msgpack()`](../api/basic_json/to_msgpack.md) if `error_handler_t::strict` is passed. !!! failure "Example message" diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index bd8f29435..e74671db2 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -95,8 +95,8 @@ class binary_writer output_vector_sink, or output_adapter_sink wrapping a type-erased output adapter) @param[in] error_handler_ how to treat a string value or object key that - is not valid UTF-8 (CBOR, UBJSON, BJData, and BSON only; never - consulted by @ref write_msgpack or @ref write_bon8) + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = binary_writer_default_error_handler()) : oa(std::move(sink)), error_handler(error_handler_) @@ -113,8 +113,8 @@ class binary_writer @param[in] adapter output adapter to write to @param[in] error_handler_ how to treat a string value or object key that - is not valid UTF-8 (CBOR, UBJSON, BJData, and BSON only; never - consulted by @ref write_msgpack or @ref write_bon8) + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ template < typename SinkType = OutputSinkType, typename std::enable_if < std::is_constructible>::value, int >::type = 0 > @@ -461,8 +461,11 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); + const auto N = to_msgpack_length(value.size(), j); if (N <= 31) { // fixstr @@ -489,8 +492,8 @@ class binary_writer // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -637,6 +640,13 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // as in write_cbor, el.first is checked here against the + // object as diagnostics context; the recursive call below + // handles keep/replace/ignore like any other string + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_msgpack(el.first); write_msgpack(el.second); } @@ -2220,12 +2230,10 @@ class binary_writer /*! @brief return @a s as it should be written, honoring @ref error_handler - Used by @ref write_cbor, @ref write_ubjson (and so @ref write_bjdata), and - the BSON writing functions for string values and object keys; never by - @ref write_msgpack or @ref write_bon8, which do not take an @ref - error_handler (MessagePack's spec allows a str object to contain - ill-formed UTF-8, and BON8 always validates, since UTF-8 lead bytes are - structural there). + Used by @ref write_cbor, @ref write_msgpack, @ref write_ubjson (and so + @ref write_bjdata), and the BSON writing functions for string values and + object keys; never by @ref write_bon8, which always validates, since UTF-8 + lead bytes are structural there. - @ref error_handler_t::keep: @a s is returned unchanged, without even checking it (the behavior of release 3.12.0 and earlier). @@ -2600,7 +2608,7 @@ class binary_writer OutputSinkType oa; /// how to treat a string value or object key that is not valid UTF-8 - /// (CBOR, UBJSON, BJData, and BSON only) + /// (CBOR, MessagePack, UBJSON, BJData, and BSON; not BON8) const error_handler_t error_handler = binary_writer_default_error_handler(); }; diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 1b374493d..8a37e8810 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -5473,26 +5473,29 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static std::vector to_msgpack(const basic_json& j) + static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_msgpack(j); + vector_writer(result, error_handler).write_msgpack(j); return result; } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a UBJSON serialization of a given JSON value diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index b4c60f53e..da09502dc 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -21418,8 +21418,8 @@ class binary_writer output_vector_sink, or output_adapter_sink wrapping a type-erased output adapter) @param[in] error_handler_ how to treat a string value or object key that - is not valid UTF-8 (CBOR, UBJSON, BJData, and BSON only; never - consulted by @ref write_msgpack or @ref write_bon8) + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = binary_writer_default_error_handler()) : oa(std::move(sink)), error_handler(error_handler_) @@ -21436,8 +21436,8 @@ class binary_writer @param[in] adapter output adapter to write to @param[in] error_handler_ how to treat a string value or object key that - is not valid UTF-8 (CBOR, UBJSON, BJData, and BSON only; never - consulted by @ref write_msgpack or @ref write_bon8) + is not valid UTF-8 (CBOR, MessagePack, UBJSON, BJData, and BSON; + never consulted by @ref write_bon8) */ template < typename SinkType = OutputSinkType, typename std::enable_if < std::is_constructible>::value, int >::type = 0 > @@ -21784,8 +21784,11 @@ class binary_writer case value_t::string: { + string_t storage; + const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); + // step 1: write control byte and the string length - const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); + const auto N = to_msgpack_length(value.size(), j); if (N <= 31) { // fixstr @@ -21812,8 +21815,8 @@ class binary_writer // step 2: write the string oa.write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + reinterpret_cast(value.data()), + value.size()); break; } @@ -21960,6 +21963,13 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // as in write_cbor, el.first is checked here against the + // object as diagnostics context; the recursive call below + // handles keep/replace/ignore like any other string + if (error_handler == error_handler_t::strict) + { + check_utf8(el.first, j); + } write_msgpack(el.first); write_msgpack(el.second); } @@ -23543,12 +23553,10 @@ class binary_writer /*! @brief return @a s as it should be written, honoring @ref error_handler - Used by @ref write_cbor, @ref write_ubjson (and so @ref write_bjdata), and - the BSON writing functions for string values and object keys; never by - @ref write_msgpack or @ref write_bon8, which do not take an @ref - error_handler (MessagePack's spec allows a str object to contain - ill-formed UTF-8, and BON8 always validates, since UTF-8 lead bytes are - structural there). + Used by @ref write_cbor, @ref write_msgpack, @ref write_ubjson (and so + @ref write_bjdata), and the BSON writing functions for string values and + object keys; never by @ref write_bon8, which always validates, since UTF-8 + lead bytes are structural there. - @ref error_handler_t::keep: @a s is returned unchanged, without even checking it (the behavior of release 3.12.0 and earlier). @@ -23923,7 +23931,7 @@ class binary_writer OutputSinkType oa; /// how to treat a string value or object key that is not valid UTF-8 - /// (CBOR, UBJSON, BJData, and BSON only) + /// (CBOR, MessagePack, UBJSON, BJData, and BSON; not BON8) const error_handler_t error_handler = binary_writer_default_error_handler(); }; @@ -32543,26 +32551,29 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static std::vector to_msgpack(const basic_json& j) + static std::vector to_msgpack(const basic_json& j, + const error_handler_t error_handler = error_handler_t::keep) { std::vector result; result.reserve(detail::binary_reserve_hint(j)); - vector_writer(result).write_msgpack(j); + vector_writer(result, error_handler).write_msgpack(j); return result; } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a MessagePack serialization of a given JSON value /// @sa https://json.nlohmann.me/api/basic_json/to_msgpack/ - static void to_msgpack(const basic_json& j, detail::output_adapter o) + static void to_msgpack(const basic_json& j, detail::output_adapter o, + const error_handler_t error_handler = error_handler_t::keep) { - binary_writer(o).write_msgpack(j); + binary_writer(o, error_handler).write_msgpack(j); } /// @brief create a UBJSON serialization of a given JSON value diff --git a/tests/src/unit-binary_utf8_error_handler.cpp b/tests/src/unit-binary_utf8_error_handler.cpp index 7212322ce..85505bffb 100644 --- a/tests/src/unit-binary_utf8_error_handler.cpp +++ b/tests/src/unit-binary_utf8_error_handler.cpp @@ -57,6 +57,7 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers") const json jval = c.bytes; CHECK_THROWS_AS(json::to_cbor(jval, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_msgpack(jval, eh::strict), json::type_error&); CHECK_THROWS_AS(json::to_ubjson(jval, false, false, eh::strict), json::type_error&); CHECK_THROWS_AS(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&); { @@ -74,6 +75,7 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers") const std::string expected = dump_and_parse(c.bytes, h); CHECK(json::from_cbor(json::to_cbor(jval, h)).get() == expected); + CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get() == expected); CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get() == expected); CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get() == expected); { @@ -87,6 +89,7 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers") // keep: the writer passes the ill-formed bytes through unchanged, // exactly as every binary writer did before this parameter existed CHECK(json::from_cbor(json::to_cbor(jval, eh::keep)).get() == c.bytes); + CHECK(json::from_msgpack(json::to_msgpack(jval, eh::keep)).get() == c.bytes); CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, eh::keep)).get() == c.bytes); CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep)).get() == c.bytes); { @@ -107,6 +110,7 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers") jobj[c.bytes] = 1; CHECK_THROWS_AS(json::to_cbor(jobj, eh::strict), json::type_error&); + CHECK_THROWS_AS(json::to_msgpack(jobj, eh::strict), json::type_error&); CHECK_THROWS_AS(json::to_ubjson(jobj, false, false, eh::strict), json::type_error&); CHECK_THROWS_AS(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&); CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&); @@ -120,6 +124,7 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers") const std::string expected = dump_and_parse(c.bytes, h); CHECK(json::from_cbor(json::to_cbor(jobj, h)).begin().key() == expected); + CHECK(json::from_msgpack(json::to_msgpack(jobj, h)).begin().key() == expected); CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, h)).begin().key() == expected); CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, h)).begin().key() == expected); CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == expected); @@ -127,6 +132,7 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers") // keep: object keys round-trip unchanged too CHECK(json::from_cbor(json::to_cbor(jobj, eh::keep)).begin().key() == c.bytes); + CHECK(json::from_msgpack(json::to_msgpack(jobj, eh::keep)).begin().key() == c.bytes); CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, eh::keep)).begin().key() == c.bytes); CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep)).begin().key() == c.bytes); CHECK(json::from_bson(json::to_bson(jobj, eh::keep)).begin().key() == c.bytes); @@ -243,6 +249,7 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers") CAPTURE(static_cast(h)); CHECK(json::from_cbor(json::to_cbor(jval, h)).get() == valid_sequence); + CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get() == valid_sequence); CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get() == valid_sequence); CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get() == valid_sequence); CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == valid_sequence); @@ -294,16 +301,22 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers") CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', true, eh::keep) == "\"\xC3\\u00e9\""); } - SECTION("to_msgpack and to_bon8 are not affected by error_handler") + SECTION("to_msgpack defaults to keep; to_bon8 is not affected by error_handler") { const json jval = ill_formed_cases[1].bytes; // lone 0xFF - // to_msgpack has no error_handler parameter; the bytes are always - // passed through, as MessagePack's spec allows + // to_msgpack's error_handler defaults to keep, as MessagePack's spec + // allows any bytes in a str, so the bytes are passed through + CHECK(json::to_msgpack(jval) == json::to_msgpack(jval, eh::keep)); CHECK(json::from_msgpack(json::to_msgpack(jval)).get() == ill_formed_cases[1].bytes); - // to_bon8 has no error_handler parameter either; UTF-8 is structural - // for BON8, so it always rejects ill-formed input + // the diagnostics context of an ill-formed key is the object + json jobj; + jobj["\xFF"] = 1; + CHECK_THROWS_WITH_AS(json::to_msgpack(jobj, eh::strict), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + + // to_bon8 has no error_handler parameter; UTF-8 is structural for + // BON8, so it always rejects ill-formed input CHECK_THROWS_AS(json::to_bon8(jval), json::type_error&); } diff --git a/tests/src/unit-binary_utf8_strict.cpp b/tests/src/unit-binary_utf8_strict.cpp index 1ed144957..2ac8aabf4 100644 --- a/tests/src/unit-binary_utf8_strict.cpp +++ b/tests/src/unit-binary_utf8_strict.cpp @@ -112,9 +112,11 @@ TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)") SECTION("MessagePack and BON8 are unaffected") { - // MessagePack allows any bytes in a str, so to_msgpack() writes them as - // is; BON8 always checks, because the lead bytes mark where strings end + // MessagePack allows any bytes in a str, so to_msgpack() still + // defaults to keep (strict only if passed explicitly); BON8 always + // checks, because the lead bytes mark where strings end CHECK(json::to_msgpack(json("\xFF")) == std::vector({0xa1, 0xff})); + CHECK_THROWS_AS(json::to_msgpack(json("\xFF"), json::error_handler_t::strict), json::type_error&); CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&); } }