From 66b0289903169fa030f5ec99687db11a0a5ebf1b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 6 Oct 2026 07:26:50 +0200 Subject: [PATCH] Add tests for error_handler_t::keep in dump() Squashed onto develop from: - Add error_handler_t::keep to copy invalid UTF-8 bytes unchanged - Mention error_handler_t::keep in the README Signed-off-by: Niels Lohmann --- README.md | 4 +-- docs/mkdocs/docs/features/serialization.md | 1 + docs/mkdocs/docs/home/exceptions.md | 1 + docs/mkdocs/docs/home/faq.md | 2 +- tests/src/unit-regression2.cpp | 9 ++++++ tests/src/unit-serialization.cpp | 37 ++++++++++++++++++++++ tests/src/unit-unicode2.cpp | 26 ++++++++++++++- tests/src/unit-unicode3.cpp | 26 ++++++++++++++- tests/src/unit-unicode4.cpp | 26 ++++++++++++++- tests/src/unit-unicode5.cpp | 26 ++++++++++++++- 10 files changed, 151 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index ae0ad12d2..19c8b8ab2 100644 --- a/README.md +++ b/README.md @@ -361,7 +361,7 @@ std::cout << j_string << " == " << serialized_string << std::endl; [`.dump()`](https://json.nlohmann.me/api/basic_json/dump/) returns the originally stored string value. -Note the library only supports UTF-8. When you store strings with different encodings in the library, calling [`dump()`](https://json.nlohmann.me/api/basic_json/dump/) may throw an exception unless `json::error_handler_t::replace` or `json::error_handler_t::ignore` are used as error handlers. +Note the library only supports UTF-8. When you store strings with different encodings in the library, calling [`dump()`](https://json.nlohmann.me/api/basic_json/dump/) may throw an exception unless `json::error_handler_t::replace`, `json::error_handler_t::ignore`, or `json::error_handler_t::keep` are used as error handlers. #### To/from streams (e.g., files, string streams) @@ -1914,7 +1914,7 @@ The library supports **Unicode input** as follows: - [Unicode noncharacters](https://www.unicode.org/faq/private_use.html#nonchar1) will not be replaced by the library. - Invalid surrogates (e.g., incomplete pairs such as `\uDEAD`) will yield parse errors. - The strings stored in the library are UTF-8 encoded. When using the default string type (`std::string`), note that its length/size functions return the number of stored bytes rather than the number of characters or glyphs. -- When you store strings with different encodings in the library, calling [`dump()`](https://json.nlohmann.me/api/basic_json/dump/) may throw an exception unless `json::error_handler_t::replace` or `json::error_handler_t::ignore` are used as error handlers. +- When you store strings with different encodings in the library, calling [`dump()`](https://json.nlohmann.me/api/basic_json/dump/) may throw an exception unless `json::error_handler_t::replace`, `json::error_handler_t::ignore`, or `json::error_handler_t::keep` are used as error handlers. - To store wide strings (e.g., `std::wstring`), you need to convert them to a UTF-8 encoded `std::string` before, see [an example](https://json.nlohmann.me/home/faq/#wide-string-handling). ### Comments in JSON diff --git a/docs/mkdocs/docs/features/serialization.md b/docs/mkdocs/docs/features/serialization.md index d9802bf08..1190c9005 100644 --- a/docs/mkdocs/docs/features/serialization.md +++ b/docs/mkdocs/docs/features/serialization.md @@ -64,6 +64,7 @@ serialization fails by default. The fourth argument of `dump` selects an - `strict` (default) — throw a [`type_error.316`](../home/exceptions.md#jsonexceptiontype_error316) exception. - `replace` — replace invalid bytes with the Unicode replacement character U+FFFD (`�`). - `ignore` — silently drop invalid bytes. +- `keep` — copy invalid bytes to the output unchanged; the result is not valid UTF-8. ??? example "Example: serialize invalid UTF-8 with different error handlers" diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 673c0c8dd..f26a330ff 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -771,6 +771,7 @@ as well for a string value or object key that is not valid UTF-8 if their `error - Pass an error handler as last parameter to the `dump()` function to avoid this exception: - `json::error_handler_t::replace` will replace invalid bytes sequences with `U+FFFD` - `json::error_handler_t::ignore` will silently ignore invalid byte sequences + - `json::error_handler_t::keep` will copy invalid byte sequences to the output unchanged ### json.exception.type_error.317 diff --git a/docs/mkdocs/docs/home/faq.md b/docs/mkdocs/docs/home/faq.md index 852fd8ffd..eec13a481 100644 --- a/docs/mkdocs/docs/home/faq.md +++ b/docs/mkdocs/docs/home/faq.md @@ -85,7 +85,7 @@ The library supports **Unicode input** as follows: - The library will not replace [Unicode noncharacters](http://www.unicode.org/faq/private_use.html#nonchar1). - Invalid surrogates (e.g., incomplete pairs such as `\uDEAD`) will yield parse errors. - The strings stored in the library are UTF-8 encoded. When using the default string type (`std::string`), note that its length/size functions return the number of stored bytes rather than the number of characters or glyphs. -- When you store strings with different encodings in the library, calling [`dump()`](../api/basic_json/dump.md) may throw an exception unless `json::error_handler_t::replace` or `json::error_handler_t::ignore` are used as error handlers. +- When you store strings with different encodings in the library, calling [`dump()`](../api/basic_json/dump.md) may throw an exception unless `json::error_handler_t::replace`, `json::error_handler_t::ignore`, or `json::error_handler_t::keep` are used as error handlers. In most cases, the parser is right to complain, because the input is not UTF-8 encoded. This is especially true for Microsoft Windows, where Latin-1 or ISO 8859-1 is often the standard encoding. diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index 1a8c73d4e..ef82b9088 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -808,6 +808,15 @@ TEST_CASE("regression tests 2") CHECK(j == k); } + SECTION("issue #4552 - UTF-8 invalid characters are not always ignored when dumping with error_handler_t::ignore") + { + json node; + node["test"] = "test\334\005"; + CHECK(node.dump(-1, ' ', false, json::error_handler_t::ignore) == "{\"test\":\"test\\u0005\"}"); + CHECK(node.dump(-1, ' ', false, json::error_handler_t::keep) == "{\"test\":\"test\334\\u0005\"}"); + CHECK(node.dump(-1, ' ', true, json::error_handler_t::keep) == "{\"test\":\"test\334\\u0005\"}"); + } + #ifdef JSON_HAS_CPP_17 SECTION("issue #5066 - MSVC converts json to std::variant via the conversion operator") { diff --git a/tests/src/unit-serialization.cpp b/tests/src/unit-serialization.cpp index 4b7a53b10..446be417f 100644 --- a/tests/src/unit-serialization.cpp +++ b/tests/src/unit-serialization.cpp @@ -102,6 +102,8 @@ TEST_CASE("serialization") CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"äü\""); CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"ä\xEF\xBF\xBDü\""); CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"\\u00e4\\ufffd\\u00fc\""); + CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"ä\xA9ü\""); + CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"\\u00e4\xA9\\u00fc\""); } SECTION("invalid character (regression guard for shared UTF-8 decoder, see #5529)") @@ -124,6 +126,8 @@ TEST_CASE("serialization") CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123\""); CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\""); CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd\""); + CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"123\xC2\""); + CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"123\xC2\""); } SECTION("unexpected character") @@ -136,6 +140,39 @@ TEST_CASE("serialization") CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123456\""); CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\x34\x35\x36\""); CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd456\""); + CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"123\xF1\xB0\x34\x35\x36\""); + CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"123\xF1\xB0\x34\x35\x36\""); + } + + SECTION("keep: valid characters are still escaped") + { + // an invalid byte followed by characters that must be escaped + const json j = "\xC2\"\\\n\xFF\x05"; + CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xC2\\\"\\\\\\n\xFF\\u0005\""); + CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xC2\\\"\\\\\\n\xFF\\u0005\""); + } + + SECTION("keep: truncated multibyte sequences") + { + CHECK(json("\xF0\x9F\x98").dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xF0\x9F\x98\""); + CHECK(json("\xF0\x9F\x98").dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xF0\x9F\x98\""); + CHECK(json("\xF0\x9F\x98" "a").dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xF0\x9F\x98" "a\""); + CHECK(json("\xF0\x9F\x98" "a").dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xF0\x9F\x98" "a\""); + } + + SECTION("keep: long string with many invalid bytes") + { + // exceeds the internal string buffer several times + std::string input; + std::string expected = "\""; + for (int i = 0; i < 2000; ++i) + { + input += "\xFF\xE2\x82\n\xC3\xA4"; + expected += "\xFF\xE2\x82\\n\xC3\xA4"; + } + expected += "\""; + const json j = input; + CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == expected); } SECTION("U+FFFD Substitution of Maximal Subparts") diff --git a/tests/src/unit-unicode2.cpp b/tests/src/unit-unicode2.cpp index a9649b4de..f77528d5a 100644 --- a/tests/src/unit-unicode2.cpp +++ b/tests/src/unit-unicode2.cpp @@ -14,6 +14,7 @@ #include using nlohmann::json; +#include #include #include #include @@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 static std::string s_replaced2; static std::string s_replaced_ascii; static std::string s_replaced2_ascii; + static std::string s_kept; + static std::string s_kept2; + static std::string s_kept_ascii; - // dumping with ignore/replace must not throw in any case + // dumping with ignore/replace/keep must not throw in any case s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore); @@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace); s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace); s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace); + s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep); if (success_expected) { @@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // all dumps should agree on the string CHECK(s_strict == s_ignored); CHECK(s_strict == s_replaced); + CHECK(s_strict == s_kept); } else { @@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // check that replace string contains a replacement character CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos); + + // ignore drops the invalid bytes, keep copies them + CHECK(s_ignored != s_kept); + CHECK(s_ignored_ascii != s_kept_ascii); + + // unless a byte needs escaping, keep copies the input unchanged + const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c) + { + return static_cast(c) < 0x20 || c == '"' || c == '\\'; + }); + if (!needs_escaping) + { + CHECK(s_kept == "\"" + json_string + "\""); + } } // check that prefix and suffix are preserved @@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz"); CHECK(s_replaced2_ascii.substr(1, 3) == "abc"); CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz"); + CHECK(s_kept2.substr(1, 3) == "abc"); + CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz"); } void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4); diff --git a/tests/src/unit-unicode3.cpp b/tests/src/unit-unicode3.cpp index 12c12eea4..e4877a0ba 100644 --- a/tests/src/unit-unicode3.cpp +++ b/tests/src/unit-unicode3.cpp @@ -14,6 +14,7 @@ #include using nlohmann::json; +#include #include #include #include @@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 static std::string s_replaced2; static std::string s_replaced_ascii; static std::string s_replaced2_ascii; + static std::string s_kept; + static std::string s_kept2; + static std::string s_kept_ascii; - // dumping with ignore/replace must not throw in any case + // dumping with ignore/replace/keep must not throw in any case s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore); @@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace); s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace); s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace); + s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep); if (success_expected) { @@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // all dumps should agree on the string CHECK(s_strict == s_ignored); CHECK(s_strict == s_replaced); + CHECK(s_strict == s_kept); } else { @@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // check that replace string contains a replacement character CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos); + + // ignore drops the invalid bytes, keep copies them + CHECK(s_ignored != s_kept); + CHECK(s_ignored_ascii != s_kept_ascii); + + // unless a byte needs escaping, keep copies the input unchanged + const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c) + { + return static_cast(c) < 0x20 || c == '"' || c == '\\'; + }); + if (!needs_escaping) + { + CHECK(s_kept == "\"" + json_string + "\""); + } } // check that prefix and suffix are preserved @@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz"); CHECK(s_replaced2_ascii.substr(1, 3) == "abc"); CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz"); + CHECK(s_kept2.substr(1, 3) == "abc"); + CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz"); } void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4); diff --git a/tests/src/unit-unicode4.cpp b/tests/src/unit-unicode4.cpp index 43cf7095e..4265a7e84 100644 --- a/tests/src/unit-unicode4.cpp +++ b/tests/src/unit-unicode4.cpp @@ -14,6 +14,7 @@ #include using nlohmann::json; +#include #include #include #include @@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 static std::string s_replaced2; static std::string s_replaced_ascii; static std::string s_replaced2_ascii; + static std::string s_kept; + static std::string s_kept2; + static std::string s_kept_ascii; - // dumping with ignore/replace must not throw in any case + // dumping with ignore/replace/keep must not throw in any case s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore); @@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace); s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace); s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace); + s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep); if (success_expected) { @@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // all dumps should agree on the string CHECK(s_strict == s_ignored); CHECK(s_strict == s_replaced); + CHECK(s_strict == s_kept); } else { @@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // check that replace string contains a replacement character CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos); + + // ignore drops the invalid bytes, keep copies them + CHECK(s_ignored != s_kept); + CHECK(s_ignored_ascii != s_kept_ascii); + + // unless a byte needs escaping, keep copies the input unchanged + const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c) + { + return static_cast(c) < 0x20 || c == '"' || c == '\\'; + }); + if (!needs_escaping) + { + CHECK(s_kept == "\"" + json_string + "\""); + } } // check that prefix and suffix are preserved @@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz"); CHECK(s_replaced2_ascii.substr(1, 3) == "abc"); CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz"); + CHECK(s_kept2.substr(1, 3) == "abc"); + CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz"); } void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4); diff --git a/tests/src/unit-unicode5.cpp b/tests/src/unit-unicode5.cpp index bc0312820..8d45b7b85 100644 --- a/tests/src/unit-unicode5.cpp +++ b/tests/src/unit-unicode5.cpp @@ -14,6 +14,7 @@ #include using nlohmann::json; +#include #include #include #include @@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 static std::string s_replaced2; static std::string s_replaced_ascii; static std::string s_replaced2_ascii; + static std::string s_kept; + static std::string s_kept2; + static std::string s_kept_ascii; - // dumping with ignore/replace must not throw in any case + // dumping with ignore/replace/keep must not throw in any case s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore); @@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace); s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace); s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace); + s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep); if (success_expected) { @@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // all dumps should agree on the string CHECK(s_strict == s_ignored); CHECK(s_strict == s_replaced); + CHECK(s_strict == s_kept); } else { @@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // check that replace string contains a replacement character CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos); + + // ignore drops the invalid bytes, keep copies them + CHECK(s_ignored != s_kept); + CHECK(s_ignored_ascii != s_kept_ascii); + + // unless a byte needs escaping, keep copies the input unchanged + const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c) + { + return static_cast(c) < 0x20 || c == '"' || c == '\\'; + }); + if (!needs_escaping) + { + CHECK(s_kept == "\"" + json_string + "\""); + } } // check that prefix and suffix are preserved @@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz"); CHECK(s_replaced2_ascii.substr(1, 3) == "abc"); CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz"); + CHECK(s_kept2.substr(1, 3) == "abc"); + CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz"); } void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);