diff --git a/tests/src/unit-deserialization.cpp b/tests/src/unit-deserialization.cpp index 4202644f6..aeb0507b7 100644 --- a/tests/src/unit-deserialization.cpp +++ b/tests/src/unit-deserialization.cpp @@ -28,11 +28,6 @@ using nlohmann::json; #include #include -#if defined(_WIN32) - #define NOMINMAX - #include // for GetACP() -#endif - namespace { struct SaxEventLogger : public nlohmann::json_sax @@ -228,24 +223,6 @@ class proxy_iterator iterator* m_it = nullptr; }; -// JSON_HAS_CPP_20 -#if defined(__cpp_char8_t) -bool check_utf8() -{ -#if defined(_WIN32) - // Runtime check of the active ANSI code page - // 65001 == UTF-8 - return GetACP() == 65001; -#elif defined(__ICC) || defined(__INTEL_COMPILER) - // classic Intel ICC does not encode narrow string literals containing - // non-ASCII source characters as UTF-8, so comparing a decoded u8 literal - // against a narrow string literal containing the same characters fails - return false; -#else - return true; -#endif -} -#endif } // namespace TEST_CASE("deserialization") @@ -1297,14 +1274,15 @@ TEST_CASE("deserialization") CHECK(j1["key"] == "value"); CHECK(j1["num"] == 42); - // UTF-8 prefixed literal (C++20 and later); - // MSVC may not set /utf-8, so we need to check - if (check_utf8()) - { - const auto j2 = u8R"({"emoji": "šŸ˜€", "msg": "hello"})"_json; - CHECK(j2["emoji"] == "šŸ˜€"); - CHECK(j2["msg"] == "hello"); - } + // UTF-8 prefixed literal (C++20 and later); the emoji is written as a + // \U escape rather than a raw multibyte character so this does not + // depend on the compiler's source-file encoding (e.g., MSVC without + // /utf-8, or classic ICC, which does not encode non-ASCII narrow + // string literals as UTF-8 - compare against a \x-escaped expectation + // for the same reason) + const auto j2 = u8"{\"emoji\": \"\U0001F600\", \"msg\": \"hello\"}"_json; + CHECK(j2["emoji"] == "\xF0\x9F\x98\x80"); + CHECK(j2["msg"] == "hello"); const auto j3 = u8R"({"key": "value", "num": 42})"_json; CHECK(j3["key"] == "value"); diff --git a/tests/src/unit-wstring.cpp b/tests/src/unit-wstring.cpp index 6e0aff29e..b553ce246 100644 --- a/tests/src/unit-wstring.cpp +++ b/tests/src/unit-wstring.cpp @@ -11,135 +11,97 @@ #include using nlohmann::json; -// ICPC errors out on multibyte character sequences in source files -#ifndef __INTEL_COMPILER -namespace -{ -bool wstring_is_utf16(); -bool wstring_is_utf16() -{ - return (std::wstring(L"šŸ’©") == std::wstring(L"\U0001F4A9")); -} - -bool u16string_is_utf16(); -bool u16string_is_utf16() -{ - return (std::u16string(u"šŸ’©") == std::u16string(u"\U0001F4A9")); -} - -bool u32string_is_utf32(); -bool u32string_is_utf32() -{ - return (std::u32string(U"šŸ’©") == std::u32string(U"\U0001F4A9")); -} -} // namespace - TEST_CASE("wide strings") { SECTION("std::wstring") { - if (wstring_is_utf16()) - { - std::wstring const w = L"[12.2,\"įƒ…aĆ¤Ć¶šŸ’¤šŸ§¢\"]"; - json const j = json::parse(w); - CHECK(j.dump() == "[12.2,\"įƒ…aĆ¤Ć¶šŸ’¤šŸ§¢\"]"); - } + // U+10C5 U+0061(a) U+00E4 U+00F6 U+1F4A4 U+1F9E2, written with \u/\U + // escapes rather than as raw multibyte characters so this file + // compiles on toolchains (e.g. classic ICC) that error out on + // multibyte character sequences in source files + std::wstring const w = L"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]"; + json const j = json::parse(w); + CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]"); } SECTION("invalid std::wstring") { - if (wstring_is_utf16()) - { - std::wstring const w = L"\"\xDBFF"; - json _; - CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&); + std::wstring const w = L"\"\xDBFF"; + json _; + CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&); - // the exact message depends on the width of wchar_t: a 16-bit - // wchar_t passes the lone surrogate to the UTF-8 decoder unchanged - // (rejected as a single ill-formed byte at column 2), while a - // 32-bit wchar_t first encodes it as an ill-formed three-byte - // sequence (rejected one byte later, at column 3) - const char* const error_low_surrogate = sizeof(wchar_t) == 2 - ? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'" - : "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'"; - const char* const error_high_surrogate = sizeof(wchar_t) == 2 - ? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'" - : "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'"; + // the exact message depends on the width of wchar_t: a 16-bit + // wchar_t passes the lone surrogate to the UTF-8 decoder unchanged + // (rejected as a single ill-formed byte at column 2), while a + // 32-bit wchar_t first encodes it as an ill-formed three-byte + // sequence (rejected one byte later, at column 3) + const char* const error_low_surrogate = sizeof(wchar_t) == 2 + ? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'" + : "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'"; + const char* const error_high_surrogate = sizeof(wchar_t) == 2 + ? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'" + : "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'"; - // a lone low surrogate cannot start a pair - CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xDC00), L'"'}), error_low_surrogate, json::parse_error&); - // a high surrogate followed by a non-low-surrogate unit is invalid - CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&); - // ... also when the unit is above the low surrogates - CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xD800), static_cast(0xE000), L'"'}), error_high_surrogate, json::parse_error&); - // a lone low surrogate must not swallow the following unit: pairing - // it with any second unit would produce valid UTF-8, so the error - // has to report an ill-formed byte at the surrogate's own position - CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&); - } + // a lone low surrogate cannot start a pair + CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xDC00), L'"'}), error_low_surrogate, json::parse_error&); + // a high surrogate followed by a non-low-surrogate unit is invalid + CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&); + // ... also when the unit is above the low surrogates + CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xD800), static_cast(0xE000), L'"'}), error_high_surrogate, json::parse_error&); + // a lone low surrogate must not swallow the following unit: pairing + // it with any second unit would produce valid UTF-8, so the error + // has to report an ill-formed byte at the surrogate's own position + CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&); } SECTION("std::u16string") { - if (u16string_is_utf16()) - { - std::u16string const w = u"[12.2,\"įƒ…aĆ¤Ć¶šŸ’¤šŸ§¢\"]"; - json const j = json::parse(w); - CHECK(j.dump() == "[12.2,\"įƒ…aĆ¤Ć¶šŸ’¤šŸ§¢\"]"); - } + std::u16string const w = u"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]"; + json const j = json::parse(w); + CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]"); } SECTION("invalid std::u16string") { - if (u16string_is_utf16()) - { - std::u16string const w = u"\"\xDBFF"; - json _; - CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&); + std::u16string const w = u"\"\xDBFF"; + json _; + CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&); - // a lone low surrogate cannot start a pair - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); - // a high surrogate followed by a non-low-surrogate unit is invalid - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); - // ... also when the unit is above the low surrogates - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); - // a lone low surrogate must not swallow the following unit: pairing - // it with any second unit would produce valid UTF-8, so the error - // has to report an ill-formed byte at the surrogate's own position - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); - // a valid surrogate pair is still decoded (U+1F600) - CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get() == "\xF0\x9F\x98\x80"); - } + // a lone low surrogate cannot start a pair + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + // a high surrogate followed by a non-low-surrogate unit is invalid + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + // ... also when the unit is above the low surrogates + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + // a lone low surrogate must not swallow the following unit: pairing + // it with any second unit would produce valid UTF-8, so the error + // has to report an ill-formed byte at the surrogate's own position + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + // a valid surrogate pair is still decoded (U+1F600) + CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get() == "\xF0\x9F\x98\x80"); } SECTION("std::u32string") { - if (u32string_is_utf32()) - { - std::u32string const w = U"[12.2,\"įƒ…aĆ¤Ć¶šŸ’¤šŸ§¢\"]"; - json const j = json::parse(w); - CHECK(j.dump() == "[12.2,\"įƒ…aĆ¤Ć¶šŸ’¤šŸ§¢\"]"); - } + std::u32string const w = U"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]"; + json const j = json::parse(w); + CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]"); } SECTION("invalid std::u32string") { - if (u32string_is_utf32()) - { - std::u32string const w = U"\"\x110000"; - json _; - CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&); + std::u32string const w = U"\"\x110000"; + json _; + CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&); - // a code unit above U+10FFFF must not be narrowed onto the EOF - // sentinel: 0xFFFFFFFF would otherwise end the document silently and - // let everything following it pass the strict end-of-input check - std::u32string const trailing{U'[', U'1', U']', static_cast(0xFFFFFFFF), U'x'}; - CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&); - CHECK(!json::accept(trailing)); + // a code unit above U+10FFFF must not be narrowed onto the EOF + // sentinel: 0xFFFFFFFF would otherwise end the document silently and + // let everything following it pass the strict end-of-input check + std::u32string const trailing{U'[', U'1', U']', static_cast(0xFFFFFFFF), U'x'}; + CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&); + CHECK(!json::accept(trailing)); - // the same unit inside a string is reported as an ill-formed byte - CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); - } + // the same unit inside a string is reported as an ill-formed byte + CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); } } -#endif