mirror of
https://github.com/nlohmann/json.git
synced 2026-10-01 12:10:32 +00:00
Stop compiling unit-wstring.cpp out entirely on classic ICC
tests/src/unit-wstring.cpp wrapped the whole file in #ifndef __INTEL_COMPILER, with the comment "ICPC errors out on multibyte character sequences in source files". The ci_icpc job (intel/oneapi-hpckit:2023.2.1) still exists, so that job ran none of the wstring/u16string/u32string input adapter tests, including the malformed-input checks #5704 (open) extends. Only 9 lines contained non-ASCII bytes: the three *_is_utf16()/ *_is_utf32() probe functions, and three std::wstring/u16string/ u32string literals plus their narrow-string dump() expectations. Rewrite all of them with \u/\U escapes in the wide/u16/u32 literals and \x escapes (split into separate string-literal tokens so a following byte is never read as part of the same hex escape, e.g. "\xE1\x83\x85" "a") in the narrow ones. Remove the #ifndef __INTEL_COMPILER/#endif guard along with it. The *_is_utf16()/*_is_utf32() probes compared a raw multibyte literal against an escape-based one to detect a compiler that misreads the source file's encoding; with no raw literals left to misread, the comparison is now tautological, so drop the probes and the "if" guards around each SECTION's body instead of leaving them in as dead checks. The same non-ASCII-in-source-and-in-a-narrow-comparison pattern existed once more in unit-deserialization.cpp's "Using _json with char8_t literals #4945" test: a raw emoji character in a u8R"(...)" literal, guarded by a check_utf8() that returned false for ICC (same reason) and for Windows without the active UTF-8 code page. Rewrite the literal with a \U escape and compare it against a \x-escaped expectation instead of a second raw literal, and drop check_utf8() and the now-unused <windows.h> include along with the guard. Verified locally (clang, -std=c++11 and -std=c++20, -fsanitize=address,undefined, and a plain build): test-wstring keeps 18/18 assertions and unit-deserialization keeps 466/466 (c++11) and 477/477 (c++20) assertions, matching this branch before the change exactly - no coverage was gained or lost, only the source-encoding dependency was removed. ci_icpc has to confirm classic ICC actually builds and passes test-wstring now; if it does not, that is a real finding, not a reason to restore the guard. Overlaps #5704 (open), which edits unit-wstring.cpp inside the previously-guarded region (an include near the top, checks in the invalid-string sections, and a new section at the end). Closes #5713 item 5. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -28,11 +28,6 @@ using nlohmann::json;
|
||||
#include <string>
|
||||
#include <valarray>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#define NOMINMAX
|
||||
#include <windows.h> // for GetACP()
|
||||
#endif
|
||||
|
||||
namespace
|
||||
{
|
||||
struct SaxEventLogger : public nlohmann::json_sax<json>
|
||||
@@ -228,24 +223,6 @@ class proxy_iterator
|
||||
iterator* m_it = nullptr;
|
||||
};
|
||||
|
||||
// JSON_HAS_CPP_20
|
||||
#if defined(__cpp_char8_t)
|
||||
bool check_utf8()
|
||||
{
|
||||
#if defined(_WIN32)
|
||||
// Runtime check of the active ANSI code page
|
||||
// 65001 == UTF-8
|
||||
return GetACP() == 65001;
|
||||
#elif defined(__ICC) || defined(__INTEL_COMPILER)
|
||||
// classic Intel ICC does not encode narrow string literals containing
|
||||
// non-ASCII source characters as UTF-8, so comparing a decoded u8 literal
|
||||
// against a narrow string literal containing the same characters fails
|
||||
return false;
|
||||
#else
|
||||
return true;
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("deserialization")
|
||||
@@ -1297,14 +1274,15 @@ TEST_CASE("deserialization")
|
||||
CHECK(j1["key"] == "value");
|
||||
CHECK(j1["num"] == 42);
|
||||
|
||||
// UTF-8 prefixed literal (C++20 and later);
|
||||
// MSVC may not set /utf-8, so we need to check
|
||||
if (check_utf8())
|
||||
{
|
||||
const auto j2 = u8R"({"emoji": "😀", "msg": "hello"})"_json;
|
||||
CHECK(j2["emoji"] == "😀");
|
||||
CHECK(j2["msg"] == "hello");
|
||||
}
|
||||
// UTF-8 prefixed literal (C++20 and later); the emoji is written as a
|
||||
// \U escape rather than a raw multibyte character so this does not
|
||||
// depend on the compiler's source-file encoding (e.g., MSVC without
|
||||
// /utf-8, or classic ICC, which does not encode non-ASCII narrow
|
||||
// string literals as UTF-8 - compare against a \x-escaped expectation
|
||||
// for the same reason)
|
||||
const auto j2 = u8"{\"emoji\": \"\U0001F600\", \"msg\": \"hello\"}"_json;
|
||||
CHECK(j2["emoji"] == "\xF0\x9F\x98\x80");
|
||||
CHECK(j2["msg"] == "hello");
|
||||
|
||||
const auto j3 = u8R"({"key": "value", "num": 42})"_json;
|
||||
CHECK(j3["key"] == "value");
|
||||
|
||||
+63
-101
@@ -11,135 +11,97 @@
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
// ICPC errors out on multibyte character sequences in source files
|
||||
#ifndef __INTEL_COMPILER
|
||||
namespace
|
||||
{
|
||||
bool wstring_is_utf16();
|
||||
bool wstring_is_utf16()
|
||||
{
|
||||
return (std::wstring(L"💩") == std::wstring(L"\U0001F4A9"));
|
||||
}
|
||||
|
||||
bool u16string_is_utf16();
|
||||
bool u16string_is_utf16()
|
||||
{
|
||||
return (std::u16string(u"💩") == std::u16string(u"\U0001F4A9"));
|
||||
}
|
||||
|
||||
bool u32string_is_utf32();
|
||||
bool u32string_is_utf32()
|
||||
{
|
||||
return (std::u32string(U"💩") == std::u32string(U"\U0001F4A9"));
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("wide strings")
|
||||
{
|
||||
SECTION("std::wstring")
|
||||
{
|
||||
if (wstring_is_utf16())
|
||||
{
|
||||
std::wstring const w = L"[12.2,\"Ⴥaäö💤🧢\"]";
|
||||
json const j = json::parse(w);
|
||||
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
|
||||
}
|
||||
// U+10C5 U+0061(a) U+00E4 U+00F6 U+1F4A4 U+1F9E2, written with \u/\U
|
||||
// escapes rather than as raw multibyte characters so this file
|
||||
// compiles on toolchains (e.g. classic ICC) that error out on
|
||||
// multibyte character sequences in source files
|
||||
std::wstring const w = L"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]";
|
||||
json const j = json::parse(w);
|
||||
CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]");
|
||||
}
|
||||
|
||||
SECTION("invalid std::wstring")
|
||||
{
|
||||
if (wstring_is_utf16())
|
||||
{
|
||||
std::wstring const w = L"\"\xDBFF";
|
||||
json _;
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
std::wstring const w = L"\"\xDBFF";
|
||||
json _;
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
|
||||
// the exact message depends on the width of wchar_t: a 16-bit
|
||||
// wchar_t passes the lone surrogate to the UTF-8 decoder unchanged
|
||||
// (rejected as a single ill-formed byte at column 2), while a
|
||||
// 32-bit wchar_t first encodes it as an ill-formed three-byte
|
||||
// sequence (rejected one byte later, at column 3)
|
||||
const char* const error_low_surrogate = sizeof(wchar_t) == 2
|
||||
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
||||
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'";
|
||||
const char* const error_high_surrogate = sizeof(wchar_t) == 2
|
||||
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
||||
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'";
|
||||
// the exact message depends on the width of wchar_t: a 16-bit
|
||||
// wchar_t passes the lone surrogate to the UTF-8 decoder unchanged
|
||||
// (rejected as a single ill-formed byte at column 2), while a
|
||||
// 32-bit wchar_t first encodes it as an ill-formed three-byte
|
||||
// sequence (rejected one byte later, at column 3)
|
||||
const char* const error_low_surrogate = sizeof(wchar_t) == 2
|
||||
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
||||
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'";
|
||||
const char* const error_high_surrogate = sizeof(wchar_t) == 2
|
||||
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
||||
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'";
|
||||
|
||||
// a lone low surrogate cannot start a pair
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'"'}), error_low_surrogate, json::parse_error&);
|
||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&);
|
||||
// ... also when the unit is above the low surrogates
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), static_cast<wchar_t>(0xE000), L'"'}), error_high_surrogate, json::parse_error&);
|
||||
// a lone low surrogate must not swallow the following unit: pairing
|
||||
// it with any second unit would produce valid UTF-8, so the error
|
||||
// has to report an ill-formed byte at the surrogate's own position
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&);
|
||||
}
|
||||
// a lone low surrogate cannot start a pair
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'"'}), error_low_surrogate, json::parse_error&);
|
||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&);
|
||||
// ... also when the unit is above the low surrogates
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), static_cast<wchar_t>(0xE000), L'"'}), error_high_surrogate, json::parse_error&);
|
||||
// a lone low surrogate must not swallow the following unit: pairing
|
||||
// it with any second unit would produce valid UTF-8, so the error
|
||||
// has to report an ill-formed byte at the surrogate's own position
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&);
|
||||
}
|
||||
|
||||
SECTION("std::u16string")
|
||||
{
|
||||
if (u16string_is_utf16())
|
||||
{
|
||||
std::u16string const w = u"[12.2,\"Ⴥaäö💤🧢\"]";
|
||||
json const j = json::parse(w);
|
||||
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
|
||||
}
|
||||
std::u16string const w = u"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]";
|
||||
json const j = json::parse(w);
|
||||
CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]");
|
||||
}
|
||||
|
||||
SECTION("invalid std::u16string")
|
||||
{
|
||||
if (u16string_is_utf16())
|
||||
{
|
||||
std::u16string const w = u"\"\xDBFF";
|
||||
json _;
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
std::u16string const w = u"\"\xDBFF";
|
||||
json _;
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
|
||||
// a lone low surrogate cannot start a pair
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// ... also when the unit is above the low surrogates
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a lone low surrogate must not swallow the following unit: pairing
|
||||
// it with any second unit would produce valid UTF-8, so the error
|
||||
// has to report an ill-formed byte at the surrogate's own position
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a valid surrogate pair is still decoded (U+1F600)
|
||||
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
|
||||
}
|
||||
// a lone low surrogate cannot start a pair
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// ... also when the unit is above the low surrogates
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a lone low surrogate must not swallow the following unit: pairing
|
||||
// it with any second unit would produce valid UTF-8, so the error
|
||||
// has to report an ill-formed byte at the surrogate's own position
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a valid surrogate pair is still decoded (U+1F600)
|
||||
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
|
||||
}
|
||||
|
||||
SECTION("std::u32string")
|
||||
{
|
||||
if (u32string_is_utf32())
|
||||
{
|
||||
std::u32string const w = U"[12.2,\"Ⴥaäö💤🧢\"]";
|
||||
json const j = json::parse(w);
|
||||
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
|
||||
}
|
||||
std::u32string const w = U"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]";
|
||||
json const j = json::parse(w);
|
||||
CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]");
|
||||
}
|
||||
|
||||
SECTION("invalid std::u32string")
|
||||
{
|
||||
if (u32string_is_utf32())
|
||||
{
|
||||
std::u32string const w = U"\"\x110000";
|
||||
json _;
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
std::u32string const w = U"\"\x110000";
|
||||
json _;
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
|
||||
// a code unit above U+10FFFF must not be narrowed onto the EOF
|
||||
// sentinel: 0xFFFFFFFF would otherwise end the document silently and
|
||||
// let everything following it pass the strict end-of-input check
|
||||
std::u32string const trailing{U'[', U'1', U']', static_cast<char32_t>(0xFFFFFFFF), U'x'};
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
|
||||
CHECK(!json::accept(trailing));
|
||||
// a code unit above U+10FFFF must not be narrowed onto the EOF
|
||||
// sentinel: 0xFFFFFFFF would otherwise end the document silently and
|
||||
// let everything following it pass the strict end-of-input check
|
||||
std::u32string const trailing{U'[', U'1', U']', static_cast<char32_t>(0xFFFFFFFF), U'x'};
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
|
||||
CHECK(!json::accept(trailing));
|
||||
|
||||
// the same unit inside a string is reported as an ill-formed byte
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
||||
}
|
||||
// the same unit inside a string is reported as an ill-formed byte
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
Reference in New Issue
Block a user