Stop compiling unit-wstring.cpp out entirely on classic ICC

tests/src/unit-wstring.cpp wrapped the whole file in
#ifndef __INTEL_COMPILER, with the comment "ICPC errors out on
multibyte character sequences in source files". The ci_icpc job
(intel/oneapi-hpckit:2023.2.1) still exists, so that job ran none of
the wstring/u16string/u32string input adapter tests, including the
malformed-input checks #5704 (open) extends.

Only 9 lines contained non-ASCII bytes: the three *_is_utf16()/
*_is_utf32() probe functions, and three std::wstring/u16string/
u32string literals plus their narrow-string dump() expectations.
Rewrite all of them with \u/\U escapes in the wide/u16/u32 literals
and \x escapes (split into separate string-literal tokens so a
following byte is never read as part of the same hex escape, e.g.
"\xE1\x83\x85" "a") in the narrow ones. Remove the
#ifndef __INTEL_COMPILER/#endif guard along with it.

The *_is_utf16()/*_is_utf32() probes compared a raw multibyte literal
against an escape-based one to detect a compiler that misreads the
source file's encoding; with no raw literals left to misread, the
comparison is now tautological, so drop the probes and the "if"
guards around each SECTION's body instead of leaving them in as dead
checks.

The same non-ASCII-in-source-and-in-a-narrow-comparison pattern
existed once more in unit-deserialization.cpp's "Using _json with
char8_t literals #4945" test: a raw emoji character in a u8R"(...)"
literal, guarded by a check_utf8() that returned false for ICC (same
reason) and for Windows without the active UTF-8 code page. Rewrite
the literal with a \U escape and compare it against a \x-escaped
expectation instead of a second raw literal, and drop check_utf8()
and the now-unused <windows.h> include along with the guard.

Verified locally (clang, -std=c++11 and -std=c++20,
-fsanitize=address,undefined, and a plain build): test-wstring keeps
18/18 assertions and unit-deserialization keeps 466/466 (c++11) and
477/477 (c++20) assertions, matching this branch before the change
exactly - no coverage was gained or lost, only the source-encoding
dependency was removed. ci_icpc has to confirm classic ICC actually
builds and passes test-wstring now; if it does not, that is a real
finding, not a reason to restore the guard.

Overlaps #5704 (open), which edits unit-wstring.cpp inside the
previously-guarded region (an include near the top, checks in the
invalid-string sections, and a new section at the end).

Closes #5713 item 5.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-30 18:30:20 +02:00
parent f07a17ce29
commit c28b76ca0c
2 changed files with 72 additions and 132 deletions
+9 -31
View File
@@ -28,11 +28,6 @@ using nlohmann::json;
#include <string>
#include <valarray>
#if defined(_WIN32)
#define NOMINMAX
#include <windows.h> // for GetACP()
#endif
namespace
{
struct SaxEventLogger : public nlohmann::json_sax<json>
@@ -228,24 +223,6 @@ class proxy_iterator
iterator* m_it = nullptr;
};
// JSON_HAS_CPP_20
#if defined(__cpp_char8_t)
bool check_utf8()
{
#if defined(_WIN32)
// Runtime check of the active ANSI code page
// 65001 == UTF-8
return GetACP() == 65001;
#elif defined(__ICC) || defined(__INTEL_COMPILER)
// classic Intel ICC does not encode narrow string literals containing
// non-ASCII source characters as UTF-8, so comparing a decoded u8 literal
// against a narrow string literal containing the same characters fails
return false;
#else
return true;
#endif
}
#endif
} // namespace
TEST_CASE("deserialization")
@@ -1297,14 +1274,15 @@ TEST_CASE("deserialization")
CHECK(j1["key"] == "value");
CHECK(j1["num"] == 42);
// UTF-8 prefixed literal (C++20 and later);
// MSVC may not set /utf-8, so we need to check
if (check_utf8())
{
const auto j2 = u8R"({"emoji": "😀", "msg": "hello"})"_json;
CHECK(j2["emoji"] == "😀");
CHECK(j2["msg"] == "hello");
}
// UTF-8 prefixed literal (C++20 and later); the emoji is written as a
// \U escape rather than a raw multibyte character so this does not
// depend on the compiler's source-file encoding (e.g., MSVC without
// /utf-8, or classic ICC, which does not encode non-ASCII narrow
// string literals as UTF-8 - compare against a \x-escaped expectation
// for the same reason)
const auto j2 = u8"{\"emoji\": \"\U0001F600\", \"msg\": \"hello\"}"_json;
CHECK(j2["emoji"] == "\xF0\x9F\x98\x80");
CHECK(j2["msg"] == "hello");
const auto j3 = u8R"({"key": "value", "num": 42})"_json;
CHECK(j3["key"] == "value");
+63 -101
View File
@@ -11,135 +11,97 @@
#include <nlohmann/json.hpp>
using nlohmann::json;
// ICPC errors out on multibyte character sequences in source files
#ifndef __INTEL_COMPILER
namespace
{
bool wstring_is_utf16();
bool wstring_is_utf16()
{
return (std::wstring(L"💩") == std::wstring(L"\U0001F4A9"));
}
bool u16string_is_utf16();
bool u16string_is_utf16()
{
return (std::u16string(u"💩") == std::u16string(u"\U0001F4A9"));
}
bool u32string_is_utf32();
bool u32string_is_utf32()
{
return (std::u32string(U"💩") == std::u32string(U"\U0001F4A9"));
}
} // namespace
TEST_CASE("wide strings")
{
SECTION("std::wstring")
{
if (wstring_is_utf16())
{
std::wstring const w = L"[12.2,\"Ⴥaäö💤🧢\"]";
json const j = json::parse(w);
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
}
// U+10C5 U+0061(a) U+00E4 U+00F6 U+1F4A4 U+1F9E2, written with \u/\U
// escapes rather than as raw multibyte characters so this file
// compiles on toolchains (e.g. classic ICC) that error out on
// multibyte character sequences in source files
std::wstring const w = L"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]";
json const j = json::parse(w);
CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]");
}
SECTION("invalid std::wstring")
{
if (wstring_is_utf16())
{
std::wstring const w = L"\"\xDBFF";
json _;
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
std::wstring const w = L"\"\xDBFF";
json _;
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// the exact message depends on the width of wchar_t: a 16-bit
// wchar_t passes the lone surrogate to the UTF-8 decoder unchanged
// (rejected as a single ill-formed byte at column 2), while a
// 32-bit wchar_t first encodes it as an ill-formed three-byte
// sequence (rejected one byte later, at column 3)
const char* const error_low_surrogate = sizeof(wchar_t) == 2
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'";
const char* const error_high_surrogate = sizeof(wchar_t) == 2
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'";
// the exact message depends on the width of wchar_t: a 16-bit
// wchar_t passes the lone surrogate to the UTF-8 decoder unchanged
// (rejected as a single ill-formed byte at column 2), while a
// 32-bit wchar_t first encodes it as an ill-formed three-byte
// sequence (rejected one byte later, at column 3)
const char* const error_low_surrogate = sizeof(wchar_t) == 2
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'";
const char* const error_high_surrogate = sizeof(wchar_t) == 2
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'";
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'"'}), error_low_surrogate, json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), static_cast<wchar_t>(0xE000), L'"'}), error_high_surrogate, json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&);
}
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'"'}), error_low_surrogate, json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), static_cast<wchar_t>(0xE000), L'"'}), error_high_surrogate, json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&);
}
SECTION("std::u16string")
{
if (u16string_is_utf16())
{
std::u16string const w = u"[12.2,\"Ⴥaäö💤🧢\"]";
json const j = json::parse(w);
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
}
std::u16string const w = u"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]";
json const j = json::parse(w);
CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]");
}
SECTION("invalid std::u16string")
{
if (u16string_is_utf16())
{
std::u16string const w = u"\"\xDBFF";
json _;
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
std::u16string const w = u"\"\xDBFF";
json _;
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a valid surrogate pair is still decoded (U+1F600)
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
}
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a valid surrogate pair is still decoded (U+1F600)
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
}
SECTION("std::u32string")
{
if (u32string_is_utf32())
{
std::u32string const w = U"[12.2,\"Ⴥaäö💤🧢\"]";
json const j = json::parse(w);
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
}
std::u32string const w = U"[12.2,\"\u10C5a\u00E4\u00F6\U0001F4A4\U0001F9E2\"]";
json const j = json::parse(w);
CHECK(j.dump() == "[12.2,\"" "\xE1\x83\x85" "a" "\xC3\xA4" "\xC3\xB6" "\xF0\x9F\x92\xA4" "\xF0\x9F\xA7\xA2" "\"]");
}
SECTION("invalid std::u32string")
{
if (u32string_is_utf32())
{
std::u32string const w = U"\"\x110000";
json _;
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
std::u32string const w = U"\"\x110000";
json _;
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// a code unit above U+10FFFF must not be narrowed onto the EOF
// sentinel: 0xFFFFFFFF would otherwise end the document silently and
// let everything following it pass the strict end-of-input check
std::u32string const trailing{U'[', U'1', U']', static_cast<char32_t>(0xFFFFFFFF), U'x'};
CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
CHECK(!json::accept(trailing));
// a code unit above U+10FFFF must not be narrowed onto the EOF
// sentinel: 0xFFFFFFFF would otherwise end the document silently and
// let everything following it pass the strict end-of-input check
std::u32string const trailing{U'[', U'1', U']', static_cast<char32_t>(0xFFFFFFFF), U'x'};
CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
CHECK(!json::accept(trailing));
// the same unit inside a string is reported as an ill-formed byte
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
}
// the same unit inside a string is reported as an ill-formed byte
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
}
}
#endif