diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index e174775c5..a67a9e428 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -445,8 +445,10 @@ struct wide_string_input_helper } else { - // get the current character - const auto wc = input.get_character(); + // get the current character; converted to an unsigned type so that + // a negative unit (wint_t is signed on some platforms) is not + // mistaken for an ASCII character or for EOF + const auto wc = static_cast(input.get_character()); // UTF-32 to UTF-8 encoding if (wc < 0x80) @@ -541,9 +543,11 @@ struct wide_string_input_helper bool valid_pair = false; if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty())) { - const auto wc2 = static_cast(input.get_character()); + // only consume the next unit if it completes the pair + const auto wc2 = static_cast(*input.current); if (0xDC00 <= wc2 && wc2 <= 0xDFFF) { + input.get_character(); const auto charcode = 0x10000u + (((static_cast(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); utf8_bytes[0] = static_cast::int_type>(0xF0u | (charcode >> 18u)); utf8_bytes[1] = static_cast::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu)); @@ -556,7 +560,8 @@ struct wide_string_input_helper if (!valid_pair) { - utf8_bytes[0] = static_cast::int_type>(wc); + // emit a byte that is never valid UTF-8 (see the UTF-32 case) + utf8_bytes[0] = 0xFF; utf8_bytes_filled = 1; } } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 576498738..cd818db6a 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7989,8 +7989,10 @@ struct wide_string_input_helper } else { - // get the current character - const auto wc = input.get_character(); + // get the current character; converted to an unsigned type so that + // a negative unit (wint_t is signed on some platforms) is not + // mistaken for an ASCII character or for EOF + const auto wc = static_cast(input.get_character()); // UTF-32 to UTF-8 encoding if (wc < 0x80) @@ -8085,9 +8087,11 @@ struct wide_string_input_helper bool valid_pair = false; if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty())) { - const auto wc2 = static_cast(input.get_character()); + // only consume the next unit if it completes the pair + const auto wc2 = static_cast(*input.current); if (0xDC00 <= wc2 && wc2 <= 0xDFFF) { + input.get_character(); const auto charcode = 0x10000u + (((static_cast(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); utf8_bytes[0] = static_cast::int_type>(0xF0u | (charcode >> 18u)); utf8_bytes[1] = static_cast::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu)); @@ -8100,7 +8104,8 @@ struct wide_string_input_helper if (!valid_pair) { - utf8_bytes[0] = static_cast::int_type>(wc); + // emit a byte that is never valid UTF-8 (see the UTF-32 case) + utf8_bytes[0] = 0xFF; utf8_bytes_filled = 1; } } diff --git a/tests/src/unit-wstring.cpp b/tests/src/unit-wstring.cpp index 6e0aff29e..281cae639 100644 --- a/tests/src/unit-wstring.cpp +++ b/tests/src/unit-wstring.cpp @@ -8,6 +8,7 @@ #include "doctest_compatibility.h" +#include #include using nlohmann::json; @@ -98,15 +99,15 @@ TEST_CASE("wide strings") CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&); // a lone low surrogate cannot start a pair - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // a high surrogate followed by a non-low-surrogate unit is invalid - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // ... also when the unit is above the low surrogates - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // a lone low surrogate must not swallow the following unit: pairing // it with any second unit would produce valid UTF-8, so the error // has to report an ill-formed byte at the surrogate's own position - CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); // a valid surrogate pair is still decoded (U+1F600) CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get() == "\xF0\x9F\x98\x80"); } @@ -141,5 +142,56 @@ TEST_CASE("wide strings") CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&); } } + + SECTION("malformed wide-string input outside strings (#5645)") + { + json _; + + // a lone low surrogate inside a literal must not be truncated to its + // low byte and mistaken for the letter the literal expects next + // (0xDC72 truncates to 'r', which is what "true" expects after 't') + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast(0xDC72), u'u', u'e'}), + "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&); + + // ... also when the lone surrogate is the last unit of the input + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'f', u'a', u'l', u's', static_cast(0xDD65)}), + "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'fals\xFF'", json::parse_error&); + + // a high surrogate followed by a unit that is not its low surrogate + // must not silently swallow that unit + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast(0xD872), u'X', u'u', u'e'}), + "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&); + + // ... in particular, if the swallowed unit is the newline that ends a + // // comment, the comment must not extend over the following line + CHECK(json::parse(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast(0xD800), u'\n', + u',', u'2', u' ', u'/', u'/', u'\n', u']'}, + nullptr, true, /*ignore_comments*/true) == json::parse("[1,2]")); + CHECK(json::accept(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast(0xD800), u'\n', + u',', u'2', u' ', u'/', u'/', u'\n', u']'}, /*ignore_comments*/true)); + + // cases 5 and 6 use a 32-bit wchar_t (Linux, macOS, the BSDs) to reach + // the UTF-32 helper tested above via u32string; the 16-bit wchar_t of + // Windows goes through the UTF-16 helper instead, already covered by + // the u16string cases above +#if WCHAR_MAX > 0xFFFFu + // a negative wchar_t must not be mistaken for + // char_traits::eof() and silently end the input, letting + // trailing garbage pass the strict end-of-input check (only observable + // where wint_t is signed, e.g. macOS/the BSDs; on Linux wint_t is + // unsigned and this was already handled by #5348) + std::wstring w = L"[1]"; + w.push_back(static_cast(-1)); + w += L"garbage"; + CHECK(!json::accept(w)); + CHECK_THROWS_WITH_AS(_ = json::parse(w), + "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&); + + // other negative wchar_t units must not be truncated to their low + // byte (0xFFFFFF72 truncates to 'r', as in the u16string case above) + CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L't', static_cast(0xFFFFFF72), L'u', L'e'}), + "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&); +#endif + } } #endif