mirror of
https://github.com/nlohmann/json.git
synced 2026-10-03 05:00:30 +00:00
Reject malformed UTF-16/UTF-32 units in wide-string input
The wide-string input adapter (used for std::u16string, std::u32string, std::wstring, and iterators over 2- or 4-byte character types) passed some malformed code units on to the lexer as values that are neither a byte (0x00..0xFF) nor char_traits<char>::eof(). As a result: - A lone UTF-16 surrogate inside true/false/null was accepted if its low byte matched the expected letter, or ended the input silently if it was the last unit. - A high surrogate followed by a unit that is not its low surrogate swallowed that unit; if the swallowed unit was the newline ending a // comment, the comment silently extended over the next line. - Where wint_t is a signed int (macOS, the BSDs), a negative wchar_t collided with char_traits<char>::eof() (ending the input early) or was truncated to its low byte, depending on its value. The UTF-32 helper now converts the code unit to std::uint32_t before the range checks, so a negative unit reaches the same "emit 0xFF" branch already used for code points above U+10FFFF. The UTF-16 helper now peeks at the next unit before consuming it, and emits 0xFF instead of the raw surrogate when no valid pair is found, matching how ill-formed UTF-8 bytes are rejected elsewhere in the lexer. Fixes #5645. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -445,8 +445,10 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
|
||||
}
|
||||
else
|
||||
{
|
||||
// get the current character
|
||||
const auto wc = input.get_character();
|
||||
// get the current character; converted to an unsigned type so that
|
||||
// a negative unit (wint_t is signed on some platforms) is not
|
||||
// mistaken for an ASCII character or for EOF
|
||||
const auto wc = static_cast<std::uint32_t>(input.get_character());
|
||||
|
||||
// UTF-32 to UTF-8 encoding
|
||||
if (wc < 0x80)
|
||||
@@ -541,9 +543,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
|
||||
bool valid_pair = false;
|
||||
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
|
||||
{
|
||||
const auto wc2 = static_cast<unsigned int>(input.get_character());
|
||||
// only consume the next unit if it completes the pair
|
||||
const auto wc2 = static_cast<unsigned int>(*input.current);
|
||||
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
|
||||
{
|
||||
input.get_character();
|
||||
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
|
||||
@@ -556,7 +560,8 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
|
||||
|
||||
if (!valid_pair)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
||||
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
|
||||
utf8_bytes[0] = 0xFF;
|
||||
utf8_bytes_filled = 1;
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user