Compare commits

..
Author SHA1 Message Date
Niels Lohmann 6b43220a50 Reject malformed UTF-16/UTF-32 units in wide-string input
The wide-string input adapter (used for std::u16string, std::u32string,
std::wstring, and iterators over 2- or 4-byte character types) passed
some malformed code units on to the lexer as values that are neither a
byte (0x00..0xFF) nor char_traits<char>::eof(). As a result:

- A lone UTF-16 surrogate inside true/false/null was accepted if its low
  byte matched the expected letter, or ended the input silently if it
  was the last unit.
- A high surrogate followed by a unit that is not its low surrogate
  swallowed that unit; if the swallowed unit was the newline ending a
  // comment, the comment silently extended over the next line.
- Where wint_t is a signed int (macOS, the BSDs), a negative wchar_t
  collided with char_traits<char>::eof() (ending the input early) or was
  truncated to its low byte, depending on its value.

The UTF-32 helper now converts the code unit to std::uint32_t before the
range checks, so a negative unit reaches the same "emit 0xFF" branch
already used for code points above U+10FFFF. The UTF-16 helper now
peeks at the next unit before consuming it, and emits 0xFF instead of
the raw surrogate when no valid pair is found, matching how ill-formed
UTF-8 bytes are rejected elsewhere in the lexer.

Fixes #5645.
Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 23:45:24 +02:00
5 changed files with 76 additions and 162 deletions
@@ -445,8 +445,10 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
}
else
{
// get the current character
const auto wc = input.get_character();
// get the current character; converted to an unsigned type so that
// a negative unit (wint_t is signed on some platforms) is not
// mistaken for an ASCII character or for EOF
const auto wc = static_cast<std::uint32_t>(input.get_character());
// UTF-32 to UTF-8 encoding
if (wc < 0x80)
@@ -541,9 +543,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
bool valid_pair = false;
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
{
const auto wc2 = static_cast<unsigned int>(input.get_character());
// only consume the next unit if it completes the pair
const auto wc2 = static_cast<unsigned int>(*input.current);
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{
input.get_character();
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
@@ -556,7 +560,8 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (!valid_pair)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
utf8_bytes[0] = 0xFF;
utf8_bytes_filled = 1;
}
}
+1 -51
View File
@@ -1307,65 +1307,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
*/
template<bool Ordered>
static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept
{
return compare_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
}
/// @brief compare two leaves that are only being checked for equality
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::false_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::false_type {});
return order_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
}
#if JSON_HAS_THREE_WAY_COMPARISON
/*!
@brief compare two leaves that are being ordered, for operator<=>
Reached only from operator<=>, so the leaves must be classified exactly
as operator<=> classifies them - which is not the same as asking
== and then order_leaves(), the way the other overload does it. The two
disagree on a binary value: == also compares the subtype, but <=> compares
only the bytes, through std::vector<std::uint8_t>::operator<=>. Using <=>
itself here keeps a leaf pair classified the same way regardless of how
deep it is nested - == first would again call operator<=> a level down
through order_leaves(), but call it after a mismatching == already ended
the comparison for a pair that <=> alone would still call equivalent.
*/
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
const std::partial_ordering order = lhs <=> rhs; // *NOPAD*
if (order == 0)
{
return compare_result::equal;
}
if (order < 0)
{
return compare_result::less;
}
if (order > 0)
{
return compare_result::greater;
}
return compare_result::unordered;
}
#else
/// @brief compare two leaves that are being ordered, for operator<
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::true_type {});
}
#endif
/*!
@brief compare two object keys
+10 -55
View File
@@ -7989,8 +7989,10 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
}
else
{
// get the current character
const auto wc = input.get_character();
// get the current character; converted to an unsigned type so that
// a negative unit (wint_t is signed on some platforms) is not
// mistaken for an ASCII character or for EOF
const auto wc = static_cast<std::uint32_t>(input.get_character());
// UTF-32 to UTF-8 encoding
if (wc < 0x80)
@@ -8085,9 +8087,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
bool valid_pair = false;
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
{
const auto wc2 = static_cast<unsigned int>(input.get_character());
// only consume the next unit if it completes the pair
const auto wc2 = static_cast<unsigned int>(*input.current);
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{
input.get_character();
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
@@ -8100,7 +8104,8 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (!valid_pair)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
utf8_bytes[0] = 0xFF;
utf8_bytes_filled = 1;
}
}
@@ -27388,65 +27393,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
*/
template<bool Ordered>
static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept
{
return compare_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
}
/// @brief compare two leaves that are only being checked for equality
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::false_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::false_type {});
return order_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
}
#if JSON_HAS_THREE_WAY_COMPARISON
/*!
@brief compare two leaves that are being ordered, for operator<=>
Reached only from operator<=>, so the leaves must be classified exactly
as operator<=> classifies them - which is not the same as asking
== and then order_leaves(), the way the other overload does it. The two
disagree on a binary value: == also compares the subtype, but <=> compares
only the bytes, through std::vector<std::uint8_t>::operator<=>. Using <=>
itself here keeps a leaf pair classified the same way regardless of how
deep it is nested - == first would again call operator<=> a level down
through order_leaves(), but call it after a mismatching == already ended
the comparison for a pair that <=> alone would still call equivalent.
*/
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
const std::partial_ordering order = lhs <=> rhs; // *NOPAD*
if (order == 0)
{
return compare_result::equal;
}
if (order < 0)
{
return compare_result::less;
}
if (order > 0)
{
return compare_result::greater;
}
return compare_result::unordered;
}
#else
/// @brief compare two leaves that are being ordered, for operator<
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::true_type {});
}
#endif
/*!
@brief compare two object keys
-48
View File
@@ -952,51 +952,3 @@ TEST_CASE("containers are compared element by element")
}
}
}
#if JSON_HAS_THREE_WAY_COMPARISON
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
TEST_CASE("operator<=> of binary values with a different subtype does not depend on nesting depth")
{
// #5654: std::vector<std::uint8_t>::operator<=>, which the binary type's
// own operator<=> uses, ignores the subtype that operator== checks. So a
// pair of binary values with the same bytes but a different subtype is
// unequal, yet <=>-equivalent - the same inconsistency between == and <=>
// that a NaN has. Within the nesting bound, an array compares itself
// with std::vector's own operator<=>, which treats an equivalent pair as
// undecided and lets the next element decide, same as
// std::lexicographical_compare_three_way does. Past the bound,
// compare_iteratively<true>() takes over and must classify the pair the
// same way, or the result of operator<=> - and of <, which C++20 derives
// from it - depends on how deeply the values are nested.
const json a = json::array({json::binary({1}, 1), 1});
const json b = json::array({json::binary({1}, 2), 2});
// the root inconsistency: unequal, yet <=>-equivalent
CHECK_FALSE(a[0] == b[0]);
CHECK((a[0] <=> b[0]) == std::partial_ordering::equivalent); // *NOPAD*
const auto deep = [](const json & j, const std::size_t depth)
{
json result = j;
for (std::size_t i = 0; i < depth; ++i)
{
result = json::array({std::move(result)});
}
return result;
};
// 127 levels stay within nesting_depth_limit() (128); 128 and 200 do not,
// and must still agree with the levels that do
for (const std::size_t depth : std::vector<std::size_t> {0, 127, 128, 200})
{
CAPTURE(depth);
const json x = deep(a, depth);
const json y = deep(b, depth);
CHECK((x <=> y) == std::partial_ordering::less); // *NOPAD*
CHECK((y <=> x) == std::partial_ordering::greater); // *NOPAD*
CHECK(x < y);
CHECK(y > x);
CHECK_FALSE(y < x);
}
}
#endif
+56 -4
View File
@@ -8,6 +8,7 @@
#include "doctest_compatibility.h"
#include <cwchar>
#include <nlohmann/json.hpp>
using nlohmann::json;
@@ -98,15 +99,15 @@ TEST_CASE("wide strings")
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// a valid surrogate pair is still decoded (U+1F600)
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
}
@@ -141,5 +142,56 @@ TEST_CASE("wide strings")
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
}
}
SECTION("malformed wide-string input outside strings (#5645)")
{
json _;
// a lone low surrogate inside a literal must not be truncated to its
// low byte and mistaken for the letter the literal expects next
// (0xDC72 truncates to 'r', which is what "true" expects after 't')
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xDC72), u'u', u'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
// ... also when the lone surrogate is the last unit of the input
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'f', u'a', u'l', u's', static_cast<char16_t>(0xDD65)}),
"[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'fals\xFF'", json::parse_error&);
// a high surrogate followed by a unit that is not its low surrogate
// must not silently swallow that unit
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xD872), u'X', u'u', u'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
// ... in particular, if the swallowed unit is the newline that ends a
// // comment, the comment must not extend over the following line
CHECK(json::parse(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
u',', u'2', u' ', u'/', u'/', u'\n', u']'},
nullptr, true, /*ignore_comments*/true) == json::parse("[1,2]"));
CHECK(json::accept(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
u',', u'2', u' ', u'/', u'/', u'\n', u']'}, /*ignore_comments*/true));
// cases 5 and 6 use a 32-bit wchar_t (Linux, macOS, the BSDs) to reach
// the UTF-32 helper tested above via u32string; the 16-bit wchar_t of
// Windows goes through the UTF-16 helper instead, already covered by
// the u16string cases above
#if WCHAR_MAX > 0xFFFFu
// a negative wchar_t must not be mistaken for
// char_traits<char>::eof() and silently end the input, letting
// trailing garbage pass the strict end-of-input check (only observable
// where wint_t is signed, e.g. macOS/the BSDs; on Linux wint_t is
// unsigned and this was already handled by #5348)
std::wstring w = L"[1]";
w.push_back(static_cast<wchar_t>(-1));
w += L"garbage";
CHECK(!json::accept(w));
CHECK_THROWS_WITH_AS(_ = json::parse(w),
"[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
// other negative wchar_t units must not be truncated to their low
// byte (0xFFFFFF72 truncates to 'r', as in the u16string case above)
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L't', static_cast<wchar_t>(0xFFFFFF72), L'u', L'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
#endif
}
}
#endif