mirror of
https://github.com/nlohmann/json.git
synced 2026-09-30 19:50:34 +00:00
Trim the compiler-appended NUL from wide/UTF string literals too
With JSON_STRICT_NUL_HANDLING defined to 1, parsing a wide, UTF-16, UTF-32, or (C++20) UTF-8 string literal (e.g. json::parse(L"[1]")) failed with parse_error.101 at the terminating NUL of the literal, and accept() returned false. The array overload of input_adapter() only dropped the compiler-added trailing '\0' for arrays of char, so for wchar_t, char16_t, char32_t, and char8_t arrays that terminator was passed to the parser as data, which the macro then rejected. Broaden the trimming to every character type that a string literal can use (char, wchar_t, char16_t, char32_t, and, since C++20, char8_t). Arrays of any other element type (unsigned char, std::uint8_t, ...), as used for CBOR/MessagePack, are unaffected: a trailing zero byte there is still read as genuine data. Fixes #5658. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -16,7 +16,8 @@ byte is still not rejected:
|
||||
[`from_msgpack`](../basic_json/from_msgpack.md), [`from_ubjson`](../basic_json/from_ubjson.md)) are never affected: there, `0x00` is ordinary data.
|
||||
- A bare `const char*` pointer has no length of its own, so its length is still determined with `strlen()`. The first
|
||||
NUL byte therefore still marks the end of the input, and nothing after it is read.
|
||||
- One trailing `'\0'` at the end of a `char` array (e.g., a string literal) is trimmed; see the warning below.
|
||||
- One trailing `'\0'` at the end of a `char`, `wchar_t`, `char16_t`, `char32_t`, or (C++20) `char8_t` array (e.g., a
|
||||
string literal) is trimmed; see the warning below.
|
||||
|
||||
## Default definition
|
||||
|
||||
@@ -57,13 +58,15 @@ The default value is `0` (disabled — existing behavior is preserved).
|
||||
This macro must be defined **before** including `<nlohmann/json.hpp>`. Defining it after the include has no
|
||||
effect.
|
||||
|
||||
Enabling it also changes how a `char` array (including a string literal, e.g. `json::parse("123")`) is read: such
|
||||
an array normally carries a trailing `'\0'` contributed by the compiler, not by the source text. With this macro
|
||||
enabled, that one trailing byte is trimmed if present so that parsing a string literal keeps working; every other
|
||||
byte in the array - including any `'\0'` that is not the very last element - is read as real data and rejected
|
||||
like any other unexpected byte. Arrays of any other element type (`unsigned char`, `std::uint8_t`, ...), as used
|
||||
for CBOR or MessagePack, are never affected by this trimming; their full extent - including a genuine trailing
|
||||
`0x00` - is always preserved, in both states of this macro.
|
||||
Enabling it also changes how an array of a text-literal element type (`char`, `wchar_t`, `char16_t`, `char32_t`,
|
||||
or, since C++20, `char8_t` - including a string literal, e.g. `json::parse("123")` or `json::parse(L"123")`) is
|
||||
read: such an array normally carries a trailing `'\0'` contributed by the compiler, not by the source text. With
|
||||
this macro enabled, that one trailing element is trimmed if present so that parsing a string literal keeps
|
||||
working, for any of these character types; every other element in the array - including any `'\0'` that is not
|
||||
the very last element - is read as real data and rejected like any other unexpected byte. Arrays of any other
|
||||
element type (`unsigned char`, `std::uint8_t`, ...), as used for CBOR or MessagePack, are never affected by this
|
||||
trimming; their full extent - including a genuine trailing `0x00` - is always preserved, in both states of this
|
||||
macro.
|
||||
|
||||
!!! note "ABI compatibility"
|
||||
|
||||
|
||||
@@ -845,16 +845,28 @@ template<typename T, std::size_t N>
|
||||
auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
|
||||
{
|
||||
#if JSON_STRICT_NUL_HANDLING
|
||||
// A `char` array from string-literal initialization (e.g. json::parse("123"))
|
||||
// carries a trailing '\0' contributed by the compiler, not by the source
|
||||
// text; drop exactly that one byte so it is not mistaken for real trailing
|
||||
// data. Every other element type (unsigned char, std::uint8_t, ...) keeps
|
||||
// the full extent unconditionally, since a trailing zero byte there is
|
||||
// genuine data (e.g. CBOR/MessagePack). This intentionally does not
|
||||
// strlen()-scan the array (as the pointer overload above does for a
|
||||
// null-delimited string): for a `char` array that is not NUL-terminated
|
||||
// within its bounds, that would read past the end of the array.
|
||||
if (std::is_same<typename std::remove_cv<T>::type, char>::value && N > 0 && array[N - 1] == 0)
|
||||
// A text-literal array from string-literal initialization (e.g.
|
||||
// json::parse("123") or json::parse(L"123")) carries a trailing '\0'
|
||||
// contributed by the compiler, not by the source text; drop exactly that
|
||||
// one byte so it is not mistaken for real trailing data. This covers all
|
||||
// character types that string literals can use: char, wchar_t, char16_t,
|
||||
// char32_t, and (C++20) char8_t. Every other element type (unsigned char,
|
||||
// std::uint8_t, ...) keeps the full extent unconditionally, since a
|
||||
// trailing zero byte there is genuine data (e.g. CBOR/MessagePack). This
|
||||
// intentionally does not strlen()-scan the array (as the pointer overload
|
||||
// above does for a null-delimited string): for an array that is not
|
||||
// NUL-terminated within its bounds, that would read past the end of the
|
||||
// array.
|
||||
using char_t = typename std::remove_cv<T>::type;
|
||||
constexpr bool is_text_literal_type = std::is_same<char_t, char>::value
|
||||
|| std::is_same<char_t, wchar_t>::value
|
||||
|| std::is_same<char_t, char16_t>::value
|
||||
|| std::is_same<char_t, char32_t>::value
|
||||
#if defined(__cpp_char8_t)
|
||||
|| std::is_same<char_t, char8_t>::value
|
||||
#endif
|
||||
;
|
||||
if (is_text_literal_type && N > 0 && array[N - 1] == 0)
|
||||
{
|
||||
return input_adapter(array, array + N - 1);
|
||||
}
|
||||
|
||||
@@ -8389,16 +8389,28 @@ template<typename T, std::size_t N>
|
||||
auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
|
||||
{
|
||||
#if JSON_STRICT_NUL_HANDLING
|
||||
// A `char` array from string-literal initialization (e.g. json::parse("123"))
|
||||
// carries a trailing '\0' contributed by the compiler, not by the source
|
||||
// text; drop exactly that one byte so it is not mistaken for real trailing
|
||||
// data. Every other element type (unsigned char, std::uint8_t, ...) keeps
|
||||
// the full extent unconditionally, since a trailing zero byte there is
|
||||
// genuine data (e.g. CBOR/MessagePack). This intentionally does not
|
||||
// strlen()-scan the array (as the pointer overload above does for a
|
||||
// null-delimited string): for a `char` array that is not NUL-terminated
|
||||
// within its bounds, that would read past the end of the array.
|
||||
if (std::is_same<typename std::remove_cv<T>::type, char>::value && N > 0 && array[N - 1] == 0)
|
||||
// A text-literal array from string-literal initialization (e.g.
|
||||
// json::parse("123") or json::parse(L"123")) carries a trailing '\0'
|
||||
// contributed by the compiler, not by the source text; drop exactly that
|
||||
// one byte so it is not mistaken for real trailing data. This covers all
|
||||
// character types that string literals can use: char, wchar_t, char16_t,
|
||||
// char32_t, and (C++20) char8_t. Every other element type (unsigned char,
|
||||
// std::uint8_t, ...) keeps the full extent unconditionally, since a
|
||||
// trailing zero byte there is genuine data (e.g. CBOR/MessagePack). This
|
||||
// intentionally does not strlen()-scan the array (as the pointer overload
|
||||
// above does for a null-delimited string): for an array that is not
|
||||
// NUL-terminated within its bounds, that would read past the end of the
|
||||
// array.
|
||||
using char_t = typename std::remove_cv<T>::type;
|
||||
constexpr bool is_text_literal_type = std::is_same<char_t, char>::value
|
||||
|| std::is_same<char_t, wchar_t>::value
|
||||
|| std::is_same<char_t, char16_t>::value
|
||||
|| std::is_same<char_t, char32_t>::value
|
||||
#if defined(__cpp_char8_t)
|
||||
|| std::is_same<char_t, char8_t>::value
|
||||
#endif
|
||||
;
|
||||
if (is_text_literal_type && N > 0 && array[N - 1] == 0)
|
||||
{
|
||||
return input_adapter(array, array + N - 1);
|
||||
}
|
||||
|
||||
@@ -645,6 +645,30 @@ TEST_CASE("parser class")
|
||||
// carries a compiler-appended trailing '\0') still works,
|
||||
// even though a NUL byte is now rejected everywhere else
|
||||
CHECK(json::parse("123") == json(123));
|
||||
|
||||
// regression test for issue #5658: the same holds for wide,
|
||||
// UTF-16, UTF-32, and (C++20) UTF-8 string literals, whose
|
||||
// compiler-appended trailing '\0' is not of type `char`
|
||||
CHECK(json::parse(L"[1]") == json({1}));
|
||||
CHECK(json::accept(L"[1]"));
|
||||
CHECK(json::parse(u"[1]") == json({1}));
|
||||
CHECK(json::accept(u"[1]"));
|
||||
CHECK(json::parse(U"[1]") == json({1}));
|
||||
CHECK(json::accept(U"[1]"));
|
||||
#if defined(__cpp_char8_t)
|
||||
CHECK(json::parse(u8"[1]") == json({1}));
|
||||
CHECK(json::accept(u8"[1]"));
|
||||
#endif
|
||||
|
||||
// a NUL byte inside such a literal, as opposed to the single
|
||||
// compiler-appended trailing one, is still rejected
|
||||
{
|
||||
json _; // NOLINT(readability-identifier-naming)
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(L"[1\0]"),
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing array - invalid literal; last read: '1<U+0000>'; expected ']'",
|
||||
json::parse_error&);
|
||||
CHECK_FALSE(json::accept(L"[1\0]"));
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user