mirror of
https://github.com/nlohmann/json.git
synced 2026-10-02 20:50:32 +00:00
Trim the compiler-appended NUL from wide/UTF string literals too (#5702)
With JSON_STRICT_NUL_HANDLING defined to 1, parsing a wide, UTF-16, UTF-32, or (C++20) UTF-8 string literal (e.g. json::parse(L"[1]")) failed with parse_error.101 at the terminating NUL of the literal, and accept() returned false. The array overload of input_adapter() only dropped the compiler-added trailing '\0' for arrays of char, so for wchar_t, char16_t, char32_t, and char8_t arrays that terminator was passed to the parser as data, which the macro then rejected. Broaden the trimming to every character type that a string literal can use (char, wchar_t, char16_t, char32_t, and, since C++20, char8_t). Arrays of any other element type (unsigned char, std::uint8_t, ...), as used for CBOR/MessagePack, are unaffected: a trailing zero byte there is still read as genuine data. Fixes #5658. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -855,16 +855,28 @@ template<typename T, std::size_t N>
|
||||
auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
|
||||
{
|
||||
#if JSON_STRICT_NUL_HANDLING
|
||||
// A `char` array from string-literal initialization (e.g. json::parse("123"))
|
||||
// carries a trailing '\0' contributed by the compiler, not by the source
|
||||
// text; drop exactly that one byte so it is not mistaken for real trailing
|
||||
// data. Every other element type (unsigned char, std::uint8_t, ...) keeps
|
||||
// the full extent unconditionally, since a trailing zero byte there is
|
||||
// genuine data (e.g. CBOR/MessagePack). This intentionally does not
|
||||
// strlen()-scan the array (as the pointer overload above does for a
|
||||
// null-delimited string): for a `char` array that is not NUL-terminated
|
||||
// within its bounds, that would read past the end of the array.
|
||||
if (std::is_same<typename std::remove_cv<T>::type, char>::value && N > 0 && array[N - 1] == 0)
|
||||
// A text-literal array from string-literal initialization (e.g.
|
||||
// json::parse("123") or json::parse(L"123")) carries a trailing '\0'
|
||||
// contributed by the compiler, not by the source text; drop exactly that
|
||||
// one byte so it is not mistaken for real trailing data. This covers all
|
||||
// character types that string literals can use: char, wchar_t, char16_t,
|
||||
// char32_t, and (C++20) char8_t. Every other element type (unsigned char,
|
||||
// std::uint8_t, ...) keeps the full extent unconditionally, since a
|
||||
// trailing zero byte there is genuine data (e.g. CBOR/MessagePack). This
|
||||
// intentionally does not strlen()-scan the array (as the pointer overload
|
||||
// above does for a null-delimited string): for an array that is not
|
||||
// NUL-terminated within its bounds, that would read past the end of the
|
||||
// array.
|
||||
using char_t = typename std::remove_cv<T>::type;
|
||||
constexpr bool is_text_literal_type = std::is_same<char_t, char>::value
|
||||
|| std::is_same<char_t, wchar_t>::value
|
||||
|| std::is_same<char_t, char16_t>::value
|
||||
|| std::is_same<char_t, char32_t>::value
|
||||
#if defined(__cpp_char8_t)
|
||||
|| std::is_same<char_t, char8_t>::value
|
||||
#endif
|
||||
;
|
||||
if (is_text_literal_type && N > 0 && array[N - 1] == 0)
|
||||
{
|
||||
return input_adapter(array, array + N - 1);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user