From d17d9c3440c80fdbc7a08580c0423a6e05a7d591 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 29 Sep 2026 23:42:11 +0200 Subject: [PATCH] Trim the compiler-appended NUL from wide/UTF string literals too With JSON_STRICT_NUL_HANDLING defined to 1, parsing a wide, UTF-16, UTF-32, or (C++20) UTF-8 string literal (e.g. json::parse(L"[1]")) failed with parse_error.101 at the terminating NUL of the literal, and accept() returned false. The array overload of input_adapter() only dropped the compiler-added trailing '\0' for arrays of char, so for wchar_t, char16_t, char32_t, and char8_t arrays that terminator was passed to the parser as data, which the macro then rejected. Broaden the trimming to every character type that a string literal can use (char, wchar_t, char16_t, char32_t, and, since C++20, char8_t). Arrays of any other element type (unsigned char, std::uint8_t, ...), as used for CBOR/MessagePack, are unaffected: a trailing zero byte there is still read as genuine data. Fixes #5658. Signed-off-by: Niels Lohmann --- .../api/macros/json_strict_nul_handling.md | 19 ++++++----- .../nlohmann/detail/input/input_adapters.hpp | 32 +++++++++++++------ single_include/nlohmann/json.hpp | 32 +++++++++++++------ tests/src/unit-class_parser.cpp | 24 ++++++++++++++ 4 files changed, 79 insertions(+), 28 deletions(-) diff --git a/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md index f22395d3b..3bad2dea1 100644 --- a/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md +++ b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md @@ -16,7 +16,8 @@ byte is still not rejected: [`from_msgpack`](../basic_json/from_msgpack.md), [`from_ubjson`](../basic_json/from_ubjson.md)) are never affected: there, `0x00` is ordinary data. - A bare `const char*` pointer has no length of its own, so its length is still determined with `strlen()`. The first NUL byte therefore still marks the end of the input, and nothing after it is read. -- One trailing `'\0'` at the end of a `char` array (e.g., a string literal) is trimmed; see the warning below. +- One trailing `'\0'` at the end of a `char`, `wchar_t`, `char16_t`, `char32_t`, or (C++20) `char8_t` array (e.g., a + string literal) is trimmed; see the warning below. ## Default definition @@ -57,13 +58,15 @@ The default value is `0` (disabled — existing behavior is preserved). This macro must be defined **before** including ``. Defining it after the include has no effect. - Enabling it also changes how a `char` array (including a string literal, e.g. `json::parse("123")`) is read: such - an array normally carries a trailing `'\0'` contributed by the compiler, not by the source text. With this macro - enabled, that one trailing byte is trimmed if present so that parsing a string literal keeps working; every other - byte in the array - including any `'\0'` that is not the very last element - is read as real data and rejected - like any other unexpected byte. Arrays of any other element type (`unsigned char`, `std::uint8_t`, ...), as used - for CBOR or MessagePack, are never affected by this trimming; their full extent - including a genuine trailing - `0x00` - is always preserved, in both states of this macro. + Enabling it also changes how an array of a text-literal element type (`char`, `wchar_t`, `char16_t`, `char32_t`, + or, since C++20, `char8_t` - including a string literal, e.g. `json::parse("123")` or `json::parse(L"123")`) is + read: such an array normally carries a trailing `'\0'` contributed by the compiler, not by the source text. With + this macro enabled, that one trailing element is trimmed if present so that parsing a string literal keeps + working, for any of these character types; every other element in the array - including any `'\0'` that is not + the very last element - is read as real data and rejected like any other unexpected byte. Arrays of any other + element type (`unsigned char`, `std::uint8_t`, ...), as used for CBOR or MessagePack, are never affected by this + trimming; their full extent - including a genuine trailing `0x00` - is always preserved, in both states of this + macro. !!! note "ABI compatibility" diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index e174775c5..b92c2e556 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -845,16 +845,28 @@ template auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { #if JSON_STRICT_NUL_HANDLING - // A `char` array from string-literal initialization (e.g. json::parse("123")) - // carries a trailing '\0' contributed by the compiler, not by the source - // text; drop exactly that one byte so it is not mistaken for real trailing - // data. Every other element type (unsigned char, std::uint8_t, ...) keeps - // the full extent unconditionally, since a trailing zero byte there is - // genuine data (e.g. CBOR/MessagePack). This intentionally does not - // strlen()-scan the array (as the pointer overload above does for a - // null-delimited string): for a `char` array that is not NUL-terminated - // within its bounds, that would read past the end of the array. - if (std::is_same::type, char>::value && N > 0 && array[N - 1] == 0) + // A text-literal array from string-literal initialization (e.g. + // json::parse("123") or json::parse(L"123")) carries a trailing '\0' + // contributed by the compiler, not by the source text; drop exactly that + // one byte so it is not mistaken for real trailing data. This covers all + // character types that string literals can use: char, wchar_t, char16_t, + // char32_t, and (C++20) char8_t. Every other element type (unsigned char, + // std::uint8_t, ...) keeps the full extent unconditionally, since a + // trailing zero byte there is genuine data (e.g. CBOR/MessagePack). This + // intentionally does not strlen()-scan the array (as the pointer overload + // above does for a null-delimited string): for an array that is not + // NUL-terminated within its bounds, that would read past the end of the + // array. + using char_t = typename std::remove_cv::type; + constexpr bool is_text_literal_type = std::is_same::value + || std::is_same::value + || std::is_same::value + || std::is_same::value +#if defined(__cpp_char8_t) + || std::is_same::value +#endif + ; + if (is_text_literal_type && N > 0 && array[N - 1] == 0) { return input_adapter(array, array + N - 1); } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 576498738..59db1c010 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8389,16 +8389,28 @@ template auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { #if JSON_STRICT_NUL_HANDLING - // A `char` array from string-literal initialization (e.g. json::parse("123")) - // carries a trailing '\0' contributed by the compiler, not by the source - // text; drop exactly that one byte so it is not mistaken for real trailing - // data. Every other element type (unsigned char, std::uint8_t, ...) keeps - // the full extent unconditionally, since a trailing zero byte there is - // genuine data (e.g. CBOR/MessagePack). This intentionally does not - // strlen()-scan the array (as the pointer overload above does for a - // null-delimited string): for a `char` array that is not NUL-terminated - // within its bounds, that would read past the end of the array. - if (std::is_same::type, char>::value && N > 0 && array[N - 1] == 0) + // A text-literal array from string-literal initialization (e.g. + // json::parse("123") or json::parse(L"123")) carries a trailing '\0' + // contributed by the compiler, not by the source text; drop exactly that + // one byte so it is not mistaken for real trailing data. This covers all + // character types that string literals can use: char, wchar_t, char16_t, + // char32_t, and (C++20) char8_t. Every other element type (unsigned char, + // std::uint8_t, ...) keeps the full extent unconditionally, since a + // trailing zero byte there is genuine data (e.g. CBOR/MessagePack). This + // intentionally does not strlen()-scan the array (as the pointer overload + // above does for a null-delimited string): for an array that is not + // NUL-terminated within its bounds, that would read past the end of the + // array. + using char_t = typename std::remove_cv::type; + constexpr bool is_text_literal_type = std::is_same::value + || std::is_same::value + || std::is_same::value + || std::is_same::value +#if defined(__cpp_char8_t) + || std::is_same::value +#endif + ; + if (is_text_literal_type && N > 0 && array[N - 1] == 0) { return input_adapter(array, array + N - 1); } diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index 3d825162e..52b49c5d9 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -645,6 +645,30 @@ TEST_CASE("parser class") // carries a compiler-appended trailing '\0') still works, // even though a NUL byte is now rejected everywhere else CHECK(json::parse("123") == json(123)); + + // regression test for issue #5658: the same holds for wide, + // UTF-16, UTF-32, and (C++20) UTF-8 string literals, whose + // compiler-appended trailing '\0' is not of type `char` + CHECK(json::parse(L"[1]") == json({1})); + CHECK(json::accept(L"[1]")); + CHECK(json::parse(u"[1]") == json({1})); + CHECK(json::accept(u"[1]")); + CHECK(json::parse(U"[1]") == json({1})); + CHECK(json::accept(U"[1]")); +#if defined(__cpp_char8_t) + CHECK(json::parse(u8"[1]") == json({1})); + CHECK(json::accept(u8"[1]")); +#endif + + // a NUL byte inside such a literal, as opposed to the single + // compiler-appended trailing one, is still rejected + { + json _; // NOLINT(readability-identifier-naming) + CHECK_THROWS_WITH_AS(_ = json::parse(L"[1\0]"), + "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing array - invalid literal; last read: '1'; expected ']'", + json::parse_error&); + CHECK_FALSE(json::accept(L"[1\0]")); + } } #endif }