mirror of
https://github.com/nlohmann/json.git
synced 2026-09-30 19:50:34 +00:00
fix: treat a NUL byte in the input as an ordinary byte, not EOF
The lexer's token dispatch had `case '\0':` fall through to the same
`end_of_input` handling as the real end-of-file sentinel, with a comment
claiming the NUL case was "needed when parsing from string literals".
That rationale no longer holds: input_adapter(const char*) already uses
strlen() to compute its range, so it never hands the lexer a trailing
NUL, and the const-char* overload is the only "string literal" path the
comment could be referring to. In practice, `case '\0':` only ever fired
on a genuine embedded or trailing NUL byte in real input data (e.g. a
std::string with '\0' appended), which was then silently swallowed as if
it were EOF instead of producing the parse_error.101 any other
unexpected byte gets. Two more spots in the comment-skipping logic had
the same NUL-as-EOF idiom, stopping a `//` or `/* */` comment scan early
at an embedded NUL instead of continuing to the real terminator.
Removing all three still left one real regression: input_adapter's
T(&array)[N] overload (used for a string literal like
json::parse("123"), as opposed to a decayed const char* pointer) passes
the array's full extent through unchanged, trailing '\0' included. That
path was relying on the lexer's old NUL-as-EOF behavior to make ordinary
literal parsing work at all. It now gets its own strlen()-like handling
for char arrays specifically: a single trailing NUL terminator is
excluded, mirroring the pointer overload, while non-char arrays (e.g.
uint8_t buffers for binary formats) are left untouched since a trailing
zero byte there may be data.
Also updates a few existing tests that (mostly incidentally) depended on
a trailing NUL being swallowed - std::array<uint8_t, 5>{"true"} left the
5th element zero-initialized - and adds an FAQ entry.
Signed-off-by: Niels Lohmann <niels.lohmann@gmail.com>
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01N4RQ1Ahan5YAGbnAQGjZTY
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
e5f84e1ebf
commit
633eef8494
@@ -7700,7 +7700,29 @@ contiguous_bytes_input_adapter input_adapter(CharT b)
|
||||
return input_adapter(ptr, ptr + length); // cppcheck-suppress[nullPointerArithmeticRedundantCheck]
|
||||
}
|
||||
|
||||
template<typename T, std::size_t N>
|
||||
// char arrays are usually string literals (e.g. json::parse("[1,2,3]")),
|
||||
// which the compiler pads with a trailing '\0' that is not part of the
|
||||
// text to parse; mirror the const char* overload above (which computes
|
||||
// its length with strlen()) and exclude a single trailing NUL terminator,
|
||||
// if present, so parsing a literal behaves the same whether the argument
|
||||
// decays to a pointer or binds directly to this array overload.
|
||||
template < typename T, std::size_t N,
|
||||
typename std::enable_if<std::is_same<typename std::remove_cv<T>::type, char>::value, int>::type = 0 >
|
||||
contiguous_bytes_input_adapter input_adapter(T (&array)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
|
||||
{
|
||||
std::size_t length = N;
|
||||
if (length > 0 && array[length - 1] == 0)
|
||||
{
|
||||
--length;
|
||||
}
|
||||
const auto* ptr = static_cast<const char*>(array);
|
||||
return input_adapter(ptr, ptr + length);
|
||||
}
|
||||
|
||||
// all other arrays (e.g. byte arrays used for binary formats) are passed
|
||||
// through unchanged, trailing zero byte included, since it may be data.
|
||||
template < typename T, std::size_t N,
|
||||
typename std::enable_if < !std::is_same<typename std::remove_cv<T>::type, char>::value, int >::type = 0 >
|
||||
auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
|
||||
{
|
||||
return input_adapter(array, array + N);
|
||||
@@ -8674,7 +8696,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
case '\n':
|
||||
case '\r':
|
||||
case char_traits<char_type>::eof():
|
||||
case '\0':
|
||||
return true;
|
||||
|
||||
default:
|
||||
@@ -8693,7 +8714,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
switch (get())
|
||||
{
|
||||
case char_traits<char_type>::eof():
|
||||
case '\0':
|
||||
{
|
||||
error_message = "invalid comment; missing closing '*/'";
|
||||
return false;
|
||||
@@ -9524,9 +9544,7 @@ scan_number_done:
|
||||
case '9':
|
||||
return scan_number();
|
||||
|
||||
// end of input (the null byte is needed when parsing from
|
||||
// string literals)
|
||||
case '\0':
|
||||
// end of input
|
||||
case char_traits<char_type>::eof():
|
||||
return token_type::end_of_input;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user