From 2add64396ab43f2dc3b0a978f3a6c6dfdade1fc0 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 29 Sep 2026 23:38:59 +0200 Subject: [PATCH] Keep a NUL byte ending a // comment as the end of input With the default NUL handling (JSON_STRICT_NUL_HANDLING not set), a NUL byte in the input is treated as the real end of input everywhere - except when it immediately ends a `//` comment: scan_comment() matched '\0' as a comment terminator like '\n', so the NUL was consumed as part of the comment and scan() never saw it as end of input; the next get() then kept reading past it. Multi-line comments and JSON_STRICT_NUL_HANDLING=1 were unaffected, since there the NUL is just part of the comment text. Fix scan_comment() to leave the NUL unconsumed (unget()) instead of returning it as part of the comment, so the following scan() reports it as end of input, exactly as for a NUL anywhere else. Fixes #5659. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/input/lexer.hpp | 7 ++++- single_include/nlohmann/json.hpp | 7 ++++- tests/src/unit-class_parser.cpp | 39 +++++++++++++++++++++++++ 3 files changed, 51 insertions(+), 2 deletions(-) diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index 00a964a17..d262bb187 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -975,10 +975,15 @@ class lexer : public lexer_base case '\n': case '\r': case char_traits::eof(): + return true; + #if !JSON_STRICT_NUL_HANDLING case '\0': -#endif + // a NUL byte is the end of the input (see scan()), + // so leave it for scan() to see + unget(); return true; +#endif default: break; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 576498738..bc415ce40 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -10067,10 +10067,15 @@ class lexer : public lexer_base case '\n': case '\r': case char_traits::eof(): + return true; + #if !JSON_STRICT_NUL_HANDLING case '\0': -#endif + // a NUL byte is the end of the input (see scan()), + // so leave it for scan() to see + unget(); return true; +#endif default: break; diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index 3d825162e..fc9c9ec23 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -592,6 +592,45 @@ TEST_CASE("parser class") // parsing from a string literal is unaffected either way CHECK(json::parse("123") == json(123)); + + // a NUL byte that ends a // comment ends the input just + // like a NUL byte anywhere else (issue #5659); before the + // fix, the NUL was consumed as part of the comment, and + // scanning continued with whatever followed it + { + // same as "//c" alone (real end of input after the + // comment), rather than continuing with "[1]" + std::string s1 = "//c"; + s1.push_back('\0'); + s1 += "[1]"; + json _; // NOLINT(readability-identifier-naming) + CHECK_THROWS_WITH_AS(_ = json::parse(s1, nullptr, true, true), + "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - unexpected end of input; expected '[', '{', or a literal", + json::parse_error&); + CHECK_FALSE(json::accept(s1, true, true)); + } + + { + // same as "[1, //c" alone, rather than continuing with " 2]" + std::string s2 = "[1, //c"; + s2.push_back('\0'); + s2 += " 2]"; + json _; // NOLINT(readability-identifier-naming) + CHECK_THROWS_WITH_AS(_ = json::parse(s2, nullptr, true, true), + "[json.exception.parse_error.101] parse error at line 1, column 8: syntax error while parsing value - unexpected end of input; expected '[', '{', or a literal", + json::parse_error&); + CHECK_FALSE(json::accept(s2, true, true)); + } + + { + // same as "1 //c" alone: the comment (and the NUL that + // ends it) is ignored, and "x" is never reached + std::string s3 = "1 //c"; + s3.push_back('\0'); + s3 += "x"; + CHECK(json::parse(s3, nullptr, true, true) == json(1)); + CHECK(json::accept(s3, true, true)); + } } #endif