diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index 052bc4de9..361baa642 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -2021,6 +2021,39 @@ scan_number_done: // read the next character and ignore whitespace skip_whitespace(); + return scan_after_whitespace(); + } + + /*! + @brief scan the next token when the caller expects a separator (':' or + ',') most of the time + + After an object key the next token is almost always ':', after a value + inside an object or array almost always ','. Testing for that character + first is a compare and a well-predicted branch, where the switch in + scan_after_whitespace() is an indirect jump through a table. Anything else + goes through the switch, so the result is the same as scan()'s. + + May only be called after scan() has run once (the BOM check is skipped). + */ + token_type scan_expecting(token_type expected_type) + { + JSON_ASSERT(expected_type == token_type::name_separator || expected_type == token_type::value_separator); + JSON_ASSERT(position.chars_read_total > 0); + const char_int_type expected_char = static_cast((expected_type == token_type::name_separator) ? ':' : ','); + skip_whitespace(); + if (JSON_HEDLEY_LIKELY(current == expected_char)) + { + return expected_type; + } + return scan_after_whitespace(); + } + + private: + /// the part of scan() after the leading whitespace: skip comments and + /// scan the token that starts with current + token_type scan_after_whitespace() + { // ignore comments while (ignore_comments && current == '/') { @@ -2100,7 +2133,6 @@ scan_number_done: } } - private: /// input adapter InputAdapterType ia; diff --git a/include/nlohmann/detail/input/parser.hpp b/include/nlohmann/detail/input/parser.hpp index 7998dc6dd..aa2b30fd7 100644 --- a/include/nlohmann/detail/input/parser.hpp +++ b/include/nlohmann/detail/input/parser.hpp @@ -260,7 +260,7 @@ class parser } // parse separator (:) - if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator)) + if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator))) { return sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), @@ -423,7 +423,7 @@ class parser { // comma -> next value // or end of array (ignore_trailing_commas = true) - if (get_token() == token_type::value_separator) + if (get_token_expecting(token_type::value_separator)) { // parse a new value get_token(); @@ -463,7 +463,7 @@ class parser // comma -> next value // or end of object (ignore_trailing_commas = true) - if (get_token() == token_type::value_separator) + if (get_token_expecting(token_type::value_separator)) { get_token(); @@ -484,7 +484,7 @@ class parser } // parse separator (:) - if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator)) + if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator))) { return sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), @@ -528,6 +528,13 @@ class parser return last_token = m_lexer.scan(); } + /// get next token from lexer; true if it is the separator @a expected_type + /// (name_separator or value_separator), which it usually is + bool get_token_expecting(token_type expected_type) + { + return (last_token = m_lexer.scan_expecting(expected_type)) == expected_type; + } + std::string exception_message(const token_type expected, const std::string& context) { std::string error_msg = "syntax error "; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 86321a36e..00e345dd3 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -12313,6 +12313,39 @@ scan_number_done: // read the next character and ignore whitespace skip_whitespace(); + return scan_after_whitespace(); + } + + /*! + @brief scan the next token when the caller expects a separator (':' or + ',') most of the time + + After an object key the next token is almost always ':', after a value + inside an object or array almost always ','. Testing for that character + first is a compare and a well-predicted branch, where the switch in + scan_after_whitespace() is an indirect jump through a table. Anything else + goes through the switch, so the result is the same as scan()'s. + + May only be called after scan() has run once (the BOM check is skipped). + */ + token_type scan_expecting(token_type expected_type) + { + JSON_ASSERT(expected_type == token_type::name_separator || expected_type == token_type::value_separator); + JSON_ASSERT(position.chars_read_total > 0); + const char_int_type expected_char = static_cast((expected_type == token_type::name_separator) ? ':' : ','); + skip_whitespace(); + if (JSON_HEDLEY_LIKELY(current == expected_char)) + { + return expected_type; + } + return scan_after_whitespace(); + } + + private: + /// the part of scan() after the leading whitespace: skip comments and + /// scan the token that starts with current + token_type scan_after_whitespace() + { // ignore comments while (ignore_comments && current == '/') { @@ -12392,7 +12425,6 @@ scan_number_done: } } - private: /// input adapter InputAdapterType ia; @@ -18422,7 +18454,7 @@ class parser } // parse separator (:) - if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator)) + if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator))) { return sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), @@ -18585,7 +18617,7 @@ class parser { // comma -> next value // or end of array (ignore_trailing_commas = true) - if (get_token() == token_type::value_separator) + if (get_token_expecting(token_type::value_separator)) { // parse a new value get_token(); @@ -18625,7 +18657,7 @@ class parser // comma -> next value // or end of object (ignore_trailing_commas = true) - if (get_token() == token_type::value_separator) + if (get_token_expecting(token_type::value_separator)) { get_token(); @@ -18646,7 +18678,7 @@ class parser } // parse separator (:) - if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator)) + if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator))) { return sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), @@ -18690,6 +18722,13 @@ class parser return last_token = m_lexer.scan(); } + /// get next token from lexer; true if it is the separator @a expected_type + /// (name_separator or value_separator), which it usually is + bool get_token_expecting(token_type expected_type) + { + return (last_token = m_lexer.scan_expecting(expected_type)) == expected_type; + } + std::string exception_message(const token_type expected, const std::string& context) { std::string error_msg = "syntax error "; diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index f2de97ea3..978d94f92 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -2317,6 +2317,58 @@ TEST_CASE("parser class") #endif } + SECTION("comments before separators") + { + // The parser first checks for the expected ':' or ',' and only then + // falls back to the full token switch, which skips comments. A comment + // directly before a separator takes that fallback. + json _; + + SECTION("ignored") + { + const std::vector> inputs = + { + {"{\"a\" /* c */ : 1}", {{"a", 1}}}, + {"{\"a\" // c\n: 1}", {{"a", 1}}}, + {R"({"a": 1, "b" /* c */ : 2})", {{"a", 1}, {"b", 2}}}, + {R"({"a": 1 /* c */ , "b": 2})", {{"a", 1}, {"b", 2}}}, + {"{\"a\": 1 // c\n, \"b\": 2}", {{"a", 1}, {"b", 2}}}, + {"[1 /* c */ , 2]", {1, 2}}, + {"[1 // c\n, 2]", {1, 2}}, + {"{\"a\" /* c */ /* d */ : [1 // c\n , 2 /**/ ] /**/ , \"b\" : 3}", {{"a", {1, 2}}, {"b", 3}}} + }; + for (const auto& input : inputs) + { + CAPTURE(input.first) + CHECK(json::parse(input.first, nullptr, true, true) == input.second); + CHECK(json::accept(input.first, true)); + } + } + + SECTION("ignored, with trailing commas") + { + CHECK(json::parse(std::string("[1 /* c */ , ]"), nullptr, true, true, true) == json({1})); + CHECK(json::parse(std::string("{\"a\": 1 /* c */ , }"), nullptr, true, true, true) == json({{"a", 1}})); + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("[1 /* c */ , ]"), nullptr, true, true), + "[json.exception.parse_error.101] parse error at line 1, column 14: syntax error while parsing value - unexpected ']'; expected '[', '{', or a literal", json::parse_error); + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\": 1 /* c */ , }"), nullptr, true, true), + "[json.exception.parse_error.101] parse error at line 1, column 19: syntax error while parsing object key - unexpected '}'; expected string literal", json::parse_error); + } + + SECTION("not ignored") + { + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\" /* c */ : 1}")), + "[json.exception.parse_error.101] parse error at line 1, column 6: syntax error while parsing object separator - invalid literal; last read: '\"a\" /'; expected ':'", json::parse_error); + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\": 1, \"b\" /* c */ : 2}")), + "[json.exception.parse_error.101] parse error at line 1, column 14: syntax error while parsing object separator - invalid literal; last read: '\"b\" /'; expected ':'", json::parse_error); + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\": 1 /* c */ , \"b\": 2}")), + "[json.exception.parse_error.101] parse error at line 1, column 9: syntax error while parsing object - invalid literal; last read: '1 /'; expected '}'", json::parse_error); + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("[1 /* c */ , 2]")), + "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing array - invalid literal; last read: '1 /'; expected ']'", json::parse_error); + CHECK(!json::accept(std::string("[1 /* c */ , 2]"))); + } + } + #if JSON_DIAGNOSTIC_POSITIONS // Macro for all test cases for start_pos and end_pos #define SETUP_TESTCASES() \