From 0490778fc30c7887db704f278941612587b0e0b7 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 6 Oct 2026 07:46:24 +0200 Subject: [PATCH 1/2] Replace retired macOS 14 runner and test all available Xcode versions (#5757) * Replace retired macOS 14 runner and test all available Xcode versions GitHub retires the macos-14 image on 2026-11-02 (brownouts from 2026-10-05). Xcode 15 is not available on any remaining hosted runner, so drop the macos-14 job and its documented compilers. Also test the Xcode versions that the images provide but CI did not use (26.1.1-26.3 on macos-15, 26.4.1-26.6 on a new macos-26 job), and pin GCC 16 explicitly next to gcc:latest. Signed-off-by: Niels Lohmann * Document new Xcode and GCC versions in the supported compilers table Versions taken from the CI logs of this PR (Xcode 26.1.1-26.6) and from the gcc:16 image (same digest as gcc:16.2.0 and gcc:latest). Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/workflows/macos.yml | 12 ++++++------ .github/workflows/ubuntu.yml | 2 +- docs/mkdocs/docs/community/quality_assurance.md | 13 +++++++------ 3 files changed, 14 insertions(+), 13 deletions(-) diff --git a/.github/workflows/macos.yml b/.github/workflows/macos.yml index 0adaf26d5..a47fbd4b0 100644 --- a/.github/workflows/macos.yml +++ b/.github/workflows/macos.yml @@ -17,11 +17,11 @@ permissions: contents: read jobs: - macos-14: - runs-on: macos-14 # https://github.com/actions/runner-images/blob/main/images/macos/macos-14-Readme.md + macos-15: + runs-on: macos-15 # https://github.com/actions/runner-images/blob/main/images/macos/macos-15-Readme.md strategy: matrix: - xcode: ['15.0.1', '15.1', '15.2', '15.3', '15.4'] + xcode: ['16.0', '16.1', '16.2', '16.3', '16.4', '26.0.1', '26.1.1', '26.2', '26.3'] env: DEVELOPER_DIR: /Applications/Xcode_${{ matrix.xcode }}.app/Contents/Developer @@ -36,11 +36,11 @@ jobs: - name: Test run: cd build ; ctest -j 10 --output-on-failure - macos-15: - runs-on: macos-15 # https://github.com/actions/runner-images/blob/main/images/macos/macos-15-Readme.md + macos-26: + runs-on: macos-26 # https://github.com/actions/runner-images/blob/main/images/macos/macos-26-arm64-Readme.md strategy: matrix: - xcode: ['16.0', '16.1', '16.2', '16.3', '16.4', '26.0.1'] + xcode: ['26.4.1', '26.5', '26.6'] env: DEVELOPER_DIR: /Applications/Xcode_${{ matrix.xcode }}.app/Contents/Developer diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index 354cf2391..1fa2f9fc6 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -209,7 +209,7 @@ jobs: strategy: matrix: # older GCC docker images (4, 5, 6) fail to check out code - compiler: ['7', '8', '9', '10', '11', '12', '13', '14', '15', 'latest'] + compiler: ['7', '8', '9', '10', '11', '12', '13', '14', '15', '16', 'latest'] container: gcc:${{ matrix.compiler }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 diff --git a/docs/mkdocs/docs/community/quality_assurance.md b/docs/mkdocs/docs/community/quality_assurance.md index d44135bf2..bd8f55790 100644 --- a/docs/mkdocs/docs/community/quality_assurance.md +++ b/docs/mkdocs/docs/community/quality_assurance.md @@ -21,17 +21,18 @@ Note: Some modern features (like C++20 ranges or filesystem support) may be disa | Compiler | Architecture | Operating System | CI | |----------------------------------------------|--------------|-----------------------------------|-----------| - | AppleClang 15.0.0.15000040; Xcode 15.0.1 | arm64 | macOS 14.7.2 (Sonoma) | GitHub | - | AppleClang 15.0.0.15000100; Xcode 15.1 | arm64 | macOS 14.7.2 (Sonoma) | GitHub | - | AppleClang 15.0.0.15000100; Xcode 15.2 | arm64 | macOS 14.7.2 (Sonoma) | GitHub | - | AppleClang 15.0.0.15000309; Xcode 15.3 | arm64 | macOS 14.7.2 (Sonoma) | GitHub | - | AppleClang 15.0.0.15000309; Xcode 15.4 | arm64 | macOS 14.7.2 (Sonoma) | GitHub | | AppleClang 16.0.0.16000026; Xcode 16 | arm64 | macOS 15.2 (Sequoia) | GitHub | | AppleClang 16.0.0.16000026; Xcode 16.1 | arm64 | macOS 15.2 (Sequoia) | GitHub | | AppleClang 16.0.0.16000026; Xcode 16.2 | arm64 | macOS 15.2 (Sequoia) | GitHub | | AppleClang 17.0.0.17000013; Xcode 16.3 | arm64 | macOS 15.5 (Sequoia) | GitHub | | AppleClang 17.0.0.17000013; Xcode 16.4 | arm64 | macOS 15.5 (Sequoia) | GitHub | | AppleClang 17.0.0.17000319; Xcode 26.0.1 | arm64 | macOS 15.5 (Sequoia) | GitHub | + | AppleClang 17.0.0.17000404; Xcode 26.1.1 | arm64 | macOS 15.7.9 (Sequoia) | GitHub | + | AppleClang 17.0.0.17000603; Xcode 26.2 | arm64 | macOS 15.7.9 (Sequoia) | GitHub | + | AppleClang 17.0.0.17000604; Xcode 26.3 | arm64 | macOS 15.7.9 (Sequoia) | GitHub | + | AppleClang 21.0.0.21000099; Xcode 26.4.1 | arm64 | macOS 26.6.2 (Tahoe) | GitHub | + | AppleClang 21.0.0.21000101; Xcode 26.5 | arm64 | macOS 26.6.2 (Tahoe) | GitHub | + | AppleClang 21.0.0.21000101; Xcode 26.6 | arm64 | macOS 26.6.2 (Tahoe) | GitHub | | Clang 3.4.2 | x86_64 | Ubuntu 22.04.1 LTS | GitHub | | Clang 3.5.2 | x86_64 | Ubuntu 22.04.1 LTS | GitHub | | Clang 3.6.2 | x86_64 | Ubuntu 22.04.1 LTS | GitHub | @@ -89,7 +90,7 @@ Note: Some modern features (like C++20 ranges or filesystem support) may be disa | GNU 13.3.0 | x86_64 | Ubuntu 22.04.1 LTS | GitHub | | GNU 14.2.0 | x86_64 | Ubuntu 22.04.1 LTS | GitHub | | GNU 15.1.0 | x86_64 | Ubuntu 22.04.1 LTS | GitHub | - | GNU 16.1.0 | x86_64 | Ubuntu 22.04.1 LTS | GitHub | + | GNU 16.2.0 | x86_64 | Ubuntu 22.04.1 LTS | GitHub | | GNU 16.1.0 | arm64 | Ubuntu 24.04 | GitHub | | icpc (ICC) 2021.10.0 20230609 | x86_64 | Ubuntu 22.04 LTS | GitHub | | icpx (Intel oneAPI DPC++/C++) 2025.3.2 | x86_64 | Ubuntu 24.04 LTS | GitHub | From 5ecb704f6b930bd5925d3f268444f0630a829ead Mon Sep 17 00:00:00 2001 From: Michiel van Slobbe Date: Tue, 6 Oct 2026 08:36:40 +0200 Subject: [PATCH 2/2] Speedup; check for the expected separator before the lexer's token switch (#5592) * Check for the expected separator before the lexer's token switch After a key the parser expects ':', after a value usually ','. Test for that character first instead of going through scan()'s switch, which compiles to an indirect jump. Any other character takes the old path, so tokens and error messages are unchanged. Parsing 6.3% faster with GCC 15.2 and 2.7% with Clang 22.1 (geomean of the ParseString, ParseFile and ParseIndented benchmarks). Signed-off-by: Michiel van Slobbe * Improvement: address PR comments Signed-off-by: Michiel van Slobbe * fix: address comments Signed-off-by: Michiel van Slobbe * Fix clang-tidy bugprone-signed-char-misuse in scan_expecting Convert the expected separator through unsigned char before storing it as char_int_type. The generated code is unchanged. Signed-off-by: Michiel van Slobbe * Use raw string literals in the separator comment tests Signed-off-by: Michiel van Slobbe --------- Signed-off-by: Michiel van Slobbe Co-authored-by: Michiel van Slobbe --- include/nlohmann/detail/input/lexer.hpp | 34 +++++++++++++++- include/nlohmann/detail/input/parser.hpp | 15 +++++-- single_include/nlohmann/json.hpp | 49 +++++++++++++++++++--- tests/src/unit-class_parser.cpp | 52 ++++++++++++++++++++++++ 4 files changed, 140 insertions(+), 10 deletions(-) diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index 052bc4de9..361baa642 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -2021,6 +2021,39 @@ scan_number_done: // read the next character and ignore whitespace skip_whitespace(); + return scan_after_whitespace(); + } + + /*! + @brief scan the next token when the caller expects a separator (':' or + ',') most of the time + + After an object key the next token is almost always ':', after a value + inside an object or array almost always ','. Testing for that character + first is a compare and a well-predicted branch, where the switch in + scan_after_whitespace() is an indirect jump through a table. Anything else + goes through the switch, so the result is the same as scan()'s. + + May only be called after scan() has run once (the BOM check is skipped). + */ + token_type scan_expecting(token_type expected_type) + { + JSON_ASSERT(expected_type == token_type::name_separator || expected_type == token_type::value_separator); + JSON_ASSERT(position.chars_read_total > 0); + const char_int_type expected_char = static_cast((expected_type == token_type::name_separator) ? ':' : ','); + skip_whitespace(); + if (JSON_HEDLEY_LIKELY(current == expected_char)) + { + return expected_type; + } + return scan_after_whitespace(); + } + + private: + /// the part of scan() after the leading whitespace: skip comments and + /// scan the token that starts with current + token_type scan_after_whitespace() + { // ignore comments while (ignore_comments && current == '/') { @@ -2100,7 +2133,6 @@ scan_number_done: } } - private: /// input adapter InputAdapterType ia; diff --git a/include/nlohmann/detail/input/parser.hpp b/include/nlohmann/detail/input/parser.hpp index 7998dc6dd..aa2b30fd7 100644 --- a/include/nlohmann/detail/input/parser.hpp +++ b/include/nlohmann/detail/input/parser.hpp @@ -260,7 +260,7 @@ class parser } // parse separator (:) - if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator)) + if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator))) { return sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), @@ -423,7 +423,7 @@ class parser { // comma -> next value // or end of array (ignore_trailing_commas = true) - if (get_token() == token_type::value_separator) + if (get_token_expecting(token_type::value_separator)) { // parse a new value get_token(); @@ -463,7 +463,7 @@ class parser // comma -> next value // or end of object (ignore_trailing_commas = true) - if (get_token() == token_type::value_separator) + if (get_token_expecting(token_type::value_separator)) { get_token(); @@ -484,7 +484,7 @@ class parser } // parse separator (:) - if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator)) + if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator))) { return sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), @@ -528,6 +528,13 @@ class parser return last_token = m_lexer.scan(); } + /// get next token from lexer; true if it is the separator @a expected_type + /// (name_separator or value_separator), which it usually is + bool get_token_expecting(token_type expected_type) + { + return (last_token = m_lexer.scan_expecting(expected_type)) == expected_type; + } + std::string exception_message(const token_type expected, const std::string& context) { std::string error_msg = "syntax error "; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 86321a36e..00e345dd3 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -12313,6 +12313,39 @@ scan_number_done: // read the next character and ignore whitespace skip_whitespace(); + return scan_after_whitespace(); + } + + /*! + @brief scan the next token when the caller expects a separator (':' or + ',') most of the time + + After an object key the next token is almost always ':', after a value + inside an object or array almost always ','. Testing for that character + first is a compare and a well-predicted branch, where the switch in + scan_after_whitespace() is an indirect jump through a table. Anything else + goes through the switch, so the result is the same as scan()'s. + + May only be called after scan() has run once (the BOM check is skipped). + */ + token_type scan_expecting(token_type expected_type) + { + JSON_ASSERT(expected_type == token_type::name_separator || expected_type == token_type::value_separator); + JSON_ASSERT(position.chars_read_total > 0); + const char_int_type expected_char = static_cast((expected_type == token_type::name_separator) ? ':' : ','); + skip_whitespace(); + if (JSON_HEDLEY_LIKELY(current == expected_char)) + { + return expected_type; + } + return scan_after_whitespace(); + } + + private: + /// the part of scan() after the leading whitespace: skip comments and + /// scan the token that starts with current + token_type scan_after_whitespace() + { // ignore comments while (ignore_comments && current == '/') { @@ -12392,7 +12425,6 @@ scan_number_done: } } - private: /// input adapter InputAdapterType ia; @@ -18422,7 +18454,7 @@ class parser } // parse separator (:) - if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator)) + if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator))) { return sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), @@ -18585,7 +18617,7 @@ class parser { // comma -> next value // or end of array (ignore_trailing_commas = true) - if (get_token() == token_type::value_separator) + if (get_token_expecting(token_type::value_separator)) { // parse a new value get_token(); @@ -18625,7 +18657,7 @@ class parser // comma -> next value // or end of object (ignore_trailing_commas = true) - if (get_token() == token_type::value_separator) + if (get_token_expecting(token_type::value_separator)) { get_token(); @@ -18646,7 +18678,7 @@ class parser } // parse separator (:) - if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator)) + if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator))) { return sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), @@ -18690,6 +18722,13 @@ class parser return last_token = m_lexer.scan(); } + /// get next token from lexer; true if it is the separator @a expected_type + /// (name_separator or value_separator), which it usually is + bool get_token_expecting(token_type expected_type) + { + return (last_token = m_lexer.scan_expecting(expected_type)) == expected_type; + } + std::string exception_message(const token_type expected, const std::string& context) { std::string error_msg = "syntax error "; diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index f2de97ea3..978d94f92 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -2317,6 +2317,58 @@ TEST_CASE("parser class") #endif } + SECTION("comments before separators") + { + // The parser first checks for the expected ':' or ',' and only then + // falls back to the full token switch, which skips comments. A comment + // directly before a separator takes that fallback. + json _; + + SECTION("ignored") + { + const std::vector> inputs = + { + {"{\"a\" /* c */ : 1}", {{"a", 1}}}, + {"{\"a\" // c\n: 1}", {{"a", 1}}}, + {R"({"a": 1, "b" /* c */ : 2})", {{"a", 1}, {"b", 2}}}, + {R"({"a": 1 /* c */ , "b": 2})", {{"a", 1}, {"b", 2}}}, + {"{\"a\": 1 // c\n, \"b\": 2}", {{"a", 1}, {"b", 2}}}, + {"[1 /* c */ , 2]", {1, 2}}, + {"[1 // c\n, 2]", {1, 2}}, + {"{\"a\" /* c */ /* d */ : [1 // c\n , 2 /**/ ] /**/ , \"b\" : 3}", {{"a", {1, 2}}, {"b", 3}}} + }; + for (const auto& input : inputs) + { + CAPTURE(input.first) + CHECK(json::parse(input.first, nullptr, true, true) == input.second); + CHECK(json::accept(input.first, true)); + } + } + + SECTION("ignored, with trailing commas") + { + CHECK(json::parse(std::string("[1 /* c */ , ]"), nullptr, true, true, true) == json({1})); + CHECK(json::parse(std::string("{\"a\": 1 /* c */ , }"), nullptr, true, true, true) == json({{"a", 1}})); + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("[1 /* c */ , ]"), nullptr, true, true), + "[json.exception.parse_error.101] parse error at line 1, column 14: syntax error while parsing value - unexpected ']'; expected '[', '{', or a literal", json::parse_error); + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\": 1 /* c */ , }"), nullptr, true, true), + "[json.exception.parse_error.101] parse error at line 1, column 19: syntax error while parsing object key - unexpected '}'; expected string literal", json::parse_error); + } + + SECTION("not ignored") + { + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\" /* c */ : 1}")), + "[json.exception.parse_error.101] parse error at line 1, column 6: syntax error while parsing object separator - invalid literal; last read: '\"a\" /'; expected ':'", json::parse_error); + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\": 1, \"b\" /* c */ : 2}")), + "[json.exception.parse_error.101] parse error at line 1, column 14: syntax error while parsing object separator - invalid literal; last read: '\"b\" /'; expected ':'", json::parse_error); + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\": 1 /* c */ , \"b\": 2}")), + "[json.exception.parse_error.101] parse error at line 1, column 9: syntax error while parsing object - invalid literal; last read: '1 /'; expected '}'", json::parse_error); + CHECK_THROWS_WITH_AS(_ = json::parse(std::string("[1 /* c */ , 2]")), + "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing array - invalid literal; last read: '1 /'; expected ']'", json::parse_error); + CHECK(!json::accept(std::string("[1 /* c */ , 2]"))); + } + } + #if JSON_DIAGNOSTIC_POSITIONS // Macro for all test cases for start_pos and end_pos #define SETUP_TESTCASES() \