From 45dc218a130624b57e3462e7f8e70f72d2cd737c Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 18:09:18 +0200 Subject: [PATCH] Replace list_missing_pages/list_removed_paths with a comm(1)-based diff The docset Makefile's list_missing_pages ran one sqlite3 query per mkdocs page, and list_removed_paths nested a loop over all mkdocs pages inside a loop over all docset index paths (O(n*m) shell iteration). Issue #5718 item 5 suggested removing or reducing these targets once #5638's check_docset() lands, but that PR is still open and covers only API pages and macros, not the full page set these targets check. Replace the loops with two sorted path lists (DOCSET_PAGE_PATHS from mkdocs' markdown sources, DOCSET_INDEX_PATHS from the built docset index) compared with a single comm(1) call each, verified to produce output identical to the old loops against the current docSet.dsidx. The sed expression used '#' as its delimiter, which GNU Make reads as a comment character even inside a variable assignment, truncating the line and orphaning the closing paren of $(shell ...) ("unterminated call to function 'shell': missing ')'"). Use '@' as the delimiter instead. Part of #5718 item 5 Signed-off-by: Niels Lohmann --- docs/docset/Makefile | 34 ++++++++++++---------------------- 1 file changed, 12 insertions(+), 22 deletions(-) diff --git a/docs/docset/Makefile b/docs/docset/Makefile index 9f4363462..b638390db 100644 --- a/docs/docset/Makefile +++ b/docs/docset/Makefile @@ -52,35 +52,25 @@ install_docset_zeal: JSON_for_Modern_C++.docset mkdir -p $$docset_root; \ cp -r JSON_for_Modern_C++.docset $$docset_root/ +# both targets below compare the docset search index with the mkdocs page +# set. They share the same normalization (docs/foo/index.md and +# docs/foo.md both become foo/index.html, the URL mkdocs itself would +# give the page; the top-level index.md is excluded, as it is not part +# of the hand-curated docSet.sql) and use comm(1) on two sorted lists +# instead of running a sqlite3 query, or an O(n*m) nested shell loop, +# once per page. +DOCSET_INDEX_PATHS=$(shell sqlite3 docSet.dsidx "SELECT DISTINCT path FROM searchIndex" | sort) +DOCSET_PAGE_PATHS=$(shell echo '$(MKDOCS_PAGES)' | tr ' ' '\n' | grep -v '^index\.md$$' | $(SED) -E 's@/index\.md$$@/index.html@; s@\.md$$@/index.html@' | sort) + # list mkdocs pages missing from the docset index .PHONY: list_missing_pages list_missing_pages: docSet.dsidx - @for page in $(MKDOCS_PAGES); do \ - case "$$page" in \ - */index.md) path=$${page/\/index.md/} ;; \ - *) path=$${page/.md/} ;; \ - esac; \ - if [ "x$$page" != "xindex.md" -a "x$$(sqlite3 docSet.dsidx "SELECT COUNT(*) FROM searchIndex WHERE path='$$path/index.html'")" = "x0" ]; then \ - echo $$page; \ - fi \ - done + @comm -23 <(echo '$(DOCSET_PAGE_PATHS)' | tr ' ' '\n') <(echo '$(DOCSET_INDEX_PATHS)' | tr ' ' '\n') # list paths in the docset index without a corresponding mkdocs page .PHONY: list_removed_paths list_removed_paths: docSet.dsidx - @for path in $$(sqlite3 docSet.dsidx "SELECT path FROM searchIndex"); do \ - page=$${path/\/index.html/.md}; \ - page_index=$${path/index.html/index.md}; \ - page_found=0; \ - for p in $(MKDOCS_PAGES); do \ - if [ "x$$p" = "x$$page" -o "x$$p" = "x$$page_index" ]; then \ - page_found=1; \ - fi \ - done; \ - if [ "x$$page_found" = "x0" ]; then \ - echo $$path; \ - fi \ - done + @comm -13 <(echo '$(DOCSET_PAGE_PATHS)' | tr ' ' '\n') <(echo '$(DOCSET_INDEX_PATHS)' | tr ' ' '\n') .PHONY: clean clean: