mirror of
https://github.com/nlohmann/json.git
synced 2026-10-03 05:00:30 +00:00
Review and extend the documentation, and check it in CI
A review of all documentation pages found factual errors, dead links, missing cross-references, and gaps in examples. This fixes them and adds checks so the same problems are caught automatically. Fixes: - wrong signatures and version histories (operator!= C++20 member, binary() subtype type, get<PointerType>(), JSON_NO_THREAD_LOCAL, ...) - stale descriptions (number parsing since #5283, UBJSON table, SAX example that no longer compiled, tsl::ordered_map advice) - dead internal and external links; repology.org badges (the domain is suspended) replaced by badges that query the registries directly - deprecation notes link the migration guide; the guide itself fixed Additions: - "See also" sections, cross-references, 25 runnable examples, 12 Mermaid diagrams, new API pages for json_pointer::operator<=> and byte_container_with_subtype::operator==/!= - landing page, guides for untrusted input and performance - "unreleased" badge after versions newer than the latest release Checks: - strict documentation build (broken links/anchors fail it); CI and the publish workflow fetch the full history the build needs - weekly external link check, Mermaid syntax check in CI - check_structure.py: example titles, heading levels, alt texts, header links, docset index coverage; its unused-example check works again - all examples produce the same output on every platform Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -4,6 +4,7 @@ import glob
|
||||
import os.path
|
||||
import re
|
||||
import sys
|
||||
import urllib.parse
|
||||
|
||||
import yaml
|
||||
|
||||
@@ -152,7 +153,7 @@ def check_structure() -> None:
|
||||
|
||||
|
||||
def check_examples() -> None:
|
||||
example_files = sorted(glob.glob("../../examples/*.cpp"))
|
||||
example_files = sorted(glob.glob("examples/*.cpp"))
|
||||
markdown_files = sorted(glob.glob("**/*.md", recursive=True))
|
||||
|
||||
# check if every example file is used in at least one markdown file
|
||||
@@ -211,11 +212,122 @@ def check_links() -> None:
|
||||
report("nav/duplicate_files", "mkdocs.yml", f'file "{duplicate_file}" is linked with multiple keys in "nav": {file_list_str}; only one is rendered properly, see #4564')
|
||||
|
||||
|
||||
FENCE_RE = re.compile(r"^\s*(`{3,}|~{3,})")
|
||||
INLINE_CODE_RE = re.compile(r"(`+).+?\1")
|
||||
|
||||
|
||||
def markdown_lines(file):
|
||||
"""Yield (lineno, line) for all lines outside fenced code blocks."""
|
||||
fence = None
|
||||
with open(file, encoding="utf-8") as content:
|
||||
for lineno, line in enumerate(content, 1):
|
||||
line = line.rstrip("\n")
|
||||
match = FENCE_RE.match(line)
|
||||
if fence is None:
|
||||
if match:
|
||||
fence = match.group(1)
|
||||
else:
|
||||
yield lineno, line
|
||||
elif match and line.strip() == match.group(1) and match.group(1)[0] == fence[0] \
|
||||
and len(match.group(1)) >= len(fence):
|
||||
fence = None
|
||||
|
||||
|
||||
def check_example_titles() -> None:
|
||||
"""On API pages with more than one example, every example needs a title of the form "Example: ..."."""
|
||||
example_re = re.compile(r'^\s*(?:\?\?\?\+?|!!!) example(?: "(.*)")?\s*$')
|
||||
for file in sorted(glob.glob("api/**/*.md", recursive=True)):
|
||||
examples = [(lineno, m.group(1)) for lineno, line in markdown_lines(file) if (m := example_re.match(line))]
|
||||
if len(examples) < 2:
|
||||
continue
|
||||
for lineno, title in examples:
|
||||
if title is None or not title.startswith("Example: "):
|
||||
report("style/example_title", f"{file}:{lineno}",
|
||||
f'pages with several examples need titles like "Example: ..." (found: {title!r})')
|
||||
|
||||
|
||||
def check_heading_levels() -> None:
|
||||
"""Headings start at level 1 and never skip a level."""
|
||||
heading_re = re.compile(r"^(#{1,6})\s|^<h([1-6])[\s>]")
|
||||
for file in sorted(glob.glob("**/*.md", recursive=True)):
|
||||
previous = 0
|
||||
for lineno, line in markdown_lines(file):
|
||||
match = heading_re.match(line)
|
||||
if not match:
|
||||
continue
|
||||
level = len(match.group(1)) if match.group(1) else int(match.group(2))
|
||||
if previous == 0 and level != 1:
|
||||
report("structure/heading_level", f"{file}:{lineno}", f"first heading should have level 1, not {level}")
|
||||
elif level > previous + 1 and previous != 0:
|
||||
report("structure/heading_level", f"{file}:{lineno}", f"heading level jumps from {previous} to {level}")
|
||||
previous = level
|
||||
|
||||
|
||||
def check_image_alt_text() -> None:
|
||||
"""Images need an alternative text."""
|
||||
empty_alt_re = re.compile(r"!\[\s*\][(\[]")
|
||||
img_re = re.compile(r"<img\b[^>]*>", re.IGNORECASE)
|
||||
alt_re = re.compile(r'\balt\s*=\s*"[^"]*\S[^"]*"', re.IGNORECASE)
|
||||
for file in sorted(glob.glob("**/*.md", recursive=True)):
|
||||
for lineno, line in markdown_lines(file):
|
||||
line = INLINE_CODE_RE.sub("", line)
|
||||
if empty_alt_re.search(line) or any(not alt_re.search(tag) for tag in img_re.findall(line)):
|
||||
report("style/image_alt_text", f"{file}:{lineno}", "image without alternative text")
|
||||
|
||||
|
||||
def check_header_links() -> None:
|
||||
"""Links to the documentation in the library's headers point to existing pages."""
|
||||
url_re = re.compile(r"https://json\.nlohmann\.me/([^\s#)>\"']*)")
|
||||
for header in sorted(glob.glob("../../../include/nlohmann/**/*.hpp", recursive=True)):
|
||||
with open(header, encoding="utf-8") as content:
|
||||
for lineno, line in enumerate(content, 1):
|
||||
for match in url_re.finditer(line):
|
||||
path = urllib.parse.unquote(match.group(1)).strip("/")
|
||||
if path and not (os.path.isfile(f"{path}.md") or os.path.isfile(f"{path}/index.md")):
|
||||
report("links/header_link", f"{os.path.relpath(header, '../../..')}:{lineno}",
|
||||
f'link to "{match.group(0)}" does not point to a documentation page')
|
||||
|
||||
|
||||
def check_docset() -> None:
|
||||
"""Every API page and every macro has an entry in the docset index; no entry points to a missing page."""
|
||||
entry_re = re.compile(r"VALUES \('((?:[^']|'')*)', '(\w+)', '([^']*)'\);")
|
||||
names_by_path = {}
|
||||
with open("../../docset/docSet.sql", encoding="utf-8") as sql:
|
||||
for name, _, path in entry_re.findall(sql.read()):
|
||||
names_by_path.setdefault(path, set()).add(name.replace("''", "'"))
|
||||
|
||||
def to_path(page):
|
||||
if os.path.basename(page) == "index.md":
|
||||
return page[:-len("index.md")] + "index.html"
|
||||
return page[:-len(".md")] + "/index.html"
|
||||
|
||||
pages = sorted(glob.glob("**/*.md", recursive=True))
|
||||
for path in sorted(set(names_by_path) - {to_path(p) for p in pages}):
|
||||
report("docset/stale_entry", "../../docset/docSet.sql", f'entry "{path}" has no documentation page')
|
||||
for page in (p for p in pages if p.startswith("api/")):
|
||||
names = names_by_path.get(to_path(page))
|
||||
if not names:
|
||||
report("docset/missing_entry", page, "page has no entry in docs/docset/docSet.sql")
|
||||
elif page.startswith("api/macros/") and os.path.basename(page) != "index.md":
|
||||
with open(page, encoding="utf-8") as content:
|
||||
text = content.read()
|
||||
match = re.search(r"^# (.+)$", text, re.MULTILINE) or re.search(r"<h1>(.*?)</h1>", text, re.DOTALL)
|
||||
title = re.sub(r"<[^>]+>|\s+", " ", match.group(1))
|
||||
for macro in filter(None, (x.strip() for x in re.split(r"[,/]", title))):
|
||||
if macro not in names:
|
||||
report("docset/missing_macro", page, f'macro "{macro}" has no entry in docs/docset/docSet.sql')
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(120 * "-")
|
||||
check_structure()
|
||||
check_examples()
|
||||
check_links()
|
||||
check_example_titles()
|
||||
check_heading_levels()
|
||||
check_image_alt_text()
|
||||
check_header_links()
|
||||
check_docset()
|
||||
print(120 * "-")
|
||||
|
||||
if warnings > 0:
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
#!/usr/bin/env python
|
||||
"""Check the "Added in version" entries of the macro pages against the git tags.
|
||||
|
||||
For every macro documented in docs/api/macros, find the first release tag whose amalgamated header mentions the macro
|
||||
and compare it with the version the page's "Version history" names. A macro documented as added *before* it appears
|
||||
in any release, or documented with a released version although no release contains it, is reported as a problem. A
|
||||
macro that appears in the header *before* its documented version is only a note: many macros existed internally before
|
||||
they were documented for users. The check is heuristic and meant to be run by hand, not in CI.
|
||||
|
||||
usage: python3 check_version_history.py (from docs/mkdocs/docs, needs the git tags)
|
||||
"""
|
||||
|
||||
import glob
|
||||
import re
|
||||
# the script only runs git with fixed arguments and without a shell
|
||||
import subprocess # nosec B404
|
||||
import sys
|
||||
|
||||
HEADER_PATHS = ["single_include/nlohmann/json.hpp", "src/json.hpp"] # older releases used src/json.hpp
|
||||
VERSION_RE = re.compile(r"[Aa]dded in (?:version )?(\d+)\.(\d+)\.(\d+)")
|
||||
NAMED_VERSION_RE = re.compile(r"[Aa]dded `([A-Z0-9_]+)` in (?:version )?(\d+)\.(\d+)\.(\d+)")
|
||||
|
||||
|
||||
def release_tags():
|
||||
# fixed git command without a shell
|
||||
tags = subprocess.run(["git", "tag", "-l", "v*"], capture_output=True, text=True, check=True).stdout.split() # nosec B603, B607
|
||||
versions = []
|
||||
for tag in tags:
|
||||
match = re.fullmatch(r"v(\d+)\.(\d+)\.(\d+)", tag)
|
||||
if match:
|
||||
versions.append((tuple(map(int, match.groups())), tag))
|
||||
return sorted(versions)
|
||||
|
||||
|
||||
def header(tag, cache={}):
|
||||
if tag not in cache:
|
||||
cache[tag] = ""
|
||||
for path in HEADER_PATHS:
|
||||
# fixed git command without a shell; the tag names come from "git tag"
|
||||
result = subprocess.run(["git", "show", f"{tag}:{path}"], capture_output=True, text=True) # nosec B603, B607
|
||||
if result.returncode == 0:
|
||||
cache[tag] = result.stdout
|
||||
break
|
||||
return cache[tag]
|
||||
|
||||
|
||||
def macros_and_versions(page):
|
||||
with open(page, encoding="utf-8") as content:
|
||||
text = content.read()
|
||||
match = re.search(r"^# (.+)$", text, re.MULTILINE) or re.search(r"<h1>(.*?)</h1>", text, re.DOTALL)
|
||||
title = re.sub(r"<[^>]+>|\s+", " ", match.group(1))
|
||||
macros = [x.strip() for x in re.split(r"[,/]", title) if x.strip()]
|
||||
history = text.split("## Version history", 1)[-1]
|
||||
entries = re.split(r"\n(?=\s*(?:\d+\.|-)\s)", history)
|
||||
specific = {} # entries like "Added `JSON_HAS_CPP_23` in version 3.12.0."
|
||||
general = []
|
||||
for entry in entries:
|
||||
named = NAMED_VERSION_RE.search(entry)
|
||||
if named:
|
||||
specific[named.group(1)] = tuple(map(int, named.groups()[1:]))
|
||||
continue
|
||||
match = VERSION_RE.search(entry)
|
||||
if match:
|
||||
general.append(tuple(map(int, match.groups())))
|
||||
rest = [macro for macro in macros if macro not in specific]
|
||||
if len(general) == len(rest): # numbered history: one entry per macro, in title order
|
||||
pairs = list(zip(rest, general))
|
||||
else:
|
||||
pairs = [(macro, general[0]) for macro in rest] if general else []
|
||||
return pairs + sorted(specific.items())
|
||||
|
||||
|
||||
def main():
|
||||
tags = release_tags()
|
||||
latest = tags[-1][0]
|
||||
problems = notes = 0
|
||||
for page in sorted(glob.glob("api/macros/*.md")):
|
||||
if page.endswith("index.md"):
|
||||
continue
|
||||
for macro, documented in macros_and_versions(page):
|
||||
pattern = re.compile(rf"\b{re.escape(macro)}\b")
|
||||
first = next((version for version, tag in tags if pattern.search(header(tag))), None)
|
||||
fmt = ".".join
|
||||
if first is None:
|
||||
if documented <= latest:
|
||||
problems += 1
|
||||
print(f"{page}: {macro} is documented as added in {fmt(map(str, documented))}, "
|
||||
f"but no release up to {fmt(map(str, latest))} contains it")
|
||||
elif documented < first:
|
||||
problems += 1
|
||||
print(f"{page}: {macro} is documented as added in {fmt(map(str, documented))}, "
|
||||
f"but first appears in {fmt(map(str, first))}")
|
||||
elif documented > first:
|
||||
notes += 1
|
||||
print(f"{page}: note: {macro} is documented as added in {fmt(map(str, documented))}, "
|
||||
f"but is mentioned in the header since {fmt(map(str, first))}")
|
||||
print(f"{problems} possible problem(s), {notes} note(s)")
|
||||
return 1 if problems else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,54 @@
|
||||
// Check that every Mermaid diagram in the documentation parses.
|
||||
//
|
||||
// MkDocs does not validate Mermaid diagrams; a syntax error only shows up as an error box in the browser. This script
|
||||
// extracts every ```mermaid block from the Markdown files and runs it through mermaid.parse(), the same parser the
|
||||
// site uses (Material for MkDocs loads mermaid@11). Mermaid needs a DOM (DOMPurify), so jsdom provides one; the globals
|
||||
// must be set before Mermaid is imported, hence the dynamic import.
|
||||
//
|
||||
// usage: node check_mermaid.mjs <docs directory>
|
||||
|
||||
import { readdirSync, readFileSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
import { JSDOM } from 'jsdom';
|
||||
|
||||
const { window } = new JSDOM('<!DOCTYPE html><html><body></body></html>', { pretendToBeVisual: true });
|
||||
globalThis.window = window;
|
||||
globalThis.document = window.document;
|
||||
globalThis.DOMParser = window.DOMParser;
|
||||
const { default: mermaid } = await import('mermaid');
|
||||
mermaid.initialize({ startOnLoad: false });
|
||||
|
||||
const docsDir = process.argv[2] ?? 'docs';
|
||||
const opening = /^(\s*)(`{3,}|~{3,})\s*mermaid\s*$/;
|
||||
let diagrams = 0;
|
||||
let errors = 0;
|
||||
|
||||
for (const file of readdirSync(docsDir, { recursive: true }).filter((f) => f.endsWith('.md')).sort()) {
|
||||
const lines = readFileSync(join(docsDir, file), 'utf8').split('\n');
|
||||
for (let i = 0; i < lines.length; ++i) {
|
||||
const match = opening.exec(lines[i]);
|
||||
if (!match) {
|
||||
continue;
|
||||
}
|
||||
// strip the indentation of the opening fence from every line (blocks inside admonitions or lists), like
|
||||
// pymdownx.superfences does
|
||||
const [, indent, fence] = match;
|
||||
const closing = new RegExp(`^\\s*\\${fence[0]}{${fence.length},}\\s*$`);
|
||||
const body = [];
|
||||
let j = i + 1;
|
||||
for (; j < lines.length && !closing.test(lines[j]); ++j) {
|
||||
body.push(lines[j].startsWith(indent) ? lines[j].slice(indent.length) : lines[j].trimStart());
|
||||
}
|
||||
++diagrams;
|
||||
try {
|
||||
await mermaid.parse(body.join('\n'));
|
||||
} catch (error) {
|
||||
++errors;
|
||||
console.log(`${join(docsDir, file)}:${i + 1}: ${String(error?.message ?? error).replaceAll('\n', '\n ')}`);
|
||||
}
|
||||
i = j;
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`checked ${diagrams} Mermaid diagrams, ${errors} invalid`);
|
||||
process.exitCode = errors ? 1 : 0;
|
||||
+1619
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,10 @@
|
||||
{
|
||||
"name": "check-mermaid",
|
||||
"private": true,
|
||||
"description": "Validate the Mermaid diagrams of the documentation (see check_mermaid.mjs)",
|
||||
"type": "module",
|
||||
"dependencies": {
|
||||
"jsdom": "30.1.1",
|
||||
"mermaid": "11.17.2"
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user