diff --git a/make.py b/make.py index 146a525ea..409ca0d3b 100755 --- a/make.py +++ b/make.py @@ -14,6 +14,7 @@ import os import subprocess +import sys from contextlib import contextmanager from pathlib import Path from shutil import rmtree, which @@ -216,6 +217,8 @@ def docs_full(): run_doc([SPHINX_BUILD, DOC_DIR, "build", "-W"]) print() print("Build finished") + # Sphinx doesn't check plain links, such as a figure's :target: + run_doc([sys.executable, "util/check_internal_links.py", "build"]) @app.command(rich_help_panel="Docs") diff --git a/util/check_internal_links.py b/util/check_internal_links.py new file mode 100644 index 000000000..5ef2cbebc --- /dev/null +++ b/util/check_internal_links.py @@ -0,0 +1,88 @@ +""" +Check a built copy of the docs for links to pages, files or anchors that +don't exist. + +Sphinx's nitpicky mode catches broken cross-references such as ``:ref:`` +and ``:py:class:``, but not plain links, such as a figure's ``:target:`` +or a named link written without its trailing underscore. This checks the +HTML Sphinx wrote instead. It doesn't check links to other sites. + +Usage:: + + python util/check_internal_links.py build + +Exits with status 1 if it finds broken links. +""" + +import os +import re +import sys +from collections import defaultdict +from pathlib import Path +from urllib.parse import unquote, urldefrag + +LINK_RE = re.compile(r'(?:href|src)="([^"]+)"') +ID_RE = re.compile(r'id="([^"]+)"') +EXTERNAL_PREFIXES = ("http:", "https:", "mailto:", "data:", "javascript:", "//") +# Generated by Sphinx, and full of links to every page +SKIPPED = ("_modules/", "_static/", "_sources/", "genindex.html", "search.html") + + +def find_broken_links(root: Path) -> dict[str, list[str]]: + """Return {link: [pages it's on]} for links that go nowhere.""" + # Every page repeats the sidebar, so the same link from the same folder + # comes up hundreds of times. Check each one once. + link_ok: dict[tuple[str, str], bool] = {} + ids: dict[str, set[str]] = {} + broken: dict[str, list[str]] = defaultdict(list) + + def is_ok(page: str, folder: str, link: str) -> bool: + url, anchor = urldefrag(unquote(link)) + url = url.split("?")[0] # Cache-busting such as ?v=1234 + # normpath, not Path.resolve(), which can hang on Windows + target = os.path.normpath(os.path.join(folder, url)) if url else page + if not os.path.exists(target): + return False + if anchor and target.endswith(".html"): + if target not in ids: + with open(target, encoding="utf-8", errors="replace") as file: + ids[target] = set(ID_RE.findall(file.read())) + return anchor in ids[target] + return True + + for path in sorted(root.rglob("*.html")): + name = path.relative_to(root).as_posix() + if name.startswith(SKIPPED): + continue + page = str(path) + folder = os.path.dirname(page) + links = set(LINK_RE.findall(path.read_text(encoding="utf-8", errors="replace"))) + for link in links: + if link.startswith(EXTERNAL_PREFIXES): + continue + # An anchor-only link depends on the page, not just the folder + key = (page if link.startswith("#") else folder, link) + if key not in link_ok: + link_ok[key] = is_ok(page, folder, link) + if not link_ok[key]: + broken[link].append(name) + return broken + + +def main() -> int: + root = Path(sys.argv[1] if len(sys.argv) > 1 else "build") + if not root.is_dir(): + print(f"No built docs at {root}") + return 1 + broken = find_broken_links(root) + if not broken: + print("No broken internal links") + return 0 + print(f"{len(broken)} broken internal links:") + for link, pages in sorted(broken.items()): + print(f" {link} (in {', '.join(pages[:3])}{', ...' if len(pages) > 3 else ''})") + return 1 + + +if __name__ == "__main__": + sys.exit(main())