Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions make.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@

import os
import subprocess
import sys
from contextlib import contextmanager
from pathlib import Path
from shutil import rmtree, which
Expand Down Expand Up @@ -216,6 +217,8 @@ def docs_full():
run_doc([SPHINX_BUILD, DOC_DIR, "build", "-W"])
print()
print("Build finished")
# Sphinx doesn't check plain links, such as a figure's :target:
run_doc([sys.executable, "util/check_internal_links.py", "build"])


@app.command(rich_help_panel="Docs")
Expand Down
88 changes: 88 additions & 0 deletions util/check_internal_links.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,88 @@
"""
Check a built copy of the docs for links to pages, files or anchors that
don't exist.

Sphinx's nitpicky mode catches broken cross-references such as ``:ref:``
and ``:py:class:``, but not plain links, such as a figure's ``:target:``
or a named link written without its trailing underscore. This checks the
HTML Sphinx wrote instead. It doesn't check links to other sites.

Usage::

python util/check_internal_links.py build

Exits with status 1 if it finds broken links.
"""

import os
import re
import sys
from collections import defaultdict
from pathlib import Path
from urllib.parse import unquote, urldefrag

LINK_RE = re.compile(r'(?:href|src)="([^"]+)"')
ID_RE = re.compile(r'id="([^"]+)"')
EXTERNAL_PREFIXES = ("http:", "https:", "mailto:", "data:", "javascript:", "//")
# Generated by Sphinx, and full of links to every page
SKIPPED = ("_modules/", "_static/", "_sources/", "genindex.html", "search.html")


def find_broken_links(root: Path) -> dict[str, list[str]]:
"""Return {link: [pages it's on]} for links that go nowhere."""
# Every page repeats the sidebar, so the same link from the same folder
# comes up hundreds of times. Check each one once.
link_ok: dict[tuple[str, str], bool] = {}
ids: dict[str, set[str]] = {}
broken: dict[str, list[str]] = defaultdict(list)

def is_ok(page: str, folder: str, link: str) -> bool:
url, anchor = urldefrag(unquote(link))
url = url.split("?")[0] # Cache-busting such as ?v=1234
# normpath, not Path.resolve(), which can hang on Windows
target = os.path.normpath(os.path.join(folder, url)) if url else page
if not os.path.exists(target):
return False
if anchor and target.endswith(".html"):
if target not in ids:
with open(target, encoding="utf-8", errors="replace") as file:
ids[target] = set(ID_RE.findall(file.read()))
return anchor in ids[target]
return True

for path in sorted(root.rglob("*.html")):
name = path.relative_to(root).as_posix()
if name.startswith(SKIPPED):
continue
page = str(path)
folder = os.path.dirname(page)
links = set(LINK_RE.findall(path.read_text(encoding="utf-8", errors="replace")))
for link in links:
if link.startswith(EXTERNAL_PREFIXES):
continue
# An anchor-only link depends on the page, not just the folder
key = (page if link.startswith("#") else folder, link)
if key not in link_ok:
link_ok[key] = is_ok(page, folder, link)
if not link_ok[key]:
broken[link].append(name)
return broken


def main() -> int:
root = Path(sys.argv[1] if len(sys.argv) > 1 else "build")
if not root.is_dir():
print(f"No built docs at {root}")
return 1
broken = find_broken_links(root)
if not broken:
print("No broken internal links")
return 0
print(f"{len(broken)} broken internal links:")
for link, pages in sorted(broken.items()):
print(f" {link} (in {', '.join(pages[:3])}{', ...' if len(pages) > 3 else ''})")
return 1


if __name__ == "__main__":
sys.exit(main())
Loading