#!/usr/bin/env python3 """check_links.py — verify that every relative markdown link in a knowledge base resolves. Starter integrity-check tool for this project's knowledge-base framework (see ../framework.md §6). Usage: python3 check_links.py # checks wiki/ (default) python3 check_links.py ... # checks the given directories Exit code 0 when every relative link target exists; 1 otherwise, with one line per broken link (file: target). External links (http/https/mailto), pure anchors (#...), and links inside fenced code blocks are ignored. Run before every commit. Known limit, disclosed rather than silently assumed away: a link's #anchor fragment (e.g. some-page.md#section-name) is not validated — only that some-page.md itself exists. A broken same-page-exists-but-wrong-section link will not be caught by this tool; check those by hand or with a model-driven read for anything load-bearing. """ import re import sys from pathlib import Path LINK_RE = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)") FENCE_RE = re.compile(r"^(```|~~~)") def iter_links(md_path: Path): in_fence = False for line in md_path.read_text(encoding="utf-8").splitlines(): if FENCE_RE.match(line.strip()): in_fence = not in_fence continue if in_fence: continue for m in LINK_RE.finditer(line): yield m.group(1) def check(dirs): broken = [] for d in dirs: root = Path(d) if not root.exists(): print(f"warning: {d} does not exist, skipping") continue for page in sorted(root.rglob("*.md")): for target in iter_links(page): if target.startswith(("http://", "https://", "mailto:", "#")): continue path_part = target.split("#", 1)[0] if not path_part: continue resolved = (page.parent / path_part).resolve() if not resolved.exists(): broken.append((page, target)) return broken def main(): dirs = sys.argv[1:] or ["wiki"] broken = check(dirs) if broken: for page, target in broken: print(f"BROKEN {page}: {target}") print(f"\n{len(broken)} broken link(s).") return 1 print("All relative links resolve.") return 0 if __name__ == "__main__": sys.exit(main())