#!/usr/bin/env python3
"""check_links.py — verify that every relative markdown link in a knowledge base resolves.
Starter integrity-check tool for this project's knowledge-base framework (see ../framework.md §6).
Usage:
python3 check_links.py # checks wiki/ (default)
python3 check_links.py
... # checks the given directories
Exit code 0 when every relative link target exists; 1 otherwise, with one line
per broken link (file: target). External links (http/https/mailto), pure
anchors (#...), and links inside fenced code blocks are ignored. Run before
every commit.
Known limit, disclosed rather than silently assumed away: a link's #anchor
fragment (e.g. some-page.md#section-name) is not validated — only that
some-page.md itself exists. A broken same-page-exists-but-wrong-section
link will not be caught by this tool; check those by hand or with a
model-driven read for anything load-bearing.
"""
import re
import sys
from pathlib import Path
LINK_RE = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)")
FENCE_RE = re.compile(r"^(```|~~~)")
def iter_links(md_path: Path):
in_fence = False
for line in md_path.read_text(encoding="utf-8").splitlines():
if FENCE_RE.match(line.strip()):
in_fence = not in_fence
continue
if in_fence:
continue
for m in LINK_RE.finditer(line):
yield m.group(1)
def check(dirs):
broken = []
for d in dirs:
root = Path(d)
if not root.exists():
print(f"warning: {d} does not exist, skipping")
continue
for page in sorted(root.rglob("*.md")):
for target in iter_links(page):
if target.startswith(("http://", "https://", "mailto:", "#")):
continue
path_part = target.split("#", 1)[0]
if not path_part:
continue
resolved = (page.parent / path_part).resolve()
if not resolved.exists():
broken.append((page, target))
return broken
def main():
dirs = sys.argv[1:] or ["wiki"]
broken = check(dirs)
if broken:
for page, target in broken:
print(f"BROKEN {page}: {target}")
print(f"\n{len(broken)} broken link(s).")
return 1
print("All relative links resolve.")
return 0
if __name__ == "__main__":
sys.exit(main())