#!/usr/bin/env python3 """Verify links in docs/ and generated/ (sources[].url). python3 scripts/verify_links.py # check external URLs (HEAD/GET, cached) + internal relative links python3 scripts/verify_links.py --offline # internal links only + external URLs must exist in the crawl manifests Writes reports/link-check.json and prints a summary. External checks use the local crawl manifests first (sources/*/pages-manifest.json): a URL whose .md twin was fetched with HTTP 200 is considered valid without a network call. Others are fetched once (GET, 20 s timeout) with a small thread pool. """ from __future__ import annotations import argparse import concurrent.futures as cf import json import re import sys import urllib.request from datetime import datetime, timezone from pathlib import Path ROOT = Path(__file__).resolve().parent.parent MD_LINK = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)") URL_IN_JSON = re.compile(r"https?://[^\s\"'<>)\]]+") UA = "doc-api-atlas-linkcheck/1.0" # API hosts appear in examples/endpoint records; they legitimately answer 401/404/405 without auth → not "broken links". SKIP_HOSTS = ("api.openai.com", "api.anthropic.com", "api.chatgpt.com", "auth.openai.com", "mtls.auth.openai.com", "eu.api.openai.com", "us.api.openai.com", "bedrock-runtime", "aiplatform.googleapis.com", "localhost", "127.0.0.1", "example.com", "your-", "mcp.deepwiki.com", "httpbin.org", "services.ai.azure.com", "openai.azure.com", "wss://", "api.x.ai", "management-api.x.ai", "generativelanguage.googleapis.com", "aistudio.google.com/apikey") def skip(url: str) -> bool: return any(h in url for h in SKIP_HOSTS) or "{" in url or "<" in url def known_ok_urls() -> set[str]: ok: set[str] = set() for prov in ("openai", "anthropic", "xai", "gemini"): mf = ROOT / "sources" / prov / "pages-manifest.json" if mf.exists(): for e in json.loads(mf.read_text()).get("pages", []): if e.get("status") == 200: u = e["url"] ok.add(u) ok.add(u[:-3] if u.endswith(".md") else u) if u.endswith(".md.txt"): ok.add(u[:-7]) ok.add(u[:-4]) return ok def check_url(url: str) -> tuple[str, int]: try: req = urllib.request.Request(url, headers={"User-Agent": UA}, method="GET") with urllib.request.urlopen(req, timeout=20) as r: return url, r.status except urllib.error.HTTPError as e: # type: ignore[attr-defined] return url, e.code except Exception: # noqa: BLE001 return url, 0 def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("--offline", action="store_true") ap.add_argument("--workers", type=int, default=8) args = ap.parse_args() internal_broken: list[dict] = [] external: dict[str, set[str]] = {} md_files = list((ROOT / "docs").rglob("*.md")) + [ROOT / "README.md"] + list((ROOT / "reports").glob("*.md")) for f in md_files: if not f.exists(): continue text = f.read_text(errors="replace") for m in MD_LINK.finditer(text): link = m.group(1).split("#")[0] if not link or link.startswith("mailto:"): continue if link.startswith("http"): if not skip(link): external.setdefault(link.rstrip("/").rstrip("."), set()).add(str(f.relative_to(ROOT))) elif "|" in link or "[" in link or link.startswith(("ws", "wss")) or len(link) < 3 or ("/" not in link and "." not in link): continue # table cell / pseudo link, not a file reference else: target = (f.parent / link).resolve() if not target.exists(): internal_broken.append({"file": str(f.relative_to(ROOT)), "link": link}) for f in (ROOT / "generated").rglob("*.json"): if "fragments" in f.parts: continue for u in URL_IN_JSON.findall(f.read_text(errors="replace")): u = u.split("#")[0].rstrip("\\;,.)").rstrip("/") if skip(u) or len(u) < 12: continue external.setdefault(u, set()).add(str(f.relative_to(ROOT))) ok = known_ok_urls() to_check = [u for u in external if u not in ok and (u + ".md") not in ok] results: dict[str, int] = {u: 200 for u in external if u not in to_check} if not args.offline and to_check: with cf.ThreadPoolExecutor(max_workers=args.workers) as ex: for u, st in ex.map(check_url, sorted(to_check)): results[u] = st else: for u in to_check: results[u] = -1 # unknown (offline) broken_ext = [{"url": u, "status": st, "referrers": sorted(external[u])[:5]} for u, st in results.items() if st not in (200, 301, 302, 303, 307, 308, -1)] report = {"checked_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), "markdown_files": len(md_files), "external_urls": len(external), "external_verified_via_crawl": len(external) - len(to_check), "external_fetched": 0 if args.offline else len(to_check), "external_unknown_offline": len(to_check) if args.offline else 0, "internal_broken": internal_broken, "external_broken": sorted(broken_ext, key=lambda x: x["url"])} (ROOT / "reports").mkdir(exist_ok=True) (ROOT / "reports" / "link-check.json").write_text(json.dumps(report, indent=1)) print(json.dumps({k: (v if not isinstance(v, list) else len(v)) for k, v in report.items()}, indent=1)) return 1 if internal_broken else 0 if __name__ == "__main__": sys.exit(main())