Python 88.3%
TypeScript 7.6%
Shell 4.1%
1#!/usr/bin/env python32"""Verify links in docs/ and generated/ (sources[].url).34 python3 scripts/verify_links.py # check external URLs (HEAD/GET, cached) + internal relative links5 python3 scripts/verify_links.py --offline # internal links only + external URLs must exist in the crawl manifests67Writes reports/link-check.json and prints a summary. External checks use the local crawl manifests first8(sources/*/pages-manifest.json): a URL whose .md twin was fetched with HTTP 200 is considered valid without a9network call. Others are fetched once (GET, 20 s timeout) with a small thread pool.10"""11from __future__ import annotations1213import argparse14import concurrent.futures as cf15import json16import re17import sys18import urllib.request19from datetime import datetime, timezone20from pathlib import Path2122ROOT = Path(__file__).resolve().parent.parent23MD_LINK = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)")24URL_IN_JSON = re.compile(r"https?://[^\s\"'<>)\]]+")25UA = "doc-api-atlas-linkcheck/1.0"26# API hosts appear in examples/endpoint records; they legitimately answer 401/404/405 without auth → not "broken links".27SKIP_HOSTS = ("api.openai.com", "api.anthropic.com", "api.chatgpt.com", "auth.openai.com", "mtls.auth.openai.com",28 "eu.api.openai.com", "us.api.openai.com", "bedrock-runtime", "aiplatform.googleapis.com",29 "localhost", "127.0.0.1", "example.com", "your-", "mcp.deepwiki.com", "httpbin.org",30 "services.ai.azure.com", "openai.azure.com", "wss://", "api.x.ai", "management-api.x.ai",31 "generativelanguage.googleapis.com", "aistudio.google.com/apikey")323334def skip(url: str) -> bool:35 return any(h in url for h in SKIP_HOSTS) or "{" in url or "<" in url363738def known_ok_urls() -> set[str]:39 ok: set[str] = set()40 for prov in ("openai", "anthropic", "xai", "gemini"):41 mf = ROOT / "sources" / prov / "pages-manifest.json"42 if mf.exists():43 for e in json.loads(mf.read_text()).get("pages", []):44 if e.get("status") == 200:45 u = e["url"]46 ok.add(u)47 ok.add(u[:-3] if u.endswith(".md") else u)48 if u.endswith(".md.txt"):49 ok.add(u[:-7])50 ok.add(u[:-4])51 return ok525354def check_url(url: str) -> tuple[str, int]:55 try:56 req = urllib.request.Request(url, headers={"User-Agent": UA}, method="GET")57 with urllib.request.urlopen(req, timeout=20) as r:58 return url, r.status59 except urllib.error.HTTPError as e: # type: ignore[attr-defined]60 return url, e.code61 except Exception: # noqa: BLE00162 return url, 0636465def main() -> int:66 ap = argparse.ArgumentParser()67 ap.add_argument("--offline", action="store_true")68 ap.add_argument("--workers", type=int, default=8)69 args = ap.parse_args()7071 internal_broken: list[dict] = []72 external: dict[str, set[str]] = {}73 md_files = list((ROOT / "docs").rglob("*.md")) + [ROOT / "README.md"] + list((ROOT / "reports").glob("*.md"))74 for f in md_files:75 if not f.exists():76 continue77 text = f.read_text(errors="replace")78 for m in MD_LINK.finditer(text):79 link = m.group(1).split("#")[0]80 if not link or link.startswith("mailto:"):81 continue82 if link.startswith("http"):83 if not skip(link):84 external.setdefault(link.rstrip("/").rstrip("."), set()).add(str(f.relative_to(ROOT)))85 elif "|" in link or "[" in link or link.startswith(("ws", "wss")) or len(link) < 3 or ("/" not in link and "." not in link):86 continue # table cell / pseudo link, not a file reference87 else:88 target = (f.parent / link).resolve()89 if not target.exists():90 internal_broken.append({"file": str(f.relative_to(ROOT)), "link": link})91 for f in (ROOT / "generated").rglob("*.json"):92 if "fragments" in f.parts:93 continue94 for u in URL_IN_JSON.findall(f.read_text(errors="replace")):95 u = u.split("#")[0].rstrip("\\;,.)").rstrip("/")96 if skip(u) or len(u) < 12:97 continue98 external.setdefault(u, set()).add(str(f.relative_to(ROOT)))99100 ok = known_ok_urls()101 to_check = [u for u in external if u not in ok and (u + ".md") not in ok]102 results: dict[str, int] = {u: 200 for u in external if u not in to_check}103 if not args.offline and to_check:104 with cf.ThreadPoolExecutor(max_workers=args.workers) as ex:105 for u, st in ex.map(check_url, sorted(to_check)):106 results[u] = st107 else:108 for u in to_check:109 results[u] = -1 # unknown (offline)110111 broken_ext = [{"url": u, "status": st, "referrers": sorted(external[u])[:5]}112 for u, st in results.items() if st not in (200, 301, 302, 303, 307, 308, -1)]113 report = {"checked_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),114 "markdown_files": len(md_files), "external_urls": len(external),115 "external_verified_via_crawl": len(external) - len(to_check),116 "external_fetched": 0 if args.offline else len(to_check),117 "external_unknown_offline": len(to_check) if args.offline else 0,118 "internal_broken": internal_broken, "external_broken": sorted(broken_ext, key=lambda x: x["url"])}119 (ROOT / "reports").mkdir(exist_ok=True)120 (ROOT / "reports" / "link-check.json").write_text(json.dumps(report, indent=1))121 print(json.dumps({k: (v if not isinstance(v, list) else len(v)) for k, v in report.items()}, indent=1))122 return 1 if internal_broken else 0123124125if __name__ == "__main__":126 sys.exit(main())127