SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
5.7 KB · 127 lines python
Raw Blame History
1#!/usr/bin/env python32"""Verify links in docs/ and generated/ (sources[].url).34    python3 scripts/verify_links.py              # check external URLs (HEAD/GET, cached) + internal relative links5    python3 scripts/verify_links.py --offline    # internal links only + external URLs must exist in the crawl manifests67Writes reports/link-check.json and prints a summary. External checks use the local crawl manifests first8(sources/*/pages-manifest.json): a URL whose .md twin was fetched with HTTP 200 is considered valid without a9network call. Others are fetched once (GET, 20 s timeout) with a small thread pool.10"""11from __future__ import annotations1213import argparse14import concurrent.futures as cf15import json16import re17import sys18import urllib.request19from datetime import datetime, timezone20from pathlib import Path2122ROOT = Path(__file__).resolve().parent.parent23MD_LINK = re.compile(r"\[[^\]]*\]\(([^)\s]+)\)")24URL_IN_JSON = re.compile(r"https?://[^\s\"'<>)\]]+")25UA = "doc-api-atlas-linkcheck/1.0"26# API hosts appear in examples/endpoint records; they legitimately answer 401/404/405 without auth → not "broken links".27SKIP_HOSTS = ("api.openai.com", "api.anthropic.com", "api.chatgpt.com", "auth.openai.com", "mtls.auth.openai.com",28              "eu.api.openai.com", "us.api.openai.com", "bedrock-runtime", "aiplatform.googleapis.com",29              "localhost", "127.0.0.1", "example.com", "your-", "mcp.deepwiki.com", "httpbin.org",30              "services.ai.azure.com", "openai.azure.com", "wss://", "api.x.ai", "management-api.x.ai",31              "generativelanguage.googleapis.com", "aistudio.google.com/apikey")323334def skip(url: str) -> bool:35    return any(h in url for h in SKIP_HOSTS) or "{" in url or "<" in url363738def known_ok_urls() -> set[str]:39    ok: set[str] = set()40    for prov in ("openai", "anthropic", "xai", "gemini"):41        mf = ROOT / "sources" / prov / "pages-manifest.json"42        if mf.exists():43            for e in json.loads(mf.read_text()).get("pages", []):44                if e.get("status") == 200:45                    u = e["url"]46                    ok.add(u)47                    ok.add(u[:-3] if u.endswith(".md") else u)48                    if u.endswith(".md.txt"):49                        ok.add(u[:-7])50                        ok.add(u[:-4])51    return ok525354def check_url(url: str) -> tuple[str, int]:55    try:56        req = urllib.request.Request(url, headers={"User-Agent": UA}, method="GET")57        with urllib.request.urlopen(req, timeout=20) as r:58            return url, r.status59    except urllib.error.HTTPError as e:  # type: ignore[attr-defined]60        return url, e.code61    except Exception:  # noqa: BLE00162        return url, 0636465def main() -> int:66    ap = argparse.ArgumentParser()67    ap.add_argument("--offline", action="store_true")68    ap.add_argument("--workers", type=int, default=8)69    args = ap.parse_args()7071    internal_broken: list[dict] = []72    external: dict[str, set[str]] = {}73    md_files = list((ROOT / "docs").rglob("*.md")) + [ROOT / "README.md"] + list((ROOT / "reports").glob("*.md"))74    for f in md_files:75        if not f.exists():76            continue77        text = f.read_text(errors="replace")78        for m in MD_LINK.finditer(text):79            link = m.group(1).split("#")[0]80            if not link or link.startswith("mailto:"):81                continue82            if link.startswith("http"):83                if not skip(link):84                    external.setdefault(link.rstrip("/").rstrip("."), set()).add(str(f.relative_to(ROOT)))85            elif "|" in link or "[" in link or link.startswith(("ws", "wss")) or len(link) < 3 or ("/" not in link and "." not in link):86                continue  # table cell / pseudo link, not a file reference87            else:88                target = (f.parent / link).resolve()89                if not target.exists():90                    internal_broken.append({"file": str(f.relative_to(ROOT)), "link": link})91    for f in (ROOT / "generated").rglob("*.json"):92        if "fragments" in f.parts:93            continue94        for u in URL_IN_JSON.findall(f.read_text(errors="replace")):95            u = u.split("#")[0].rstrip("\\;,.)").rstrip("/")96            if skip(u) or len(u) < 12:97                continue98            external.setdefault(u, set()).add(str(f.relative_to(ROOT)))99100    ok = known_ok_urls()101    to_check = [u for u in external if u not in ok and (u + ".md") not in ok]102    results: dict[str, int] = {u: 200 for u in external if u not in to_check}103    if not args.offline and to_check:104        with cf.ThreadPoolExecutor(max_workers=args.workers) as ex:105            for u, st in ex.map(check_url, sorted(to_check)):106                results[u] = st107    else:108        for u in to_check:109            results[u] = -1  # unknown (offline)110111    broken_ext = [{"url": u, "status": st, "referrers": sorted(external[u])[:5]}112                  for u, st in results.items() if st not in (200, 301, 302, 303, 307, 308, -1)]113    report = {"checked_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),114              "markdown_files": len(md_files), "external_urls": len(external),115              "external_verified_via_crawl": len(external) - len(to_check),116              "external_fetched": 0 if args.offline else len(to_check),117              "external_unknown_offline": len(to_check) if args.offline else 0,118              "internal_broken": internal_broken, "external_broken": sorted(broken_ext, key=lambda x: x["url"])}119    (ROOT / "reports").mkdir(exist_ok=True)120    (ROOT / "reports" / "link-check.json").write_text(json.dumps(report, indent=1))121    print(json.dumps({k: (v if not isinstance(v, list) else len(v)) for k, v in report.items()}, indent=1))122    return 1 if internal_broken else 0123124125if __name__ == "__main__":126    sys.exit(main())127