SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
9.6 KB · 236 lines python
Raw Blame History
1#!/usr/bin/env python32"""Crawl official documentation pages listed in llms.txt indexes.34Usage:5  python3 scripts/crawl_docs.py openai6  python3 scripts/crawl_docs.py anthropic7  python3 scripts/crawl_docs.py openai --extra-urls file.txt   # add URLs not in the index89Downloads every Markdown page referenced by the provider's llms.txt index files into10sources/<provider>/pages/<slug>.md and writes sources/<provider>/pages-manifest.json11(url, local path, sha256, size, http status, retrieved_at). Re-running detects changed12pages (sha256 diff) and records them in sources/<provider>/pages-changes.json so that13scripts/update_atlas.py can produce reports/changes.md.1415No credentials are used. Only public documentation is fetched.16"""17from __future__ import annotations1819import argparse20import concurrent.futures as cf21import hashlib22import json23import re24import sys25import time26import urllib.request27from datetime import datetime, timezone28from pathlib import Path2930ROOT = Path(__file__).resolve().parent.parent31UA = "Mozilla/5.0 (compatible; doc-api-atlas/1.0; documentation crawler)"3233INDEXES = {34    "openai": {35        "index_files": [36            "sources/openai/docs-llms.txt",37            "sources/openai/reference-llms.txt",38            "sources/openai/workspace-agents-llms.txt",39        ],40        "index_urls": [41            "https://developers.openai.com/api/docs/llms.txt",42            "https://developers.openai.com/api/reference/llms.txt",43            "https://developers.openai.com/workspace-agents/llms.txt",44        ],45        "url_prefix": "https://developers.openai.com/",46        "allow": re.compile(r"^https://developers\.openai\.com/(api|workspace-agents)/.*\.md$"),47    },48    "anthropic": {49        "index_files": ["sources/anthropic/llms.txt"],50        "index_urls": ["https://platform.claude.com/llms.txt"],51        "url_prefix": "https://platform.claude.com/docs/en/",52        "allow": re.compile(r"^https://platform\.claude\.com/docs/en/.*\.md$"),53    },54    "xai": {55        "index_files": ["sources/xai/llms.txt"],56        "index_urls": ["https://docs.x.ai/llms.txt"],57        "url_prefix": "https://docs.x.ai/",58        "allow": re.compile(r"^https://docs\.x\.ai/.*\.md$"),59    },60    "gemini": {61        "index_files": ["sources/gemini/llms.txt"],62        "index_urls": ["https://ai.google.dev/gemini-api/docs/llms.txt"],63        "url_prefix": "https://ai.google.dev/",64        "allow": re.compile(r"^https://ai\.google\.dev/(gemini-api|api)/.*\.md\.txt$"),65    },66}6768LINK_RE = re.compile(r"\((https?://[^)\s]+)\)")697071def now() -> str:72    return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")737475def fetch(url: str, retries: int = 3, timeout: int = 60) -> tuple[int, bytes]:76    last_exc: Exception | None = None77    for attempt in range(retries):78        try:79            req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept": "text/markdown, text/plain, */*"})80            with urllib.request.urlopen(req, timeout=timeout) as r:81                return r.status, r.read()82        except urllib.error.HTTPError as e:  # type: ignore[attr-defined]83            if e.code in (429, 500, 502, 503, 504) and attempt < retries - 1:84                time.sleep(1.5 * (attempt + 1))85                continue86            return e.code, b""87        except Exception as e:  # noqa: BLE00188            last_exc = e89            time.sleep(1.5 * (attempt + 1))90    print(f"!! {url}: {last_exc}", file=sys.stderr)91    return 0, b""929394def slug_for(url: str, prefix: str) -> str:95    s = url[len(prefix):] if url.startswith(prefix) else re.sub(r"^https?://", "", url)96    s = s.strip("/")97    if s.endswith(".md.txt"):98        s = s[:-4]99    if not s.endswith(".md"):100        s += ".md"101    return s102103104def main() -> int:105    ap = argparse.ArgumentParser()106    ap.add_argument("provider", choices=INDEXES)107    ap.add_argument("--extra-urls", type=Path, help="text file with one extra URL per line")108    ap.add_argument("--refresh-index", action="store_true", help="re-download the llms.txt indexes first")109    ap.add_argument("--workers", type=int, default=12)110    ap.add_argument("--follow", action="store_true", help="also fetch doc pages linked from fetched pages")111    ap.add_argument("--max-depth", type=int, default=3)112    args = ap.parse_args()113114    cfg = INDEXES[args.provider]115    out_dir = ROOT / "sources" / args.provider / "pages"116    out_dir.mkdir(parents=True, exist_ok=True)117    manifest_path = ROOT / "sources" / args.provider / "pages-manifest.json"118    changes_path = ROOT / "sources" / args.provider / "pages-changes.json"119120    if args.refresh_index:121        for f, u in zip(cfg["index_files"], cfg["index_urls"]):122            st, body = fetch(u)123            if st == 200:124                (ROOT / f).write_bytes(body)125                print(f"index {u} -> {f} ({len(body)} bytes)")126127    urls: set[str] = set()128    for f in cfg["index_files"]:129        p = ROOT / f130        if not p.exists():131            print(f"missing index {p}", file=sys.stderr)132            continue133        for m in LINK_RE.finditer(p.read_text(errors="replace")):134            u = m.group(1)135            if cfg["allow"].match(u):136                urls.add(u)137    if args.extra_urls and args.extra_urls.exists():138        for line in args.extra_urls.read_text().splitlines():139            line = line.strip()140            if line and not line.startswith("#"):141                urls.add(line)142143    prev: dict[str, dict] = {}144    if manifest_path.exists():145        prev = {e["url"]: e for e in json.loads(manifest_path.read_text()).get("pages", [])}146147    # Links inside pages that point to other doc pages (with or without .md, with anchors/queries).148    site_link_re = {149        "openai": re.compile(r"\(((?:https://developers\.openai\.com)?/(?:api|workspace-agents)/[^)\s#?]+)"),150        "anthropic": re.compile(r"\(((?:https://platform\.claude\.com)?/docs/en/[^)\s#?]+)"),151        "xai": re.compile(r"\(((?:https://docs\.x\.ai)?/(?:developers|api|build|overview|grok)[^)\s#?]*)"),152        "gemini": re.compile(r"\(((?:https://ai\.google\.dev)?/(?:gemini-api|api)/[^)\s#?]+)"),153    }[args.provider]154    host = {"openai": "https://developers.openai.com", "anthropic": "https://platform.claude.com",155            "xai": "https://docs.x.ai", "gemini": "https://ai.google.dev"}[args.provider]156    suffix = ".md.txt" if args.provider == "gemini" else ".md"157158    def normalize(link: str) -> str | None:159        if link.startswith("/"):160            link = host + link161        link = link.rstrip("/")162        if link.endswith(".md.txt"):163            pass164        elif link.endswith(".md") and suffix == ".md.txt":165            link += ".txt"166        elif not link.endswith(suffix):167            link += suffix168        # skip non-doc assets169        if re.search(r"\.(png|jpg|jpeg|gif|svg|pdf|zip|json|yaml|yml)\.md(\.txt)?$", link):170            return None171        return link if cfg["allow"].match(link) else None172173    print(f"{args.provider}: {len(urls)} pages to fetch (index)")174    entries: list[dict] = []175    changes = {"run_at": now(), "new": [], "changed": [], "removed": [], "failed": []}176    discovered: set[str] = set()177178    def work(url: str) -> dict:179        st, body = fetch(url)180        slug = slug_for(url, cfg["url_prefix"])181        entry = {"url": url, "path": f"sources/{args.provider}/pages/{slug}", "status": st,182                 "size": len(body), "sha256": hashlib.sha256(body).hexdigest() if body else None,183                 "retrieved_at": now()}184        if st == 200 and body:185            dest = out_dir / slug186            dest.parent.mkdir(parents=True, exist_ok=True)187            dest.write_bytes(body)188            if args.follow:189                for m in site_link_re.finditer(body.decode("utf-8", "replace")):190                    n = normalize(m.group(1))191                    if n:192                        discovered.add(n)193        return entry194195    pending = sorted(urls)196    done: set[str] = set()197    depth = 0198    while pending and depth <= args.max_depth:199        with cf.ThreadPoolExecutor(max_workers=args.workers) as ex:200            for i, entry in enumerate(ex.map(work, pending), 1):201                entries.append(entry)202                done.add(entry["url"])203                if entry["status"] != 200:204                    changes["failed"].append({"url": entry["url"], "status": entry["status"], "depth": depth})205                elif entry["url"] not in prev:206                    changes["new"].append(entry["url"])207                elif prev[entry["url"]].get("sha256") != entry["sha256"]:208                    changes["changed"].append(entry["url"])209                if i % 50 == 0:210                    print(f"  depth {depth}: {i}/{len(pending)}")211        if not args.follow:212            break213        pending = sorted(u for u in discovered if u not in done)214        depth += 1215        if pending:216            print(f"{args.provider}: depth {depth}: {len(pending)} newly linked pages")217    urls = done218    # Pages that 404 at depth>0 are just dangling links, not index regressions; keep only 200s in manifest.219    entries = [e for e in entries if e["status"] == 200 or e["url"] in set(prev)]220221    for u in prev:222        if u not in urls:223            changes["removed"].append(u)224225    manifest = {"provider": args.provider, "generated_at": now(), "page_count": len(entries),226                "ok_count": sum(1 for e in entries if e["status"] == 200), "pages": sorted(entries, key=lambda e: e["url"])}227    manifest_path.write_text(json.dumps(manifest, indent=2))228    changes_path.write_text(json.dumps(changes, indent=2))229    print(f"ok={manifest['ok_count']} failed={len(changes['failed'])} new={len(changes['new'])} "230          f"changed={len(changes['changed'])} removed={len(changes['removed'])}")231    return 0232233234if __name__ == "__main__":235    sys.exit(main())236