#!/usr/bin/env python3 """Crawl official documentation pages listed in llms.txt indexes. Usage: python3 scripts/crawl_docs.py openai python3 scripts/crawl_docs.py anthropic python3 scripts/crawl_docs.py openai --extra-urls file.txt # add URLs not in the index Downloads every Markdown page referenced by the provider's llms.txt index files into sources//pages/.md and writes sources//pages-manifest.json (url, local path, sha256, size, http status, retrieved_at). Re-running detects changed pages (sha256 diff) and records them in sources//pages-changes.json so that scripts/update_atlas.py can produce reports/changes.md. No credentials are used. Only public documentation is fetched. """ from __future__ import annotations import argparse import concurrent.futures as cf import hashlib import json import re import sys import time import urllib.request from datetime import datetime, timezone from pathlib import Path ROOT = Path(__file__).resolve().parent.parent UA = "Mozilla/5.0 (compatible; doc-api-atlas/1.0; documentation crawler)" INDEXES = { "openai": { "index_files": [ "sources/openai/docs-llms.txt", "sources/openai/reference-llms.txt", "sources/openai/workspace-agents-llms.txt", ], "index_urls": [ "https://developers.openai.com/api/docs/llms.txt", "https://developers.openai.com/api/reference/llms.txt", "https://developers.openai.com/workspace-agents/llms.txt", ], "url_prefix": "https://developers.openai.com/", "allow": re.compile(r"^https://developers\.openai\.com/(api|workspace-agents)/.*\.md$"), }, "anthropic": { "index_files": ["sources/anthropic/llms.txt"], "index_urls": ["https://platform.claude.com/llms.txt"], "url_prefix": "https://platform.claude.com/docs/en/", "allow": re.compile(r"^https://platform\.claude\.com/docs/en/.*\.md$"), }, "xai": { "index_files": ["sources/xai/llms.txt"], "index_urls": ["https://docs.x.ai/llms.txt"], "url_prefix": "https://docs.x.ai/", "allow": re.compile(r"^https://docs\.x\.ai/.*\.md$"), }, "gemini": { "index_files": ["sources/gemini/llms.txt"], "index_urls": ["https://ai.google.dev/gemini-api/docs/llms.txt"], "url_prefix": "https://ai.google.dev/", "allow": re.compile(r"^https://ai\.google\.dev/(gemini-api|api)/.*\.md\.txt$"), }, } LINK_RE = re.compile(r"\((https?://[^)\s]+)\)") def now() -> str: return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") def fetch(url: str, retries: int = 3, timeout: int = 60) -> tuple[int, bytes]: last_exc: Exception | None = None for attempt in range(retries): try: req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept": "text/markdown, text/plain, */*"}) with urllib.request.urlopen(req, timeout=timeout) as r: return r.status, r.read() except urllib.error.HTTPError as e: # type: ignore[attr-defined] if e.code in (429, 500, 502, 503, 504) and attempt < retries - 1: time.sleep(1.5 * (attempt + 1)) continue return e.code, b"" except Exception as e: # noqa: BLE001 last_exc = e time.sleep(1.5 * (attempt + 1)) print(f"!! {url}: {last_exc}", file=sys.stderr) return 0, b"" def slug_for(url: str, prefix: str) -> str: s = url[len(prefix):] if url.startswith(prefix) else re.sub(r"^https?://", "", url) s = s.strip("/") if s.endswith(".md.txt"): s = s[:-4] if not s.endswith(".md"): s += ".md" return s def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("provider", choices=INDEXES) ap.add_argument("--extra-urls", type=Path, help="text file with one extra URL per line") ap.add_argument("--refresh-index", action="store_true", help="re-download the llms.txt indexes first") ap.add_argument("--workers", type=int, default=12) ap.add_argument("--follow", action="store_true", help="also fetch doc pages linked from fetched pages") ap.add_argument("--max-depth", type=int, default=3) args = ap.parse_args() cfg = INDEXES[args.provider] out_dir = ROOT / "sources" / args.provider / "pages" out_dir.mkdir(parents=True, exist_ok=True) manifest_path = ROOT / "sources" / args.provider / "pages-manifest.json" changes_path = ROOT / "sources" / args.provider / "pages-changes.json" if args.refresh_index: for f, u in zip(cfg["index_files"], cfg["index_urls"]): st, body = fetch(u) if st == 200: (ROOT / f).write_bytes(body) print(f"index {u} -> {f} ({len(body)} bytes)") urls: set[str] = set() for f in cfg["index_files"]: p = ROOT / f if not p.exists(): print(f"missing index {p}", file=sys.stderr) continue for m in LINK_RE.finditer(p.read_text(errors="replace")): u = m.group(1) if cfg["allow"].match(u): urls.add(u) if args.extra_urls and args.extra_urls.exists(): for line in args.extra_urls.read_text().splitlines(): line = line.strip() if line and not line.startswith("#"): urls.add(line) prev: dict[str, dict] = {} if manifest_path.exists(): prev = {e["url"]: e for e in json.loads(manifest_path.read_text()).get("pages", [])} # Links inside pages that point to other doc pages (with or without .md, with anchors/queries). site_link_re = { "openai": re.compile(r"\(((?:https://developers\.openai\.com)?/(?:api|workspace-agents)/[^)\s#?]+)"), "anthropic": re.compile(r"\(((?:https://platform\.claude\.com)?/docs/en/[^)\s#?]+)"), "xai": re.compile(r"\(((?:https://docs\.x\.ai)?/(?:developers|api|build|overview|grok)[^)\s#?]*)"), "gemini": re.compile(r"\(((?:https://ai\.google\.dev)?/(?:gemini-api|api)/[^)\s#?]+)"), }[args.provider] host = {"openai": "https://developers.openai.com", "anthropic": "https://platform.claude.com", "xai": "https://docs.x.ai", "gemini": "https://ai.google.dev"}[args.provider] suffix = ".md.txt" if args.provider == "gemini" else ".md" def normalize(link: str) -> str | None: if link.startswith("/"): link = host + link link = link.rstrip("/") if link.endswith(".md.txt"): pass elif link.endswith(".md") and suffix == ".md.txt": link += ".txt" elif not link.endswith(suffix): link += suffix # skip non-doc assets if re.search(r"\.(png|jpg|jpeg|gif|svg|pdf|zip|json|yaml|yml)\.md(\.txt)?$", link): return None return link if cfg["allow"].match(link) else None print(f"{args.provider}: {len(urls)} pages to fetch (index)") entries: list[dict] = [] changes = {"run_at": now(), "new": [], "changed": [], "removed": [], "failed": []} discovered: set[str] = set() def work(url: str) -> dict: st, body = fetch(url) slug = slug_for(url, cfg["url_prefix"]) entry = {"url": url, "path": f"sources/{args.provider}/pages/{slug}", "status": st, "size": len(body), "sha256": hashlib.sha256(body).hexdigest() if body else None, "retrieved_at": now()} if st == 200 and body: dest = out_dir / slug dest.parent.mkdir(parents=True, exist_ok=True) dest.write_bytes(body) if args.follow: for m in site_link_re.finditer(body.decode("utf-8", "replace")): n = normalize(m.group(1)) if n: discovered.add(n) return entry pending = sorted(urls) done: set[str] = set() depth = 0 while pending and depth <= args.max_depth: with cf.ThreadPoolExecutor(max_workers=args.workers) as ex: for i, entry in enumerate(ex.map(work, pending), 1): entries.append(entry) done.add(entry["url"]) if entry["status"] != 200: changes["failed"].append({"url": entry["url"], "status": entry["status"], "depth": depth}) elif entry["url"] not in prev: changes["new"].append(entry["url"]) elif prev[entry["url"]].get("sha256") != entry["sha256"]: changes["changed"].append(entry["url"]) if i % 50 == 0: print(f" depth {depth}: {i}/{len(pending)}") if not args.follow: break pending = sorted(u for u in discovered if u not in done) depth += 1 if pending: print(f"{args.provider}: depth {depth}: {len(pending)} newly linked pages") urls = done # Pages that 404 at depth>0 are just dangling links, not index regressions; keep only 200s in manifest. entries = [e for e in entries if e["status"] == 200 or e["url"] in set(prev)] for u in prev: if u not in urls: changes["removed"].append(u) manifest = {"provider": args.provider, "generated_at": now(), "page_count": len(entries), "ok_count": sum(1 for e in entries if e["status"] == 200), "pages": sorted(entries, key=lambda e: e["url"])} manifest_path.write_text(json.dumps(manifest, indent=2)) changes_path.write_text(json.dumps(changes, indent=2)) print(f"ok={manifest['ok_count']} failed={len(changes['failed'])} new={len(changes['new'])} " f"changed={len(changes['changed'])} removed={len(changes['removed'])}") return 0 if __name__ == "__main__": sys.exit(main())