Python 88.3%
TypeScript 7.6%
Shell 4.1%
1#!/usr/bin/env python32"""Crawl official documentation pages listed in llms.txt indexes.34Usage:5 python3 scripts/crawl_docs.py openai6 python3 scripts/crawl_docs.py anthropic7 python3 scripts/crawl_docs.py openai --extra-urls file.txt # add URLs not in the index89Downloads every Markdown page referenced by the provider's llms.txt index files into10sources/<provider>/pages/<slug>.md and writes sources/<provider>/pages-manifest.json11(url, local path, sha256, size, http status, retrieved_at). Re-running detects changed12pages (sha256 diff) and records them in sources/<provider>/pages-changes.json so that13scripts/update_atlas.py can produce reports/changes.md.1415No credentials are used. Only public documentation is fetched.16"""17from __future__ import annotations1819import argparse20import concurrent.futures as cf21import hashlib22import json23import re24import sys25import time26import urllib.request27from datetime import datetime, timezone28from pathlib import Path2930ROOT = Path(__file__).resolve().parent.parent31UA = "Mozilla/5.0 (compatible; doc-api-atlas/1.0; documentation crawler)"3233INDEXES = {34 "openai": {35 "index_files": [36 "sources/openai/docs-llms.txt",37 "sources/openai/reference-llms.txt",38 "sources/openai/workspace-agents-llms.txt",39 ],40 "index_urls": [41 "https://developers.openai.com/api/docs/llms.txt",42 "https://developers.openai.com/api/reference/llms.txt",43 "https://developers.openai.com/workspace-agents/llms.txt",44 ],45 "url_prefix": "https://developers.openai.com/",46 "allow": re.compile(r"^https://developers\.openai\.com/(api|workspace-agents)/.*\.md$"),47 },48 "anthropic": {49 "index_files": ["sources/anthropic/llms.txt"],50 "index_urls": ["https://platform.claude.com/llms.txt"],51 "url_prefix": "https://platform.claude.com/docs/en/",52 "allow": re.compile(r"^https://platform\.claude\.com/docs/en/.*\.md$"),53 },54 "xai": {55 "index_files": ["sources/xai/llms.txt"],56 "index_urls": ["https://docs.x.ai/llms.txt"],57 "url_prefix": "https://docs.x.ai/",58 "allow": re.compile(r"^https://docs\.x\.ai/.*\.md$"),59 },60 "gemini": {61 "index_files": ["sources/gemini/llms.txt"],62 "index_urls": ["https://ai.google.dev/gemini-api/docs/llms.txt"],63 "url_prefix": "https://ai.google.dev/",64 "allow": re.compile(r"^https://ai\.google\.dev/(gemini-api|api)/.*\.md\.txt$"),65 },66}6768LINK_RE = re.compile(r"\((https?://[^)\s]+)\)")697071def now() -> str:72 return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")737475def fetch(url: str, retries: int = 3, timeout: int = 60) -> tuple[int, bytes]:76 last_exc: Exception | None = None77 for attempt in range(retries):78 try:79 req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept": "text/markdown, text/plain, */*"})80 with urllib.request.urlopen(req, timeout=timeout) as r:81 return r.status, r.read()82 except urllib.error.HTTPError as e: # type: ignore[attr-defined]83 if e.code in (429, 500, 502, 503, 504) and attempt < retries - 1:84 time.sleep(1.5 * (attempt + 1))85 continue86 return e.code, b""87 except Exception as e: # noqa: BLE00188 last_exc = e89 time.sleep(1.5 * (attempt + 1))90 print(f"!! {url}: {last_exc}", file=sys.stderr)91 return 0, b""929394def slug_for(url: str, prefix: str) -> str:95 s = url[len(prefix):] if url.startswith(prefix) else re.sub(r"^https?://", "", url)96 s = s.strip("/")97 if s.endswith(".md.txt"):98 s = s[:-4]99 if not s.endswith(".md"):100 s += ".md"101 return s102103104def main() -> int:105 ap = argparse.ArgumentParser()106 ap.add_argument("provider", choices=INDEXES)107 ap.add_argument("--extra-urls", type=Path, help="text file with one extra URL per line")108 ap.add_argument("--refresh-index", action="store_true", help="re-download the llms.txt indexes first")109 ap.add_argument("--workers", type=int, default=12)110 ap.add_argument("--follow", action="store_true", help="also fetch doc pages linked from fetched pages")111 ap.add_argument("--max-depth", type=int, default=3)112 args = ap.parse_args()113114 cfg = INDEXES[args.provider]115 out_dir = ROOT / "sources" / args.provider / "pages"116 out_dir.mkdir(parents=True, exist_ok=True)117 manifest_path = ROOT / "sources" / args.provider / "pages-manifest.json"118 changes_path = ROOT / "sources" / args.provider / "pages-changes.json"119120 if args.refresh_index:121 for f, u in zip(cfg["index_files"], cfg["index_urls"]):122 st, body = fetch(u)123 if st == 200:124 (ROOT / f).write_bytes(body)125 print(f"index {u} -> {f} ({len(body)} bytes)")126127 urls: set[str] = set()128 for f in cfg["index_files"]:129 p = ROOT / f130 if not p.exists():131 print(f"missing index {p}", file=sys.stderr)132 continue133 for m in LINK_RE.finditer(p.read_text(errors="replace")):134 u = m.group(1)135 if cfg["allow"].match(u):136 urls.add(u)137 if args.extra_urls and args.extra_urls.exists():138 for line in args.extra_urls.read_text().splitlines():139 line = line.strip()140 if line and not line.startswith("#"):141 urls.add(line)142143 prev: dict[str, dict] = {}144 if manifest_path.exists():145 prev = {e["url"]: e for e in json.loads(manifest_path.read_text()).get("pages", [])}146147 # Links inside pages that point to other doc pages (with or without .md, with anchors/queries).148 site_link_re = {149 "openai": re.compile(r"\(((?:https://developers\.openai\.com)?/(?:api|workspace-agents)/[^)\s#?]+)"),150 "anthropic": re.compile(r"\(((?:https://platform\.claude\.com)?/docs/en/[^)\s#?]+)"),151 "xai": re.compile(r"\(((?:https://docs\.x\.ai)?/(?:developers|api|build|overview|grok)[^)\s#?]*)"),152 "gemini": re.compile(r"\(((?:https://ai\.google\.dev)?/(?:gemini-api|api)/[^)\s#?]+)"),153 }[args.provider]154 host = {"openai": "https://developers.openai.com", "anthropic": "https://platform.claude.com",155 "xai": "https://docs.x.ai", "gemini": "https://ai.google.dev"}[args.provider]156 suffix = ".md.txt" if args.provider == "gemini" else ".md"157158 def normalize(link: str) -> str | None:159 if link.startswith("/"):160 link = host + link161 link = link.rstrip("/")162 if link.endswith(".md.txt"):163 pass164 elif link.endswith(".md") and suffix == ".md.txt":165 link += ".txt"166 elif not link.endswith(suffix):167 link += suffix168 # skip non-doc assets169 if re.search(r"\.(png|jpg|jpeg|gif|svg|pdf|zip|json|yaml|yml)\.md(\.txt)?$", link):170 return None171 return link if cfg["allow"].match(link) else None172173 print(f"{args.provider}: {len(urls)} pages to fetch (index)")174 entries: list[dict] = []175 changes = {"run_at": now(), "new": [], "changed": [], "removed": [], "failed": []}176 discovered: set[str] = set()177178 def work(url: str) -> dict:179 st, body = fetch(url)180 slug = slug_for(url, cfg["url_prefix"])181 entry = {"url": url, "path": f"sources/{args.provider}/pages/{slug}", "status": st,182 "size": len(body), "sha256": hashlib.sha256(body).hexdigest() if body else None,183 "retrieved_at": now()}184 if st == 200 and body:185 dest = out_dir / slug186 dest.parent.mkdir(parents=True, exist_ok=True)187 dest.write_bytes(body)188 if args.follow:189 for m in site_link_re.finditer(body.decode("utf-8", "replace")):190 n = normalize(m.group(1))191 if n:192 discovered.add(n)193 return entry194195 pending = sorted(urls)196 done: set[str] = set()197 depth = 0198 while pending and depth <= args.max_depth:199 with cf.ThreadPoolExecutor(max_workers=args.workers) as ex:200 for i, entry in enumerate(ex.map(work, pending), 1):201 entries.append(entry)202 done.add(entry["url"])203 if entry["status"] != 200:204 changes["failed"].append({"url": entry["url"], "status": entry["status"], "depth": depth})205 elif entry["url"] not in prev:206 changes["new"].append(entry["url"])207 elif prev[entry["url"]].get("sha256") != entry["sha256"]:208 changes["changed"].append(entry["url"])209 if i % 50 == 0:210 print(f" depth {depth}: {i}/{len(pending)}")211 if not args.follow:212 break213 pending = sorted(u for u in discovered if u not in done)214 depth += 1215 if pending:216 print(f"{args.provider}: depth {depth}: {len(pending)} newly linked pages")217 urls = done218 # Pages that 404 at depth>0 are just dangling links, not index regressions; keep only 200s in manifest.219 entries = [e for e in entries if e["status"] == 200 or e["url"] in set(prev)]220221 for u in prev:222 if u not in urls:223 changes["removed"].append(u)224225 manifest = {"provider": args.provider, "generated_at": now(), "page_count": len(entries),226 "ok_count": sum(1 for e in entries if e["status"] == 200), "pages": sorted(entries, key=lambda e: e["url"])}227 manifest_path.write_text(json.dumps(manifest, indent=2))228 changes_path.write_text(json.dumps(changes, indent=2))229 print(f"ok={manifest['ok_count']} failed={len(changes['failed'])} new={len(changes['new'])} "230 f"changed={len(changes['changed'])} removed={len(changes['removed'])}")231 return 0232233234if __name__ == "__main__":235 sys.exit(main())236