#!/usr/bin/env python3 """Download https://docs.x.ai/llms-full.txt and split it into per-page files under sources/xai/pages/. docs.x.ai refuses direct `.md` page fetches from non-browser HTTP clients (404 for Python urllib on 2026-09-19), but publishes the complete documentation as a single export with `===/path===` page markers. This script rebuilds the same layout that scripts/crawl_docs.py produces for the other providers (pages/, pages-manifest.json, pages-changes.json with new/changed/removed) so that update_atlas.py can diff runs. python3 scripts/split_xai_llms_full.py # download + split python3 scripts/split_xai_llms_full.py --offline # split the already-downloaded sources/xai/llms-full.txt """ from __future__ import annotations import argparse import hashlib import json import re import subprocess import sys from datetime import datetime, timezone from pathlib import Path ROOT = Path(__file__).resolve().parent.parent SRC = ROOT / "sources" / "xai" / "llms-full.txt" INDEX = ROOT / "sources" / "xai" / "llms.txt" OUT = ROOT / "sources" / "xai" / "pages" MANIFEST = ROOT / "sources" / "xai" / "pages-manifest.json" CHANGES = ROOT / "sources" / "xai" / "pages-changes.json" def now() -> str: return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") def download() -> None: # curl with a browser UA works where urllib does not for url, dest in (("https://docs.x.ai/llms-full.txt", SRC), ("https://docs.x.ai/llms.txt", INDEX)): r = subprocess.run(["curl", "-sL", "-A", "Mozilla/5.0", "-o", str(dest), "-w", "%{http_code}", url], capture_output=True, text=True) print(f"{r.stdout} {url} -> {dest.relative_to(ROOT)} ({dest.stat().st_size if dest.exists() else 0} bytes)") def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("--offline", action="store_true") args = ap.parse_args() if not args.offline: download() if not SRC.exists(): print("missing sources/xai/llms-full.txt", file=sys.stderr) return 1 prev = {} if MANIFEST.exists(): prev = {e["url"]: e for e in json.loads(MANIFEST.read_text()).get("pages", [])} text = SRC.read_text(errors="replace") parts = re.split(r"^===(/[^=\n]*)===\s*$", text, flags=re.M) ts = now() entries, seen = [], set() changes = {"run_at": ts, "new": [], "changed": [], "removed": [], "failed": []} OUT.mkdir(parents=True, exist_ok=True) for i in range(1, len(parts), 2): path = parts[i].strip() body = parts[i + 1].strip() + "\n" url = f"https://docs.x.ai{path}" slug = path.strip("/") + ".md" dest = OUT / slug dest.parent.mkdir(parents=True, exist_ok=True) dest.write_text(f"\n" + body) sha = hashlib.sha256(body.encode()).hexdigest() entries.append({"url": url, "path": f"sources/xai/pages/{slug}", "status": 200, "size": len(body), "sha256": sha, "retrieved_at": ts, "via": "llms-full.txt"}) seen.add(url) if url not in prev: changes["new"].append(url) elif prev[url].get("sha256") != sha: changes["changed"].append(url) changes["removed"] = sorted(u for u in prev if u not in seen) MANIFEST.write_text(json.dumps({"provider": "xai", "generated_at": ts, "page_count": len(entries), "ok_count": len(entries), "source": "https://docs.x.ai/llms-full.txt", "pages": sorted(entries, key=lambda e: e["url"])}, indent=2)) CHANGES.write_text(json.dumps(changes, indent=2)) print(f"xai pages={len(entries)} new={len(changes['new'])} changed={len(changes['changed'])} removed={len(changes['removed'])}") return 0 if __name__ == "__main__": sys.exit(main())