SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
3.8 KB · 90 lines python
Raw Blame History
1#!/usr/bin/env python32"""Download https://docs.x.ai/llms-full.txt and split it into per-page files under sources/xai/pages/.34docs.x.ai refuses direct `.md` page fetches from non-browser HTTP clients (404 for Python urllib on 2026-09-19),5but publishes the complete documentation as a single export with `===/path===` page markers. This script rebuilds6the same layout that scripts/crawl_docs.py produces for the other providers (pages/, pages-manifest.json,7pages-changes.json with new/changed/removed) so that update_atlas.py can diff runs.89    python3 scripts/split_xai_llms_full.py            # download + split10    python3 scripts/split_xai_llms_full.py --offline  # split the already-downloaded sources/xai/llms-full.txt11"""12from __future__ import annotations1314import argparse15import hashlib16import json17import re18import subprocess19import sys20from datetime import datetime, timezone21from pathlib import Path2223ROOT = Path(__file__).resolve().parent.parent24SRC = ROOT / "sources" / "xai" / "llms-full.txt"25INDEX = ROOT / "sources" / "xai" / "llms.txt"26OUT = ROOT / "sources" / "xai" / "pages"27MANIFEST = ROOT / "sources" / "xai" / "pages-manifest.json"28CHANGES = ROOT / "sources" / "xai" / "pages-changes.json"293031def now() -> str:32    return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")333435def download() -> None:36    # curl with a browser UA works where urllib does not37    for url, dest in (("https://docs.x.ai/llms-full.txt", SRC), ("https://docs.x.ai/llms.txt", INDEX)):38        r = subprocess.run(["curl", "-sL", "-A", "Mozilla/5.0", "-o", str(dest), "-w", "%{http_code}", url],39                           capture_output=True, text=True)40        print(f"{r.stdout} {url} -> {dest.relative_to(ROOT)} ({dest.stat().st_size if dest.exists() else 0} bytes)")414243def main() -> int:44    ap = argparse.ArgumentParser()45    ap.add_argument("--offline", action="store_true")46    args = ap.parse_args()47    if not args.offline:48        download()49    if not SRC.exists():50        print("missing sources/xai/llms-full.txt", file=sys.stderr)51        return 15253    prev = {}54    if MANIFEST.exists():55        prev = {e["url"]: e for e in json.loads(MANIFEST.read_text()).get("pages", [])}5657    text = SRC.read_text(errors="replace")58    parts = re.split(r"^===(/[^=\n]*)===\s*$", text, flags=re.M)59    ts = now()60    entries, seen = [], set()61    changes = {"run_at": ts, "new": [], "changed": [], "removed": [], "failed": []}62    OUT.mkdir(parents=True, exist_ok=True)63    for i in range(1, len(parts), 2):64        path = parts[i].strip()65        body = parts[i + 1].strip() + "\n"66        url = f"https://docs.x.ai{path}"67        slug = path.strip("/") + ".md"68        dest = OUT / slug69        dest.parent.mkdir(parents=True, exist_ok=True)70        dest.write_text(f"<!-- source: {url} (from https://docs.x.ai/llms-full.txt, retrieved {ts}) -->\n" + body)71        sha = hashlib.sha256(body.encode()).hexdigest()72        entries.append({"url": url, "path": f"sources/xai/pages/{slug}", "status": 200, "size": len(body),73                        "sha256": sha, "retrieved_at": ts, "via": "llms-full.txt"})74        seen.add(url)75        if url not in prev:76            changes["new"].append(url)77        elif prev[url].get("sha256") != sha:78            changes["changed"].append(url)79    changes["removed"] = sorted(u for u in prev if u not in seen)80    MANIFEST.write_text(json.dumps({"provider": "xai", "generated_at": ts, "page_count": len(entries),81                                    "ok_count": len(entries), "source": "https://docs.x.ai/llms-full.txt",82                                    "pages": sorted(entries, key=lambda e: e["url"])}, indent=2))83    CHANGES.write_text(json.dumps(changes, indent=2))84    print(f"xai pages={len(entries)} new={len(changes['new'])} changed={len(changes['changed'])} removed={len(changes['removed'])}")85    return 0868788if __name__ == "__main__":89    sys.exit(main())90