SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
6.8 KB · 178 lines python
Raw Blame History
1#!/usr/bin/env python32"""Parse Anthropic reference leaf pages (sources/anthropic/pages/api/**) into structured JSON.34Output: tmp/platform-anthropic/ref/<slug>.json  +  tmp/platform-anthropic/ref/_index.json5Each record: {title, url, method, path, description, headers[], path_params[], query_params[],6              body_params[] (nested tree), body_content_type, returns{name, fields[] tree},7              example_request, example_response}8Usage: python3 tmp/platform-anthropic/extract_ref.py <dir-or-file>...9"""10from __future__ import annotations11import json, re, sys12from pathlib import Path1314ROOT = Path(__file__).resolve().parents[2]15PAGES = ROOT / "sources/anthropic/pages"16OUT = ROOT / "tmp/platform-anthropic/ref"17OUT.mkdir(parents=True, exist_ok=True)1819ITEM_RE = re.compile(r"^(\s*)- `(.+?)`\s*$")20CONSTRAINT_RE = re.compile(r"^(default|minimum|maximum|minLength|maxLength|format|pattern|minItems|maxItems):\s*(.+)$")212223def parse_items(lines: list[str]) -> list[dict]:24    """Parse a nested bullet list of `name: type` items into a tree."""25    roots: list[dict] = []26    stack: list[tuple[int, dict]] = []27    cur: dict | None = None28    for ln in lines:29        m = ITEM_RE.match(ln)30        if m:31            indent = len(m.group(1)) // 232            raw = m.group(2)33            item = {"raw": raw, "desc": [], "children": [], "constraints": {}}34            nm = re.match(r'^("?)([A-Za-z_][\w\[\]\.\-]*)\1:\s+(.+)$', raw)35            if nm:36                name, typ = nm.group(2), nm.group(3)37                item["name"] = name.strip()38                typ = typ.strip()39                item["required"] = not typ.startswith("optional")40                item["type"] = re.sub(r"^optional\s+", "", typ)41            else:42                # enum value or object variant line like `BetaSkill object` / `"custom"`43                item["name"] = None44                item["type"] = raw45                item["required"] = None46            while stack and stack[-1][0] >= indent:47                stack.pop()48            if stack:49                stack[-1][1]["children"].append(item)50            else:51                roots.append(item)52            stack.append((indent, item))53            cur = item54            continue55        if cur is None:56            continue57        s = ln.strip()58        if not s:59            continue60        cm = CONSTRAINT_RE.match(s)61        if cm and "," not in s or (cm and all(CONSTRAINT_RE.match(p.strip()) for p in s.split(","))):62            for part in s.split(","):63                k, _, v = part.strip().partition(":")64                cur["constraints"][k.strip()] = v.strip()65            continue66        # description line belongs to the innermost open item67        cur["desc"].append(s)68    return roots697071def simplify(node: dict) -> dict:72    out = {"name": node.get("name"), "type": node.get("type"), "required": node.get("required"),73           "description": " ".join(node["desc"]).strip()}74    if node["constraints"]:75        out["constraints"] = node["constraints"]76    enum_vals = [c["type"] for c in node["children"] if c.get("name") is None and c["type"].startswith('"') and not c["children"]]77    if enum_vals and len(enum_vals) == len(node["children"]):78        out["enum"] = [e.strip('"') for e in enum_vals]79    else:80        kids = [simplify(c) for c in node["children"]]81        if kids:82            out["children"] = kids83    return out848586def sections(text: str) -> dict[str, list[str]]:87    secs: dict[str, list[str]] = {}88    cur = "_pre"89    secs[cur] = []90    for ln in text.splitlines():91        if ln.startswith("## "):92            cur = ln[3:].strip()93            secs.setdefault(cur, [])94            continue95        secs[cur].append(ln)96    return secs979899def parse_page(p: Path) -> dict | None:100    text = p.read_text()101    fm = {}102    if text.startswith("---"):103        _, fmtxt, text = text.split("---", 2)104        for l in fmtxt.strip().splitlines():105            k, _, v = l.partition(":")106            fm[k.strip()] = v.strip()107    m = re.search(r"^\*\*(GET|POST|PUT|PATCH|DELETE)\*\*\s+`([^`]+)`", text, re.M)108    if not m:109        return None110    secs = sections(text)111    pre = "\n".join(secs.get("_pre", []))112    desc_lines = [l for l in pre.splitlines() if l.strip() and not l.startswith("#") and not l.startswith("**")]113    rec = {114        "title": fm.get("title") or p.stem,115        "url": fm.get("url"),116        "source_file": str(p.relative_to(ROOT)),117        "method": m.group(1),118        "path": m.group(2),119        "description": " ".join(desc_lines).strip(),120    }121    for key, sec in (("path_params", "Path parameters"), ("query_params", "Query parameters"), ("headers", "Headers")):122        if sec in secs:123            rec[key] = [simplify(n) for n in parse_items(secs[sec])]124    body_key = next((k for k in secs if k.startswith("Body parameters")), None)125    if body_key:126        rec["body_content_type"] = "multipart/form-data" if "form-data" in body_key else "application/json"127        rec["body_params"] = [simplify(n) for n in parse_items(secs[body_key])]128    if "Returns" in secs:129        ret = parse_items(secs["Returns"])130        if ret:131            r0 = ret[0]132            rec["returns"] = {"type": r0["type"], "fields": [simplify(c) for c in r0["children"]],133                              "description": " ".join(r0["desc"]).strip()}134    ex = secs.get("Example", [])135    ex_txt = "\n".join(ex)136    cm = re.search(r"```bash\n(.*?)```", ex_txt, re.S)137    if cm:138        rec["example_request"] = cm.group(1).strip()139    rm = re.search(r"### Response.*?```json\n(.*?)```", ex_txt, re.S)140    if rm:141        try:142            rec["example_response"] = json.loads(rm.group(1))143        except Exception:144            rec["example_response_raw"] = rm.group(1)[:4000]145    # streaming pages: capture text/event-stream marker146    if "text/event-stream" in text or "Server-Sent" in text:147        rec["streaming_hint"] = True148    return rec149150151def main(argv: list[str]) -> None:152    files: list[Path] = []153    for a in argv:154        pa = PAGES / a if not a.startswith("/") else Path(a)155        if pa.is_dir():156            files += sorted(pa.rglob("*.md"))157        elif pa.exists():158            files.append(pa)159    index = []160    for f in files:161        rec = parse_page(f)162        if not rec:163            continue164        slug = str(f.relative_to(PAGES)).replace("/", "__").replace(".md", "")165        (OUT / f"{slug}.json").write_text(json.dumps(rec, indent=1, ensure_ascii=False))166        index.append({"slug": slug, "method": rec["method"], "path": rec["path"], "title": rec["title"], "url": rec["url"]})167    idx_path = OUT / "_index.json"168    old = json.loads(idx_path.read_text()) if idx_path.exists() else []169    seen = {(o["slug"]) for o in index}170    merged = [o for o in old if o["slug"] not in seen] + index171    idx_path.write_text(json.dumps(merged, indent=1))172    for o in index:173        print(f'{o["method"]:6} {o["path"]:70} {o["title"]}  [{o["slug"]}]')174175176if __name__ == "__main__":177    main(sys.argv[1:])178