Python 88.3%
TypeScript 7.6%
Shell 4.1%
1#!/usr/bin/env python32"""Parse Anthropic reference leaf pages (sources/anthropic/pages/api/**) into structured JSON.34Output: tmp/platform-anthropic/ref/<slug>.json + tmp/platform-anthropic/ref/_index.json5Each record: {title, url, method, path, description, headers[], path_params[], query_params[],6 body_params[] (nested tree), body_content_type, returns{name, fields[] tree},7 example_request, example_response}8Usage: python3 tmp/platform-anthropic/extract_ref.py <dir-or-file>...9"""10from __future__ import annotations11import json, re, sys12from pathlib import Path1314ROOT = Path(__file__).resolve().parents[2]15PAGES = ROOT / "sources/anthropic/pages"16OUT = ROOT / "tmp/platform-anthropic/ref"17OUT.mkdir(parents=True, exist_ok=True)1819ITEM_RE = re.compile(r"^(\s*)- `(.+?)`\s*$")20CONSTRAINT_RE = re.compile(r"^(default|minimum|maximum|minLength|maxLength|format|pattern|minItems|maxItems):\s*(.+)$")212223def parse_items(lines: list[str]) -> list[dict]:24 """Parse a nested bullet list of `name: type` items into a tree."""25 roots: list[dict] = []26 stack: list[tuple[int, dict]] = []27 cur: dict | None = None28 for ln in lines:29 m = ITEM_RE.match(ln)30 if m:31 indent = len(m.group(1)) // 232 raw = m.group(2)33 item = {"raw": raw, "desc": [], "children": [], "constraints": {}}34 nm = re.match(r'^("?)([A-Za-z_][\w\[\]\.\-]*)\1:\s+(.+)$', raw)35 if nm:36 name, typ = nm.group(2), nm.group(3)37 item["name"] = name.strip()38 typ = typ.strip()39 item["required"] = not typ.startswith("optional")40 item["type"] = re.sub(r"^optional\s+", "", typ)41 else:42 # enum value or object variant line like `BetaSkill object` / `"custom"`43 item["name"] = None44 item["type"] = raw45 item["required"] = None46 while stack and stack[-1][0] >= indent:47 stack.pop()48 if stack:49 stack[-1][1]["children"].append(item)50 else:51 roots.append(item)52 stack.append((indent, item))53 cur = item54 continue55 if cur is None:56 continue57 s = ln.strip()58 if not s:59 continue60 cm = CONSTRAINT_RE.match(s)61 if cm and "," not in s or (cm and all(CONSTRAINT_RE.match(p.strip()) for p in s.split(","))):62 for part in s.split(","):63 k, _, v = part.strip().partition(":")64 cur["constraints"][k.strip()] = v.strip()65 continue66 # description line belongs to the innermost open item67 cur["desc"].append(s)68 return roots697071def simplify(node: dict) -> dict:72 out = {"name": node.get("name"), "type": node.get("type"), "required": node.get("required"),73 "description": " ".join(node["desc"]).strip()}74 if node["constraints"]:75 out["constraints"] = node["constraints"]76 enum_vals = [c["type"] for c in node["children"] if c.get("name") is None and c["type"].startswith('"') and not c["children"]]77 if enum_vals and len(enum_vals) == len(node["children"]):78 out["enum"] = [e.strip('"') for e in enum_vals]79 else:80 kids = [simplify(c) for c in node["children"]]81 if kids:82 out["children"] = kids83 return out848586def sections(text: str) -> dict[str, list[str]]:87 secs: dict[str, list[str]] = {}88 cur = "_pre"89 secs[cur] = []90 for ln in text.splitlines():91 if ln.startswith("## "):92 cur = ln[3:].strip()93 secs.setdefault(cur, [])94 continue95 secs[cur].append(ln)96 return secs979899def parse_page(p: Path) -> dict | None:100 text = p.read_text()101 fm = {}102 if text.startswith("---"):103 _, fmtxt, text = text.split("---", 2)104 for l in fmtxt.strip().splitlines():105 k, _, v = l.partition(":")106 fm[k.strip()] = v.strip()107 m = re.search(r"^\*\*(GET|POST|PUT|PATCH|DELETE)\*\*\s+`([^`]+)`", text, re.M)108 if not m:109 return None110 secs = sections(text)111 pre = "\n".join(secs.get("_pre", []))112 desc_lines = [l for l in pre.splitlines() if l.strip() and not l.startswith("#") and not l.startswith("**")]113 rec = {114 "title": fm.get("title") or p.stem,115 "url": fm.get("url"),116 "source_file": str(p.relative_to(ROOT)),117 "method": m.group(1),118 "path": m.group(2),119 "description": " ".join(desc_lines).strip(),120 }121 for key, sec in (("path_params", "Path parameters"), ("query_params", "Query parameters"), ("headers", "Headers")):122 if sec in secs:123 rec[key] = [simplify(n) for n in parse_items(secs[sec])]124 body_key = next((k for k in secs if k.startswith("Body parameters")), None)125 if body_key:126 rec["body_content_type"] = "multipart/form-data" if "form-data" in body_key else "application/json"127 rec["body_params"] = [simplify(n) for n in parse_items(secs[body_key])]128 if "Returns" in secs:129 ret = parse_items(secs["Returns"])130 if ret:131 r0 = ret[0]132 rec["returns"] = {"type": r0["type"], "fields": [simplify(c) for c in r0["children"]],133 "description": " ".join(r0["desc"]).strip()}134 ex = secs.get("Example", [])135 ex_txt = "\n".join(ex)136 cm = re.search(r"```bash\n(.*?)```", ex_txt, re.S)137 if cm:138 rec["example_request"] = cm.group(1).strip()139 rm = re.search(r"### Response.*?```json\n(.*?)```", ex_txt, re.S)140 if rm:141 try:142 rec["example_response"] = json.loads(rm.group(1))143 except Exception:144 rec["example_response_raw"] = rm.group(1)[:4000]145 # streaming pages: capture text/event-stream marker146 if "text/event-stream" in text or "Server-Sent" in text:147 rec["streaming_hint"] = True148 return rec149150151def main(argv: list[str]) -> None:152 files: list[Path] = []153 for a in argv:154 pa = PAGES / a if not a.startswith("/") else Path(a)155 if pa.is_dir():156 files += sorted(pa.rglob("*.md"))157 elif pa.exists():158 files.append(pa)159 index = []160 for f in files:161 rec = parse_page(f)162 if not rec:163 continue164 slug = str(f.relative_to(PAGES)).replace("/", "__").replace(".md", "")165 (OUT / f"{slug}.json").write_text(json.dumps(rec, indent=1, ensure_ascii=False))166 index.append({"slug": slug, "method": rec["method"], "path": rec["path"], "title": rec["title"], "url": rec["url"]})167 idx_path = OUT / "_index.json"168 old = json.loads(idx_path.read_text()) if idx_path.exists() else []169 seen = {(o["slug"]) for o in index}170 merged = [o for o in old if o["slug"] not in seen] + index171 idx_path.write_text(json.dumps(merged, indent=1))172 for o in index:173 print(f'{o["method"]:6} {o["path"]:70} {o["title"]} [{o["slug"]}]')174175176if __name__ == "__main__":177 main(sys.argv[1:])178