#!/usr/bin/env python3 """Export the API Atlas capability graph from the merged files. python3 scripts/export_capability_graph.py # generated/{models,endpoints,tools}.json → generated/capability-graph.{json,mmd,dot} python3 scripts/export_capability_graph.py --models X --endpoints Y --tools Z --out-dir DIR [--max-mermaid-nodes N] Inputs may be partially missing or malformed: each is skipped gracefully (recorded in `meta.inputs`). Node types: provider · api_family · endpoint · model · capability · tool Edge types: provider→api_family HAS_FAMILY · api_family→endpoint HAS_ENDPOINT · provider→model OFFERS · model→endpoint AVAILABLE_ON · model→capability SUPPORTS (value true) / SUPPORTS_UNKNOWN (value "unknown") · model→tool SUPPORTS_TOOL · tool→endpoint USABLE_ON · tool→model COMPATIBLE_WITH · model→model ALIAS_OF / SNAPSHOT_OF / REDIRECTS_TO (xAI `kind: "retired_redirect"` records whose live target is in verification.request_note) Capabilities that are `false` are NOT edges (absence = false when the model has a capabilities object; see graph meta). Provider-agnostic (openai · anthropic · xai · gemini): every provider found in the records becomes a node. Alias entries may be strings (optionally with a trailing parenthetical note, stripped) or dicts `{alias, resolves_to_live}` (Gemini `-latest` records → ALIAS_OF the live target when it exists); model `tools[]` may be strings or dicts `{type, category, support}` (Gemini; `support: false` entries produce no edge, string values such as "Supported (Preview)" count as supported). Record schemas: CLAUDE.md (model, endpoint, tool). Accepts a list of records or {"records"|"models"|"endpoints"|"tools": [...]}. Stdlib only. Tested offline by tests/shared/test_capability_graph.py against tests/shared/fixtures/*.json (4 providers). """ from __future__ import annotations import argparse import json import re import sys from datetime import datetime, timezone from pathlib import Path from typing import Any, Optional ROOT = Path(__file__).resolve().parent.parent GEN = ROOT / "generated" def load_records(path: Optional[Path], list_keys: tuple[str, ...]) -> tuple[list[dict[str, Any]], str]: """Return (records, status) where status ∈ {loaded, missing, invalid, empty}.""" if path is None or not path.exists(): return [], "missing" try: doc = json.loads(path.read_text()) except (json.JSONDecodeError, OSError): return [], "invalid" if isinstance(doc, dict): for k in list_keys: if isinstance(doc.get(k), list): doc = doc[k] break else: doc = [({"id": k, **v} if isinstance(v, dict) and "id" not in v and "name" not in v else v) for k, v in doc.items() if isinstance(v, dict)] if not isinstance(doc, list): return [], "invalid" recs = [r for r in doc if isinstance(r, dict)] return recs, ("loaded" if recs else "empty") def _slug(s: str) -> str: return re.sub(r"[^A-Za-z0-9_]", "_", s) _PAREN_NOTE = re.compile(r"\s*\(.*\)\s*$") _REDIRECT_NOTE = re.compile(r"with id ['\"]([^'\"]+)['\"]") def _alias_name(a: Any) -> Optional[str]: """'grok-voice-latest (routes here since …)' → 'grok-voice-latest'; {alias: …} dicts → their alias.""" if isinstance(a, dict): a = a.get("alias") or a.get("id") if not isinstance(a, str): return None a = _PAREN_NOTE.sub("", a).strip() return a or None def _redirect_target(m: dict[str, Any]) -> Optional[str]: for k in ("resolves_to", "redirects_to", "redirect_to", "alias_of"): if isinstance(m.get(k), str) and m[k].strip(): return _alias_name(m[k]) if m.get("kind") == "retired_redirect": mm = _REDIRECT_NOTE.search(str((m.get("verification") or {}).get("request_note") or "")) if mm: return mm.group(1) return None def _model_tool_types(m: dict[str, Any]) -> list[str]: """Model tools[] as type strings; dict entries (Gemini) with support == false are dropped.""" out = [] for t in m.get("tools") or []: if isinstance(t, str): out.append(t) elif isinstance(t, dict): typ = t.get("type") or t.get("name") sup = t.get("support", True) if isinstance(sup, str): sup = not (sup.lower().startswith("not") or sup.lower() in ("no", "unsupported")) if isinstance(typ, str) and sup is not False: out.append(typ) return out class Graph: def __init__(self) -> None: self.nodes: dict[str, dict[str, Any]] = {} self.edges: list[dict[str, Any]] = [] self._edge_keys: set[tuple[str, str, str]] = set() def node(self, ntype: str, key: str, label: Optional[str] = None, **attrs: Any) -> str: nid = f"{ntype}:{key}" if nid not in self.nodes: self.nodes[nid] = {"id": nid, "type": ntype, "label": label or key, **attrs} else: for k, v in attrs.items(): self.nodes[nid].setdefault(k, v) return nid def edge(self, src: str, etype: str, dst: str, **attrs: Any) -> None: k = (src, etype, dst) if k in self._edge_keys: return self._edge_keys.add(k) self.edges.append({"source": src, "type": etype, "target": dst, **attrs}) def build_graph(models: list[dict[str, Any]], endpoints: list[dict[str, Any]], tools: list[dict[str, Any]]) -> Graph: g = Graph() endpoint_ids: dict[tuple[str, str], str] = {} # (provider, "METHOD /path") -> node id # --- endpoints (provider → api_family → endpoint) for e in endpoints: prov, path = str(e.get("provider", "")).lower(), e.get("path") if not prov or not isinstance(path, str): continue method = str(e.get("method", "GET")).upper() fam = str(e.get("api_family") or "unknown") p = g.node("provider", prov) f = g.node("api_family", f"{prov}/{fam}", label=fam, provider=prov) key = f"{method} {path}" ep = g.node("endpoint", f"{prov}/{key}", label=key, provider=prov, api_family=fam, method=method, path=path, status=e.get("status") or [], streaming=(e.get("streaming") or {}).get("supported"), idempotency=e.get("idempotency")) g.edge(p, "HAS_FAMILY", f) g.edge(f, "HAS_ENDPOINT", ep) endpoint_ids[(prov, key)] = ep endpoint_ids[(prov, path)] = ep # tolerate path-only references def endpoint_ref(prov: str, ref: Any) -> Optional[str]: if not isinstance(ref, str): return None ref = ref.strip() if (prov, ref) in endpoint_ids: return endpoint_ids[(prov, ref)] # create a placeholder endpoint node when endpoints.json is missing or incomplete method, _, path = ref.partition(" ") if " " in ref else ("POST", "", ref) key = f"{method.upper()} {path}" if path else ref ep = g.node("endpoint", f"{prov}/{key}", label=key, provider=prov, method=method.upper(), path=path or ref, placeholder=True) endpoint_ids[(prov, key)] = ep return ep # --- models model_ids: dict[tuple[str, str], str] = {} for m in models: prov, mid = str(m.get("provider", "")).lower(), m.get("id") if not prov or not isinstance(mid, str): continue p = g.node("provider", prov) # record_kind: OpenAI emits it (model | snapshot | id_only | alias); xAI/Gemini use `kind` (alias / retired_redirect are pointers). # Anthropic's `kind: "snapshot"` denotes a real dated model record and stays "model". kind = m.get("record_kind") or {"retired_redirect": "redirect", "alias": "alias"}.get(str(m.get("kind")), "model") mn = g.node("model", f"{prov}/{mid}", label=mid, provider=prov, status=m.get("status") or [], family=m.get("family"), record_kind=kind, context_window=m.get("context_window"), max_output=m.get("max_output")) model_ids[(prov, mid.lower())] = mn g.edge(p, "OFFERS", mn) for m in models: prov, mid = str(m.get("provider", "")).lower(), m.get("id") if not prov or not isinstance(mid, str): continue mn = model_ids[(prov, mid.lower())] canon = m.get("canonical_model") if isinstance(canon, str) and canon.lower() != mid.lower() and (prov, canon.lower()) in model_ids: g.edge(mn, "SNAPSHOT_OF", model_ids[(prov, canon.lower())]) tgt = _redirect_target(m) if tgt and tgt.lower() != mid.lower() and (prov, tgt.lower()) in model_ids: g.edge(mn, "REDIRECTS_TO", model_ids[(prov, tgt.lower())]) for a in m.get("aliases") or []: name = _alias_name(a) if not name: continue if isinstance(a, dict) and name.lower() == mid.lower(): # Gemini `-latest` alias records: {alias: , resolves_to_live: } → this node ALIAS_OF the live target live = a.get("resolves_to_live") or a.get("resolves_to") if isinstance(live, str) and (prov, _alias_name(live).lower()) in model_ids and _alias_name(live).lower() != mid.lower(): g.nodes[mn].setdefault("record_kind", "alias") g.nodes[mn]["record_kind"] = "alias" g.edge(mn, "ALIAS_OF", model_ids[(prov, _alias_name(live).lower())]) continue if name.lower() == mid.lower(): continue an = model_ids.get((prov, name.lower())) or g.node("model", f"{prov}/{name}", label=name, provider=prov, record_kind="alias") g.edge(an, "ALIAS_OF", mn) for s in m.get("snapshots") or []: name = _alias_name(s) if name and name.lower() != mid.lower(): sn = model_ids.get((prov, name.lower())) or g.node("model", f"{prov}/{name}", label=name, provider=prov, record_kind="snapshot") g.edge(sn, "SNAPSHOT_OF", mn) for ref in m.get("endpoints") or []: ep = endpoint_ref(prov, ref) if ep: g.edge(mn, "AVAILABLE_ON", ep) caps = m.get("capabilities") if isinstance(caps, dict): for cap, val in caps.items(): if val is True: g.edge(mn, "SUPPORTS", g.node("capability", cap)) elif val == "unknown": g.edge(mn, "SUPPORTS_UNKNOWN", g.node("capability", cap)) # False → no edge; non-boolean metadata (lists, notes) → ignored for t in _model_tool_types(m): g.edge(mn, "SUPPORTS_TOOL", g.node("tool", f"{prov}/{t}", label=t, provider=prov)) # --- tools for t in tools: prov = str(t.get("provider", "")).lower() typ = t.get("type") or t.get("name") if not prov or not isinstance(typ, str): continue tn = g.node("tool", f"{prov}/{typ}", label=typ, provider=prov, name=t.get("name"), category=t.get("category"), status=t.get("status") or [], beta_header=t.get("beta_header"), security=t.get("security")) g.edge(g.node("provider", prov), "PROVIDES_TOOL", tn) for ref in t.get("compatible_endpoints") or []: ep = endpoint_ref(prov, ref) if ep: g.edge(tn, "USABLE_ON", ep) for mref in t.get("compatible_models") or []: if isinstance(mref, str): mn = model_ids.get((prov, mref.lower())) or g.node("model", f"{prov}/{mref}", label=mref, provider=prov, placeholder=True) g.edge(tn, "COMPATIBLE_WITH", mn) return g # ----------------------------------------------------------------------------- exporters def to_json(g: Graph, meta: dict[str, Any]) -> dict[str, Any]: counts: dict[str, int] = {} for n in g.nodes.values(): counts[n["type"]] = counts.get(n["type"], 0) + 1 ecounts: dict[str, int] = {} for e in g.edges: ecounts[e["type"]] = ecounts.get(e["type"], 0) + 1 providers = sorted(n["label"] for n in g.nodes.values() if n["type"] == "provider") per_provider = {p: {"models": sum(1 for n in g.nodes.values() if n["type"] == "model" and n.get("provider") == p and n.get("record_kind", "model") not in ("alias", "snapshot", "redirect") and not n.get("placeholder")), "endpoints": sum(1 for n in g.nodes.values() if n["type"] == "endpoint" and n.get("provider") == p), "tools": sum(1 for n in g.nodes.values() if n["type"] == "tool" and n.get("provider") == p)} for p in providers} return {"meta": {**meta, "providers": providers, "per_provider": per_provider, "node_counts": counts, "edge_counts": ecounts, "semantics": {"SUPPORTS": "capability value true", "SUPPORTS_UNKNOWN": "capability value \"unknown\"", "absent": "capability false, or model without a capabilities object", "REDIRECTS_TO": "retired id whose requests are served by the target (xAI retired_redirect)", "ALIAS_OF": "alias id → canonical record (incl. Gemini -latest records → live target)"}}, "nodes": list(g.nodes.values()), "edges": g.edges} _MMD_SHAPE = {"provider": ("[[", "]]"), "api_family": ("(", ")"), "endpoint": ("[", "]"), "model": ("([", "])"), "capability": ("{{", "}}"), "tool": (">", "]")} def to_mermaid(g: Graph, max_nodes: int = 400) -> str: """Mermaid gets unreadable past a few hundred nodes: skip alias/snapshot/redirect-only model nodes first, then truncate.""" lines = ["graph LR"] nodes = [n for n in g.nodes.values() if n.get("record_kind") not in ("alias", "snapshot", "redirect")] keep = {n["id"] for n in nodes[:max_nodes]} truncated = len(g.nodes) - len(keep) for n in nodes[:max_nodes]: o, c = _MMD_SHAPE.get(n["type"], ("[", "]")) label = n["label"].replace('"', "'") lines.append(f' {_slug(n["id"])}{o}"{label}"{c}') for e in g.edges: if e["source"] in keep and e["target"] in keep: lines.append(f' {_slug(e["source"])} -->|{e["type"]}| {_slug(e["target"])}') for t in _MMD_SHAPE: ids = [_slug(n["id"]) for n in nodes[:max_nodes] if n["type"] == t] if ids: lines.append(f" classDef {t} stroke-width:2px;") lines.append(f" class {','.join(ids)} {t};") if truncated > 0: lines.append(f" %% {truncated} node(s) omitted (aliases/snapshots or beyond --max-mermaid-nodes); see capability-graph.json") return "\n".join(lines) + "\n" _DOT_SHAPE = {"provider": "doubleoctagon", "api_family": "ellipse", "endpoint": "box", "model": "component", "capability": "hexagon", "tool": "cds"} def to_dot(g: Graph) -> str: lines = ["digraph capability_graph {", " rankdir=LR;", " node [fontname=Helvetica fontsize=10];", " edge [fontname=Helvetica fontsize=8];"] for n in g.nodes.values(): label = n["label"].replace('"', '\\"') lines.append(f' "{n["id"]}" [label="{label}" shape={_DOT_SHAPE.get(n["type"], "box")} class="{n["type"]}"];') for e in g.edges: lines.append(f' "{e["source"]}" -> "{e["target"]}" [label="{e["type"]}"];') lines.append("}") return "\n".join(lines) + "\n" def export(models_path: Optional[Path], endpoints_path: Optional[Path], tools_path: Optional[Path], out_dir: Path, max_mermaid_nodes: int = 400) -> dict[str, Any]: models, ms = load_records(models_path, ("records", "models")) endpoints, es = load_records(endpoints_path, ("records", "endpoints")) tools, ts = load_records(tools_path, ("records", "tools")) g = build_graph(models, endpoints, tools) meta = {"generated_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), "inputs": {"models": {"path": str(models_path), "status": ms, "records": len(models)}, "endpoints": {"path": str(endpoints_path), "status": es, "records": len(endpoints)}, "tools": {"path": str(tools_path), "status": ts, "records": len(tools)}}} out_dir.mkdir(parents=True, exist_ok=True) doc = to_json(g, meta) (out_dir / "capability-graph.json").write_text(json.dumps(doc, indent=1, ensure_ascii=False) + "\n") (out_dir / "capability-graph.mmd").write_text(to_mermaid(g, max_mermaid_nodes)) (out_dir / "capability-graph.dot").write_text(to_dot(g)) return doc def main(argv: Optional[list[str]] = None) -> int: ap = argparse.ArgumentParser(description=__doc__.split("\n")[0]) ap.add_argument("--models", type=Path, default=GEN / "models.json") ap.add_argument("--endpoints", type=Path, default=GEN / "endpoints.json") ap.add_argument("--tools", type=Path, default=GEN / "tools.json") ap.add_argument("--out-dir", type=Path, default=GEN) ap.add_argument("--max-mermaid-nodes", type=int, default=400) a = ap.parse_args(argv) doc = export(a.models, a.endpoints, a.tools, a.out_dir, a.max_mermaid_nodes) m = doc["meta"] print(f"inputs: " + ", ".join(f"{k}={v['status']}({v['records']})" for k, v in m["inputs"].items()), file=sys.stderr) print(f"providers: {m['providers']} per_provider: {m['per_provider']}", file=sys.stderr) print(f"nodes: {m['node_counts']}\nedges: {m['edge_counts']}\n→ {a.out_dir}/capability-graph.{{json,mmd,dot}}", file=sys.stderr) return 0 if __name__ == "__main__": sys.exit(main())