#!/usr/bin/env python3 """Merge generated/fragments/**/*.json into the canonical machine-readable outputs. python3 scripts/build_generated.py # build everything python3 scripts/build_generated.py --check # validate fragments only (no write) Inputs : generated/fragments//*.json — each file is a JSON array of records (or {"records":[...]}) Outputs : generated/models.json|csv, endpoints.json|csv, parameters.json|csv, tools.json|csv, streaming-events.json, errors.json|csv, headers.json, pricing.json|csv, rate-limits.json, deprecations.json, objects.json, sdks.json, webhook-events.json, audit-log-events.json, status-lifecycles.json, beta-headers.json, examples-manifest.json, compatibility/model-capability-matrix.{json,csv}, compatibility/model-endpoint-matrix.{json,csv}, compatibility/model-tool-matrix.{json,csv}, compatibility/tool-model-matrix.csv, compatibility/feature-platform-matrix.json (anthropic clouds), index.json (counts + provenance) Fragment kinds are inferred from the directory name (generated/fragments//). Unknown kinds are merged into generated/.json verbatim so agents can add new record types without touching this script. Records are de-duplicated on a per-kind natural key; on conflict the later file (alphabetical) wins but the conflict is reported in generated/build-report.json. """ from __future__ import annotations import argparse import csv import json import sys from collections import defaultdict from datetime import datetime, timezone from pathlib import Path ROOT = Path(__file__).resolve().parent.parent FRAG = ROOT / "generated" / "fragments" OUT = ROOT / "generated" STATUSES = {"DOCUMENTED", "LIVE_DISCOVERED", "LIVE_VERIFIED", "BETA", "PREVIEW", "LEGACY", "DEPRECATED", "RETIRED", "ACCOUNT_RESTRICTED", "UNVERIFIED", "FAILED_VERIFICATION", "DOCUMENTATION_INCOMPLETE", "GA"} # directory-name aliases (agents sometimes pluralise differently) KIND_ALIASES = {"prices": "pricing", "price": "pricing", "model": "models", "endpoint": "endpoints", "parameter": "parameters", "tool": "tools", "streaming_events": "streaming-events", "events": "streaming-events", "error": "errors", "header": "headers", "rate_limits": "rate-limits", "ratelimits": "rate-limits", "deprecation": "deprecations", "object": "objects", "sdk": "sdks", "webhook_events": "webhook-events", "webhooks": "webhook-events", "audit_log_events": "audit-log-events", "lifecycles": "status-lifecycles", "status_lifecycles": "status-lifecycles", "beta_headers": "beta-headers", "compat": "compatibility"} # natural keys per kind (tuple of field names); None → no dedup KEYS: dict[str, tuple[str, ...] | None] = { "models": ("provider", "id"), "endpoints": ("provider", "method", "path"), "parameters": ("provider", "endpoint", "parameter", "location", "variant", "type"), "tools": ("provider", "type", "name"), "streaming-events": ("provider", "api", "event", "direction"), "errors": ("provider", "http_status", "type", "code"), "headers": ("provider", "name", "direction"), "pricing": ("provider", "model_or_service", "dimension", "tier", "unit"), "rate-limits": None, "deprecations": ("provider", "subject", "shutdown_date"), "objects": ("provider", "name"), "sdks": ("provider", "language", "package"), "webhook-events": ("provider", "event"), "audit-log-events": ("provider", "event"), "status-lifecycles": ("provider", "resource"), "beta-headers": ("provider", "value"), "compatibility": None, } CSV_COLUMNS: dict[str, list[str]] = { "models": ["provider", "id", "display_name", "family", "status", "release_date", "knowledge_cutoff", "context_window", "max_output", "input_modalities", "output_modalities", "reasoning", "streaming", "structured_outputs", "function_calling", "prompt_caching", "batch", "fine_tuning", "vision", "audio_in", "audio_out", "image_out", "video_out", "realtime", "computer_use", "web_search", "code_execution", "file_search", "mcp", "price_input", "price_cached_input", "price_output", "endpoints", "tools", "last_verified"], "endpoints": ["provider", "api_family", "method", "path", "name", "status", "auth", "beta_header", "streaming", "sdk_python", "sdk_node", "verification_result", "source"], "parameters": ["provider", "endpoint", "parameter", "location", "type", "required", "default", "minimum", "maximum", "enum", "compatible_models", "beta_header", "status", "description", "source"], "tools": ["provider", "type", "name", "category", "status", "compatible_models", "compatible_endpoints", "beta_header", "billing", "verification_result", "source"], "errors": ["provider", "http_status", "type", "code", "retryable", "recommended_action", "message_semantics", "source"], "pricing": ["provider", "model_or_service", "dimension", "tier", "price", "currency", "unit", "effective_notes", "source", "retrieved_at"], } def now() -> str: return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") def load_fragment(path: Path) -> list[dict]: data = json.loads(path.read_text()) if isinstance(data, dict): for k in ("records", "items", "data"): if k in data and isinstance(data[k], list): data = data[k] break else: data = [data] if not isinstance(data, list): raise ValueError(f"{path}: expected a JSON array") out = [] for rec in data: if not isinstance(rec, dict): raise ValueError(f"{path}: record is not an object") rec.setdefault("_fragment", str(path.relative_to(ROOT))) out.append(rec) return out API_CANON = [ # (substring test on lowercased `api`, canonical value) ("lyria", "lyria-realtime"), ("music", "lyria-realtime"), ("interaction", "interactions"), ("bidigeneratecontent", "live"), ("generatecontent", "generate-content"), ("voice", "voice"), ("deferred", "chat_completions"), ("websocket", "responses-websocket"), ("responses", "responses"), ("chat", "chat_completions"), ("/v1/completions", "completions"), ("completions", "completions"), ("translation", "realtime-translation"), ("realtime", "realtime"), ("live", "live"), ("transcription", "audio-transcriptions"), ("speech", "audio-speech"), ("images", "images"), ("webhook", "agents-webhooks"), ("agents/sessions", "agents"), ("agents", "agents"), ("managed", "managed-agents"), ("/v1/complete", "complete-legacy"), ("messages", "messages"), ] def canon_api(rec: dict) -> None: """Normalize streaming_event.api to a small canonical vocabulary; keep the original in api_raw.""" if rec.get("provider") == "gemini" and "live" in str(rec.get("api", "")).lower() and "lyria" not in str(rec.get("api", "")).lower(): rec["api_raw"], rec["api"] = rec.get("api"), "live" return if rec.get("provider") == "anthropic" and "managed" in str(rec.get("api", "")).lower(): rec["api_raw"], rec["api"] = rec.get("api"), "managed-agents" return a = str(rec.get("api") or "").lower() for needle, canon in API_CANON: if needle in a: if rec.get("api") != canon: rec["api_raw"] = rec.get("api") rec["api"] = canon return def key_of(kind: str, rec: dict) -> tuple | None: fields = KEYS.get(kind) if fields is None: return None if kind == "objects": name = next((rec.get(k) for k in ("name", "object", "type", "title", "id") if rec.get(k)), None) if name is None: return None # cannot identify → keep as is return (json.dumps(rec.get("provider")), json.dumps(name, default=str)) return tuple(json.dumps(rec.get(f), sort_keys=True, default=str) for f in fields) def flatten(v) -> str: if v is None: return "" if isinstance(v, bool): return "true" if v else "false" if isinstance(v, (list, tuple)): return "|".join(flatten(x) for x in v) if isinstance(v, dict): return json.dumps(v, ensure_ascii=False, sort_keys=True) return str(v) def model_row(m: dict) -> dict: caps = m.get("capabilities") or {} mod = m.get("modalities") or {} pr = m.get("pricing") if isinstance(m.get("pricing"), dict) else {} def cap(*names): for n in names: if n in caps: return caps[n] return "unknown" def price(*names): for n in names: v = pr.get(n) if isinstance(v, dict): v = v.get("price", v.get("value")) if v is not None: return v return "" return { "provider": m.get("provider"), "id": m.get("id"), "display_name": m.get("display_name"), "family": m.get("family"), "status": m.get("status"), "release_date": m.get("release_date"), "knowledge_cutoff": m.get("knowledge_cutoff"), "context_window": m.get("context_window"), "max_output": m.get("max_output"), "input_modalities": mod.get("input"), "output_modalities": mod.get("output"), "reasoning": cap("reasoning", "thinking", "extended_thinking"), "streaming": cap("streaming"), "structured_outputs": cap("structured_outputs"), "function_calling": cap("function_calling", "tool_use"), "prompt_caching": cap("prompt_caching"), "batch": cap("batch"), "fine_tuning": cap("fine_tuning"), "vision": cap("image_in", "image_input", "vision"), "audio_in": cap("audio_in", "audio_input"), "audio_out": cap("audio_out", "audio_output"), "image_out": cap("image_out", "image_output", "image_generation"), "video_out": cap("video_out", "video_output", "video_generation"), "realtime": cap("realtime"), "computer_use": cap("computer_use"), "web_search": cap("web_search"), "code_execution": cap("code_execution", "code_interpreter"), "file_search": cap("file_search"), "mcp": cap("mcp"), "price_input": price("input"), "price_cached_input": price("cached_input", "cache_read"), "price_output": price("output"), "endpoints": sorted(names(m.get("endpoints"), "route", "path", "name")), "tools": sorted(names(m.get("tools"), "type", "name")), "last_verified": m.get("last_verified"), } def generic_row(kind: str, rec: dict) -> dict: cols = CSV_COLUMNS[kind] row = {} for c in cols: if c == "verification_result": row[c] = (rec.get("verification") or {}).get("result") elif c == "sdk_python": row[c] = (rec.get("sdk") or {}).get("python") elif c == "sdk_node": row[c] = (rec.get("sdk") or {}).get("node") elif c == "streaming" and isinstance(rec.get("streaming"), dict): row[c] = rec["streaming"].get("supported") elif c == "source": src = rec.get("source") or rec.get("sources") if isinstance(src, list): src = "|".join((s.get("url") if isinstance(s, dict) else str(s)) for s in src) row[c] = src else: row[c] = rec.get(c) return row def write_csv(path: Path, rows: list[dict], columns: list[str]) -> None: with path.open("w", newline="") as f: w = csv.DictWriter(f, fieldnames=columns, extrasaction="ignore") w.writeheader() for r in rows: w.writerow({k: flatten(r.get(k)) for k in columns}) def names(items, *keys: str) -> set[str]: """Normalize a list of strings/objects into a set of identifier strings.""" out: set[str] = set() for x in items or []: if isinstance(x, str): out.add(x) elif isinstance(x, dict): for k in keys: v = x.get(k) if isinstance(v, str): out.add(v) break return out def build_matrices(models: list[dict], tools: list[dict], endpoints: list[dict]) -> dict[str, list[dict]]: cap_names: list[str] = [] seen = set() for m in models: for c in (m.get("capabilities") or {}): if c not in seen: seen.add(c) cap_names.append(c) cap_matrix = [] for m in models: row = {"provider": m.get("provider"), "model": m.get("id")} caps = m.get("capabilities") or {} for c in cap_names: row[c] = caps.get(c, "unknown") cap_matrix.append(row) ep_names = sorted({e for m in models for e in names(m.get("endpoints"), "route", "path", "name")}) ep_matrix = [] for m in models: row = {"provider": m.get("provider"), "model": m.get("id")} mine = names(m.get("endpoints"), "route", "path", "name") for e in ep_names: row[e] = e in mine ep_matrix.append(row) tool_types = sorted({t.get("type") for t in tools if t.get("type")}) tool_matrix = [] for m in models: row = {"provider": m.get("provider"), "model": m.get("id")} mine = names(m.get("tools"), "type", "name") for t in tool_types: # tool record may list compatible models explicitly; union both directions row[t] = (t in mine) or any( tr.get("type") == t and tr.get("provider") == m.get("provider") and m.get("id") in (tr.get("compatible_models") or []) for tr in tools) tool_matrix.append(row) tool_model = [] for t in tools: tool_model.append({"provider": t.get("provider"), "tool_type": t.get("type"), "name": t.get("name"), "category": t.get("category"), "status": t.get("status"), "compatible_models": t.get("compatible_models"), "compatible_endpoints": t.get("compatible_endpoints"), "beta_header": t.get("beta_header")}) return {"model-capability-matrix": cap_matrix, "model-endpoint-matrix": ep_matrix, "model-tool-matrix": tool_matrix, "tool-model-matrix": tool_model} def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("--check", action="store_true") args = ap.parse_args() if not FRAG.exists(): print("no fragments directory", file=sys.stderr) return 1 merged: dict[str, list[dict]] = defaultdict(list) report = {"built_at": now(), "fragments": [], "conflicts": [], "warnings": [], "counts": {}} indexes: dict[str, dict[tuple, int]] = defaultdict(dict) for kind_dir in sorted(p for p in FRAG.iterdir() if p.is_dir()): kind = KIND_ALIASES.get(kind_dir.name, kind_dir.name) index = indexes[kind] for f in sorted(kind_dir.glob("*.json")): try: recs = load_fragment(f) except Exception as e: # noqa: BLE001 report["warnings"].append(f"{f}: {e}") continue report["fragments"].append({"kind": kind, "file": str(f.relative_to(ROOT)), "records": len(recs)}) for r in recs: st = r.get("status") if isinstance(st, str): r["status"] = [st] if kind == "streaming-events": canon_api(r) for s in (r.get("status") or []): if s not in STATUSES: report["warnings"].append(f"{f}: unknown status {s!r}") k = key_of(kind, r) if k is None: merged[kind].append(r) continue if k in index: prev = merged[kind][index[k]] report["conflicts"].append({"kind": kind, "key": k, "kept": r["_fragment"], "dropped": prev["_fragment"]}) merged[kind][index[k]] = r else: index[k] = len(merged[kind]) merged[kind].append(r) for kind, recs in merged.items(): report["counts"][kind] = len(recs) if args.check: print(json.dumps(report, indent=1, default=str)) return 0 OUT.mkdir(exist_ok=True) (OUT / "compatibility").mkdir(exist_ok=True) for kind, recs in merged.items(): if kind == "compatibility": continue (OUT / f"{kind}.json").write_text(json.dumps(recs, indent=1, ensure_ascii=False, default=str)) if kind in CSV_COLUMNS: rows = [model_row(r) for r in recs] if kind == "models" else [generic_row(kind, r) for r in recs] write_csv(OUT / f"{kind}.csv", rows, CSV_COLUMNS[kind]) # compatibility fragments are copied through by file name for f in sorted((FRAG / "compatibility").glob("*.json")) if (FRAG / "compatibility").exists() else []: (OUT / "compatibility" / f.name).write_text(f.read_text()) matrices = build_matrices(merged.get("models", []), merged.get("tools", []), merged.get("endpoints", [])) for name, rows in matrices.items(): (OUT / "compatibility" / f"{name}.json").write_text(json.dumps(rows, indent=1, ensure_ascii=False, default=str)) if rows: cols = list(rows[0].keys()) for r in rows[1:]: for c in r: if c not in cols: cols.append(c) write_csv(OUT / "compatibility" / f"{name}.csv", rows, cols) # examples manifest: merge examples/manifest-*.json ex = [] for f in sorted((ROOT / "examples").glob("manifest-*.json")): try: data = json.loads(f.read_text()) items = data.get("examples", data.get("files", data.get("items"))) if isinstance(data, dict) else data if isinstance(items, dict): items = [dict(v, file=k) if isinstance(v, dict) else {"file": k, "status": v} for k, v in items.items()] if items is None: items = [v for v in data.values() if isinstance(v, dict)] if isinstance(data, dict) else [] norm = [] for it in items: if isinstance(it, str): it = {"file": it} elif not isinstance(it, dict): continue it.setdefault("_manifest", f.name) norm.append(it) ex.extend(norm) except Exception as e: # noqa: BLE001 report["warnings"].append(f"{f}: {e}") (OUT / "examples-manifest.json").write_text(json.dumps(ex, indent=1, ensure_ascii=False, default=str)) report["counts"]["examples"] = len(ex) # live request log summary log = ROOT / "reports" / "live-requests.jsonl" if log.exists(): calls = [json.loads(l) for l in log.read_text().splitlines() if l.strip()] summary = defaultdict(lambda: {"calls": 0, "est_cost_usd": 0.0, "statuses": defaultdict(int)}) for c in calls: s = summary[c.get("provider", "?")] s["calls"] += 1 s["est_cost_usd"] += float(c.get("est_cost_usd") or 0) s["statuses"][str(c.get("status"))] += 1 report["live_requests"] = {k: {"calls": v["calls"], "est_cost_usd": round(v["est_cost_usd"], 4), "statuses": dict(v["statuses"])} for k, v in summary.items()} (OUT / "build-report.json").write_text(json.dumps(report, indent=1, default=str)) index = {"built_at": report["built_at"], "counts": report["counts"], "files": sorted(str(p.relative_to(ROOT)) for p in OUT.rglob("*") if p.is_file() and "fragments" not in p.parts)} (OUT / "index.json").write_text(json.dumps(index, indent=1)) print(json.dumps({"counts": report["counts"], "conflicts": len(report["conflicts"]), "warnings": len(report["warnings"])}, indent=1)) return 0 if __name__ == "__main__": sys.exit(main())