Python 88.3%
TypeScript 7.6%
Shell 4.1%
1#!/usr/bin/env python32"""Generate the OpenAI Realtime / Live / Audio fragments (API Atlas).34Inputs (offline): sources/openai/pages/api/reference/resources/{realtime,live,audio}/**, openapi-master.yaml, pages-manifest.json,5tmp-live/realtime-audio/*.json (live probe results, sanitized).6Outputs: generated/fragments/streaming-events/openai-{realtime,live,audio-transcription}.json7 generated/fragments/parameters/openai-{realtime,live,audio}.json8 generated/fragments/endpoints/openai-realtime-live-audio.json9Run: .venv/bin/python scripts/gen_openai_realtime_audio.py10"""11from __future__ import annotations1213import json14import re15from pathlib import Path1617import yaml1819ROOT = Path(__file__).resolve().parent.parent20SRC = ROOT / "sources" / "openai"21REF = SRC / "pages" / "api" / "reference" / "resources"22OUT = ROOT / "generated" / "fragments"23PROBE = ROOT / "tmp-live" / "realtime-audio"24VERIFIED_AT = "2026-09-18"25MANIFEST = {p["path"]: p for p in json.load(open(SRC / "pages-manifest.json"))["pages"]}26SPEC = yaml.safe_load(open(SRC / "openapi" / "openapi-master.yaml"))27SCHEMAS = SPEC["components"]["schemas"]282930def src(path: Path, anchor: str | None = None) -> dict:31 rel = str(path.relative_to(ROOT))32 m = MANIFEST.get(rel, {})33 url = m.get("url", "https://developers.openai.com/api/" + str(path.relative_to(SRC / "pages" / "api")))34 url = url[:-3] if url.endswith(".md") else url35 if anchor:36 url += "#" + anchor37 return {"url": url, "retrieved_at": m.get("retrieved_at", "2026-09-19T01:31:00Z")}383940# --------------------------------------------------------------------------- markdown bullet-tree parser41BULLET = re.compile(r"^(\s*)- `(.*)`\s*$")42FIELD = re.compile(r"^([A-Za-z_][\w\[\].]*): (.*)$")434445def parse_bullets(lines: list[str]) -> list[dict]:46 """Parse the nested `- \\`name: type\\`` bullet list used by the OpenAI reference pages."""47 root: list[dict] = []48 stack: list[tuple[int, dict]] = []49 cur: dict | None = None50 cur_indent = -151 for raw in lines:52 m = BULLET.match(raw)53 if m:54 indent = len(m.group(1))55 body = m.group(2)56 f = FIELD.match(body)57 node: dict58 if f:59 t = f.group(2).strip()60 req = not t.startswith("optional ")61 t = t[len("optional "):] if not req else t62 node = {"name": f.group(1), "type": t, "required": req, "description": "", "children": []}63 else:64 node = {"variant": body, "description": "", "children": []}65 while stack and stack[-1][0] >= indent:66 stack.pop()67 (stack[-1][1]["children"] if stack else root).append(node)68 stack.append((indent, node))69 cur, cur_indent = node, indent70 continue71 if cur is not None and raw.strip():72 if len(raw) - len(raw.lstrip()) > cur_indent:73 cur["description"] = (cur["description"] + " " + raw.strip()).strip()74 return root757677def literal(v: str):78 if re.fullmatch(r'"[^"]*"', v):79 return v.strip('"')80 if re.fullmatch(r"-?\d+(\.\d+)?", v):81 return float(v) if "." in v else int(v)82 return None838485def compact(node: dict, depth: int, max_depth: int) -> dict:86 """Field node -> compact JSON schema-ish description (enum literals folded, variants folded)."""87 out = {"name": node["name"], "type": node["type"], "required": node["required"], "description": node["description"]}88 enum, children, variants = [], [], []89 for c in node["children"]:90 if "variant" in c:91 lit = literal(c["variant"])92 if lit is not None:93 enum.append(lit)94 elif c["children"]:95 variants.append({"variant": c["variant"], "description": c["description"],96 "fields": [compact(g, depth + 1, max_depth) for g in c["children"] if "name" in g] if depth < max_depth else "…"})97 else:98 children.append(c)99 if enum:100 out["enum"] = enum101 if variants:102 out["variants"] = variants103 if children:104 out["fields"] = [compact(c, depth + 1, max_depth) for c in children] if depth < max_depth else f"… {len(children)} nested fields (see source)"105 return out106107108def parse_event_file(path: Path, level: str, direction_by_section: bool) -> list[dict]:109 """Split a reference page into event sections. `level` is '##' or '###'."""110 text = path.read_text().splitlines()111 events, cur, direction = [], None, None112 schema_level = level + "#"113 i = 0114 while i < len(text):115 line = text[i]116 if direction_by_section and re.match(r"^## (Client|Server) events", line):117 direction = "client→server" if "Client" in line else "server→client"118 if re.match(rf"^{level} ", line) and not re.match(rf"^{level}#", line):119 title = line[len(level) + 1:].strip()120 if re.fullmatch(r"[a-z_]+(\.[a-z_]+)+", title):121 cur = {"event": title, "direction": direction, "description": [], "schema_lines": [], "example": None, "section": None}122 events.append(cur)123 else:124 cur = None125 i += 1126 continue127 if cur is not None:128 if line.startswith(f"{schema_level} Schema"):129 cur["section"] = "schema"130 elif line.startswith(f"{schema_level} Example"):131 cur["section"] = "example"132 elif line.startswith("```") and cur["section"] == "example":133 j = i + 1134 buf = []135 while j < len(text) and not text[j].startswith("```"):136 buf.append(text[j]); j += 1137 try:138 cur["example"] = json.loads("\n".join(buf))139 except Exception: # noqa: BLE001140 cur["example"] = "\n".join(buf)141 i = j142 elif cur["section"] == "schema":143 cur["schema_lines"].append(line)144 elif cur["section"] is None and line.strip() and not line.startswith("<a id"):145 cur["description"].append(line.strip())146 i += 1147 out = []148 for e in events:149 sname = next((re.search(r"`([^`]+)`", l).group(1) for l in e["schema_lines"] if l.startswith("Schema name:")), None)150 tree = parse_bullets(e["schema_lines"])151 fields = [compact(n, 1, 2) for n in tree if "name" in n]152 out.append({"event": e["event"], "direction": e["direction"], "description": " ".join(e["description"]),153 "schema": {"schema_name": sname, "fields": fields}, "example": e["example"]})154 return out155156157def parse_body_params(path: Path) -> list[dict]:158 text = path.read_text().splitlines()159 buf, on = [], False160 for line in text:161 if line.startswith("### Body Parameters"):162 on = True; continue163 if on and line.startswith("### "):164 break165 if on:166 buf.append(line)167 return parse_bullets(buf)168169170# --------------------------------------------------------------------------- OpenAPI flattener -> parameter records171def deref(s: dict) -> dict:172 while "$ref" in s:173 s = SCHEMAS[s["$ref"].split("/")[-1]]174 return s175176177def type_of(v: dict) -> str:178 v2 = deref(v)179 if "oneOf" in v2 or "anyOf" in v2:180 subs = v2.get("oneOf") or v2.get("anyOf")181 ts = []182 for s in subs:183 sd = deref(s)184 if "$ref" in s:185 ts.append(s["$ref"].split("/")[-1])186 elif sd.get("enum"):187 ts.append("enum")188 else:189 ts.append(sd.get("type", "object") if sd.get("type") != "array" else "array")190 return " | ".join(dict.fromkeys(ts))191 t = v2.get("type", "object" if "properties" in v2 else "any")192 if t == "array":193 it = v2.get("items", {})194 return "array<" + (it["$ref"].split("/")[-1] if "$ref" in it else type_of(it)) + ">"195 if isinstance(t, list):196 return " | ".join(t)197 return t198199200def flatten(schema: dict, prefix: str = "", depth: int = 0, out: list | None = None, seen: tuple = (),201 required_parent: bool = True, variant: str | None = None) -> list[dict]:202 out = out if out is not None else []203 if depth > 9:204 return out205 ref_name = schema.get("$ref", "").split("/")[-1] if "$ref" in schema else None206 if ref_name and ref_name in seen:207 return out208 s = deref(schema)209 seen = seen + ((ref_name,) if ref_name else ())210 for comb in ("oneOf", "anyOf", "allOf"):211 if comb in s:212 for sub in s[comb]:213 sd = deref(sub)214 vname = sd.get("title") or (sub.get("$ref", "").split("/")[-1] if "$ref" in sub else None)215 flatten(sub, prefix, depth, out, seen, required_parent, vname if comb != "allOf" else variant)216 if "properties" not in s:217 return out218 req = set(s.get("required", []))219 for k, v in (s.get("properties") or {}).items():220 vd = deref(v)221 path = prefix + k222 enum = vd.get("enum")223 if not enum and ("anyOf" in vd or "oneOf" in vd): # collect literal members of unions (e.g. "inf", model ids)224 vals = []225 for sub in vd.get("anyOf") or vd.get("oneOf"):226 sd = deref(sub)227 if sd.get("enum"):228 vals += sd["enum"]229 enum = vals or None230 rec = {"parameter": path, "type": type_of(v), "required": k in req and required_parent and depth == 0,231 "nullable": bool(vd.get("nullable")), "default": vd.get("default"), "minimum": vd.get("minimum"), "maximum": vd.get("maximum"),232 "enum": enum, "description": (vd.get("description") or v.get("description") or "").strip()}233 if variant:234 rec["variant"] = variant235 # merge duplicates (same path from several oneOf variants)236 dup = next((r for r in out if r["parameter"] == path), None)237 if dup:238 if rec["enum"] and dup.get("enum") and rec["enum"] != dup["enum"]:239 dup["enum"] = list(dict.fromkeys(dup["enum"] + rec["enum"]))240 if variant and dup.get("variant") and variant not in dup["variant"]:241 dup["variant"] += " | " + variant242 if not dup["description"] and rec["description"]:243 dup["description"] = rec["description"]244 else:245 out.append(rec)246 # recurse into objects / arrays / unions247 if vd.get("type") == "array":248 it = vd.get("items", {})249 itd = deref(it)250 if "properties" in itd or "oneOf" in itd or "anyOf" in itd or "allOf" in itd:251 flatten(it, path + "[].", depth + 1, out, seen)252 elif "properties" in vd or "oneOf" in vd or "anyOf" in vd or "allOf" in vd or "$ref" in v:253 flatten(v, path + ".", depth + 1, out, seen)254 return out255256257def param_records(endpoint: str, schema_name: str | None, source: dict, verified: set[str] = frozenset(),258 status: list[str] | None = None, location: str = "body", compatible_models: list[str] | None = None,259 extra: list[dict] | None = None, prefix: str = "") -> list[dict]:260 recs = flatten({"$ref": f"#/components/schemas/{schema_name}"}, prefix) if schema_name else []261 recs += extra or []262 out = []263 for r in recs:264 st = list(status or ["DOCUMENTED"])265 if r["parameter"] in verified or r["parameter"].split("[")[0] in verified:266 st.append("LIVE_VERIFIED")267 rec = {"provider": "openai", "endpoint": endpoint, "parameter": r["parameter"], "location": r.get("location", location),268 "type": r["type"], "required": r["required"], "default": r.get("default"), "minimum": r.get("minimum"),269 "maximum": r.get("maximum"), "enum": r.get("enum"), "nullable": r.get("nullable", False), "description": r["description"],270 "compatible_models": compatible_models or [], "beta_header": None, "status": st, "source": source}271 if r.get("variant"):272 rec["variant"] = r["variant"]273 out.append(rec)274 return out275276277def q(name, typ, desc, default=None, minimum=None, maximum=None, required=False, enum=None):278 return {"parameter": name, "type": typ, "required": required, "default": default, "minimum": minimum, "maximum": maximum,279 "enum": enum, "description": desc, "location": "query"}280281282# --------------------------------------------------------------------------- probe results283def probe(name: str) -> dict | None:284 p = PROBE / f"{name}.json"285 return json.load(open(p)) if p.exists() else None286287288def verification(name: str, note: str = "") -> dict:289 p = probe(name)290 if not p:291 return {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": note}292 st = p["status"]293 res = "success" if 200 <= st < 300 else ("restricted" if st in (401, 403) else "failure")294 body = p.get("body")295 err = body.get("error", {}) if isinstance(body, dict) else {}296 n = note or ""297 if err:298 n += f" error.message={err.get('message')!r} code={err.get('code')!r} param={err.get('param')!r}"299 return {"method": "live_api", "verified_at": VERIFIED_AT, "result": res, "http_status": st, "request_note": n.strip()}300301302def st_from(name: str, base: list[str]) -> list[str]:303 p = probe(name)304 if not p:305 return base306 if "status" not in p: # WebSocket probe file: success if events were captured307 return base + (["LIVE_VERIFIED"] if p.get("server_event_types") else ["FAILED_VERIFICATION"])308 s = p["status"]309 if 200 <= s < 300:310 return base + ["LIVE_VERIFIED"]311 if s in (401, 403):312 return base + ["ACCOUNT_RESTRICTED"]313 return base + ["FAILED_VERIFICATION"]314315316# =========================================================================== main317def main() -> None:318 ws_probe = probe("ws_realtime_text_only") or {}319 observed = set(ws_probe.get("server_event_types", [])) | set(ws_probe.get("client_events", []))320321 # ---------------- streaming events: realtime322 rt = REF / "realtime"323 ce = parse_event_file(rt / "client-events.md", "##", False)324 se = parse_event_file(rt / "server-events.md", "##", False)325 tce = parse_event_file(rt / "translation-client-events.md", "##", False)326 tse = parse_event_file(rt / "translation-server-events.md", "##", False)327 realtime_events = []328 for lst, direction, fname, api in ((ce, "client→server", "client-events.md", "realtime"), (se, "server→client", "server-events.md", "realtime"),329 (tce, "client→server", "translation-client-events.md", "realtime-translation"),330 (tse, "server→client", "translation-server-events.md", "realtime-translation")):331 for e in lst:332 rec = {"provider": "openai", "api": api, "direction": direction, "event": e["event"], "description": e["description"],333 "schema": e["schema"], "example": e["example"],334 "status": ["DOCUMENTED"] + (["LIVE_VERIFIED"] if e["event"] in observed and api == "realtime" else []),335 "transports": ["websocket", "webrtc-datachannel", "sip-sideband-websocket"] if api == "realtime" else ["websocket", "webrtc-datachannel"],336 "source": src(rt / fname, e["event"])}337 if e["event"] in ("conversation.created", "conversation.item.created"):338 rec["notes"] = "Listed at the end of the server-events reference; `conversation.item.created` is the pre-GA name of `conversation.item.added` and was NOT observed in the GA WebSocket probe (only `.added`/`.done` were)."339 if e["event"] in ("session.update",) and api == "realtime":340 rec["notes"] = "Payload `session` accepts either a realtime session object (type=realtime) or a transcription session object (type=transcription); same object as POST /v1/realtime/client_secrets `session`."341 realtime_events.append(rec)342 # translation-specific top-level note343 for r in realtime_events:344 if r["api"] == "realtime-translation":345 r["connection"] = "wss://api.openai.com/v1/realtime/translations?model=gpt-realtime-translate (or POST /v1/realtime/translations/calls for WebRTC)"346 (OUT / "streaming-events").mkdir(parents=True, exist_ok=True)347 json.dump({"_meta": {"domain": "openai-realtime", "generated_by": "scripts/gen_openai_realtime_audio.py", "last_verified": VERIFIED_AT,348 "client_event_count": sum(1 for r in realtime_events if r["direction"] == "client→server"),349 "server_event_count": sum(1 for r in realtime_events if r["direction"] == "server→client"),350 "observed_server_sequence_text_only_probe": ws_probe.get("server_event_types"),351 "probe_model": "gpt-realtime-mini", "probe_url": ws_probe.get("url")},352 "records": realtime_events}, open(OUT / "streaming-events" / "openai-realtime.json", "w"), indent=1, ensure_ascii=False)353354 # ---------------- streaming events: live (primary / fork / sideband merged)355 live_dir = REF / "live"356 merged: dict[tuple, dict] = {}357 conn_urls = {"primary": "wss://api.openai.com/v1/live/sessions", "fork": "wss://api.openai.com/v1/live/sessions/{session_id}/fork",358 "sideband": "wss://api.openai.com/v1/live/sessions/{session_id}/attach"}359 for conn in ("primary", "fork", "sideband"):360 f = live_dir / f"{conn}-websocket.md"361 for e in parse_event_file(f, "###", True):362 if not e["direction"]:363 continue364 key = (e["direction"], e["event"])365 if key in merged:366 merged[key]["connections"].append(conn)367 merged[key]["sources"].append(src(f, e["event"]))368 else:369 merged[key] = {"provider": "openai", "api": "live", "direction": e["direction"], "event": e["event"], "description": e["description"],370 "schema": e["schema"], "example": e["example"], "connections": [conn], "status": ["DOCUMENTED"],371 "transports": ["websocket", "webrtc-datachannel (primary events except audio)", "sideband websocket"],372 "sources": src(f, e["event"]) and [src(f, e["event"])]}373 live_events = list(merged.values())374 for r in live_events:375 r["source"] = r["sources"][0]376 r["connection_urls"] = {c: conn_urls[c] for c in r["connections"]}377 json.dump({"_meta": {"domain": "openai-live", "generated_by": "scripts/gen_openai_realtime_audio.py", "last_verified": VERIFIED_AT,378 "client_event_count": sum(1 for r in live_events if r["direction"] == "client→server"),379 "server_event_count": sum(1 for r in live_events if r["direction"] == "server→client"),380 "note": "No Live WebSocket was opened (billed per second, $0.05/min, 15 s billed at WebRTC creation). All events DOCUMENTED only."},381 "records": live_events}, open(OUT / "streaming-events" / "openai-live.json", "w"), indent=1, ensure_ascii=False)382383 # ---------------- streaming events: audio (transcription SSE + speech SSE)384 tr_stream = REF / "audio" / "subresources" / "transcriptions" / "streaming-events.md"385 stt_probe = probe("stt_stream_events") or {}386 stt_seen = set(stt_probe.get("body", {}).get("event_types", [])) if isinstance(stt_probe.get("body"), dict) else set()387 diar = probe("stt_diarize") or {}388 audio_events = []389 for e in parse_event_file(tr_stream, "##", False):390 st = ["DOCUMENTED"]391 if e["event"] in stt_seen:392 st.append("LIVE_VERIFIED")393 elif e["event"] == "transcript.text.segment" and diar.get("status") == 200:394 st.append("LIVE_VERIFIED") # observed as segment objects in the non-streaming diarized_json response395 audio_events.append({"provider": "openai", "api": "audio-transcriptions", "endpoint": "POST /v1/audio/transcriptions (stream=true, SSE)",396 "direction": "server→client", "event": e["event"], "description": e["description"], "schema": e["schema"],397 "example": e["example"], "status": st, "source": src(tr_stream, e["event"])})398 tts_probe = probe("tts_sse_events") or {}399 tts_seen = set(tts_probe.get("body", {}).get("event_types", [])) if isinstance(tts_probe.get("body"), dict) else set()400 speech_src = {"url": "https://developers.openai.com/api/reference/resources/audio/subresources/speech/methods/create", "retrieved_at": MANIFEST.get(401 "sources/openai/pages/api/reference/resources/audio/subresources/speech/methods/create.md", {}).get("retrieved_at", "2026-09-19T01:31:00Z")}402 for name, desc, fields, ex in (403 ("speech.audio.delta", "Emitted for each chunk of audio data generated during speech synthesis (stream_format=sse).",404 [{"name": "type", "type": '"speech.audio.delta"', "required": True, "description": "Always `speech.audio.delta`."},405 {"name": "audio", "type": "string", "required": True, "description": "A chunk of Base64-encoded audio data."}],406 {"type": "speech.audio.delta", "audio": "<base64>"}),407 ("speech.audio.done", "Emitted when the speech synthesis is complete and all audio has been streamed.",408 [{"name": "type", "type": '"speech.audio.done"', "required": True, "description": "Always `speech.audio.done`."},409 {"name": "usage", "type": "object { input_tokens, output_tokens, total_tokens }", "required": True, "description": "Token usage statistics for the request."}],410 {"type": "speech.audio.done", "usage": {"input_tokens": 14, "output_tokens": 101, "total_tokens": 115}}),411 ):412 audio_events.append({"provider": "openai", "api": "audio-speech", "endpoint": "POST /v1/audio/speech (stream_format=sse, SSE)", "direction": "server→client",413 "event": name, "description": desc, "schema": {"schema_name": "CreateSpeechResponseStreamEvent", "fields": fields}, "example": ex,414 "status": ["DOCUMENTED"] + (["LIVE_VERIFIED"] if name in tts_seen else []),415 "notes": "Observed 2026-09-18: 4× speech.audio.delta then speech.audio.done then `data: [DONE]` for input 'OK' (gpt-4o-mini-tts, response_format=pcm). Not supported for tts-1/tts-1-hd." if name in tts_seen else None,416 "source": speech_src, "openapi_schema": "CreateSpeechResponseStreamEvent"})417 json.dump({"_meta": {"domain": "openai-audio", "generated_by": "scripts/gen_openai_realtime_audio.py", "last_verified": VERIFIED_AT,418 "observed_stt_stream_sequence": stt_probe.get("body", {}).get("event_types") if isinstance(stt_probe.get("body"), dict) else None,419 "observed_tts_sse_sequence": tts_probe.get("body", {}).get("event_types") if isinstance(tts_probe.get("body"), dict) else None},420 "records": audio_events}, open(OUT / "streaming-events" / "openai-audio-transcription.json", "w"), indent=1, ensure_ascii=False)421422 # ---------------- parameters: realtime423 (OUT / "parameters").mkdir(parents=True, exist_ok=True)424 cs_src = src(rt / "subresources" / "client_secrets" / "methods" / "create.md")425 ce_src = src(rt / "client-events.md")426 verified_cs = {"session", "session.type", "session.model", "expires_after", "expires_after.anchor", "expires_after.seconds",427 "session.audio", "session.audio.input", "session.audio.input.transcription", "session.audio.input.transcription.model"}428 realtime_params = []429 realtime_params += param_records("POST /v1/realtime/client_secrets", "RealtimeCreateClientSecretRequest", cs_src, verified_cs,430 compatible_models=["gpt-realtime-2.1", "gpt-realtime-2.1-mini", "gpt-realtime-2", "gpt-realtime-1.5", "gpt-realtime", "gpt-realtime-mini",431 "gpt-4o-realtime-preview", "gpt-4o-mini-realtime-preview"])432 realtime_params += param_records("WS /v1/realtime session.update", "RealtimeClientEventSessionUpdate", {**ce_src, "url": ce_src["url"] + "#session.update"},433 {"session", "session.type", "session.output_modalities", "session.instructions"})434 realtime_params += param_records("WS /v1/realtime response.create", "RealtimeClientEventResponseCreate", {**ce_src, "url": ce_src["url"] + "#response.create"},435 {"response", "response.output_modalities", "response.max_output_tokens"})436 realtime_params += param_records("WS /v1/realtime conversation.item.create", "RealtimeClientEventConversationItemCreate", {**ce_src, "url": ce_src["url"] + "#conversation.item.create"},437 {"item", "item.type", "item.role", "item.content", "item.content[].type", "item.content[].text"})438 for ev, sch in (("input_audio_buffer.append", "RealtimeClientEventInputAudioBufferAppend"), ("input_audio_buffer.commit", "RealtimeClientEventInputAudioBufferCommit"),439 ("input_audio_buffer.clear", "RealtimeClientEventInputAudioBufferClear"), ("conversation.item.retrieve", "RealtimeClientEventConversationItemRetrieve"),440 ("conversation.item.truncate", "RealtimeClientEventConversationItemTruncate"), ("conversation.item.delete", "RealtimeClientEventConversationItemDelete"),441 ("response.cancel", "RealtimeClientEventResponseCancel"), ("output_audio_buffer.clear", "RealtimeClientEventOutputAudioBufferClear")):442 realtime_params += param_records(f"WS /v1/realtime {ev}", sch, {**ce_src, "url": ce_src["url"] + "#" + ev})443 calls_src = src(rt / "subresources" / "calls" / "methods" / "create.md")444 realtime_params += param_records("POST /v1/realtime/calls", "RealtimeCallCreateRequest", calls_src, extra=[445 {"parameter": "session (JSON string)", "type": "RealtimeSessionCreateRequestGA | RealtimeTranscriptionSessionCreateRequestGA", "required": False, "description":446 "multipart/form-data part `session` (type=application/json) carrying the same session object as POST /v1/realtime/client_secrets `session` (see those records). When the request body is raw `application/sdp`, no session config can be attached (use an ephemeral client secret minted with the config instead)."}])447 realtime_params += param_records("POST /v1/realtime/calls/{call_id}/accept", "RealtimeSessionCreateRequestGA", src(rt / "subresources" / "calls" / "methods" / "accept.md"),448 extra=[{"parameter": "call_id", "type": "string", "required": True, "description": "Path parameter: call ID from the `realtime.call.incoming` webhook.", "location": "path"}])449 realtime_params += param_records("POST /v1/realtime/calls/{call_id}/refer", "RealtimeCallReferRequest", src(rt / "subresources" / "calls" / "methods" / "refer.md"),450 extra=[{"parameter": "call_id", "type": "string", "required": True, "description": "Path parameter.", "location": "path"}])451 realtime_params += param_records("POST /v1/realtime/calls/{call_id}/reject", "RealtimeCallRejectRequest", src(rt / "subresources" / "calls" / "methods" / "reject.md"),452 extra=[{"parameter": "call_id", "type": "string", "required": True, "description": "Path parameter.", "location": "path"}])453 realtime_params += param_records("POST /v1/realtime/calls/{call_id}/hangup", None, src(rt / "subresources" / "calls" / "methods" / "hangup.md"),454 extra=[{"parameter": "call_id", "type": "string", "required": True, "description": "Path parameter. No request body.", "location": "path"}])455 tr_src = src(rt / "translation-client-events.md")456 realtime_params += param_records("POST /v1/realtime/translations/client_secrets", "RealtimeTranslationClientSecretCreateRequest",457 {"url": "https://developers.openai.com/api/reference/resources/realtime/subresources/client_secrets", "retrieved_at": cs_src["retrieved_at"]},458 {"session", "session.model"}, compatible_models=["gpt-realtime-translate"])459 realtime_params += param_records("WS /v1/realtime/translations session.update", "RealtimeTranslationClientEventSessionUpdate", {**tr_src, "url": tr_src["url"] + "#session.update"},460 compatible_models=["gpt-realtime-translate"])461 realtime_params += param_records("WS /v1/realtime/translations session.input_audio_buffer.append", "RealtimeTranslationClientEventInputAudioBufferAppend",462 {**tr_src, "url": tr_src["url"] + "#session.input_audio_buffer.append"}, compatible_models=["gpt-realtime-translate"])463 realtime_params += param_records("WS /v1/realtime/translations session.close", "RealtimeTranslationClientEventSessionClose", {**tr_src, "url": tr_src["url"] + "#session.close"},464 compatible_models=["gpt-realtime-translate"])465 legacy_src = {"url": "https://developers.openai.com/api/reference/realtime-beta/overview", "retrieved_at": cs_src["retrieved_at"]}466 realtime_params += param_records("POST /v1/realtime/sessions", "RealtimeSessionCreateRequest", legacy_src, status=["DOCUMENTED", "LEGACY", "BETA", "FAILED_VERIFICATION"])467 realtime_params += param_records("POST /v1/realtime/transcription_sessions", "RealtimeTranscriptionSessionCreateRequest", legacy_src, status=["DOCUMENTED", "LEGACY", "BETA", "FAILED_VERIFICATION"])468 # connection (query/header) parameters for the WebSocket endpoints469 ws_src = {"url": "https://developers.openai.com/api/docs/guides/voice-websockets?api=realtime", "retrieved_at": cs_src["retrieved_at"]}470 for rec in (471 q("model", "string", "Realtime model to use when opening a fresh WebSocket session with a standard API key (`wss://api.openai.com/v1/realtime?model=…`). Ignored when `call_id` is given."),472 q("call_id", "string", "Attach a sideband WebSocket to an existing WebRTC/SIP call (`wss://api.openai.com/v1/realtime?call_id=rtc_…`). The model is already configured by the call/accept step."),473 {"parameter": "Authorization", "type": "string", "required": True, "default": None, "minimum": None, "maximum": None, "enum": None, "location": "header",474 "description": "`Bearer <standard API key>` (server) or `Bearer <ek_… client secret>` (browser/mobile). In browsers without header support, use WebSocket subprotocols `realtime`, `openai-insecure-api-key.<ek_…>`, optional `openai-organization.<org>`, `openai-project.<proj>`."},475 {"parameter": "OpenAI-Safety-Identifier", "type": "string", "required": False, "default": None, "minimum": None, "maximum": None, "enum": None, "location": "header",476 "description": "Stable, privacy-preserving end-user identifier (hashed). Set on the server-side connection request or on the client_secrets request (bound to the ephemeral token)."},477 {"parameter": "OpenAI-Beta", "type": "string", "required": False, "default": None, "minimum": None, "maximum": None, "enum": ["realtime=v1"], "location": "header",478 "description": "BETA-ERA header only. Remove for the GA interface (docs: 'Remove the OpenAI-Beta: realtime=v1 header when calling the GA interface')."},479 ):480 st = ["DOCUMENTED"] + (["LIVE_VERIFIED"] if rec["parameter"] in ("model", "Authorization") else [])481 realtime_params.append({"provider": "openai", "endpoint": "WS /v1/realtime (connect)", **{k: rec.get(k) for k in ("parameter", "location", "type", "required", "default", "minimum", "maximum", "enum", "description")},482 "nullable": False, "compatible_models": [], "beta_header": "realtime=v1" if rec["parameter"] == "OpenAI-Beta" else None, "status": st, "source": ws_src})483 json.dump({"_meta": {"domain": "openai-realtime", "generated_by": "scripts/gen_openai_realtime_audio.py", "last_verified": VERIFIED_AT, "count": len(realtime_params),484 "note": "Session object flattened from the official OpenAPI spec (RealtimeCreateClientSecretRequest). `variant` marks fields that belong to one oneOf branch (realtime vs transcription session; PCM vs G.711 audio formats; server_vad vs semantic_vad; function vs mcp tools)."},485 "records": realtime_params}, open(OUT / "parameters" / "openai-realtime.json", "w"), indent=1, ensure_ascii=False)486487 # ---------------- parameters: live488 live_params = []489 pw_src = src(live_dir / "primary-websocket.md")490 live_ref = "https://developers.openai.com/api/reference/resources/live/subresources/sessions/methods/"491 live_params += param_records("POST /v1/live/sessions", "LiveCreateRequest", {"url": live_ref + "create", "retrieved_at": pw_src["retrieved_at"]},492 {"session", "session.model", "transport", "transport.type", "transport.sdp"}, compatible_models=["gpt-live-1"])493 live_params += param_records("POST /v1/live/sessions/{session_id}/fork", "LiveForkRequest", {"url": live_ref + "fork", "retrieved_at": pw_src["retrieved_at"]}, compatible_models=["gpt-live-1"],494 extra=[{"parameter": "session_id", "type": "string", "required": True, "description": "Path parameter: stored source session to fork.", "location": "path"}])495 live_params += param_records("POST /v1/live/sessions/{session_id}/accept", "LiveCallAcceptRequest", {"url": live_ref + "accept", "retrieved_at": pw_src["retrieved_at"]}, compatible_models=["gpt-live-1"],496 extra=[{"parameter": "session_id", "type": "string", "required": True, "description": "Path parameter: `data.session_id` from the `live.transport.incoming` webhook.", "location": "path"}])497 live_params += param_records("POST /v1/live/sessions/{session_id}/refer", "LiveCallReferRequest", {"url": live_ref + "refer", "retrieved_at": pw_src["retrieved_at"]},498 extra=[{"parameter": "session_id", "type": "string", "required": True, "description": "Path parameter.", "location": "path"}])499 live_params += param_records("POST /v1/live/sessions/{session_id}/reject", "LiveCallRejectRequest", {"url": live_ref + "reject", "retrieved_at": pw_src["retrieved_at"]},500 extra=[{"parameter": "session_id", "type": "string", "required": True, "description": "Path parameter.", "location": "path"}])501 live_params += param_records("POST /v1/live/sessions/{session_id}/hangup", None, {"url": live_ref + "hangup", "retrieved_at": pw_src["retrieved_at"]},502 extra=[{"parameter": "session_id", "type": "string", "required": True, "description": "Path parameter. No body.", "location": "path"}])503 live_params += param_records("GET /v1/live/sessions/{session_id}/content", None, {"url": live_ref + "content", "retrieved_at": pw_src["retrieved_at"]},504 extra=[{"parameter": "session_id", "type": "string", "required": True, "description": "Path parameter. Returns audio/wav (stereo: left=input, right=output) of a finalized stored session (store=true).", "location": "path"}])505 live_params += param_records("WS /v1/live/sessions session.start", "LiveSessionStartEvent", {**pw_src, "url": pw_src["url"] + "#session.start"}, compatible_models=["gpt-live-1"])506 live_params += param_records("WS /v1/live/sessions/{session_id}/fork session.start", "LiveForkSessionStartEvent", {**src(live_dir / "fork-websocket.md"), "url": src(live_dir / "fork-websocket.md")["url"] + "#session.start"})507 for ev, sch in (("session.update", "LiveSessionUpdateParam"), ("session.input_audio.append", "LiveInputAudioAppendEvent"), ("session.input_audio.mute", "LiveInputAudioMuteParam"),508 ("session.input_audio.unmute", "LiveInputAudioUnmuteParam"), ("session.instructions.append", "LiveInstructionsAppendParam"), ("session.thinking.append", "LiveThinkingAppendParam"),509 ("session.commentary.append", "LiveCommentaryAppendParam"), ("response.item.create", "LiveResponseItemCreateParam"), ("response.create", "LiveResponseCreateParam"),510 ("session.close", "LiveSessionCloseParam")):511 if sch in SCHEMAS:512 live_params += param_records(f"WS /v1/live/sessions {ev}", sch, {**pw_src, "url": pw_src["url"] + "#" + ev})513 json.dump({"_meta": {"domain": "openai-live", "generated_by": "scripts/gen_openai_realtime_audio.py", "last_verified": VERIFIED_AT, "count": len(live_params)},514 "records": live_params}, open(OUT / "parameters" / "openai-live.json", "w"), indent=1, ensure_ascii=False)515516 # ---------------- parameters: audio517 au = REF / "audio" / "subresources"518 audio_params = []519 audio_params += param_records("POST /v1/audio/speech", "CreateSpeechRequest", src(au / "speech" / "methods" / "create.md"),520 {"model", "input", "voice", "response_format", "stream_format"}, compatible_models=["gpt-4o-mini-tts", "gpt-4o-mini-tts-2025-12-15", "tts-1", "tts-1-hd"])521 audio_params += param_records("POST /v1/audio/transcriptions", "CreateTranscriptionRequest", src(au / "transcriptions" / "methods" / "create.md"),522 {"file", "model", "response_format", "timestamp_granularities", "include", "stream"},523 compatible_models=["gpt-transcribe", "gpt-4o-transcribe", "gpt-4o-mini-transcribe", "gpt-4o-transcribe-diarize", "whisper-1"])524 audio_params += param_records("POST /v1/audio/translations", "CreateTranslationRequest", src(au / "translations" / "methods" / "create.md"), {"file", "model", "response_format"}, compatible_models=["whisper-1"])525 audio_params += param_records("POST /v1/audio/voices", "CreateVoiceRequest", src(au / "voices" / "methods" / "create.md"))526 audio_params += param_records("POST /v1/audio/voice_consents", "CreateVoiceConsentRequest", src(au / "voice_consents" / "methods" / "create.md"))527 audio_params += param_records("POST /v1/audio/voice_consents/{consent_id}", "UpdateVoiceConsentRequest", src(au / "voice_consents" / "methods" / "update.md"),528 extra=[{"parameter": "consent_id", "type": "string", "required": True, "description": "Path parameter.", "location": "path"}])529 audio_params += param_records("GET /v1/audio/voice_consents", None, src(au / "voice_consents" / "methods" / "list.md"), extra=[530 q("after", "string", "Cursor for pagination (object ID)."), q("limit", "integer", "1–100, default 20.", default=20, minimum=1, maximum=100)])531 audio_params += param_records("GET /v1/audio/voice_consents/{consent_id}", None, src(au / "voice_consents" / "methods" / "retrieve.md"),532 extra=[{"parameter": "consent_id", "type": "string", "required": True, "description": "Path parameter.", "location": "path"}])533 audio_params += param_records("DELETE /v1/audio/voice_consents/{consent_id}", None, src(au / "voice_consents" / "methods" / "delete.md"),534 extra=[{"parameter": "consent_id", "type": "string", "required": True, "description": "Path parameter.", "location": "path"}])535 # chat completions audio subset536 chat_src = {"url": "https://developers.openai.com/api/docs/guides/audio-chat-completions", "retrieved_at": cs_src["retrieved_at"]}537 cc = SCHEMAS["CreateChatCompletionRequest"]538 props = {}539 for sub in cc.get("allOf", []):540 props.update(deref(sub).get("properties", {}))541 chat_extra = [{"parameter": "modalities", "type": "array<string>", "required": False, "enum": ["text", "audio"], "description": (deref(props["modalities"]).get("description") or "").strip()542 + " For audio output use `[\"text\",\"audio\"]`. LIVE 2026-09-18: gpt-audio-mini with `modalities:[\"text\"]` and text-only input → 400 `This model requires that either input content or output modality contain audio.`"}]543 chat_extra += flatten(props["audio"], "audio.")544 chat_extra += flatten({"$ref": "#/components/schemas/ChatCompletionRequestMessageContentPartAudio"}, "messages[].content[].")545 audio_params += param_records("POST /v1/chat/completions (audio subset)", None, chat_src, {"modalities"}, compatible_models=["gpt-audio-1.5", "gpt-audio", "gpt-audio-mini", "gpt-4o-audio-preview", "gpt-4o-mini-audio-preview"], extra=chat_extra)546 json.dump({"_meta": {"domain": "openai-audio", "generated_by": "scripts/gen_openai_realtime_audio.py", "last_verified": VERIFIED_AT, "count": len(audio_params)},547 "records": audio_params}, open(OUT / "parameters" / "openai-audio.json", "w"), indent=1, ensure_ascii=False)548549 # ---------------- endpoints550 (OUT / "endpoints").mkdir(parents=True, exist_ok=True)551 R = "https://developers.openai.com/api/reference/resources/"552 G = "https://developers.openai.com/api/docs/guides/"553 ra = cs_src["retrieved_at"]554555 def ep(method, path, name, desc, status, auth, req, resp, streaming, sdk, ver, srcs, **kw):556 rec = {"provider": "openai", "api_family": kw.pop("api_family", "realtime"), "method": method, "path": path, "name": name, "description": desc,557 "status": status, "auth": auth, "beta_header": kw.pop("beta_header", None), "request": req, "response": resp, "streaming": streaming,558 "pagination": kw.pop("pagination", None), "idempotency": kw.pop("idempotency", "not documented"), "sdk": sdk, "verification": ver,559 "sources": [{"url": u, "retrieved_at": ra} for u in srcs]}560 rec.update(kw)561 return rec562563 std = "Bearer standard API key (server-side)"564 eph = "Bearer ephemeral client secret `ek_…` (browser/mobile) OR standard API key (server)"565 endpoints = [566 ep("POST", "/v1/realtime/client_secrets", "Create client secret", "Mint a short-lived ephemeral key (`ek_…`, default TTL 600 s, 10–7200 s) bound to a realtime or transcription session configuration. Returns `{value, expires_at, session}`; the session object shows all effective defaults.",567 st_from("client_secrets_minimal", ["DOCUMENTED"]), std, {"content_type": "application/json", "body_ref": "RealtimeCreateClientSecretRequest"},568 {"content_type": "application/json", "body_ref": "RealtimeCreateClientSecretResponse", "observed_example": (probe("client_secrets_minimal") or {}).get("body")},569 {"supported": False, "events_ref": None}, {"python": "client.realtime.client_secrets.create(session={...})", "node": "client.realtime.clientSecrets.create({ session })"},570 verification("client_secrets_minimal", "session={type:realtime, model:gpt-realtime-mini} → 200; expires_at = now+600 s; also tested transcription session + expires_after 60 s (200)"),571 [R + "realtime/subresources/client_secrets/methods/create", G + "voice-webrtc?api=realtime"], safety_identifier_header="OpenAI-Safety-Identifier bound to the token"),572 ep("POST", "/v1/realtime/translations/client_secrets", "Create translation client secret", "Ephemeral key for a `type: translation` session (`gpt-realtime-translate`). Session fields: model, audio.input.{transcription,noise_reduction}, audio.output.language.",573 st_from("translation_client_secret", ["DOCUMENTED"]), std, {"content_type": "application/json", "body_ref": "RealtimeTranslationClientSecretCreateRequest"},574 {"content_type": "application/json", "body_ref": "RealtimeTranslationClientSecretCreateResponse", "observed_example": (probe("translation_client_secret") or {}).get("body")},575 {"supported": False, "events_ref": None}, {"python": "client.realtime.translations.client_secrets.create(...)", "node": "client.realtime.translations.clientSecrets.create(...)"},576 verification("translation_client_secret", "session={model:gpt-realtime-translate} → 200; default output language observed 'es'"),577 [G + "realtime-translation", R + "realtime/translation-client-events"]),578 ep("POST", "/v1/realtime/calls", "Create call (WebRTC unified interface)", "Exchange a WebRTC SDP offer for the SDP answer. Body is either raw `application/sdp` (ephemeral-key flow from the browser) or `multipart/form-data` with `sdp` + optional `session` JSON part (server flow with standard key). Response 201 `application/sdp`; `Location` header carries the `call_id` (`rtc_…`) usable for a sideband WebSocket `wss://api.openai.com/v1/realtime?call_id=…`.",579 ["DOCUMENTED", "UNVERIFIED"], eph, {"content_type": "application/sdp | multipart/form-data", "body_ref": "RealtimeCallCreateRequest"},580 {"content_type": "application/sdp", "status_code": 201, "headers": ["Location: /v1/realtime/calls/rtc_…"]}, {"supported": False, "events_ref": "streaming-events/openai-realtime.json (data channel `oai-events`)"},581 {"python": "client.realtime.calls.create(sdp=..., session=...)", "node": "client.realtime.calls.create({ sdp, session })"},582 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "Needs a real WebRTC peer/SDP offer; not attempted."},583 [R + "realtime/subresources/calls/methods/create", G + "voice-webrtc?api=realtime", G + "voice-server-controls?api=realtime"], transport="webrtc"),584 ep("POST", "/v1/realtime/calls/{call_id}/accept", "Accept call (SIP)", "Accept an inbound SIP call announced by the `realtime.call.incoming` webhook and configure the session (same fields as a client-secret `session`, sent at top level: type, model, instructions, audio, tools…). 200 with empty body once the SIP leg is ringing.",585 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "application/json", "body_ref": "RealtimeSessionCreateRequestGA"}, {"content_type": None, "status_code": 200, "body": "empty"},586 {"supported": False, "events_ref": None}, {"python": "client.realtime.calls.accept(call_id, type='realtime', model=...)", "node": "client.realtime.calls.accept(callId, {...})"},587 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "Requires a SIP trunk + webhook; not attempted."},588 [R + "realtime/subresources/calls/methods/accept", G + "voice-sip?api=realtime"], transport="sip", webhook="realtime.call.incoming"),589 ep("POST", "/v1/realtime/calls/{call_id}/reject", "Reject call (SIP)", "Decline an inbound SIP call with an optional SIP `status_code` (default 603 Decline).",590 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "application/json", "body_ref": "RealtimeCallRejectRequest"}, {"content_type": None, "status_code": 200, "body": "empty"},591 {"supported": False, "events_ref": None}, {"python": "client.realtime.calls.reject(call_id, status_code=486)", "node": "client.realtime.calls.reject(callId, { status_code: 486 })"},592 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "not attempted"}, [R + "realtime/subresources/calls/methods/reject", G + "voice-sip?api=realtime"], transport="sip"),593 ep("POST", "/v1/realtime/calls/{call_id}/refer", "Refer call (SIP transfer)", "Transfer an active SIP call via SIP REFER; `target_uri` goes into the Refer-To header (`tel:+1…` or `sip:agent@example.com`).",594 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "application/json", "body_ref": "RealtimeCallReferRequest"}, {"content_type": None, "status_code": 200, "body": "empty"},595 {"supported": False, "events_ref": None}, {"python": "client.realtime.calls.refer(call_id, target_uri='tel:+1…')", "node": "client.realtime.calls.refer(callId, { target_uri })"},596 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "not attempted"}, [R + "realtime/subresources/calls/methods/refer"], transport="sip"),597 ep("POST", "/v1/realtime/calls/{call_id}/hangup", "Hang up call", "End an active SIP or WebRTC call. No body. LIVE: bogus call_id → 404 `call_id_not_found` ('No session found for the provided call_id'), which confirms the route exists.",598 ["DOCUMENTED", "LIVE_DISCOVERED"], std, {"content_type": None, "body_ref": None}, {"content_type": None, "status_code": 200, "body": "empty"},599 {"supported": False, "events_ref": None}, {"python": "client.realtime.calls.hangup(call_id)", "node": "client.realtime.calls.hangup(callId)"},600 verification("calls_hangup_bogus", "POST with call_id=rtc_bogus"), [R + "realtime/subresources/calls/methods/hangup"], transport="sip|webrtc"),601 ep("POST", "/v1/realtime/sessions", "Create session (LEGACY beta)", "Beta-era endpoint returning an ephemeral `client_secret` for a realtime session. Still present in the OpenAPI spec but LIVE 2026-09-18 → 404 `Invalid URL (POST /v1/realtime/sessions)` even with `OpenAI-Beta: realtime=v1`. Use /v1/realtime/client_secrets.",602 ["DOCUMENTED", "LEGACY", "BETA", "FAILED_VERIFICATION"], std, {"content_type": "application/json", "body_ref": "RealtimeSessionCreateRequest"}, {"content_type": "application/json", "body_ref": "RealtimeSessionCreateResponse"},603 {"supported": False, "events_ref": None}, {"python": "client.beta.realtime.sessions.create(...)", "node": "client.beta.realtime.sessions.create(...)"},604 verification("legacy_sessions", "body {model: gpt-realtime-mini} + OpenAI-Beta: realtime=v1"), ["https://developers.openai.com/api/reference/realtime-beta/overview"], beta_header="OpenAI-Beta: realtime=v1"),605 ep("POST", "/v1/realtime/transcription_sessions", "Create transcription session (LEGACY beta)", "Beta-era ephemeral token for transcription-only sessions. LIVE 2026-09-18 → 404 `Invalid URL`. Use /v1/realtime/client_secrets with `session.type: transcription`. Note: model pages still label the route `v1/realtime/transcription_sessions` in their endpoint-support tables.",606 ["DOCUMENTED", "LEGACY", "BETA", "FAILED_VERIFICATION"], std, {"content_type": "application/json", "body_ref": "RealtimeTranscriptionSessionCreateRequest"}, {"content_type": "application/json", "body_ref": "RealtimeTranscriptionSessionCreateResponse"},607 {"supported": False, "events_ref": None}, {"python": "client.beta.realtime.transcription_sessions.create(...)", "node": "client.beta.realtime.transcriptionSessions.create(...)"},608 verification("legacy_transcription_sessions", "body {} + OpenAI-Beta: realtime=v1"), ["https://developers.openai.com/api/reference/realtime-beta/overview"], beta_header="OpenAI-Beta: realtime=v1"),609 # ---- pseudo endpoints: connections610 ep("WS", "wss://api.openai.com/v1/realtime?model={model}", "Realtime WebSocket (new session)", "Bidirectional JSON events. Server sends `session.created` first. Audio travels as base64 in `input_audio_buffer.append` / `response.output_audio.delta`. Max session 60 min. LIVE probe: gpt-realtime-mini, text-only → 14 server events in 3.6 s, usage 13 in / 3 out text tokens.",611 st_from("ws_realtime_text_only", ["DOCUMENTED"]), eph + "; browser subprotocols: `realtime`, `openai-insecure-api-key.<ek>`, `openai-organization.<org>`, `openai-project.<proj>`",612 {"content_type": "websocket text frames (JSON client events)", "body_ref": "RealtimeClientEvent"}, {"content_type": "websocket text frames (JSON server events)", "body_ref": "RealtimeServerEvent"},613 {"supported": True, "events_ref": "streaming-events/openai-realtime.json"}, {"python": "client.realtime.connect(model=...) (async with … as conn)", "node": "new OpenAIRealtimeWS({ model }, client) from 'openai/realtime/ws'"},614 {"method": "live_api", "verified_at": VERIFIED_AT, "result": "success", "http_status": 101, "request_note": "session.update(output_modalities=[text]) + conversation.item.create + response.create(max_output_tokens=16) → response.done; observed order: " + ", ".join(ws_probe.get("server_event_types", []))},615 [G + "voice-websockets?api=realtime", R + "realtime/client-events", R + "realtime/server-events"], transport="websocket",616 headers={"Authorization": "Bearer …", "OpenAI-Safety-Identifier": "optional", "OpenAI-Beta": "NOT needed for GA (beta only)"}, query={"model": "required unless call_id", "call_id": "attach to existing call"}),617 ep("WS", "wss://api.openai.com/v1/realtime?call_id={call_id}", "Realtime WebSocket sideband (existing WebRTC/SIP call)", "Second connection to an existing call (call_id from the `Location` header of POST /v1/realtime/calls or from the `realtime.call.incoming` webhook). `model` is ignored. Used by the application server to monitor, update instructions and answer tool calls while the client keeps the media.",618 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "websocket JSON", "body_ref": "RealtimeClientEvent"}, {"content_type": "websocket JSON", "body_ref": "RealtimeServerEvent"},619 {"supported": True, "events_ref": "streaming-events/openai-realtime.json"}, {"python": "client.realtime.connect(call_id=...)", "node": "new OpenAIRealtimeWS({ callId })"},620 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "needs an active call"}, [G + "voice-server-controls?api=realtime", G + "voice-sip?api=realtime"], transport="websocket-sideband"),621 ep("WS", "wss://api.openai.com/v1/realtime/translations?model=gpt-realtime-translate", "Realtime translation WebSocket", "Dedicated continuous translation session: no conversation/response lifecycle, no `response.create`. Client: session.update{audio.output.language}, session.input_audio_buffer.append (24 kHz PCM16 base64), session.close. Server: session.created/updated, session.input_transcript.delta, session.output_transcript.delta, session.output_audio.delta, session.closed, error. WebRTC variant: POST /v1/realtime/translations/calls with the SDP offer + ek_ token.",622 ["DOCUMENTED", "UNVERIFIED"], eph, {"content_type": "websocket JSON", "body_ref": "RealtimeTranslationClientEvent"}, {"content_type": "websocket JSON", "body_ref": "RealtimeTranslationServerEvent"},623 {"supported": True, "events_ref": "streaming-events/openai-realtime.json (api=realtime-translation)"}, {"python": "not in SDK surface checked", "node": "not in SDK surface checked"},624 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "billed $0.034/min; not opened"}, [G + "realtime-translation", R + "realtime/translation-server-events"], transport="websocket"),625 ep("POST", "/v1/realtime/translations/calls", "Create translation call (WebRTC)", "WebRTC SDP exchange for a translation session (documented in the translation guide only; NOT present in openapi-master.yaml).",626 ["DOCUMENTED", "UNVERIFIED"], "Bearer ephemeral client secret from /v1/realtime/translations/client_secrets", {"content_type": "application/sdp", "body_ref": None}, {"content_type": "application/sdp"},627 {"supported": False, "events_ref": None}, {"python": None, "node": None}, {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "not attempted"}, [G + "realtime-translation"], transport="webrtc"),628 ep("SIP", "sip:{PROJECT_ID}@sip.api.openai.com;transport=tls", "Realtime SIP endpoint", "Point a SIP trunk (Twilio, Telnyx…) at this URI (EU residency: `sip-eu.api.openai.com`). Signaling TLS on TCP 5061; media SRTP/UDP from 13.79.45.80/28, 23.98.140.64/28, 40.67.149.176/28, 40.83.204.240/28. Inbound INVITE fires the `realtime.call.incoming` webhook (`data.call_id`, `data.sip_headers`); then accept/reject/refer/hangup via REST and attach a WebSocket with `call_id`.",629 ["DOCUMENTED", "UNVERIFIED"], "project webhook + standard API key for call control", {"content_type": "SIP INVITE", "body_ref": None}, {"content_type": "SIP", "body_ref": None},630 {"supported": True, "events_ref": "webhook realtime.call.incoming + streaming-events/openai-realtime.json"}, {"python": "client.webhooks.unwrap(...)", "node": "client.webhooks.unwrap(...)"},631 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "no SIP trunk available"}, [G + "voice-sip?api=realtime"], transport="sip"),632 # ---- Live API633 ep("POST", "/v1/live/sessions", "Create Live session (WebRTC)", "Start a GPT-Live session from an SDP offer: body `{session:{model, instructions, audio.output.voice, delegation, store, input, client}, transport:{type:'webrtc', sdp}}` → 201 `{session:{id}, transport:{type, sdp}}`. Bills 15 s of voice at creation (credited later). LIVE: without transport → 400 `invalid_value` param `transport.type` 'Only the webrtc transport is supported.'; with a bogus SDP → 400 `invalid_offer` 'Offer did not have an audio media section.' — confirms the route and that our key has Live access.",634 ["DOCUMENTED", "LIVE_DISCOVERED"], std, {"content_type": "application/json", "body_ref": "LiveCreateRequest"}, {"content_type": "application/json", "status_code": 201, "body_ref": "LiveCreateResponse"},635 {"supported": False, "events_ref": "streaming-events/openai-live.json (data channel `oai-events`)"}, {"python": "client.live.create(session=..., transport={'type':'webrtc','sdp':...})", "node": "client.live.create({ session, transport })"},636 verification("live_sessions_bad_sdp", "two probes: no transport → 400 transport.type; bogus sdp → 400 invalid_offer"), [live_ref + "create", G + "voice-webrtc?api=live"], api_family="live", transport="webrtc"),637 ep("POST", "/v1/live/sessions/{session_id}/fork", "Fork stored Live session (WebRTC)", "New session continuing a finalized stored (`store: true`) session; overrides limited to store, delegation.responses, client permissions. 201 LiveCreateResponse.",638 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "application/json", "body_ref": "LiveForkRequest"}, {"content_type": "application/json", "status_code": 201, "body_ref": "LiveCreateResponse"},639 {"supported": False, "events_ref": None}, {"python": "client.live.fork(session_id, ...)", "node": "client.live.fork(sessionId, {...})"},640 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "requires a stored session"}, [live_ref + "fork", G + "live-conversations#store-and-fork-a-session"], api_family="live", transport="webrtc"),641 ep("POST", "/v1/live/sessions/{session_id}/accept", "Accept Live SIP call", "Accept an inbound SIP call announced by the `live.transport.incoming` webhook (`data.type: sip`, `data.session_id`). Body `{session:{type:'live', model, instructions, audio.output.voice, delegation, store, input}}`; 200 empty.",642 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "application/json", "body_ref": "LiveCallAcceptRequest"}, {"content_type": None, "status_code": 200, "body": "empty"},643 {"supported": False, "events_ref": None}, {"python": "client.live.accept(session_id, session={...})", "node": "client.live.accept(sessionId, { session })"},644 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "requires SIP"}, [live_ref + "accept", G + "voice-sip?api=live"], api_family="live", transport="sip", webhook="live.transport.incoming (deprecated alias live.call.incoming)"),645 ep("POST", "/v1/live/sessions/{session_id}/reject", "Reject Live SIP call", "Reject with a required SIP `status_code` (300–699). First accept/reject wins; later → `decision_already_made`.",646 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "application/json", "body_ref": "LiveCallRejectRequest"}, {"content_type": None, "status_code": 200, "body": "empty"},647 {"supported": False, "events_ref": None}, {"python": "client.live.reject(session_id, status_code=486)", "node": "client.live.reject(sessionId, { status_code: 486 })"},648 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "requires SIP"}, [live_ref + "reject"], api_family="live", transport="sip"),649 ep("POST", "/v1/live/sessions/{session_id}/refer", "Transfer Live SIP call", "SIP REFER to `target_uri`.", ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "application/json", "body_ref": "LiveCallReferRequest"},650 {"content_type": None, "status_code": 200, "body": "empty"}, {"supported": False, "events_ref": None}, {"python": "client.live.refer(session_id, target_uri=...)", "node": "client.live.refer(sessionId, { target_uri })"},651 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "requires SIP"}, [live_ref + "refer"], api_family="live", transport="sip"),652 ep("POST", "/v1/live/sessions/{session_id}/hangup", "Hang up Live session", "End a Live session (SIP or WebRTC). No body; 200 empty. `session.closed` reason `close_requested`.", ["DOCUMENTED", "UNVERIFIED"], std,653 {"content_type": None, "body_ref": None}, {"content_type": None, "status_code": 200, "body": "empty"}, {"supported": False, "events_ref": None},654 {"python": "client.live.hangup(session_id)", "node": "client.live.hangup(sessionId)"}, {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "no live session"}, [live_ref + "hangup"], api_family="live"),655 ep("GET", "/v1/live/sessions/{session_id}/content", "Download Live recording", "Binary stereo WAV of a finalized stored session (left=input, right=output). Stored recordings kept 30 days; unavailable under ZDR.", ["DOCUMENTED", "UNVERIFIED"], std,656 {"content_type": None, "body_ref": None}, {"content_type": "audio/wav"}, {"supported": False, "events_ref": None}, {"python": "client.live.content(session_id)", "node": "client.live.content(sessionId)"},657 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "no stored session"}, [live_ref + "content", G + "live-conversations#download-a-recording"], api_family="live"),658 ep("WS", "wss://api.openai.com/v1/live/sessions", "Live primary WebSocket", "Server-side audio + events. No query params; first message MUST be `session.start` {session:{model:'gpt-live-1', instructions, audio.format, audio.output.voice, delegation, store, input}}; wait for `session.started`. Audio: `session.input_audio.append` (base64 raw, formats audio/pcm 24k|16k, audio/pcmu 8k, audio/pcma 8k) / `session.output_audio.delta`. Close with `session.close` → `session.closed` (final usage.seconds, reason).",659 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "websocket JSON", "body_ref": "LiveClientEvent"}, {"content_type": "websocket JSON", "body_ref": "LiveServerEvent"},660 {"supported": True, "events_ref": "streaming-events/openai-live.json"}, {"python": "client.live.connect() (AsyncOpenAI)", "node": "new LiveWS(client) from 'openai/resources/live/ws'"},661 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "not opened: billed per second"}, [R + "live/primary-websocket", G + "voice-websockets?api=live"], api_family="live", transport="websocket"),662 ep("WS", "wss://api.openai.com/v1/live/sessions/{session_id}/fork", "Live fork WebSocket", "Fork a stored session on a WebSocket; first message `session.start` with `{session: {}}` (or overrides: store, delegation.responses, audio.format); do not supply a new model.",663 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "websocket JSON", "body_ref": "LiveForkClientEvent"}, {"content_type": "websocket JSON", "body_ref": "LiveForkServerEvent"},664 {"supported": True, "events_ref": "streaming-events/openai-live.json"}, {"python": "client.live.connect(fork=session_id) (see SDK)", "node": "LiveWS fork (see SDK)"},665 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "requires stored session"}, [R + "live/fork-websocket"], api_family="live", transport="websocket"),666 ep("WS", "wss://api.openai.com/v1/live/sessions/{session_id}/attach", "Live sideband WebSocket", "Attach a backend to a running WebRTC/SIP session: receives all session events (incl. reflected `session.input_audio.append` and `session.output_audio.delta` as 24 kHz PCM16 with start_ms/end_ms per the guide) and can send commands (session.update, *.append, response.item.create, response.create, mute/unmute, session.close). Do NOT send session.start or input audio.",667 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "websocket JSON", "body_ref": "LiveSidebandClientEvent"}, {"content_type": "websocket JSON", "body_ref": "LiveSidebandServerEvent"},668 {"supported": True, "events_ref": "streaming-events/openai-live.json"}, {"python": "AsyncSidebandConnection (see SDK)", "node": "see SDK"},669 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "requires running session"}, [R + "live/sideband-websocket", G + "voice-server-controls?api=live"], api_family="live", transport="websocket-sideband"),670 # ---- Audio REST671 ep("POST", "/v1/audio/speech", "Create speech (TTS)", "Text → audio. Models gpt-4o-mini-tts (instructions supported, ≤2000 input tokens, 13 voices), tts-1/tts-1-hd (9 voices, no instructions/SSE). `response_format` mp3|opus|aac|flac|wav|pcm; `speed` 0.25–4.0; `stream_format` audio (chunked bytes, default) | sse (speech.audio.delta/done). LIVE: 'OK' → 17,664-byte mp3; SSE → 4 deltas + done.",672 st_from("tts_mp3", ["DOCUMENTED"]), std, {"content_type": "application/json", "body_ref": "CreateSpeechRequest"}, {"content_type": "application/octet-stream (audio) | text/event-stream", "observed": "17664 bytes audio/mpeg"},673 {"supported": True, "events_ref": "streaming-events/openai-audio-transcription.json (api=audio-speech)"}, {"python": "client.audio.speech.create(model, voice, input, ...) / .with_streaming_response", "node": "client.audio.speech.create({...})"},674 verification("tts_mp3", "gpt-4o-mini-tts, voice alloy, input 'OK', mp3 → 200; also stream_format=sse+pcm → 200 SSE"), [R + "audio/subresources/speech/methods/create", G + "text-to-speech"], api_family="audio"),675 ep("POST", "/v1/audio/transcriptions", "Create transcription (STT)", "multipart/form-data: file (≤25 MB; flac, mp3, mp4, mpeg, mpga, m4a, ogg, wav, webm), model, language|languages[], prompt, keywords[], response_format json|text|srt|verbose_json|vtt|diarized_json, temperature, timestamp_granularities[] (whisper-1 + verbose_json), include[]=logprobs (gpt-4o-*-transcribe, json), stream, chunking_strategy auto|server_vad{...}, known_speaker_names[]/references[] (diarize). LIVE: 5 successful variants.",676 st_from("stt_mini_json", ["DOCUMENTED"]), std, {"content_type": "multipart/form-data", "body_ref": "CreateTranscriptionRequest"},677 {"content_type": "application/json | text/plain | text/event-stream", "body_ref": "Transcription | TranscriptionVerbose | TranscriptionDiarized", "observed_examples": {k: (probe(k) or {}).get("body") for k in ("stt_mini_json", "stt_whisper_verbose", "stt_diarize", "stt_logprobs")}},678 {"supported": True, "events_ref": "streaming-events/openai-audio-transcription.json"}, {"python": "client.audio.transcriptions.create(file=..., model=...)", "node": "client.audio.transcriptions.create({ file, model })"},679 verification("stt_mini_json", "gpt-4o-mini-transcribe json; whisper-1 verbose_json word+segment; gpt-4o-transcribe-diarize diarized_json; include[]=logprobs; stream=true — all 200"),680 [R + "audio/subresources/transcriptions/methods/create", G + "speech-to-text"], api_family="audio"),681 ep("POST", "/v1/audio/translations", "Create translation (→ English)", "whisper-1 only. multipart: file, model, prompt, response_format json|text|srt|verbose_json|vtt, temperature. Output always English.",682 st_from("translation_whisper", ["DOCUMENTED"]), std, {"content_type": "multipart/form-data", "body_ref": "CreateTranslationRequest"}, {"content_type": "application/json", "body_ref": "Translation | TranslationVerbose", "observed_example": (probe("translation_whisper") or {}).get("body")},683 {"supported": False, "events_ref": None}, {"python": "client.audio.translations.create(file=..., model='whisper-1')", "node": "client.audio.translations.create({...})"},684 verification("translation_whisper", "whisper-1 json on ok.mp3 → {'text': 'OK.'}"), [R + "audio/subresources/translations/methods/create", G + "speech-to-text#translations"], api_family="audio"),685 ep("POST", "/v1/audio/voices", "Create custom voice", "multipart: name, consent (cons_… id), audio_sample (≤10 MiB; audio/mpeg, wav, x-wav, ogg, aac, flac, webm, mp4; ≤30 s, ≥5 s speech). Returns `audio.voice` {id, name, created_at}. Max 20 voices/org; requires custom-voice access (`api.voices.write`). Not called (creates a resource).",686 ["DOCUMENTED", "ACCOUNT_RESTRICTED", "UNVERIFIED"], std + " (project-scoped key with custom voice access)", {"content_type": "multipart/form-data", "body_ref": "CreateVoiceRequest"}, {"content_type": "application/json", "body_ref": "VoiceResource"},687 {"supported": False, "events_ref": None}, {"python": "client.audio.voices.create(name=..., consent=..., audio_sample=...)", "node": "client.audio.voices.create({...})"},688 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "not attempted (creates a voice; access gated). GET /v1/audio/voices probe → 404 'Endpoint not found.' (no list endpoint in spec either)"},689 [R + "audio/subresources/voices/methods/create", G + "custom-voices"], api_family="audio"),690 ep("GET", "/v1/audio/voices", "List voices (NOT DOCUMENTED)", "Probe only: the spec has no GET on /audio/voices; LIVE → 404 `Endpoint not found.` Built-in voice names are enumerated in the speech `voice` parameter instead.",691 ["UNVERIFIED", "FAILED_VERIFICATION"], std, {"content_type": None, "body_ref": None}, {"content_type": "application/json", "observed": (probe("voices_get") or {}).get("body")}, {"supported": False, "events_ref": None}, {"python": None, "node": None},692 verification("voices_get", "GET"), [R + "audio/subresources/voices/methods/create"], api_family="audio"),693 ep("POST", "/v1/audio/voice_consents", "Create voice consent", "multipart: name, language (BCP 47 e.g. en-US), recording (≤10 MiB) of one of the 17 exact consent phrases (de,en,es,fr,hi,id,it,ja,ko,nl,pl,pt,ru,uk,vi,zh). Returns `audio.voice_consent`. Not called.",694 ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "multipart/form-data", "body_ref": "CreateVoiceConsentRequest"}, {"content_type": "application/json", "body_ref": "VoiceConsentResource"},695 {"supported": False, "events_ref": None}, {"python": "client.audio.voice_consents.create(...)", "node": "client.audio.voiceConsents.create(...)"},696 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "not attempted (creates a resource)"}, [R + "audio/subresources/voice_consents/methods/create", G + "custom-voices"], api_family="audio"),697 ep("GET", "/v1/audio/voice_consents", "List voice consents", "Cursor list (`after`, `limit` 1–100 default 20) → {object:'list', data:[…], first_id, last_id, has_more}. LIVE 2026-09-18 with our key: 404 `Endpoint not found.` — most likely gated behind custom-voice access (docs: 'Custom voices are limited to eligible customers'); a 404 does not mean the route does not exist.",698 ["DOCUMENTED", "FAILED_VERIFICATION"], std, {"content_type": None, "body_ref": None}, {"content_type": "application/json", "body_ref": "VoiceConsentListResource", "observed": (probe("voice_consents_list") or {}).get("body")},699 {"supported": False, "events_ref": None}, {"python": "client.audio.voice_consents.list(limit=20)", "node": "client.audio.voiceConsents.list({ limit: 20 })"},700 verification("voice_consents_list", "GET ?limit=5"), [R + "audio/subresources/voice_consents/methods/list"], api_family="audio", pagination={"style": "cursor", "params": ["after", "limit"], "response": ["first_id", "last_id", "has_more"]}),701 ep("GET", "/v1/audio/voice_consents/{consent_id}", "Retrieve voice consent", "Returns VoiceConsentResource.", ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": None, "body_ref": None}, {"content_type": "application/json", "body_ref": "VoiceConsentResource"},702 {"supported": False, "events_ref": None}, {"python": "client.audio.voice_consents.retrieve(consent_id)", "node": "client.audio.voiceConsents.retrieve(consentId)"},703 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "list returned 404; no id to retrieve"}, [R + "audio/subresources/voice_consents/methods/retrieve"], api_family="audio"),704 ep("POST", "/v1/audio/voice_consents/{consent_id}", "Update voice consent (metadata)", "Body {name}. Returns VoiceConsentResource.", ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": "application/json", "body_ref": "UpdateVoiceConsentRequest"}, {"content_type": "application/json", "body_ref": "VoiceConsentResource"},705 {"supported": False, "events_ref": None}, {"python": "client.audio.voice_consents.update(consent_id, name=...)", "node": "client.audio.voiceConsents.update(consentId, { name })"},706 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "not attempted"}, [R + "audio/subresources/voice_consents/methods/update"], api_family="audio"),707 ep("DELETE", "/v1/audio/voice_consents/{consent_id}", "Delete voice consent", "Returns {id, object:'audio.voice_consent', deleted:true}. Destructive — never run by the atlas.", ["DOCUMENTED", "UNVERIFIED"], std, {"content_type": None, "body_ref": None},708 {"content_type": "application/json", "body_ref": "VoiceConsentDeletedResource"}, {"supported": False, "events_ref": None}, {"python": "client.audio.voice_consents.delete(consent_id)", "node": "client.audio.voiceConsents.delete(consentId)"},709 {"method": "docs_only", "verified_at": VERIFIED_AT, "result": None, "http_status": None, "request_note": "destructive; not attempted"}, [R + "audio/subresources/voice_consents/methods/delete"], api_family="audio"),710 ep("POST", "/v1/chat/completions (audio)", "Chat Completions with audio (pointer)", "Audio in/out for gpt-audio-1.5 / gpt-audio / gpt-audio-mini / gpt-4o-(mini-)audio-preview: `modalities: [\"text\",\"audio\"]`, `audio: {voice, format wav|aac|mp3|flac|opus|pcm16}`, input part `{type:'input_audio', input_audio:{data (base64), format wav|mp3}}`. LIVE: gpt-audio-mini with modalities ['text'] and text input only → 400 invalid_value (param model): 'This model requires that either input content or output modality contain audio.' Full endpoint owned by the chat-completions agent.",711 ["DOCUMENTED", "LIVE_DISCOVERED"], std, {"content_type": "application/json", "body_ref": "CreateChatCompletionRequest (audio, modalities, input_audio part)"}, {"content_type": "application/json", "body_ref": "ChatCompletion (message.audio {id, data, transcript, expires_at})"},712 {"supported": True, "events_ref": "chat completions SSE (owned by chat agent)"}, {"python": "client.chat.completions.create(model='gpt-audio-mini', modalities=['text','audio'], audio={'voice':'alloy','format':'wav'}, ...)", "node": "client.chat.completions.create({...})"},713 verification("chat_gpt_audio_mini_text", "modalities=[text], max_completion_tokens=8"), [G + "audio-chat-completions"], api_family="chat-audio"),714 ]715 json.dump({"_meta": {"domain": "openai-realtime-live-audio", "generated_by": "scripts/gen_openai_realtime_audio.py", "last_verified": VERIFIED_AT, "count": len(endpoints),716 "live_calls_logged": "reports/live-requests.jsonl (2026-09-19T01:4x Z, provider openai)", "raw_probe_dir": "tmp-live/realtime-audio/ (gitignored, ek_ values masked)"},717 "records": endpoints}, open(OUT / "endpoints" / "openai-realtime-live-audio.json", "w"), indent=1, ensure_ascii=False, default=str)718719 render_events_doc(realtime_events, live_events, audio_events, ws_probe)720 print("realtime events", len(realtime_events), "| live events", len(live_events), "| audio events", len(audio_events))721 print("params realtime", len(realtime_params), "| live", len(live_params), "| audio", len(audio_params), "| endpoints", len(endpoints))722723724def _fields(rec: dict) -> str:725 names = []726 for f in rec["schema"]["fields"]:727 n = f["name"]728 if isinstance(f.get("fields"), list):729 n += "{" + ", ".join(g["name"] for g in f["fields"][:8]) + ("…" if len(f["fields"]) > 8 else "") + "}"730 elif f.get("variants"):731 n += "{" + " | ".join(v["variant"].split(" ")[0] for v in f["variants"]) + "}"732 names.append(f"`{n}`" + ("" if f["required"] else "?"))733 return ", ".join(names)734735736def _esc(s: str) -> str:737 return (s or "").replace("|", "\\|").replace("\n", " ")738739740def render_events_doc(rt: list[dict], lv: list[dict], au: list[dict], ws_probe: dict) -> None:741 L = []742 L.append("# OpenAI Realtime / Live / Audio — full event reference\n")743 L.append("**Status**: DOCUMENTED (all events, parsed from the official reference pages) · LIVE_VERIFIED where marked ✅ (observed 2026-09-18 with our key). "744 "Machine-readable twin: `generated/fragments/streaming-events/openai-{realtime,live,audio-transcription}.json` (each record carries the full field tree to depth 2 + the official example).\n")745 L.append("**Sources**: https://developers.openai.com/api/reference/resources/realtime/client-events · …/realtime/server-events · …/realtime/translation-client-events · …/realtime/translation-server-events · "746 "https://developers.openai.com/api/reference/resources/live/primary-websocket · …/live/fork-websocket · …/live/sideband-websocket · https://developers.openai.com/api/reference/resources/audio/subresources/transcriptions/streaming-events · …/audio/subresources/speech/methods/create\n")747 L.append(f"**Last verified**: {VERIFIED_AT}\n")748 L.append("Legend: `field?` = optional; `a{b, c}` = object with listed sub-fields; ✅ = observed live.\n")749 n_c = sum(1 for r in rt if r["direction"] == "client→server" and r["api"] == "realtime")750 n_s = sum(1 for r in rt if r["direction"] == "server→client" and r["api"] == "realtime")751 L.append(f"\n## 1. Realtime API (`wss://api.openai.com/v1/realtime`) — {n_c} client events, {n_s} server events\n")752 L.append("### 1.1 Observed server-event sequence (text-only turn, gpt-realtime-mini, 2026-09-18)\n")753 L.append("Client sent: `session.update` → `conversation.item.create` → `response.create`. Server sent, in order:\n")754 L.append("```\n" + "\n".join(f"{i+1:2d}. {t}" for i, t in enumerate(ws_probe.get("server_event_types", []))) + "\n```\n")755 L.append("Note: `conversation.item.added` is emitted twice (once for the user item, once for the assistant item, the latter right after `response.output_item.added`). "756 "`rate_limits.updated` was NOT emitted in this run (it is documented as emitted at the start of a response). No `response.output_audio*` events because `output_modalities` was `[\"text\"]`.\n")757 for api, direction, title in (("realtime", "client→server", "1.2 Client events (client → server)"), ("realtime", "server→client", "1.3 Server events (server → client)"),758 ("realtime-translation", "client→server", "1.4 Translation session client events (`/v1/realtime/translations`)"),759 ("realtime-translation", "server→client", "1.5 Translation session server events")):760 L.append(f"\n### {title}\n")761 L.append("| Event | Live | Description | Payload fields |\n|---|---|---|---|")762 for r in rt:763 if r["api"] == api and r["direction"] == direction:764 L.append(f"| `{r['event']}` | {'✅' if 'LIVE_VERIFIED' in r['status'] else ''} | {_esc(r['description'])[:400]} | {_fields(r)} |")765 L.append("\n### 1.6 Realtime event families at a glance\n")766 L.append("| Family | Client events | Server events |\n|---|---|---|")767 fam = {}768 for r in rt:769 if r["api"] != "realtime":770 continue771 k = r["event"].split(".")[0]772 fam.setdefault(k, {"c": [], "s": []})["c" if r["direction"] == "client→server" else "s"].append(r["event"])773 for k, v in fam.items():774 L.append(f"| `{k}.*` | {', '.join('`'+e+'`' for e in v['c']) or '—'} | {', '.join('`'+e+'`' for e in v['s']) or '—'} |")775 L.append("\n## 2. Live API (GPT-Live) WebSocket events — {} client, {} server (DOCUMENTED only; no Live session was opened)\n".format(776 sum(1 for r in lv if r["direction"] == "client→server"), sum(1 for r in lv if r["direction"] == "server→client")))777 L.append("Connections: **primary** `wss://api.openai.com/v1/live/sessions` · **fork** `wss://api.openai.com/v1/live/sessions/{session_id}/fork` · **sideband** `wss://api.openai.com/v1/live/sessions/{session_id}/attach`. "778 "WebRTC sessions receive the same server events (minus audio) on the data channel `oai-events`. The `connections` column shows which reference pages list the event.\n")779 for direction, title in (("client→server", "2.1 Client events"), ("server→client", "2.2 Server events")):780 L.append(f"\n### {title}\n")781 L.append("| Event | Connections | Description | Payload fields |\n|---|---|---|---|")782 for r in lv:783 if r["direction"] == direction:784 L.append(f"| `{r['event']}` | {', '.join(r['connections'])} | {_esc(r['description'])[:400]} | {_fields(r)} |")785 L.append("\n## 3. Audio REST streaming events (SSE)\n")786 L.append("| Endpoint | Event | Live | Description | Payload fields |\n|---|---|---|---|---|")787 for r in au:788 L.append(f"| `{r['endpoint']}` | `{r['event']}` | {'✅' if 'LIVE_VERIFIED' in r['status'] else ''} | {_esc(r['description'])[:300]} | {_fields(r)} |")789 L.append("\nObserved 2026-09-18: `POST /v1/audio/transcriptions` `stream=true` (gpt-4o-mini-transcribe, 1.1 s clip) → `transcript.text.delta` ×2 → `transcript.text.done` (with `usage.type: tokens`). "790 "`POST /v1/audio/speech` `stream_format: sse` (gpt-4o-mini-tts, pcm) → `speech.audio.delta` ×4 → `speech.audio.done` → `data: [DONE]`.\n")791 (ROOT / "docs" / "openai").mkdir(parents=True, exist_ok=True)792 (ROOT / "docs" / "openai" / "realtime-events.md").write_text("\n".join(L) + "\n")793794795if __name__ == "__main__":796 main()797