SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
95.2 KB · 769 lines python
Raw Blame History
1#!/usr/bin/env python32"""Generate the xAI media / voice / skills fragments (endpoints, parameters, streaming events, objects, lifecycles).34Sources: sources/xai/pages/developers/rest-api-reference/inference/{images,videos,voice}.md, model-capabilities/**,5sources/xai/openapi/openapi.json, the official WebSocket schema files sources/xai/openapi/{voice-realtime,tts-streaming,6stt-streaming}.ws.json (fetched 2026-09-18 from docs.x.ai) and the live probes saved in tmp-live/xai-media/ (2026-09-18).78Run: python3 scripts/gen_xai_media.py   (rewrites generated/fragments/**/xai-*.json for this domain)9"""10from __future__ import annotations1112import json13from pathlib import Path1415ROOT = Path(__file__).resolve().parent.parent16FRAG = ROOT / "generated" / "fragments"17V = "2026-09-18"18RET = "2026-09-18"19D = "https://docs.x.ai/developers"20SRC_IMG = f"{D}/rest-api-reference/inference/images"21SRC_VID = f"{D}/rest-api-reference/inference/videos"22SRC_VOICE = f"{D}/rest-api-reference/inference/voice"23SRC_S2S = f"{D}/model-capabilities/audio/speech-to-speech"24SRC_TTS = f"{D}/model-capabilities/audio/text-to-speech"25SRC_STT = f"{D}/model-capabilities/audio/speech-to-text"26SRC_CV = f"{D}/model-capabilities/audio/custom-voices"27SRC_IMG_GEN = f"{D}/model-capabilities/images/generation"28SRC_VID_GEN = f"{D}/model-capabilities/video/generation"29SRC_PRICING = f"{D}/pricing"30SRC_OPENAPI = "https://docs.x.ai/openapi.json (local copy sources/xai/openapi/openapi.json)"31SRC_WS_RT = "https://docs.x.ai/voice-realtime.ws.json"32SRC_WS_TTS = "https://docs.x.ai/tts-streaming.ws.json"33SRC_WS_STT = "https://docs.x.ai/stt-streaming.ws.json"34RAW = "tmp-live/xai-media/"353637def src(*urls: str) -> list[dict]:38    return [{"url": u, "retrieved_at": RET} for u in urls]394041def ver(result: str, http: int | None, note: str, method: str = "live_api") -> dict:42    return {"method": method, "verified_at": V, "result": result, "http_status": http, "request_note": note}434445DOCS_ONLY = ver("not_tested", None, "documentation only (paid, needs media input, or enterprise-gated)", "docs_only")4647# --------------------------------------------------------------------------------------------------------------------48# ENDPOINTS49# --------------------------------------------------------------------------------------------------------------------50MEDIA_USAGE_NOTE = ("usage.cost_in_usd_ticks (1 USD = 10,000,000,000 ticks) is always present; the documented token detail fields "51                    "(input_tokens, output_tokens, *_details) were absent on every live response.")525354def ep(method, path, name, desc, status, *, family, req=None, resp=None, streaming=None, pagination=None, sdk=None,55       verification=None, sources=None, idempotency="not documented", auth="Bearer xAI API key", relations=None) -> dict:56    return {57        "provider": "xai", "api_family": family, "method": method, "path": path, "name": name, "description": desc,58        "status": status, "auth": auth, "beta_header": None, "request": req, "response": resp,59        "streaming": streaming or {"supported": False, "events_ref": None}, "pagination": pagination, "idempotency": idempotency,60        "sdk": sdk, "relations": relations, "verification": verification or DOCS_ONLY, "sources": sources or src(SRC_OPENAPI),61    }626364ENDPOINTS = [65    # ---------------- Images ----------------66    ep("POST", "/v1/images/generations", "Generate image(s)",67       "Text → 1–10 images (Grok Imagine). JSON body, OpenAI-compatible (`client.images.generate` works). Response `data[]` items carry `url` "68       "(ephemeral https://imgen.x.ai/xai-imgen/xai-<uuid>.jpeg) or `b64_json`, plus `mime_type` (live: image/jpeg). Optional `storage_options` "69       "persists the file to the Files API (`file_output`). Synchronous (live: ~5–10 s).",70       ["DOCUMENTED", "LIVE_VERIFIED"], family="images",71       req={"content_type": "application/json", "body_ref": "GenerateImageRequest", "params_ref": "generated/fragments/parameters/xai-images.json"},72       resp={"content_type": "application/json", "body_ref": "GeneratedImageResponse",73             "observed_example": {"data": [{"b64_json": "<146796 base64 chars → 110096-byte JPEG>", "mime_type": "image/jpeg"}],74                                  "usage": {"cost_in_usd_ticks": 200000000}},75             "notes": MEDIA_USAGE_NOTE},76       sdk={"python": "openai: client.images.generate(model=..., prompt=..., response_format='b64_json', extra_body={'aspect_ratio': '1:1'}) · xai_sdk: client.image.sample(...) / sample_batch(n=...)",77            "node": "openai: client.images.generate({model, prompt, aspect_ratio}) (base_url https://api.x.ai/v1) · @ai-sdk/xai: generateImage({model: xai.image(id)})"},78       verification=ver("success", 200, "grok-imagine-image, prompt 'a plain white square', n=1, aspect_ratio 1:1, resolution 1k, response_format b64_json → 200, JPEG 110 KB, cost 200000000 ticks = $0.02. "79                        "Also: quality='low' on grok-imagine-image is silently ACCEPTED (200, billed $0.02) although docs say quality is 2.0-only; aspect_ratio '7:7' → 422 text/plain serde error; "80                        "n=0 → 400 {code:'invalid-argument'}; response_format 'png' → 400 'Invalid format.'; model grok-2-image → 404 {code:'not-found'}. Raw: " + RAW + "images-generations-*.json"),81       sources=src(SRC_IMG, SRC_IMG_GEN, SRC_PRICING)),82    ep("POST", "/v1/images/edits", "Edit image(s)",83       "Prompt + 1 source image (`image`) or up to 5 (`images[]`, referenced as <IMAGE_0>… in the prompt) → edited image(s). JSON body only "84       "(public URL, base64 data URL or Files API `file_id`) — the OpenAI SDK `images.edit()` (multipart) is NOT supported. Output aspect ratio follows "85       "the (first) input image unless `aspect_ratio` is set for multi-image edits. Billed input image + output image.",86       ["DOCUMENTED", "LIVE_VERIFIED"], family="images",87       req={"content_type": "application/json", "body_ref": "EditImageRequest", "params_ref": "generated/fragments/parameters/xai-images.json"},88       resp={"content_type": "application/json", "body_ref": "GeneratedImageResponse",89             "observed_example": {"data": [{"b64_json": "<123193-byte JPEG>", "mime_type": "image/jpeg"}], "usage": {"cost_in_usd_ticks": 220000000}}},90       sdk={"python": "xai_sdk: client.image.sample(prompt=..., model=..., image_url='data:image/jpeg;base64,...') · openai SDK: NOT supported (multipart) — use requests/httpx",91            "node": "fetch POST https://api.x.ai/v1/images/edits · @ai-sdk/xai: generateImage({prompt: {text, images: [...]}})"},92       verification=ver("success", 200, "grok-imagine-image, image={url: data:image/jpeg;base64,...} (the generated white square), prompt 'draw a small red circle in the center', b64_json → 200, "93                        "cost 220000000 ticks = $0.022 ($0.02 output + $0.002 input image). Quirk: POST /v1/images/edits WITHOUT `image`/`images` (prompt only) returned 200 and generated a fresh image "94                        "billed $0.02 — behaves like /generations instead of rejecting. Raw: " + RAW + "images-edits-*.json"),95       sources=src(SRC_IMG, f"{D}/model-capabilities/images/editing", f"{D}/model-capabilities/images/multi-image-editing")),96    ep("GET", "/v1/image-generation-models", "List image generation models",97       "Image models available to the key with modalities, fingerprint, aliases, `image_price` (USD ticks per image, default tier) and, for grok-imagine-image-2.0, "98       "a `pricing[]` matrix by (quality, resolution).",99       ["DOCUMENTED", "LIVE_VERIFIED"], family="images",100       resp={"content_type": "application/json", "body_ref": "ListImageGenerationModelsResponse",101             "observed_example": {"models": [{"id": "grok-imagine-image", "image_price": 200000000, "max_prompt_length": 16000, "aliases": ["grok-imagine-image-2026-03-02"]},102                                             {"id": "grok-imagine-image-2.0", "image_price": 600000000, "max_prompt_length": 64000, "pricing": "[6 tiers low/medium × 1k/1.5k/2k: 4e8…8e8 ticks]"},103                                             {"id": "grok-imagine-image-quality", "image_price": 500000000, "aliases": ["grok-imagine-image-quality-20260403", "grok-imagine-image-quality-latest", "grok-imagine-image-pro"]}]}},104       sdk={"python": "xai_sdk: client.models.list_image_generation_models()", "node": "fetch"},105       verification=ver("success", 200, "3 models; raw " + RAW + "image-generation-models.json"),106       sources=src(SRC_OPENAPI, f"{D}/rest-api-reference/inference/models")),107    ep("GET", "/v1/image-generation-models/{model_id}", "Get image generation model", "Single ImageGenerationModel record (aliases resolve).",108       ["DOCUMENTED", "LIVE_VERIFIED"], family="images",109       resp={"content_type": "application/json", "body_ref": "ImageGenerationModel"},110       verification=ver("success", 200, "grok-imagine-image → id, fingerprint fp_574cc24a75, version 1.0, image_price 200000000"),111       sources=src(SRC_OPENAPI, f"{D}/rest-api-reference/inference/models")),112    # ---------------- Videos ----------------113    ep("POST", "/v1/videos/generations", "Start video generation (async)",114       "Text-to-video, image-to-video (`image` = first frame), reference-to-video (`reference_images`/`reference_audios`, grok-imagine-video-1.5) and first/last-frame "115       "(`last_frame`, 1.5 only). Returns immediately with `{request_id}`; poll GET /v1/videos/{request_id}. 1–15 s, 480p/720p/1080p, audio track by default "116       "(`generate_audio: false` for silent). Billed per output second ($0.05/s grok-imagine-video, $0.08/s grok-imagine-video-1.5). Also usable in the Batch API (standard rates).",117       ["DOCUMENTED", "LIVE_VERIFIED"], family="videos",118       req={"content_type": "application/json", "body_ref": "GenerateVideoRequest", "params_ref": "generated/fragments/parameters/xai-videos.json"},119       resp={"content_type": "application/json", "body_ref": "{request_id: string}", "observed_example": {"request_id": "60407e87-a722-92de-8eeb-9a71491aa8e1"},120             "observed_headers": ["x-request-id (= request_id)", "x-ratelimit-limit-requests: 480", "x-zero-data-retention: false", "x-data-retention: general"]},121       sdk={"python": "xai_sdk: client.video.generate(...) (polls) / client.video.start(...) + client.video.get(id)",122            "node": "@ai-sdk/xai: experimental_generateVideo({model: xai.video(id), prompt, duration, aspectRatio, providerOptions: {xai: {resolution, pollTimeoutMs, pollIntervalMs}}})"},123       verification=ver("success", 200, "grok-imagine-video, prompt 'a plain white square, static camera', duration 1, resolution 480p, aspect_ratio 1:1 → 200 {request_id}; job done after 11.4 s; "124                        "cost 500000000 ticks = $0.05 (1 s × $0.05). duration 99 → 400 'Duration must be between 1 and 15 seconds'; resolution '4k' → 422 text/plain serde error. Raw " + RAW + "video-job.json"),125       sources=src(SRC_VID, SRC_VID_GEN, f"{D}/model-capabilities/video/reference-to-video", SRC_PRICING)),126    ep("POST", "/v1/videos/edits", "Start video edit (async)",127       "Prompt + source `video` (public .mp4 URL, base64 data URL or Files API file_id) → edited video with the same duration (input capped at 8.7 s), aspect ratio and "128       "resolution (capped at 720p). No `duration`/`aspect_ratio`/`resolution` params. Returns `{request_id}`; poll GET /v1/videos/{request_id}.",129       ["DOCUMENTED"], family="videos",130       req={"content_type": "application/json", "body_ref": "EditVideoRequest"},131       resp={"content_type": "application/json", "body_ref": "{request_id: string}"},132       sdk={"python": "xai_sdk: client.video.generate(prompt=..., video_url=...)", "node": "@ai-sdk/xai providerOptions.xai.{mode: 'edit-video', videoUrl}"},133       verification=ver("failure", 422, "Only the validation path was exercised: body without `video` → 422 text/plain 'Failed to deserialize the JSON body into the target type: missing field `video`'. No paid edit run."),134       sources=src(SRC_VID, f"{D}/model-capabilities/video/editing")),135    ep("POST", "/v1/videos/extensions", "Start video extension (async)",136       "Prompt + source `video` → the original clip continued from its last frame; `duration` (2–10 s, default 6) is the length of the ADDED segment; output = original + extension. "137       "Returns `{request_id}`.",138       ["DOCUMENTED"], family="videos",139       req={"content_type": "application/json", "body_ref": "ExtendVideoRequest"},140       resp={"content_type": "application/json", "body_ref": "{request_id: string}"},141       sdk={"python": "xai_sdk: client.video.extend(...) / client.video.extend_start(...)", "node": "@ai-sdk/xai providerOptions.xai.{mode: 'extend-video', videoUrl}"},142       sources=src(SRC_VID, f"{D}/model-capabilities/video/extension")),143    ep("GET", "/v1/videos/{request_id}", "Get deferred video result",144       "Poll a video job. Live: HTTP **202** with `{status:'pending', progress:0–99}` while running, HTTP 200 with `{status:'done', video:{url, duration, respect_moderation}, model, usage, progress:100}` "145       "when ready; `status:'failed'` carries `error:{code,message}`; `expired` after the retention window. `video.url` (https://vidgen.x.ai/xai-vidgen-bucket/xai-video-<request_id>.mp4) is temporary — download promptly.",146       ["DOCUMENTED", "LIVE_VERIFIED"], family="videos",147       resp={"content_type": "application/json", "body_ref": "VideoResponse (flattened; the OpenAPI GetDeferredVideoResponse wrapper {status, response} is NOT what the REST endpoint returns)",148             "observed_examples": [{"http": 202, "body": {"status": "pending", "progress": 1}},149                                   {"http": 200, "body": {"status": "done", "video": {"url": "https://vidgen.x.ai/xai-vidgen-bucket/xai-video-<request_id>.mp4", "duration": 1, "respect_moderation": True},150                                                          "model": "grok-imagine-video", "usage": {"cost_in_usd_ticks": 500000000}, "progress": 100}}],151             "states_ref": "generated/fragments/status-lifecycles/xai-lifecycles.json#video_generation_job"},152       verification=ver("success", 200, "3 polls at 0.6 s (202 pending/1%), 6.0 s (202 pending/87%), 11.4 s (200 done); MP4 53608 bytes downloaded (video/mp4, ftypisom). "153                        "Unknown UUID → 404 {code:'not-found', error:'Failed to read static file.'}; non-UUID id → 400 {code:'invalid-argument', error:'Malformed request ID'}."),154       sources=src(SRC_VID, SRC_VID_GEN)),155    ep("GET", "/v1/video-generation-models", "List video generation models",156       "Video models with input/output modalities and aliases (no price field — pricing is per second on the pricing page).",157       ["DOCUMENTED", "LIVE_VERIFIED"], family="videos",158       resp={"content_type": "application/json", "body_ref": "ListVideoGenerationModelsResponse",159             "observed_example": {"models": [{"id": "grok-imagine-video", "input_modalities": ["text", "image", "video"], "output_modalities": ["video"], "aliases": []},160                                             {"id": "grok-imagine-video-1.5", "input_modalities": ["text", "image", "audio"], "aliases": ["grok-imagine-video-1.5-preview", "grok-imagine-video-1.5-2026-05-30"]}]}},161       verification=ver("success", 200, "2 models; raw " + RAW + "video-generation-models.json"),162       sources=src(SRC_OPENAPI, f"{D}/rest-api-reference/inference/models")),163    ep("GET", "/v1/video-generation-models/{model_id}", "Get video generation model", "Single VideoGenerationModel record.",164       ["DOCUMENTED", "LIVE_VERIFIED"], family="videos", resp={"content_type": "application/json", "body_ref": "VideoGenerationModel"},165       verification=ver("success", 200, "grok-imagine-video → fingerprint fp_5f37b474c6, version 1.0"),166       sources=src(SRC_OPENAPI)),167    # ---------------- Voice: Speech to Speech ----------------168    ep("POST", "/v1/realtime/client_secrets", "Create ephemeral client secret (voice)",169       "Mint a short-lived token (`xai-realtime-…`, default TTL 600 s, max 3600 s) for browser/mobile clients of the Speech-to-Speech WebSocket. Optional `session` "170       "(model, reasoning.effort) is bound to the secret. Use as `Authorization: Bearer <value>` or as WebSocket subprotocol `xai-client-secret.<value>`.",171       ["DOCUMENTED", "LIVE_VERIFIED"], family="voice",172       req={"content_type": "application/json", "body_ref": "{expires_after:{seconds}, session:{model, reasoning:{effort}}}"},173       resp={"content_type": "application/json", "body_ref": "{value: string, expires_at: int}", "observed_example": {"value": "xai-real…(111 chars)", "expires_at": 1789789324}},174       verification=ver("success", 200, "expires_after.seconds=60 → value (111 chars, prefix 'xai-real'), expires_at = now+60; with session {model: grok-voice-latest, reasoning: {effort: none}} → 200 same shape. "175                        "Secret then used as Sec-WebSocket-Protocol 'xai-client-secret.<value>' → 101, session.created + conversation.created received."),176       sources=src(SRC_VOICE, f"{D}/model-capabilities/audio/ephemeral-tokens")),177    ep("WS", "wss://api.x.ai/v1/realtime", "Speech to Speech realtime session (WebSocket)",178       "Bidirectional JSON (or binary-frame audio) session with grok-voice-latest (= grok-voice-think-fast-2.0). Query: `model`, `reasoning.effort` (high|none), `call_id` (SIP), "179       "`conversation_id` (resumption). Client events: session.update, input_audio_buffer.append/commit/clear, conversation.item.create/delete/truncate, response.create/cancel. "180       "OpenAI Realtime-compatible naming with xAI extensions (force_message, resumption, replace). Billing: $0.08/min audio (billable_audio_seconds, rounded up) + $0.004 per text `conversation.item.create`.",181       ["DOCUMENTED", "LIVE_VERIFIED"], family="voice", auth="Bearer API key (server) or ephemeral client secret (header or Sec-WebSocket-Protocol xai-client-secret.<token>)",182       req={"content_type": "websocket json + optional binary frames", "body_ref": "client events → generated/fragments/streaming-events/xai-voice.json (api=realtime, client→server)"},183       resp={"content_type": "websocket json + optional binary frames", "body_ref": "server events → generated/fragments/streaming-events/xai-voice.json (api=realtime, server→client)",184             "observed_sequence_text_only": ["session.created", "conversation.created", "ping", "session.updated", "conversation.item.added", "response.created", "response.output_item.added",185                                             "conversation.item.added", "response.content_part.added", "response.output_audio.delta", "response.output_audio_transcript.delta",186                                             "response.output_audio.delta", "response.output_audio.delta", "response.output_audio_transcript.done", "response.content_part.done",187                                             "response.output_audio.done", "response.output_item.done", "response.done"]},188       streaming={"supported": True, "events_ref": "generated/fragments/streaming-events/xai-voice.json"},189       sdk={"python": "websockets (raw) · openai AsyncOpenAI(base_url='https://api.x.ai/v1').realtime.connect(model='grok-voice-latest')",190            "node": "ws (raw) · openai OpenAIRealtimeWS({model:'grok-voice-latest'}, client with baseURL https://api.x.ai/v1)"},191       verification=ver("success", 101, "wss://api.x.ai/v1/realtime?model=grok-voice-latest, API key header. session.update {turn_detection:null, reasoning:{effort:'none'}, voice:'eve', pcm 16 kHz} → "192                        "conversation.item.create input_text 'Reply with OK.' → response.create → 18 events in 1.23 s, transcript 'OK', 22.7 KB PCM, response.done.usage "193                        "{input_tokens:4, output_tokens:37 (audio 36, text 1), output_audio_seconds:0.71, billable_audio_seconds:1}. Undocumented server event `ping` observed. "194                        "Default session.created shows voice 'xai_ara', model grok-voice-think-fast-2.0, modalities ['audio'], turn_detection {type:null}. Raw " + RAW + "voice-ws-session.json"),195       sources=src(SRC_VOICE, SRC_S2S, SRC_WS_RT, SRC_PRICING)),196    ep("POST", "/v2/phone-numbers", "Create SIP phone number (voice)",197       "Register a Direct SIP number (`origin: byo_trunk`, customer-owned E.164 number) routed to an `agent_id` or a `webhook` that receives signed `realtime.call.incoming` events "198       "(Standard Webhooks v1 HMAC-SHA256; `dispatch_signing_secret` returned once). Provisioning xAI numbers (`xai_provisioned`) via API is not supported per the SIP guide. Note the /v2 prefix.",199       ["DOCUMENTED"], family="voice", req={"content_type": "application/json", "body_ref": "CreatePhoneNumberV2 {origin, name, agent_id|webhook, phone_number, area_code, sip_auth}"},200       resp={"content_type": "application/json", "body_ref": "{phone_number:{phone_number_id, team_id, phone_number, name, agent_id, webhook_id, origin, sip_host:'sip.voice.x.ai', inbound_trunk_id, sip_auth, created_at, updated_at}, webhook:{webhook_id, dispatch_signing_secret}}"},201       sources=src(SRC_VOICE, f"{SRC_S2S}/sip")),202    ep("POST", "/v1/realtime/calls/{call_id}/refer", "Transfer SIP call (REFER)", "Transfer an active SIP call to `target_uri` (tel:+E.164 or sip:user@host). Blocks until the transfer resolves; HTTP status reports the outcome. Returns `{}`.",203       ["DOCUMENTED"], family="voice", req={"content_type": "application/json", "body_ref": "{target_uri: string}"}, resp={"content_type": "application/json", "body_ref": "{}"},204       sources=src(SRC_VOICE, f"{SRC_S2S}/sip")),205    ep("POST", "/v1/realtime/calls/{call_id}/hangup", "Hang up SIP call", "End an active SIP call. Returns `{}`.", ["DOCUMENTED"], family="voice",206       resp={"content_type": "application/json", "body_ref": "{}"}, sources=src(SRC_VOICE, f"{SRC_S2S}/sip")),207    # ---------------- Voice: TTS ----------------208    ep("POST", "/v1/tts", "Text to speech (unary)",209       "Text (≤ 60,000 chars, inline speech tags like [laugh], <whisper>) → audio bytes (mp3 default 24 kHz/128 kbps; wav, pcm, mulaw, alaw). `language` is REQUIRED (BCP-47 or 'auto'). "210       "With `with_timestamps: true` the response becomes a JSON envelope {audio (base64), content_type, duration, audio_timestamps}. $15 / 1M characters.",211       ["DOCUMENTED", "LIVE_VERIFIED"], family="voice",212       req={"content_type": "application/json", "body_ref": "TTS request {text, voice_id, language, output_format:{codec, sample_rate, bit_rate}, speed, optimize_streaming_latency, text_normalization, with_timestamps, replace}"},213       resp={"content_type": "audio/* (or application/json with with_timestamps)", "body_ref": "audio bytes | {audio, content_type, duration, audio_timestamps:{graph_chars[], graph_times[]}}",214             "observed_examples": [{"http": 200, "content_type": "audio/wav", "bytes": 25324, "request": "text 'OK.', wav 16 kHz"},215                                   {"http": 200, "content_type": "application/json", "body": {"audio": "<16384 base64 chars>", "content_type": "audio/mpeg", "duration": 0.71,216                                                                                               "audio_timestamps": {"graph_chars": ["O", "K", "."], "graph_times": [[0.04, 0.06], [0.2, 0.22], [0.22, 0.71]]}}}],217             "notes": "Live: graph_times is an array of [start, end] pairs (as in the TTS guide), not the {start, end} objects shown in the REST reference page."},218       verification=ver("success", 200, "'OK.' voice eve language en → wav 25 KB; with_timestamps → JSON envelope; missing `language` → 422 text/plain 'missing field `language`'. ~$0.00005 each."),219       sources=src(SRC_VOICE, SRC_TTS, SRC_PRICING)),220    ep("WS", "wss://api.x.ai/v1/tts", "Text to speech streaming (WebSocket)",221       "Bidirectional streaming TTS: configure by query (voice, language [required], codec, sample_rate, bit_rate, optimize_streaming_latency, speed 0.7–1.5, text_normalization, with_timestamps); "222       "send `text.delta`… `text.done`, receive `audio.delta`… `audio.done`; multi-utterance on one connection; no text length limit.",223       ["DOCUMENTED"], family="voice", streaming={"supported": True, "events_ref": "generated/fragments/streaming-events/xai-voice.json (api=tts)"},224       req={"content_type": "websocket json", "body_ref": "text.delta / text.done"}, resp={"content_type": "websocket json", "body_ref": "audio.delta / audio.done / error"},225       sources=src(SRC_VOICE, SRC_TTS, SRC_WS_TTS)),226    ep("GET", "/v1/tts/voices", "List built-in voices",227       "Roster shared by TTS, Speech-to-Speech (`voice`) and video `reference_audios.voice_id`. Live: 28 voices, each {voice_id, name, language:'multilingual', gender} — the `gender` field and the "228       "'multilingual' language value are not in the REST reference (which shows language 'en'); 'aurora' and 'liora' are absent from the documented example list.",229       ["DOCUMENTED", "LIVE_VERIFIED"], family="voice",230       resp={"content_type": "application/json", "body_ref": "{voices: [{voice_id, name, language, gender}]}",231             "observed_example": {"voices": [{"voice_id": "altair", "name": "Altair", "language": "multilingual", "gender": "male"}, {"voice_id": "eve", "name": "Eve", "language": "multilingual", "gender": "female"}, "… 28 total"]}},232       verification=ver("success", 200, "28 voices; raw " + RAW + "tts-voices.json"), sources=src(SRC_VOICE, SRC_TTS)),233    ep("GET", "/v1/tts/voices/{voice_id}", "Get built-in voice", "Single voice record.", ["DOCUMENTED", "LIVE_VERIFIED"], family="voice",234       resp={"content_type": "application/json", "body_ref": "{voice_id, name, language, gender}", "observed_example": {"voice_id": "eve", "name": "Eve", "language": "multilingual", "gender": "female"}},235       verification=ver("success", 200, "eve"), sources=src(SRC_VOICE)),236    # ---------------- Voice: STT ----------------237    ep("POST", "/v1/stt", "Speech to text (batch)",238       "multipart/form-data: `file` (≤ 500 MB; wav, mp3, ogg, opus, flac, aac, mp4, m4a, mkv auto-detected; raw pcm/mulaw/alaw need `audio_format` + `sample_rate`) or `url`; options language, format, "239       "multichannel, channels, diarize, keyterm[], filler_words, vad_threshold. Returns {text, language, duration, words[], channels[]}. $0.10 / hour.",240       ["DOCUMENTED", "LIVE_VERIFIED"], family="voice",241       req={"content_type": "multipart/form-data", "body_ref": "STT multipart fields (file must be the last field)"},242       resp={"content_type": "application/json", "body_ref": "{text, language, duration, words:[{text,start,end,confidence?,speaker?}], channels?}",243             "observed_example": {"text": "Okay.", "language": "en", "duration": 0.79, "words": [{"text": "Okay.", "start": 0.122, "end": 0.466}]}},244       verification=ver("success", 200, "0.79 s WAV (the TTS 'OK.' clip), language=en → text 'Okay.'; `confidence` omitted (docs: omitted when 0). ~$0.00002."),245       sources=src(SRC_VOICE, SRC_STT, f"{D}/rest-api-reference/inference/speech-to-text", SRC_PRICING)),246    ep("WS", "wss://api.x.ai/v1/stt", "Speech to text streaming (WebSocket)",247       "Raw audio binary frames in (pcm|mulaw|alaw|opus; query sample_rate, encoding, interim_results, endpointing, language, model grok-voice-transcribe-1.0|2.0, multichannel, channels, diarize, keyterm, "248       "filler_words, smart_turn, smart_turn_timeout, vad_threshold) → transcript.created, transcript.partial (is_final / speech_final), transcript.done. Client JSON: finalize, audio.done. $0.20 / hour.",249       ["DOCUMENTED"], family="voice", streaming={"supported": True, "events_ref": "generated/fragments/streaming-events/xai-voice.json (api=stt)"},250       req={"content_type": "websocket binary frames + json", "body_ref": "binary audio | finalize | audio.done"}, resp={"content_type": "websocket json", "body_ref": "transcript.created | transcript.partial | transcript.done | error"},251       sources=src(SRC_VOICE, SRC_STT, SRC_WS_STT)),252    # ---------------- Voice: custom voices ----------------253    ep("POST", "/v1/custom-voices", "Create custom voice (clone)",254       "multipart: reference `file` (≤ 120 s; wav/mp3/flac/ogg/opus/m4a/aac/mkv/mp4) + labels (name, description, gender, accent, age, language, use_case, tone) → {voice_id (8 lowercase alnum), …}. "255       "API creation is gated to Enterprise teams (403 otherwise); console cloning is free for up to 30 voices/team; US-only (except Illinois).",256       ["DOCUMENTED", "ACCOUNT_RESTRICTED"], family="voice",257       req={"content_type": "multipart/form-data", "body_ref": "custom voice fields"}, resp={"content_type": "application/json", "body_ref": "CustomVoice"},258       verification=ver("not_tested", None, "Not called: documented as Enterprise-only for API creation (403 'Custom voices not enabled for this team'); no reference clip and no consent basis for cloning.", "docs_only"),259       sources=src(SRC_VOICE, SRC_CV)),260    ep("GET", "/v1/custom-voices", "List custom voices", "Team-scoped custom voices; `limit` 1–1000 (default 100), `pagination_token`. Live shape adds `total_count` and `cap` (30) to the documented {voices, pagination_token}.",261       ["DOCUMENTED", "LIVE_VERIFIED"], family="voice", pagination={"style": "token", "params": ["limit", "pagination_token"], "response_field": "pagination_token"},262       resp={"content_type": "application/json", "body_ref": "{voices: [CustomVoice], pagination_token?, total_count, cap}", "observed_example": {"voices": [], "total_count": 0, "cap": 30}},263       verification=ver("success", 200, "empty list; note undocumented total_count/cap fields and absent pagination_token key"), sources=src(SRC_VOICE, SRC_CV)),264    ep("GET", "/v1/custom-voices/{voice_id}", "Get custom voice", "Single CustomVoice.", ["DOCUMENTED"], family="voice", resp={"content_type": "application/json", "body_ref": "CustomVoice"}, sources=src(SRC_VOICE, SRC_CV)),265    ep("PATCH", "/v1/custom-voices/{voice_id}", "Update custom voice metadata", "Partial update of labels (name, description, gender, accent, age, language, use_case, tone); empty strings rejected (400).",266       ["DOCUMENTED"], family="voice", req={"content_type": "application/json", "body_ref": "CustomVoice labels (all optional)"}, resp={"content_type": "application/json", "body_ref": "CustomVoice"}, sources=src(SRC_VOICE, SRC_CV)),267    ep("DELETE", "/v1/custom-voices/{voice_id}", "Delete custom voice", "Returns {deleted: true}.", ["DOCUMENTED"], family="voice", resp={"content_type": "application/json", "body_ref": "{deleted: true}"}, sources=src(SRC_VOICE, SRC_CV)),268    ep("GET", "/v1/custom-voices/{voice_id}/audio", "Download custom voice reference audio", "Raw reference clip bytes.", ["DOCUMENTED"], family="voice", resp={"content_type": "audio/*", "body_ref": "bytes"}, sources=src(SRC_VOICE, SRC_CV)),269    # ---------------- Skills ----------------270    ep("GET", "/v1/skills", "List hosted skills",271       "OpenAPI-only endpoint (no public docs page): paginated SkillList {object:'list', data:[Skill], first_id, last_id, has_more}; query limit 1–100 (default 100), after, order asc|desc (default desc). "272       "Live with our key: 404 {error:{code:404, message:'The requested resource was not found…'}} — the resource is not reachable for this team (gated or not yet deployed).",273       ["DOCUMENTED", "ACCOUNT_RESTRICTED"], family="skills", pagination={"style": "cursor", "params": ["limit", "after", "order"], "response_fields": ["first_id", "last_id", "has_more"]},274       resp={"content_type": "application/json", "body_ref": "SkillList", "observed_example": {"error": {"code": 404, "message": "The requested resource was not found. Please check the URL and try again. Documentation is available at https://docs.x.ai/"}}},275       verification=ver("restricted", 404, "GET /v1/skills?limit=5 → 404 generic JSON error (Content-Type application/json). Same for GET /v1/skills/{id}. Raw " + RAW + "skills-*.json"),276       sources=src(SRC_OPENAPI)),277    ep("POST", "/v1/skills", "Upload hosted skill",278       "multipart/form-data field `files` (a zip of the skill directory, or the field repeated once per file of a directory upload). The skill's `name`/`description` come from SKILL.md YAML frontmatter. "279       "Returns Skill {id, object:'skill', name, description, created_at, default_version:'1', latest_version:'1'}; 400 invalid upload, 413 too large. Live: 404 text/plain with our key.",280       ["DOCUMENTED", "ACCOUNT_RESTRICTED"], family="skills",281       req={"content_type": "multipart/form-data", "body_ref": "UploadSkillMultipartRequest {files[]}"}, resp={"content_type": "application/json", "body_ref": "Skill"},282       verification=ver("restricted", 404, "zip(atlas-probe-skill/SKILL.md) as `files` → 404 text/plain (122 bytes); single SKILL.md file → 404; JSON body → 404 JSON error."),283       sources=src(SRC_OPENAPI)),284    ep("GET", "/v1/skills/{skill_id}", "Retrieve hosted skill", "Skill record; 404 'Skill not found.'", ["DOCUMENTED", "ACCOUNT_RESTRICTED"], family="skills",285       resp={"content_type": "application/json", "body_ref": "Skill"}, verification=ver("restricted", 404, "unknown id → 404 generic JSON error"), sources=src(SRC_OPENAPI)),286    ep("DELETE", "/v1/skills/{skill_id}", "Delete hosted skill", "Returns DeletedSkill {id, object:'skill.deleted', deleted:true}.", ["DOCUMENTED", "ACCOUNT_RESTRICTED"], family="skills",287       resp={"content_type": "application/json", "body_ref": "DeletedSkill"}, verification=ver("not_tested", None, "not attempted (no skill could be created)", "docs_only"), sources=src(SRC_OPENAPI)),288    ep("GET", "/v1/skills/{skill_id}/content", "Download hosted skill content", "Raw skill zip bytes streamed in chunks (application/zip).", ["DOCUMENTED", "ACCOUNT_RESTRICTED"], family="skills",289       resp={"content_type": "application/zip", "body_ref": "zip bytes"}, verification=ver("not_tested", None, "not attempted (no skill could be created)", "docs_only"), sources=src(SRC_OPENAPI)),290]291292# --------------------------------------------------------------------------------------------------------------------293# PARAMETERS294# --------------------------------------------------------------------------------------------------------------------295IMG_MODELS = ["grok-imagine-image", "grok-imagine-image-2.0", "grok-imagine-image-quality"]296IMG2 = ["grok-imagine-image-2.0"]297VID_MODELS = ["grok-imagine-video", "grok-imagine-video-1.5"]298VID15 = ["grok-imagine-video-1.5"]299ASPECTS = ["1:1", "3:4", "4:3", "9:16", "16:9", "2:3", "3:2", "9:19.5", "19.5:9", "9:20", "20:9", "1:2", "2:1", "21:9", "5:2", "auto"]300301302def prm(endpoint, name, typ, desc, *, loc="body", req=False, default=None, mn=None, mx=None, enum=None, models=None, status=None, source=None, notes=None) -> dict:303    return {"provider": "xai", "endpoint": endpoint, "parameter": name, "location": loc, "type": typ, "required": req, "default": default,304            "minimum": mn, "maximum": mx, "enum": enum, "description": desc, "compatible_models": models, "beta_header": None,305            "status": status or ["DOCUMENTED"], "source": source or SRC_IMG, "notes": notes}306307308LV = ["DOCUMENTED", "LIVE_VERIFIED"]309STORAGE = [310    ("storage_options", "object", "Persist the output to the Files API and return `file_output` alongside the ephemeral URL/b64.", {}),311    ("storage_options.filename", "string", "Filename for the stored file (required when storage_options is present).", {"req": True}),312    ("storage_options.expires_after", "integer|null", "Seconds until the stored file auto-expires; max 2592000 (30 days); omitted = no expiry.", {"mx": 2592000}),313    ("storage_options.public_url", "boolean|object", "Also create a permanent shareable public URL (`file_output.public_url`); object form accepts an `expires_after`.", {}),314]315316317def storage_params(endpoint, source):318    return [prm(endpoint, n, t, d, source=source, **kw) for n, t, d, kw in STORAGE]319320321def image_params(endpoint):322    common = [323        prm(endpoint, "model", "string|null", "Image model id or alias (grok-imagine-image, grok-imagine-image-2.0, grok-imagine-image-quality → served by 2.0/low from 2026-11-02). Effectively required.",324            models=IMG_MODELS, status=LV, notes="grok-2-image → 404 not-found (retired slug)."),325        prm(endpoint, "prompt", "string", "Prompt (max_prompt_length: 16000 chars on grok-imagine-image / -quality, 64000 on 2.0 per the models endpoint). Required.", req=True, models=IMG_MODELS, status=LV),326        prm(endpoint, "n", "integer|null", "Number of images (1–10).", default=1, mn=1, mx=10, models=IMG_MODELS, status=LV, notes="Live: n=0 → 400 {code:'invalid-argument', error:'The number of images to generate (n) must be between 1 and 10 inclusive.'}"),327        prm(endpoint, "response_format", "string|null", "`url` (ephemeral hosted URL) or `b64_json` (bare base64, no data-URI prefix).", default="url", enum=["url", "b64_json"], models=IMG_MODELS, status=LV,328            notes="Live: 'png' → 400 'Invalid format.'"),329        prm(endpoint, "resolution", "string|null", "Output resolution; grok-imagine models only. Pricing on 2.0 rises with resolution (1k < 1.5k < 2k).", default="1k", enum=["1k", "1.5k", "2k"], models=IMG_MODELS, status=LV,330            source=SRC_IMG_GEN, notes="The guide lists only 1k and 2k; the OpenAPI enum and REST reference also include 1.5k (and the 2.0 pricing matrix prices it)."),331        prm(endpoint, "quality", "string|null", "Quality tier for grok-imagine-image-2.0: low, medium or auto (default; currently low for generation, medium for editing). Billed at the quality served.",332            default="auto", enum=["low", "medium", "auto"], models=IMG2, source=SRC_IMG_GEN, status=LV,333            notes="Not in the REST reference page nor the OpenAPI GenerateImageRequest (docs-guide only). Live: quality='low' on grok-imagine-image (1.0) was accepted (200, $0.02) rather than rejected."),334        prm(endpoint, "user", "string|null", "Opaque end-user id for abuse monitoring.", models=IMG_MODELS),335    ]336    return common + storage_params(endpoint, SRC_IMG)337338339PARAMS_IMAGES = (340    image_params("POST /v1/images/generations")341    + [prm("POST /v1/images/generations", "aspect_ratio", "string|null", "Aspect ratio; `auto` (default) lets the model pick. 21:9 and 5:2 added 2026-08.", default="auto", enum=ASPECTS, models=IMG_MODELS, status=LV,342           notes="Live: '7:7' → 422 text/plain 'Failed to deserialize the JSON body into the target type: aspect_ratio: unknown variant `7:7`, expected one of …'")]343    + image_params("POST /v1/images/edits")344    + [prm("POST /v1/images/edits", "aspect_ratio", "string|null", "Only for multi-image edits (`images`): output ratio, otherwise auto-detected from the first input. Do not set for single-image edits.", enum=ASPECTS, models=IMG_MODELS),345       prm("POST /v1/images/edits", "image", "object|null", "Single source image. Mutually exclusive with `images`.", models=IMG_MODELS, status=LV,346           notes="Live quirk: omitting both `image` and `images` does NOT fail — the request behaves like a generation (200, billed $0.02)."),347       prm("POST /v1/images/edits", "image.url", "string", "Public URL or base64 data URL (JPEG/PNG/WebP). Also accepted as `image_url`. Required when file_id is absent. The docs examples add `type: 'image_url'` (ignored/optional).", models=IMG_MODELS, status=LV),348       prm("POST /v1/images/edits", "image.file_id", "string|null", "Files API file id (must be a fully uploaded image). Mutually exclusive with url.", models=IMG_MODELS, source=f"{D}/model-capabilities/imagine/files/inputs"),349       prm("POST /v1/images/edits", "images", "array<object>", "2–5 reference images (was 3 before 2026-08); refer to them as <IMAGE_0>, <IMAGE_1>… in the prompt. Each item {url | file_id}. Mutually exclusive with `image`.", mx=5, models=IMG_MODELS,350           source=f"{D}/model-capabilities/images/multi-image-editing"),351       prm("POST /v1/images/edits", "images[].url", "string", "Public URL or base64 data URL.", models=IMG_MODELS),352       prm("POST /v1/images/edits", "images[].file_id", "string|null", "Files API file id.", models=IMG_MODELS)]353    + [prm("GET /v1/image-generation-models/{model_id}", "model_id", "string", "Model id or alias.", loc="path", req=True, status=LV, source=SRC_OPENAPI),354       prm("GET /v1/skills", "x", "x", "placeholder", )]  # placeholder removed below355)356PARAMS_IMAGES = [p for p in PARAMS_IMAGES if p["parameter"] != "x"]357358VID_GEN = "POST /v1/videos/generations"359PARAMS_VIDEOS = [360    prm(VID_GEN, "model", "string|null", "grok-imagine-video ($0.05/s; text/image/video inputs) or grok-imagine-video-1.5 ($0.08/s; text/image/audio inputs, native 1080p, reference audio, last_frame).", models=VID_MODELS, status=LV, source=SRC_VID),361    prm(VID_GEN, "prompt", "string", "Required for text-to-video; optional when `image`, `reference_images` or `last_frame` is present.", models=VID_MODELS, status=LV, source=SRC_VID),362    prm(VID_GEN, "duration", "integer|null", "Output length in seconds, 1–15 (also accepts `seconds` and string values for OpenAI compatibility).", default=8, mn=1, mx=15, models=VID_MODELS, status=LV, source=SRC_VID,363        notes="Live: duration 1 accepted (billed 1 s); 99 → 400 'Duration must be between 1 and 15 seconds'."),364    prm(VID_GEN, "aspect_ratio", "string|null", "Ignored for image-to-video (derived from the first frame — guide says it can override and stretch).", default="16:9", enum=["1:1", "16:9", "9:16", "4:3", "3:4", "3:2", "2:3"], models=VID_MODELS, status=LV, source=SRC_VID),365    prm(VID_GEN, "resolution", "string|null", "480p (default, fastest), 720p, 1080p (grok-imagine-video-1.5 T2V/I2V only; reference-to-video capped at 720p).", default="480p", enum=["480p", "720p", "1080p"], models=VID_MODELS, status=LV, source=SRC_VID,366        notes="Live: '4k' → 422 text/plain serde error listing the enum."),367    prm(VID_GEN, "image", "object|null", "First-frame image for image-to-video ({url | file_id}; also `input_reference`). With reference_images/reference_audios/last_frame on 1.5 it pins the first frame.", models=VID_MODELS, source=SRC_VID),368    prm(VID_GEN, "image.url", "string", "Public URL or base64 data URL (JPEG/PNG/WebP). Also `image_url`.", models=VID_MODELS, source=SRC_VID),369    prm(VID_GEN, "image.file_id", "string|null", "Files API file id.", models=VID_MODELS, source=SRC_VID),370    prm(VID_GEN, "reference_images", "array<object>", "Reference-to-video: style/content references ({url | file_id}), tagged <IMAGE_0>… in the prompt.", models=VID_MODELS, source=f"{D}/model-capabilities/video/reference-to-video"),371    prm(VID_GEN, "reference_audios", "array<object>", "Up to 3 voice references ({voice_id} preset from GET /v1/tts/voices, or {url} clip ≤ 15 s — own-clip uploads are trusted-partner only). Tag <AUDIO_0>… in the prompt. grok-imagine-video-1.5 only.",372        mx=3, models=VID15, source=f"{D}/model-capabilities/video/reference-to-video"),373    prm(VID_GEN, "reference_audios[].voice_id", "string|null", "Preset voice id (case-insensitive); unknown → 400 with the list of voices.", models=VID15, source=f"{D}/model-capabilities/video/reference-to-video"),374    prm(VID_GEN, "reference_audios[].url", "string|null", "data:audio/wav;base64,… or http(s) clip ≤ 15 s (trusted partners).", models=VID15, source=f"{D}/model-capabilities/video/reference-to-video"),375    prm(VID_GEN, "last_frame", "object", "Pin the exact last frame ({url | file_id}); alone or with `image` (first frame → interpolation). grok-imagine-video-1.5 only; classic grok-imagine-video rejects it.", models=VID15,376        source=SRC_VID_GEN, notes="Documented in the guide's Request Modes table, absent from the REST reference/OpenAPI."),377    prm(VID_GEN, "generate_audio", "boolean", "Videos include an audio track by default; false → silent video.", default=True, models=VID_MODELS, source=SRC_VID_GEN, notes="Guide-only parameter (not in REST reference/OpenAPI)."),378    prm(VID_GEN, "output", "object|null", "Bring-your-own storage: {upload_url} presigned URL the service PUTs the MP4 to (used by Grok Build under ZDR).", models=VID_MODELS, source=SRC_VID),379    prm(VID_GEN, "output.upload_url", "string", "Signed URL for HTTP PUT of the generated video.", req=True, models=VID_MODELS, source=SRC_VID),380    prm(VID_GEN, "user", "string|null", "Opaque end-user id.", models=VID_MODELS, source=SRC_VID),381    *storage_params(VID_GEN, SRC_VID),382    prm("POST /v1/videos/edits", "model", "string|null", "Video model (docs examples use grok-imagine-video).", models=VID_MODELS, source=SRC_VID),383    prm("POST /v1/videos/edits", "prompt", "string", "Edit instruction.", req=True, models=VID_MODELS, source=SRC_VID),384    prm("POST /v1/videos/edits", "video", "object", "Source video {url | file_id}; .mp4 (H.264/H.265/AV1…), ≤ 8.7 s, output capped at 720p.", req=True, models=VID_MODELS, source=SRC_VID,385        status=["DOCUMENTED", "LIVE_VERIFIED"], notes="Live: missing → 422 text/plain 'missing field `video`'."),386    prm("POST /v1/videos/edits", "video.url", "string", "Public .mp4 URL or base64 data URL.", models=VID_MODELS, source=SRC_VID),387    prm("POST /v1/videos/edits", "video.file_id", "string|null", "Files API file id (video).", models=VID_MODELS, source=SRC_VID),388    prm("POST /v1/videos/edits", "output", "object|null", "{upload_url} presigned PUT target.", models=VID_MODELS, source=SRC_VID),389    prm("POST /v1/videos/edits", "user", "string|null", "Opaque end-user id.", models=VID_MODELS, source=SRC_VID),390    *storage_params("POST /v1/videos/edits", SRC_VID),391    prm("POST /v1/videos/extensions", "model", "string|null", "Video model.", models=VID_MODELS, source=SRC_VID),392    prm("POST /v1/videos/extensions", "prompt", "string", "What happens next.", req=True, models=VID_MODELS, source=SRC_VID),393    prm("POST /v1/videos/extensions", "video", "object", "Source video {url | file_id}.", req=True, models=VID_MODELS, source=SRC_VID),394    prm("POST /v1/videos/extensions", "duration", "integer|null", "Length of the ADDED segment in seconds (2–10); total output = original + extension.", default=6, mn=2, mx=10, models=VID_MODELS, source=SRC_VID),395    prm("POST /v1/videos/extensions", "output", "object|null", "{upload_url} presigned PUT target.", models=VID_MODELS, source=SRC_VID),396    *storage_params("POST /v1/videos/extensions", SRC_VID),397    prm("GET /v1/videos/{request_id}", "request_id", "string (UUID)", "Deferred request id from a generations/edits/extensions call.", loc="path", req=True, status=LV, source=SRC_VID,398        notes="Live: non-UUID → 400 'Malformed request ID'; unknown UUID → 404 'Failed to read static file.'"),399    prm("GET /v1/video-generation-models/{model_id}", "model_id", "string", "Model id or alias.", loc="path", req=True, status=LV, source=SRC_OPENAPI),400]401402RT = "WS wss://api.x.ai/v1/realtime"403SU = "WS wss://api.x.ai/v1/realtime session.update"404VOICE_MODELS = ["grok-voice-latest", "grok-voice-think-fast-2.0"]405PARAMS_VOICE = [406    # client secrets407    prm("POST /v1/realtime/client_secrets", "expires_after.seconds", "integer", "TTL of the secret; max 3600; default 600.", default=600, mx=3600, status=LV, source=SRC_VOICE),408    prm("POST /v1/realtime/client_secrets", "session", "object|null", "Initial session config stored with the secret and applied when the WebSocket opens.", status=LV, source=SRC_VOICE,409        notes="The ephemeral-tokens guide says 'session' and 'expires_after.anchor' are not supported; the REST reference documents `session` and it was accepted live (200)."),410    prm("POST /v1/realtime/client_secrets", "session.model", "string", "Voice model bound to the secret.", enum=VOICE_MODELS, status=LV, source=SRC_VOICE),411    prm("POST /v1/realtime/client_secrets", "session.reasoning.effort", "string", "high (default) or none.", default="high", enum=["high", "none"], status=LV, source=SRC_VOICE),412    # realtime query413    prm(RT, "model", "string", "Voice model; ignored with call_id.", loc="query", default="grok-voice-latest", enum=VOICE_MODELS, models=VOICE_MODELS, status=LV, source=SRC_VOICE),414    prm(RT, "reasoning.effort", "string", "Enable/disable reasoning for the session.", loc="query", default="high", enum=["high", "none"], source=SRC_VOICE),415    prm(RT, "call_id", "string", "SIP call id from a realtime.call.incoming webhook; binds the socket to that call (API key auth only).", loc="query", source=SRC_VOICE),416    prm(RT, "conversation_id", "string", "Resume a cached conversation (requires session.resumption.enabled on both sessions; 30 min inactivity expiry).", loc="query", source=SRC_S2S),417    prm(RT, "Authorization", "string", "Bearer <api key> (server) or Bearer <ephemeral secret>.", loc="header", status=LV, source=SRC_VOICE),418    prm(RT, "Sec-WebSocket-Protocol", "string", "Browser auth: `xai-client-secret.<ephemeral token>`; replaces the Authorization header.", loc="header", status=LV, source=SRC_VOICE),419    # session.update fields420    prm(SU, "session.model", "string", "Model (also settable by query).", enum=VOICE_MODELS, source=SRC_WS_RT),421    prm(SU, "session.instructions", "string", "System prompt (second person, fixed section order recommended).", status=LV, source=SRC_S2S),422    prm(SU, "session.reasoning.effort", "string", "high (default) or none.", default="high", enum=["high", "none"], status=LV, source=SRC_S2S),423    prm(SU, "session.voice", "string", "Built-in voice id (GET /v1/tts/voices) or 8-char custom voice id.", default="xai_ara (observed in session.created; docs say eve)", status=LV, source=SRC_S2S),424    prm(SU, "session.turn_detection.type", "string|null", "`server_vad` (auto turns) or null (manual: commit + response.create).", enum=["server_vad", None], status=LV, source=SRC_S2S),425    prm(SU, "session.turn_detection.threshold", "number", "VAD activation threshold 0.1–0.9.", default=0.85, mn=0.1, mx=0.9, source=SRC_S2S),426    prm(SU, "session.turn_detection.silence_duration_ms", "number", "Silence before the turn ends (0–10000).", mn=0, mx=10000, source=SRC_S2S),427    prm(SU, "session.turn_detection.prefix_padding_ms", "number", "Audio kept before detected speech start (0–10000).", default=333, mn=0, mx=10000, source=SRC_S2S),428    prm(SU, "session.turn_detection.idle_timeout_ms", "number|null", "Proactive re-engagement timer after each response (emits input_audio_buffer.timeout_triggered).", default=None, source=SRC_S2S),429    prm(SU, "session.resumption.enabled", "boolean", "Cache turns keyed by conversation_id for replay on reconnect (xAI extension).", default=False, source=SRC_S2S),430    prm(SU, "session.audio.input.format.type", "string", "Input codec.", default="audio/pcm", enum=["audio/pcm", "audio/pcmu", "audio/pcma", "audio/opus"], status=LV, source=SRC_S2S),431    prm(SU, "session.audio.input.format.rate", "integer", "PCM sample rate (Hz).", default=24000, enum=[8000, 11025, 16000, 22050, 24000, 32000, 44100, 48000], status=LV, source=SRC_S2S,432        notes="11025 appears only in the ws.json schema enum, not in the guide table."),433    prm(SU, "session.audio.input.transport", "string", "json (base64 in input_audio_buffer.append) or binary (raw frames; dual-accept).", default="json", enum=["json", "binary"], status=LV, source=SRC_S2S),434    prm(SU, "session.audio.input.transcription.language_hint", "string", "BCP-47 hint for ASR (es/pt need a regional variant).", source=SRC_S2S),435    prm(SU, "session.audio.input.transcription.keyterms", "array<string>", "≤ 100 terms, ≤ 50 chars each.", mx=100, source=SRC_S2S),436    prm(SU, "session.audio.input.transcription.model", "string", "Set to `grok-transcribe` to receive conversation.item.input_audio_transcription.updated events (live captions).", enum=["grok-transcribe"], source=SRC_VOICE),437    prm(SU, "session.audio.output.format.type", "string", "Output codec.", default="audio/pcm", enum=["audio/pcm", "audio/pcmu", "audio/pcma", "audio/opus"], status=LV, source=SRC_S2S),438    prm(SU, "session.audio.output.format.rate", "integer", "PCM sample rate (Hz).", default=24000, enum=[8000, 11025, 16000, 22050, 24000, 32000, 44100, 48000], status=LV, source=SRC_S2S),439    prm(SU, "session.audio.output.transport", "string", "json or binary; strict (output only on the configured path); changes apply at the next response boundary.", default="json", enum=["json", "binary"], status=LV, source=SRC_S2S),440    prm(SU, "session.audio.output.speed", "number", "Playback speed 0.7–1.5.", default=1.0, mn=0.7, mx=1.5, source=SRC_S2S),441    prm(SU, "session.replace", "object", "Phrase → spoken substitution map applied before TTS (transcript unchanged); echoed on session.updated.", source=SRC_S2S),442    prm(SU, "session.tools", "array<object>", "file_search {vector_store_ids, max_num_results} · web_search {allowed_domains|excluded_domains ≤5, enable_image_understanding, location} · x_search {allowed_x_handles|excluded_x_handles ≤20, from_date, to_date, enable_image_understanding, enable_video_understanding} · mcp {server_url, server_label, server_description, allowed_tools, authorization, headers} · function {name, description, parameters}.",443        source=SRC_S2S, notes="Server-side tools are executed by xAI; only `function` requires client handling via function_call_output."),444    prm(SU, "session.tool_choice", "string", "Observed in session.updated ('auto'); not documented.", default="auto", status=["LIVE_DISCOVERED"], source=RAW + "voice-ws-session.json"),445    prm(SU, "session.enable_noise_suppression", "boolean", "Observed in session.updated (false); not documented.", default=False, status=["LIVE_DISCOVERED"], source=RAW + "voice-ws-session.json"),446    prm(SU, "session.enable_phonetic_spelling", "boolean", "Observed in session.updated (false); not documented.", default=False, status=["LIVE_DISCOVERED"], source=RAW + "voice-ws-session.json"),447    prm(SU, "session.keep_context", "boolean", "Observed in session.updated (false); not documented.", default=False, status=["LIVE_DISCOVERED"], source=RAW + "voice-ws-session.json"),448    prm(SU, "session.temperature", "number", "Observed in session.updated as -1.0 (unset); not documented.", default=-1.0, status=["LIVE_DISCOVERED"], source=RAW + "voice-ws-session.json"),449    prm(SU, "session.max_response_output_tokens", "string|integer", "Observed in session.updated as 'inf'; not documented.", default="inf", status=["LIVE_DISCOVERED"], source=RAW + "voice-ws-session.json"),450    # response.create / conversation.item.create extras451    prm("WS wss://api.x.ai/v1/realtime response.create", "response.instructions", "string", "Per-response override of the session system prompt.", source=SRC_S2S),452    prm("WS wss://api.x.ai/v1/realtime response.create", "response.metadata", "object|null", "Developer key-values echoed on response.done (schema).", source=SRC_WS_RT),453    prm("WS wss://api.x.ai/v1/realtime conversation.item.create", "item.type", "string", "message (user/assistant, content input_text|input_audio|output_text) · function_call (seed history) · function_call_output {call_id, output} · force_message (xAI: verbatim TTS line, {role:'assistant', interruptible, content:[{type:'output_text', text}]}).",454        enum=["message", "function_call", "function_call_output", "force_message"], status=LV, source=SRC_S2S),455    prm("WS wss://api.x.ai/v1/realtime conversation.item.create", "item.interruptible", "boolean", "force_message only: false drops caller audio until playback completes.", default=True, source=SRC_S2S),456    # TTS unary457    prm("POST /v1/tts", "text", "string", "≤ 60,000 chars; speech tags [pause] [laugh] … and wrappers <whisper> <soft> …", req=True, mx=60000, status=LV, source=SRC_VOICE),458    prm("POST /v1/tts", "voice_id", "string", "Built-in (GET /v1/tts/voices) or custom voice id.", default="eve", status=LV, source=SRC_VOICE),459    prm("POST /v1/tts", "language", "string", "BCP-47 (en, zh, pt-BR…) or auto. REQUIRED.", req=True, status=LV, source=SRC_VOICE, notes="Live: missing → 422 text/plain 'missing field `language`'."),460    prm("POST /v1/tts", "output_format.codec", "string", "Audio codec.", default="mp3", enum=["mp3", "wav", "pcm", "mulaw", "alaw"], status=LV, source=SRC_VOICE),461    prm("POST /v1/tts", "output_format.sample_rate", "integer|null", "Hz.", default=24000, enum=[8000, 16000, 22050, 24000, 44100, 48000], status=LV, source=SRC_VOICE),462    prm("POST /v1/tts", "output_format.bit_rate", "integer|null", "bps, mp3 only.", default=128000, enum=[32000, 64000, 96000, 128000, 192000], source=SRC_VOICE),463    prm("POST /v1/tts", "speed", "number", "0.7–1.5.", default=1.0, mn=0.7, mx=1.5, source=SRC_VOICE),464    prm("POST /v1/tts", "optimize_streaming_latency", "string|integer", "0 (quality) | 1 (smaller first chunk); the guide also lists 2.", default="0", enum=["0", "1", "2"], source=SRC_VOICE),465    prm("POST /v1/tts", "text_normalization", "boolean", "Expand numbers/abbreviations to spoken form.", default=False, source=SRC_VOICE),466    prm("POST /v1/tts", "with_timestamps", "boolean", "true → JSON envelope with per-character timings (adds latency).", default=False, status=LV, source=SRC_VOICE),467    prm("POST /v1/tts", "replace", "object", "Phrase → respelling or /IPA/ map; ≤ 200 entries, keys ≤ 100 chars (letters/digits/apostrophes/spaces), values ≤ 128; expanded text ≤ 240,000 chars.", mx=200, source=SRC_TTS),468    # TTS ws query469    *[prm("WS wss://api.x.ai/v1/tts", n, t, d, loc="query", default=df, enum=en, source=SRC_VOICE) for n, t, d, df, en in [470        ("voice", "string", "Voice id.", "eve", None), ("language", "string", "BCP-47 or auto (required).", None, None), ("codec", "string", "Output codec.", "mp3", ["mp3", "wav", "pcm", "mulaw", "alaw"]),471        ("sample_rate", "integer", "Hz.", 24000, None), ("bit_rate", "integer", "bps (mp3).", 128000, None), ("optimize_streaming_latency", "integer", "0 or 1.", 0, [0, 1]),472        ("speed", "number", "0.7–1.5.", 1.0, None), ("text_normalization", "boolean", "Normalize text.", False, None), ("with_timestamps", "boolean", "audio_timestamps on each audio.delta.", False, None)]],473    # STT batch474    prm("POST /v1/stt", "file", "binary", "Audio ≤ 500 MB; must be the LAST multipart field. Either file or url.", status=LV, source=SRC_VOICE),475    prm("POST /v1/stt", "url", "string", "Server-side download URL (alternative to file).", source=SRC_VOICE),476    prm("POST /v1/stt", "audio_format", "string", "Only for raw formats.", enum=["pcm", "mulaw", "alaw", "wav", "mp3", "ogg", "opus", "flac", "aac", "mp4", "m4a", "mkv"], source=SRC_VOICE),477    prm("POST /v1/stt", "sample_rate", "string", "Required for raw formats (also sample_rate_hertz).", enum=["8000", "16000", "22050", "24000", "44100", "48000"], source=SRC_VOICE),478    prm("POST /v1/stt", "language", "string", "Language code; with format=true enables inverse text normalization.", status=LV, source=SRC_VOICE),479    prm("POST /v1/stt", "format", "string", "'true' enables formatting (requires language).", default="false", enum=["true", "false"], source=SRC_VOICE),480    prm("POST /v1/stt", "multichannel", "string", "Per-channel transcription → channels[].", default="false", enum=["true", "false"], source=SRC_VOICE),481    prm("POST /v1/stt", "channels", "integer", "2–8, required for multichannel raw audio.", mn=2, mx=8, source=SRC_VOICE),482    prm("POST /v1/stt", "diarize", "string", "Speaker diarization → words[].speaker.", default="false", enum=["true", "false"], source=SRC_VOICE),483    prm("POST /v1/stt", "keyterm", "array<string>", "Repeatable bias terms, ≤ 100 × 50 chars.", mx=100, source=SRC_VOICE),484    prm("POST /v1/stt", "filler_words", "string", "Keep uh/um.", default="false", enum=["true", "false"], source=SRC_VOICE),485    prm("POST /v1/stt", "vad_threshold", "number", "Speech-probability gate 0–1; 0 disables.", default=0.5, mn=0, mx=1, source=SRC_VOICE),486    prm("POST /v1/stt", "model", "string", "grok-voice-transcribe-1.0 | 2.0 (default 2.0 per the guide; the REST page does not list it for batch).", default="grok-voice-transcribe-2.0", enum=["grok-voice-transcribe-1.0", "grok-voice-transcribe-2.0"], source=f"{D}/model-capabilities/audio/voice"),487    # STT ws query488    *[prm("WS wss://api.x.ai/v1/stt", n, t, d, loc="query", default=df, enum=en, source=SRC_VOICE) for n, t, d, df, en in [489        ("sample_rate", "integer", "Hz (ignored for opus).", 16000, [8000, 16000, 22050, 24000, 44100, 48000]), ("encoding", "string", "pcm s16le | mulaw | alaw | opus (one packet per binary frame, mono).", "pcm", ["pcm", "mulaw", "alaw", "opus"]),490        ("interim_results", "boolean", "Emit is_final=false partials ~every 500 ms.", False, None), ("endpointing", "integer", "Silence ms before speech_final (0–5000).", 400, None),491        ("language", "string", "Enables inverse text normalization.", "", None), ("model", "string", "Transcribe model.", "grok-voice-transcribe-2.0", ["grok-voice-transcribe-1.0", "grok-voice-transcribe-2.0"]),492        ("multichannel", "boolean", "Requires channels ≥ 2; not with opus.", False, None), ("channels", "integer", "1–8.", 1, None), ("diarize", "boolean", "Speaker field on words.", False, None),493        ("keyterm", "string (repeatable)", "Bias terms.", None, None), ("filler_words", "boolean", "Keep fillers.", False, None),494        ("smart_turn", "number", "End-of-turn confidence threshold 0–1 (adds end_of_turn_confidence to partials).", None, None), ("smart_turn_timeout", "integer", "Max silence ms before forced speech_final (1–5000).", None, None),495        ("vad_threshold", "number", "Speech gate 0–1.", 0.08, None)]],496    # custom voices497    prm("POST /v1/custom-voices", "file", "binary", "Reference clip ≤ 120 s (90–120 s recommended).", req=True, source=SRC_VOICE),498    *[prm("POST /v1/custom-voices", n, "string", d, enum=en, source=SRC_VOICE) for n, d, en in [499        ("name", "Display name.", None), ("description", "Free text.", None), ("gender", "Label.", ["male", "female", "neutral"]), ("accent", "Free text.", None), ("age", "Label.", ["young", "middle-aged", "old"]),500        ("language", "ISO 639 / BCP-47 (region uppercase).", None), ("use_case", "Label.", ["conversational", "narration", "characters", "educational", "advertisement", "social_media", "entertainment"]),501        ("tone", "Label.", ["warm", "casual", "professional", "friendly", "authoritative", "expressive", "calm"])]],502    prm("GET /v1/custom-voices", "limit", "integer", "1–1000.", loc="query", default=100, mn=1, mx=1000, status=LV, source=SRC_VOICE),503    prm("GET /v1/custom-voices", "pagination_token", "string", "Next-page token from the previous response.", loc="query", source=SRC_VOICE),504    prm("GET /v1/custom-voices/{voice_id}", "voice_id", "string", "8-char lowercase alphanumeric id.", loc="path", req=True, source=SRC_VOICE),505    # SIP506    prm("POST /v2/phone-numbers", "origin", "string", "byo_trunk for customer-owned Direct SIP numbers; xai_provisioned not available via API per the SIP guide.", req=True, enum=["xai_provisioned", "byo_trunk"], source=SRC_VOICE),507    prm("POST /v2/phone-numbers", "name", "string", "Display name.", req=True, source=SRC_VOICE),508    prm("POST /v2/phone-numbers", "agent_id", "string", "Route to an agent (mutually exclusive with webhook).", source=SRC_VOICE),509    prm("POST /v2/phone-numbers", "phone_number", "string", "E.164 number (byo_trunk).", source=SRC_VOICE),510    prm("POST /v2/phone-numbers", "area_code", "string", "3-digit US area code filter (xai_provisioned).", source=SRC_VOICE),511    prm("POST /v2/phone-numbers", "sip_auth.auth_username", "string", "SIP digest username (with auth_password).", source=SRC_VOICE),512    prm("POST /v2/phone-numbers", "sip_auth.auth_password", "string", "Stored encrypted, never returned.", source=SRC_VOICE),513    prm("POST /v2/phone-numbers", "sip_auth.allowed_addresses", "array<string>", "Source CIDRs allowed to INVITE.", source=SRC_VOICE),514    prm("POST /v2/phone-numbers", "webhook.url", "string", "Receives signed realtime.call.incoming POSTs.", req=True, source=SRC_VOICE),515    prm("POST /v2/phone-numbers", "webhook.name", "string", "Optional display name.", source=SRC_VOICE),516    prm("POST /v2/phone-numbers", "webhook.auth_url", "string", "OAuth token-exchange URL (with auth_token).", source=SRC_VOICE),517    prm("POST /v2/phone-numbers", "webhook.auth_token", "string", "Bearer credential or OAuth client credential.", source=SRC_VOICE),518    prm("POST /v1/realtime/calls/{call_id}/refer", "call_id", "string", "From realtime.call.incoming.", loc="path", req=True, source=SRC_VOICE),519    prm("POST /v1/realtime/calls/{call_id}/refer", "target_uri", "string", "tel:+E.164 or sip:user@host.", req=True, source=SRC_VOICE),520    prm("POST /v1/realtime/calls/{call_id}/hangup", "call_id", "string", "From realtime.call.incoming.", loc="path", req=True, source=SRC_VOICE),521]522523PARAMS_SKILLS = [524    prm("GET /v1/skills", "limit", "integer|null", "Page size 1–100.", loc="query", default=100, mn=1, mx=100, status=["DOCUMENTED", "ACCOUNT_RESTRICTED"], source=SRC_OPENAPI, notes="Live 404 with our key."),525    prm("GET /v1/skills", "after", "string|null", "Cursor: id of the last item of the previous page.", loc="query", status=["DOCUMENTED", "ACCOUNT_RESTRICTED"], source=SRC_OPENAPI),526    prm("GET /v1/skills", "order", "string|null", "asc | desc.", loc="query", default="desc", enum=["asc", "desc"], status=["DOCUMENTED", "ACCOUNT_RESTRICTED"], source=SRC_OPENAPI),527    prm("POST /v1/skills", "files", "array<binary>", "Skill zip, or the field repeated once per file of a directory upload. SKILL.md frontmatter (name, description) is parsed server-side.", req=True,528        status=["DOCUMENTED", "ACCOUNT_RESTRICTED"], source=SRC_OPENAPI, notes="Live: zip and single-file uploads both → 404 text/plain."),529    prm("GET /v1/skills/{skill_id}", "skill_id", "string", "Skill id.", loc="path", req=True, status=["DOCUMENTED", "ACCOUNT_RESTRICTED"], source=SRC_OPENAPI),530    prm("DELETE /v1/skills/{skill_id}", "skill_id", "string", "Skill id.", loc="path", req=True, status=["DOCUMENTED", "ACCOUNT_RESTRICTED"], source=SRC_OPENAPI),531    prm("GET /v1/skills/{skill_id}/content", "skill_id", "string", "Skill id.", loc="path", req=True, status=["DOCUMENTED", "ACCOUNT_RESTRICTED"], source=SRC_OPENAPI),532    prm("POST /v1/responses (tools[] shell.environment)", "environment.skills[]", "array<LocalShellSkill>", "Skills available in a local shell environment: {name, description, path (directory with SKILL.md)}. `environment.type` is 'local'. "533        "This is the only place the OpenAPI wires skills into inference (ShellCall.environment).", status=["DOCUMENTED"], source=SRC_OPENAPI),534]535536# --------------------------------------------------------------------------------------------------------------------537# STREAMING EVENTS (from the official ws.json files + live observations)538# --------------------------------------------------------------------------------------------------------------------539OBSERVED_SEQUENCE = ["session.created", "conversation.created", "ping", "session.updated", "conversation.item.added", "response.created", "response.output_item.added",540                     "conversation.item.added", "response.content_part.added", "response.output_audio.delta", "response.output_audio_transcript.delta", "response.output_audio.delta",541                     "response.output_audio.delta", "response.output_audio_transcript.done", "response.content_part.done", "response.output_audio.done", "response.output_item.done", "response.done"]542SENT_CLIENT = ["session.update", "conversation.item.create", "response.create"]543544545def load_observed() -> dict[str, dict]:546    p = ROOT / "tmp-live" / "xai-media" / "voice-ws-session.json"547    out: dict[str, dict] = {}548    if p.exists():549        for e in json.loads(p.read_text())["main"]["events"]:550            t = e.get("type")551            if t and t not in out:552                e = dict(e)553                e.pop("t", None)554                out[t] = e555    return out556557558def build_streaming() -> dict:559    observed = load_observed()560    recs = []561    for fname, api, srcurl in [("voice-realtime.ws.json", "realtime", SRC_WS_RT), ("tts-streaming.ws.json", "tts", SRC_WS_TTS), ("stt-streaming.ws.json", "stt", SRC_WS_STT)]:562        d = json.loads((ROOT / "sources" / "xai" / "openapi" / fname).read_text())563        order = [f"{m['direction']}:{m['type']}" for m in d.get("exampleFlow", [])]564        for direction, key in [("client→server", "clientMessages"), ("server→client", "serverMessages")]:565            for i, m in enumerate(d.get(key, [])):566                t = m["type"]567                status = ["DOCUMENTED"]568                extra = {}569                if api == "realtime" and direction == "server→client" and t in observed:570                    status.append("LIVE_VERIFIED")571                    extra["observed_example"] = observed[t]572                if api == "realtime" and direction == "client→server" and t in SENT_CLIENT:573                    status.append("LIVE_VERIFIED")574                hint = None575                k = f"{'client' if direction.startswith('client') else 'server'}:{t}"576                if k in order:577                    hint = order.index(k) + 1578                recs.append({"provider": "xai", "api": api, "direction": direction, "event": t, "description": m.get("description"), "schema": m.get("schema"),579                             "example": m.get("example"), **extra, "order_hint": hint, "status": status, "source": srcurl})580    # undocumented events / fields observed live581    recs.append({"provider": "xai", "api": "realtime", "direction": "server→client", "event": "ping",582                 "description": "Keep-alive / clock event sent right after conversation.created (and presumably periodically). Not in the official ws.json schema.",583                 "schema": {"type": "object", "properties": {"type": "ping", "event_id": "string (uuid)", "timestamp": "integer (unix ms)", "previous_item_id": "string|null"}},584                 "example": observed.get("ping"), "order_hint": None, "status": ["LIVE_DISCOVERED"], "source": RAW + "voice-ws-session.json"})585    recs.append({"provider": "xai", "api": "realtime", "direction": "server→client", "event": "response.audio.delta",586                 "description": "Legacy alias of response.output_audio.delta mentioned in the Speech-to-Speech guide (audio transport table); not emitted in the live probe (only response.output_audio.delta was).",587                 "schema": None, "example": None, "order_hint": None, "status": ["DOCUMENTED", "UNVERIFIED"], "source": SRC_S2S})588    recs.append({"provider": "xai", "api": "realtime", "direction": "server→client (webhook)", "event": "realtime.call.incoming",589                 "description": "Signed webhook (Standard Webhooks v1: webhook-id, webhook-timestamp, webhook-signature headers, HMAC-SHA256 with dispatch_signing_secret) POSTed to the phone number's webhook URL when a SIP call arrives; `data.call_id` is then used as ?call_id= on wss://api.x.ai/v1/realtime.",590                 "schema": {"type": "object", "properties": {"type": "realtime.call.incoming", "data": {"call_id": "string", "…": "caller/callee SIP headers (see SIP guide)"}}}, "example": None,591                 "order_hint": None, "status": ["DOCUMENTED"], "source": f"{SRC_S2S}/sip"})592    meta = {"domain": "xai-voice", "generated_by": "scripts/gen_xai_media.py", "last_verified": V,593            "apis": {"realtime": "wss://api.x.ai/v1/realtime (Speech to Speech)", "tts": "wss://api.x.ai/v1/tts (streaming TTS)", "stt": "wss://api.x.ai/v1/stt (streaming STT)"},594            "schema_sources": [SRC_WS_RT, SRC_WS_TTS, SRC_WS_STT, "local copies sources/xai/openapi/*.ws.json (fetched 2026-09-18)"],595            "observed_server_sequence_text_only_probe": OBSERVED_SEQUENCE, "probe_url": "wss://api.x.ai/v1/realtime?model=grok-voice-latest",596            "probe_model_reported": "grok-voice-think-fast-2.0", "raw_probe": RAW + "voice-ws-session.json",597            "openai_realtime_compat": {"renamed": {"conversation.item.input_audio_transcription.delta": "conversation.item.input_audio_transcription.updated (cumulative transcript, needs audio.input.transcription.model=grok-transcribe)"},598                                       "unsupported_client": ["conversation.item.retrieve", "output_audio_buffer.clear (WebRTC/SIP only)"],599                                       "not_emitted_server": ["conversation.item.done", "conversation.item.input_audio_transcription.failed", "conversation.item.input_audio_transcription.segment", "conversation.item.retrieved",600                                                              "output_audio_buffer.started/stopped/cleared", "rate_limits.updated"],601                                       "xai_extensions": ["force_message item type", "session.resumption", "session.replace", "audio.*.transport binary", "input_audio_buffer.timeout_triggered", "input_audio_buffer.dtmf_event_received (SIP)"]},602            "count": len(recs)}603    return {"_meta": meta, "records": recs}604605606# --------------------------------------------------------------------------------------------------------------------607# OBJECTS608# --------------------------------------------------------------------------------------------------------------------609def obj(name, kind, desc, fields, *, status=None, example=None, notes=None, verification=None, sources=None) -> dict:610    return {"provider": "xai", "name": name, "kind": kind, "description": desc, "schema": fields, "status": status or ["DOCUMENTED"], "example": example, "notes": notes,611            "verification": verification or DOCS_ONLY, "sources": sources or src(SRC_OPENAPI), "last_verified": V}612613614OBJECTS = [615    obj("GeneratedImageResponse", "response", "Body of POST /v1/images/generations and /v1/images/edits.",616        {"data": "GeneratedImage[] (required)", "usage": "MediaUsage|null"}, status=LV,617        example={"data": [{"b64_json": "<base64 JPEG>", "mime_type": "image/jpeg"}], "usage": {"cost_in_usd_ticks": 200000000}},618        verification=ver("success", 200, "generation + edit"), sources=src(SRC_IMG)),619    obj("GeneratedImage", "object", "One generated image.",620        {"url": "string|null (ephemeral https://imgen.x.ai/xai-imgen/xai-<uuid>.jpeg; default)", "b64_json": "string|null (bare base64, no data-URI prefix; when response_format=b64_json)",621         "mime_type": "string|null (image/png | image/jpeg | image/webp; live always image/jpeg)", "file_output": "FileOutput (only with storage_options)", "storage_error": "string|null",622         "respect_moderation": "boolean (xAI SDK / gRPC field; not present on the REST body observed)"}, status=LV,623        example={"url": "https://imgen.x.ai/xai-imgen/xai-<uuid>.jpeg", "mime_type": "image/jpeg"}, verification=ver("success", 200, "url and b64_json variants both observed"), sources=src(SRC_IMG)),624    obj("MediaUsage", "object", "Billing block on image and video responses (`usage`).",625        {"cost_in_usd_ticks": "integer (required; 1 USD = 1e10 ticks, 1 cent = 1e8)", "input_tokens": "integer|null", "input_tokens_details": "{cached_tokens, image_tokens, text_tokens}",626         "output_tokens": "integer|null", "output_tokens_details": "{image_tokens, reasoning_tokens, text_tokens}", "total_tokens": "integer|null"}, status=LV,627        example={"cost_in_usd_ticks": 220000000}, notes="Live responses contained only cost_in_usd_ticks (image $0.02 = 2e8, edit $0.022 = 2.2e8, 1-s video $0.05 = 5e8).",628        verification=ver("success", 200, "3 media responses"), sources=src(SRC_IMG, f"{D}/cost-tracking")),629    obj("FileOutput", "object", "Stored-file reference returned when `storage_options` is set (images: data[].file_output; videos: video.file_output).",630        {"file_id": "string (Files API id)", "filename": "string", "expires_at": "integer|null (unix s)", "public_url": "string|null", "public_url_error": "string|null", "public_url_expires_at": "integer|null"},631        sources=src(SRC_IMG, f"{D}/model-capabilities/imagine/files/outputs")),632    obj("StorageOptions", "request object", "`storage_options` on every Imagine request.", {"filename": "string (required)", "expires_after": "integer|null (≤ 2592000 s)", "public_url": "boolean|object"},633        sources=src(SRC_IMG, f"{D}/model-capabilities/imagine/files/outputs")),634    obj("ImageGenerationModel", "model record", "Item of GET /v1/image-generation-models.",635        {"id": "string", "object": "'model'", "owned_by": "'xai'", "version": "string", "fingerprint": "string (fp_…)", "created": "integer", "max_prompt_length": "integer", "input_modalities": "string[]",636         "output_modalities": "string[]", "image_price": "integer (USD ticks per image, default tier)", "pricing": "ImagePricingTier[] (only grok-imagine-image-2.0)", "aliases": "string[]"}, status=LV,637        example={"id": "grok-imagine-image-2.0", "image_price": 600000000, "pricing": [{"quality": "low", "resolution": "1k", "price_per_image": 400000000}], "max_prompt_length": 64000},638        verification=ver("success", 200, "3 models"), sources=src(SRC_OPENAPI)),639    obj("ImagePricingTier", "object", "Per-image price for one (quality, resolution) tier.", {"quality": "'low'|'medium'|'high'", "resolution": "'1k'|'1.5k'|'2k'", "price_per_image": "integer (USD ticks)"}, status=LV,640        example={"quality": "medium", "resolution": "2k", "price_per_image": 800000000}, verification=ver("success", 200, "6 tiers on 2.0: low 4e8/5e8/6e8, medium 6e8/7e8/8e8"), sources=src(SRC_OPENAPI)),641    obj("VideoGenerationModel", "model record", "Item of GET /v1/video-generation-models (no price field).",642        {"id": "string", "object": "'model'", "owned_by": "'xai'", "version": "string", "fingerprint": "string", "created": "integer", "input_modalities": "string[] (text|image|video|audio)", "output_modalities": "['video']", "aliases": "string[]"},643        status=LV, example={"id": "grok-imagine-video-1.5", "input_modalities": ["text", "image", "audio"], "aliases": ["grok-imagine-video-1.5-preview", "grok-imagine-video-1.5-2026-05-30"]},644        verification=ver("success", 200, "2 models"), sources=src(SRC_OPENAPI)),645    obj("VideoStartResponse", "response", "Body of POST /v1/videos/{generations,edits,extensions}.", {"request_id": "string (UUID; also echoed as x-request-id header)"}, status=LV,646        example={"request_id": "60407e87-a722-92de-8eeb-9a71491aa8e1"}, verification=ver("success", 200, "generation"), sources=src(SRC_VID)),647    obj("VideoResponse (poll result)", "response", "Body of GET /v1/videos/{request_id}. HTTP 202 while pending, 200 when done. The OpenAPI wrapper GetDeferredVideoResponse {status, response} is NOT used by REST.",648        {"status": "'pending'|'done'|'failed'|'expired'", "progress": "integer|null 0–100 (omitted when failed)", "video": "GeneratedVideo (done only)", "model": "string|null (done only)", "usage": "MediaUsage|null", "error": "VideoError (failed only)"},649        status=LV, example={"status": "done", "video": {"url": "https://vidgen.x.ai/xai-vidgen-bucket/xai-video-<request_id>.mp4", "duration": 1, "respect_moderation": True}, "model": "grok-imagine-video", "usage": {"cost_in_usd_ticks": 500000000}, "progress": 100},650        verification=ver("success", 200, "202 {status:pending, progress:1} → 202 {pending, 87} → 200 done"), sources=src(SRC_VID, SRC_VID_GEN)),651    obj("GeneratedVideo", "object", "`video` of a done poll.", {"url": "string|null (temporary; empty when respect_moderation=false)", "duration": "integer (s)", "respect_moderation": "boolean", "file_output": "FileOutput (with storage_options)", "storage_error": "string|null"},652        status=LV, example={"url": "https://vidgen.x.ai/…/xai-video-<id>.mp4", "duration": 1, "respect_moderation": True}, verification=ver("success", 200, "MP4 53 KB, video/mp4"), sources=src(SRC_VID)),653    obj("VideoError", "object", "`error` of a failed poll.", {"code": "'invalid_argument'|'permission_denied'|'failed_precondition'|'service_unavailable'|'internal_error'", "message": "string"},654        example={"code": "invalid_argument", "message": "Prompt cannot be empty. Please provide a prompt."}, notes="Auth, model-not-found and synchronous rate-limit errors are HTTP errors, never VideoError.", sources=src(SRC_VID, SRC_VID_GEN)),655    obj("ClientSecret", "response", "Body of POST /v1/realtime/client_secrets.", {"value": "string (prefix xai-realtime-… / observed 'xai-real', 111 chars)", "expires_at": "integer (unix s)"}, status=LV,656        example={"value": "xai-realtime-…", "expires_at": 1789789324}, verification=ver("success", 200, "60 s TTL"), sources=src(SRC_VOICE)),657    obj("realtime.session", "object", "`session` in session.created / session.updated.",658        {"id": "string (32 hex)", "object": "'realtime.session'", "model": "string", "instructions": "string", "voice": "string", "modalities": "['audio'] observed", "turn_detection": "{type: 'server_vad'|null, …}", "tools": "array",659         "reasoning": "{effort}", "audio": "{input:{format:{type,rate}, transport}, output:{…}}", "resumption": "{enabled}", "replace": "object",660         "observed_only_in_session.updated": ["enable_noise_suppression", "enable_phonetic_spelling", "keep_context", "input_audio_format", "output_audio_format", "tool_choice", "temperature", "max_response_output_tokens", "input_audio_transcription"]},661        status=LV, example={"id": "e4af06be…", "object": "realtime.session", "instructions": "", "voice": "xai_ara", "modalities": ["audio"], "turn_detection": {"type": None}, "tools": [], "model": "grok-voice-think-fast-2.0"},662        notes="Default voice reported as 'xai_ara' (docs say eve). session.updated echoes the effective config in a flatter, partly undocumented shape.",663        verification=ver("success", 101, "text-only probe"), sources=src(SRC_WS_RT)),664    obj("realtime.conversation", "object", "`conversation` in conversation.created.", {"id": "string (UUID; use as ?conversation_id= for resumption)", "object": "'realtime.conversation'"}, status=LV,665        example={"id": "08b1471f-87d5-4963-b94e-092537a06b95", "object": "realtime.conversation"}, verification=ver("success", 101, "both probes"), sources=src(SRC_WS_RT)),666    obj("realtime.item", "object", "Conversation item (conversation.item.added, response.output_item.*, response.done.output[]).",667        {"id": "string (UUID)", "object": "'realtime.item'", "type": "'message'|'function_call'|'function_call_output'|'force_message'", "status": "'in_progress'|'completed'", "role": "'user'|'assistant'",668         "content": "[{type:'input_text'|'output_text'|'audio'|'input_audio', text?, transcript?}]", "replayed": "boolean (true when replayed by session resumption)", "call_id/name/arguments/output": "function items"},669        status=LV, example={"id": "95be5a32-…", "object": "realtime.item", "type": "message", "status": "completed", "role": "user", "content": [{"type": "input_text", "text": "Reply with OK."}], "replayed": False},670        verification=ver("success", 101, "user + assistant items observed"), sources=src(SRC_WS_RT)),671    obj("realtime.response", "object", "`response` in response.created / response.done.",672        {"id": "string (UUID)", "object": "'realtime.response'", "status": "'in_progress'|'completed'|'cancelled'|'incomplete'", "status_details": "string ('unimplemented' observed)", "output": "realtime.item[]", "usage": "{} (empty; real usage is top-level on response.done)", "metadata": "object|null"},673        status=LV, example={"id": "5e885467-…", "object": "realtime.response", "status": "completed", "status_details": "unimplemented", "output": [{"type": "message", "role": "assistant", "content": [{"type": "audio", "transcript": "OK"}]}], "usage": {}},674        verification=ver("success", 101, "response.done"), sources=src(SRC_WS_RT)),675    obj("realtime response usage (response.done.usage)", "object", "Top-level `usage` on response.done — the billing signal for Speech to Speech.",676        {"input_tokens": "integer", "input_token_details": "{text_tokens, audio_tokens, grok_tokens}", "output_tokens": "integer", "output_token_details": "{text_tokens, audio_tokens, grok_tokens}", "total_tokens": "integer",677         "output_audio_seconds": "number", "billable_audio_seconds": "integer (rounded up; × $0.08/60)"}, status=["LIVE_DISCOVERED"],678        example={"input_tokens": 4, "input_token_details": {"text_tokens": 4, "audio_tokens": 0, "grok_tokens": 0}, "output_tokens": 37, "output_token_details": {"text_tokens": 1, "audio_tokens": 36, "grok_tokens": 0}, "total_tokens": 41, "output_audio_seconds": 0.71, "billable_audio_seconds": 1},679        notes="Not in the ws.json schema (which documents only input/output/total tokens inside response.usage).", verification=ver("success", 101, "text-only probe"), sources=src(RAW + "voice-ws-session.json")),680    obj("realtime audio delta extras", "object", "Undocumented fields observed on response.output_audio.delta / transcript.delta.",681        {"rid": "string (UUID; groups deltas of one synthesis run)", "latency": "string seconds ('0.49' on first delta)", "audio_duration_ms": "integer (ms of PCM in this delta)", "ts": "integer unix ms", "start_time": "number (transcript delta)", "previous_item_id": "string|null (on every event)"},682        status=["LIVE_DISCOVERED"], example={"audio_duration_ms": 550, "latency": "0.49", "rid": "07193536-…"}, verification=ver("success", 101, "probe"), sources=src(RAW + "voice-ws-session.json")),683    obj("Voice (built-in)", "object", "Item of GET /v1/tts/voices.", {"voice_id": "string (lowercase)", "name": "string", "language": "string|null ('multilingual' live; 'en' in docs)", "gender": "'male'|'female' (undocumented)"}, status=LV,684        example={"voice_id": "eve", "name": "Eve", "language": "multilingual", "gender": "female"}, verification=ver("success", 200, "28 voices"), sources=src(SRC_VOICE)),685    obj("TTS JSON envelope", "response", "POST /v1/tts response when with_timestamps=true.",686        {"audio": "string base64", "content_type": "string (audio/mpeg …)", "duration": "number s", "audio_timestamps": "{graph_chars: string[], graph_times: [start, end][]}"}, status=LV,687        example={"audio": "<base64>", "content_type": "audio/mpeg", "duration": 0.71, "audio_timestamps": {"graph_chars": ["O", "K", "."], "graph_times": [[0.04, 0.06], [0.2, 0.22], [0.22, 0.71]]}},688        notes="graph_times items are [start, end] arrays (guide) — the REST reference's {start, end} objects do not match the live shape.", verification=ver("success", 200, "'OK.'"), sources=src(SRC_VOICE, SRC_TTS)),689    obj("STT transcript", "response", "Body of POST /v1/stt.", {"text": "string", "language": "string (BCP-47)", "duration": "number s", "words": "[{text, start, end, confidence?, speaker?}] (omitted when empty)", "channels": "[{index, language, text, words[]}] (multichannel)"},690        status=LV, example={"text": "Okay.", "language": "en", "duration": 0.79, "words": [{"text": "Okay.", "start": 0.122, "end": 0.466}]}, verification=ver("success", 200, "0.79 s clip"), sources=src(SRC_VOICE)),691    obj("STT streaming transcript.partial", "event payload", "State flags: interim (is_final=false) → chunk final (is_final=true, speech_final=false) → utterance final (is_final=true, speech_final=true); end_of_turn_confidence with smart_turn.",692        {"type": "'transcript.partial'", "text": "string", "is_final": "boolean", "speech_final": "boolean", "words": "[{text,start,end,speaker?}]", "channel": "integer (multichannel)", "end_of_turn_confidence": "number (smart_turn)"}, sources=src(SRC_WS_STT)),693    obj("CustomVoice", "object", "Custom (cloned) voice.", {"voice_id": "string (8 lowercase alnum)", "name": "string|null", "description": "string|null", "gender": "'male'|'female'|'neutral'|null", "accent": "string|null", "age": "'young'|'middle-aged'|'old'|null",694        "language": "string|null", "use_case": "string|null", "tone": "string|null", "created_at": "RFC 3339"}, example={"voice_id": "nlbqfwie", "name": "Friendly Narrator", "gender": "female", "language": "en", "use_case": "narration", "tone": "warm", "created_at": "2026-04-26T18:56:34.872993+00:00"}, sources=src(SRC_VOICE, SRC_CV)),695    obj("CustomVoiceList", "response", "GET /v1/custom-voices.", {"voices": "CustomVoice[]", "pagination_token": "string|null (absent when no more)", "total_count": "integer (undocumented)", "cap": "integer (undocumented; 30)"}, status=LV,696        example={"voices": [], "total_count": 0, "cap": 30}, verification=ver("success", 200, "empty"), sources=src(SRC_VOICE)),697    obj("PhoneNumber (SIP)", "object", "POST /v2/phone-numbers response.", {"phone_number": "{phone_number_id, team_id, phone_number (E.164), name, agent_id, webhook_id, origin, sip_host ('sip.voice.x.ai'), inbound_trunk_id, sip_auth:{auth_username, allowed_addresses[]}, created_at, updated_at, agent_name}",698        "webhook": "{webhook_id, dispatch_signing_secret (returned once)}"}, sources=src(SRC_VOICE, f"{SRC_S2S}/sip")),699    obj("Skill", "object", "Hosted skill (OpenAPI).", {"id": "string", "object": "'skill'", "name": "string (SKILL.md frontmatter)", "description": "string (frontmatter)", "created_at": "integer unix s", "default_version": "'1'", "latest_version": "'1'"},700        status=["DOCUMENTED", "ACCOUNT_RESTRICTED"], verification=ver("restricted", 404, "all /v1/skills routes 404 with our key"), sources=src(SRC_OPENAPI)),701    obj("SkillList", "response", "GET /v1/skills.", {"object": "'list'", "data": "Skill[]", "first_id": "string", "last_id": "string", "has_more": "boolean"}, status=["DOCUMENTED", "ACCOUNT_RESTRICTED"], sources=src(SRC_OPENAPI)),702    obj("DeletedSkill", "response", "DELETE /v1/skills/{id}.", {"id": "string", "object": "'skill.deleted'", "deleted": "boolean"}, status=["DOCUMENTED", "ACCOUNT_RESTRICTED"], sources=src(SRC_OPENAPI)),703    obj("LocalShellSkill", "request object", "Skill made available to a local `shell` tool environment (ShellCall.environment.skills[]).", {"name": "string", "description": "string", "path": "string (directory containing SKILL.md)"}, sources=src(SRC_OPENAPI)),704    obj("xAI error envelopes (media/voice)", "error", "Three shapes observed on this surface.",705        {"A (most 4xx, application/json)": "{code: 'invalid-argument'|'not-found'|…, error: string}", "B (routing 404, application/json)": "{error: {code: 404, message: string}}",706         "C (body deserialization, 422 text/plain)": "'Failed to deserialize the JSON body into the target type: <field>: unknown variant `x`, expected one of …' / 'missing field `y`'"},707        status=["LIVE_VERIFIED"], example={"A": {"code": "invalid-argument", "error": "The number of images to generate (n) must be between 1 and 10 inclusive."}, "B": {"error": {"code": 404, "message": "The requested resource was not found. Please check the URL and try again. Documentation is available at https://docs.x.ai/"}},708                                            "C": "Failed to deserialize the JSON body into the target type: missing field `language` at line 1 column 34"},709        verification=ver("success", 422, "images 7:7, n=0, png; videos 99 s, 4k, missing video; tts missing language; skills 404"), sources=src(RAW + "*.json")),710]711712# --------------------------------------------------------------------------------------------------------------------713# LIFECYCLES714# --------------------------------------------------------------------------------------------------------------------715LIFECYCLES = [716    {"provider": "xai", "object": "video_generation_job", "api_family": "videos", "status_field": "status",717     "states": {"pending": "Job running. GET returns HTTP 202 with progress 0–99 (observed 1 → 87).", "done": "Terminal. HTTP 200; `video.url` (temporary), `video.duration`, `video.respect_moderation`, `model`, `usage.cost_in_usd_ticks`, progress 100. If respect_moderation=false the url is empty.",718                "failed": "Terminal. `error.code` ∈ invalid_argument | permission_denied | failed_precondition | service_unavailable | internal_error, `error.message`; progress/model/video omitted.",719                "expired": "Terminal. Result no longer retrievable (retention window elapsed; the docs do not state the window — the Batch API notes media URLs expire after 1 hour)."},720     "initial": "pending", "terminal": ["done", "failed", "expired"],721     "transitions": [["pending", "done"], ["pending", "failed"], ["pending", "expired"], ["done", "expired"]],722     "http_status_by_state": {"pending": 202, "done": 200, "failed": "200 (documented as body-level failure; HTTP not observed)", "expired": "unobserved"},723     "timestamps": [], "polling": {"documented_interval": "SDK default 100 ms, examples use 5 s", "sdk_timeout_default": "10 min", "observed_completion": "1 s / 480p / grok-imagine-video → done in 11.4 s"},724     "status": ["DOCUMENTED", "LIVE_VERIFIED"], "sources": src(SRC_VID, SRC_VID_GEN), "last_verified": V},725    {"provider": "xai", "object": "realtime.response", "api_family": "voice", "status_field": "response.status",726     "states": {"in_progress": "Emitted on response.created; deltas follow.", "completed": "Terminal (response.done).", "cancelled": "Terminal: interrupted by VAD barge-in or response.cancel.", "incomplete": "Terminal: cut short (max tokens / duration)."},727     "initial": "in_progress", "terminal": ["completed", "cancelled", "incomplete"], "transitions": [["in_progress", "completed"], ["in_progress", "cancelled"], ["in_progress", "incomplete"]],728     "status": ["DOCUMENTED", "LIVE_VERIFIED"], "sources": src(SRC_WS_RT), "last_verified": V},729    {"provider": "xai", "object": "realtime.item", "api_family": "voice", "status_field": "item.status",730     "states": {"in_progress": "Assistant item being generated (response.output_item.added / conversation.item.added).", "completed": "Item finished (user items are created completed; assistant items on response.output_item.done)."},731     "initial": "in_progress (assistant) | completed (user)", "terminal": ["completed"], "transitions": [["in_progress", "completed"]],732     "status": ["LIVE_VERIFIED"], "sources": src(SRC_WS_RT, RAW + "voice-ws-session.json"), "last_verified": V},733    {"provider": "xai", "object": "stt_streaming_transcript", "api_family": "voice", "status_field": "transcript.partial flags (is_final, speech_final)",734     "states": {"interim": "is_final=false — text may change (only with interim_results=true).", "chunk_final": "is_final=true, speech_final=false — chunk locked (also the demoted state when smart_turn confidence is below threshold).",735                "utterance_final": "is_final=true, speech_final=true — speaker stopped (endpointing / smart_turn / finalize).", "done": "transcript.done after audio.done; connection closes."},736     "initial": "transcript.created (ready)", "terminal": ["done"], "transitions": [["interim", "chunk_final"], ["chunk_final", "utterance_final"], ["interim", "utterance_final"], ["utterance_final", "done"]],737     "status": ["DOCUMENTED"], "sources": src(SRC_WS_STT, SRC_STT), "last_verified": V},738    {"provider": "xai", "object": "image_generation_call (Responses tool output)", "api_family": "responses", "status_field": "status",739     "states": {"in_progress": "Tool call accepted.", "generating": "Image being rendered.", "completed": "`result` holds bare base64 (JPEG/PNG by magic bytes).", "failed": "`result` null."},740     "initial": "in_progress", "terminal": ["completed", "failed"], "transitions": [["in_progress", "generating"], ["generating", "completed"], ["generating", "failed"], ["in_progress", "failed"]],741     "status": ["DOCUMENTED"], "sources": src(SRC_OPENAPI, f"{D}/tools/image-generation"), "notes": "Documented by the tools agent (docs/tools); listed here for the status machine only.", "last_verified": V},742]743744745def write(path: Path, payload) -> None:746    path.parent.mkdir(parents=True, exist_ok=True)747    path.write_text(json.dumps(payload, indent=1, ensure_ascii=False) + "\n")748    n = len(payload["records"]) if isinstance(payload, dict) else len(payload)749    print(f"wrote {path.relative_to(ROOT)} ({n} records)")750751752def main() -> None:753    meta = {"domain": "xai-media-voice-skills", "generated_by": "scripts/gen_xai_media.py", "last_verified": V,754            "live_calls_logged": "reports/live-requests.jsonl (2026-09-19T04:xx–05:xxZ, provider xai, notes prefixed 'media-agent')", "raw_probe_dir": RAW + "(gitignored)",755            "estimated_spend_usd": 0.16, "models_touched": ["grok-imagine-image", "grok-imagine-video", "grok-voice-think-fast-2.0 (grok-voice-latest)", "eve (TTS)", "grok-voice-transcribe-2.0 (default STT)", "grok-build-0.1"],756            "count": len(ENDPOINTS)}757    write(FRAG / "endpoints" / "xai-media-voice-skills.json", {"_meta": meta, "records": ENDPOINTS})758    write(FRAG / "parameters" / "xai-images.json", PARAMS_IMAGES)759    write(FRAG / "parameters" / "xai-videos.json", PARAMS_VIDEOS)760    write(FRAG / "parameters" / "xai-voice.json", PARAMS_VOICE)761    write(FRAG / "parameters" / "xai-skills.json", PARAMS_SKILLS)762    write(FRAG / "streaming-events" / "xai-voice.json", build_streaming())763    write(FRAG / "objects" / "xai-media-objects.json", {"domain": "xai-media", "generated_at": V, "records": OBJECTS})764    write(FRAG / "status-lifecycles" / "xai-lifecycles.json", LIFECYCLES)765766767if __name__ == "__main__":768    main()769