Python 88.3%
TypeScript 7.6%
Shell 4.1%
1"""xAI Voice tests. Voice roster, client secret and TTS validation are free and always run. TTS/STT bytes are gated by RUN_AUDIO_TESTS2(~$0.0001), the Speech-to-Speech text turn by RUN_REALTIME_TESTS (~$0.005)."""3from __future__ import annotations45import asyncio6import json7import os8import time9import uuid1011import pytest121314def test_tts_voices_roster(xai):15 st, body, _ = xai("GET", "/v1/tts/voices", note="test_voice roster (free)")16 assert st == 200, body17 ids = {v["voice_id"] for v in body["voices"]}18 assert {"eve", "ara", "leo", "rex", "sal"} <= ids and len(ids) >= 2619 v = next(v for v in body["voices"] if v["voice_id"] == "eve")20 assert v["name"] == "Eve" and v["gender"] == "female" # `gender` is undocumented but stable212223def test_get_voice(xai):24 st, body, _ = xai("GET", "/v1/tts/voices/eve", note="test_voice get eve (free)")25 assert st == 200 and body["voice_id"] == "eve"262728def test_tts_requires_language(xai):29 st, body, hdrs = xai("POST", "/v1/tts", {"text": "OK.", "voice_id": "eve"}, note="test_voice tts missing language (free)")30 assert st == 422 and hdrs.get("Content-Type", "").startswith("text/plain") and b"missing field `language`" in body313233def test_client_secret(xai):34 st, body, _ = xai("POST", "/v1/realtime/client_secrets", {"expires_after": {"seconds": 60}}, note="test_voice client secret (free)")35 assert st == 200, body36 assert body["value"].startswith("xai-") and len(body["value"]) > 4037 assert 50 <= body["expires_at"] - int(time.time()) <= 61383940def test_custom_voices_list(xai):41 st, body, _ = xai("GET", "/v1/custom-voices", note="test_voice custom voices list (free)")42 assert st == 200 and isinstance(body["voices"], list) and body.get("cap", 30) >= 1434445@pytest.mark.run_audio_tests46def test_tts_then_stt_roundtrip(xai):47 st, wav, hdrs = xai("POST", "/v1/tts", {"text": "OK.", "voice_id": "eve", "language": "en", "output_format": {"codec": "wav", "sample_rate": 16000}},48 est_cost_usd=0.00005, note="test_voice tts wav")49 assert st == 200 and hdrs["Content-Type"].startswith("audio/wav") and wav[:4] == b"RIFF"50 bnd = uuid.uuid4().hex51 form = (f'--{bnd}\r\nContent-Disposition: form-data; name="language"\r\n\r\nen\r\n'52 f'--{bnd}\r\nContent-Disposition: form-data; name="file"; filename="ok.wav"\r\nContent-Type: audio/wav\r\n\r\n').encode() + wav + f"\r\n--{bnd}--\r\n".encode()53 st, body, _ = xai("POST", "/v1/stt", data=form, content_type=f"multipart/form-data; boundary={bnd}", est_cost_usd=0.0001, note="test_voice stt")54 assert st == 200, body55 assert "ok" in body["text"].lower() and body["language"] == "en" and body["duration"] > 0 and body["words"][0]["end"] > body["words"][0]["start"]565758@pytest.mark.run_realtime_tests59def test_speech_to_speech_text_turn(xai):60 """~$0.005: one text turn on wss://api.x.ai/v1/realtime; asserts the documented lifecycle events and the usage block."""61 import websockets6263 async def run() -> tuple[list[str], dict]:64 types, usage = [], {}65 async with websockets.connect("wss://api.x.ai/v1/realtime?model=grok-voice-latest",66 additional_headers={"Authorization": f"Bearer {os.environ['XAI_API_KEY']}"}, max_size=None) as ws:67 await ws.send(json.dumps({"type": "session.update", "session": {"instructions": "Answer with the single word OK.", "voice": "eve",68 "reasoning": {"effort": "none"}, "turn_detection": None}}))69 await ws.send(json.dumps({"type": "conversation.item.create", "item": {"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Reply with OK."}]}}))70 await ws.send(json.dumps({"type": "response.create"}))71 while True:72 ev = json.loads(await asyncio.wait_for(ws.recv(), timeout=60))73 types.append(ev["type"])74 if ev["type"] == "response.done":75 usage = ev.get("usage", {})76 break77 assert ev["type"] != "error", ev78 return types, usage7980 types, usage = asyncio.run(run())81 from scripts import live82 live.log_request("xai", "WS", "/v1/realtime text turn", 0, 0.005, "test_voice speech-to-speech text turn")83 for must in ("session.created", "conversation.created", "session.updated", "response.created", "response.output_audio.delta",84 "response.output_audio_transcript.done", "response.done"):85 assert must in types, types86 assert usage["billable_audio_seconds"] >= 1 and usage["output_token_details"]["audio_tokens"] > 087