SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
4.6 KB · 87 lines python
Raw Blame History
1"""xAI Voice tests. Voice roster, client secret and TTS validation are free and always run. TTS/STT bytes are gated by RUN_AUDIO_TESTS2(~$0.0001), the Speech-to-Speech text turn by RUN_REALTIME_TESTS (~$0.005)."""3from __future__ import annotations45import asyncio6import json7import os8import time9import uuid1011import pytest121314def test_tts_voices_roster(xai):15    st, body, _ = xai("GET", "/v1/tts/voices", note="test_voice roster (free)")16    assert st == 200, body17    ids = {v["voice_id"] for v in body["voices"]}18    assert {"eve", "ara", "leo", "rex", "sal"} <= ids and len(ids) >= 2619    v = next(v for v in body["voices"] if v["voice_id"] == "eve")20    assert v["name"] == "Eve" and v["gender"] == "female"   # `gender` is undocumented but stable212223def test_get_voice(xai):24    st, body, _ = xai("GET", "/v1/tts/voices/eve", note="test_voice get eve (free)")25    assert st == 200 and body["voice_id"] == "eve"262728def test_tts_requires_language(xai):29    st, body, hdrs = xai("POST", "/v1/tts", {"text": "OK.", "voice_id": "eve"}, note="test_voice tts missing language (free)")30    assert st == 422 and hdrs.get("Content-Type", "").startswith("text/plain") and b"missing field `language`" in body313233def test_client_secret(xai):34    st, body, _ = xai("POST", "/v1/realtime/client_secrets", {"expires_after": {"seconds": 60}}, note="test_voice client secret (free)")35    assert st == 200, body36    assert body["value"].startswith("xai-") and len(body["value"]) > 4037    assert 50 <= body["expires_at"] - int(time.time()) <= 61383940def test_custom_voices_list(xai):41    st, body, _ = xai("GET", "/v1/custom-voices", note="test_voice custom voices list (free)")42    assert st == 200 and isinstance(body["voices"], list) and body.get("cap", 30) >= 1434445@pytest.mark.run_audio_tests46def test_tts_then_stt_roundtrip(xai):47    st, wav, hdrs = xai("POST", "/v1/tts", {"text": "OK.", "voice_id": "eve", "language": "en", "output_format": {"codec": "wav", "sample_rate": 16000}},48                        est_cost_usd=0.00005, note="test_voice tts wav")49    assert st == 200 and hdrs["Content-Type"].startswith("audio/wav") and wav[:4] == b"RIFF"50    bnd = uuid.uuid4().hex51    form = (f'--{bnd}\r\nContent-Disposition: form-data; name="language"\r\n\r\nen\r\n'52            f'--{bnd}\r\nContent-Disposition: form-data; name="file"; filename="ok.wav"\r\nContent-Type: audio/wav\r\n\r\n').encode() + wav + f"\r\n--{bnd}--\r\n".encode()53    st, body, _ = xai("POST", "/v1/stt", data=form, content_type=f"multipart/form-data; boundary={bnd}", est_cost_usd=0.0001, note="test_voice stt")54    assert st == 200, body55    assert "ok" in body["text"].lower() and body["language"] == "en" and body["duration"] > 0 and body["words"][0]["end"] > body["words"][0]["start"]565758@pytest.mark.run_realtime_tests59def test_speech_to_speech_text_turn(xai):60    """~$0.005: one text turn on wss://api.x.ai/v1/realtime; asserts the documented lifecycle events and the usage block."""61    import websockets6263    async def run() -> tuple[list[str], dict]:64        types, usage = [], {}65        async with websockets.connect("wss://api.x.ai/v1/realtime?model=grok-voice-latest",66                                      additional_headers={"Authorization": f"Bearer {os.environ['XAI_API_KEY']}"}, max_size=None) as ws:67            await ws.send(json.dumps({"type": "session.update", "session": {"instructions": "Answer with the single word OK.", "voice": "eve",68                                                                            "reasoning": {"effort": "none"}, "turn_detection": None}}))69            await ws.send(json.dumps({"type": "conversation.item.create", "item": {"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Reply with OK."}]}}))70            await ws.send(json.dumps({"type": "response.create"}))71            while True:72                ev = json.loads(await asyncio.wait_for(ws.recv(), timeout=60))73                types.append(ev["type"])74                if ev["type"] == "response.done":75                    usage = ev.get("usage", {})76                    break77                assert ev["type"] != "error", ev78        return types, usage7980    types, usage = asyncio.run(run())81    from scripts import live82    live.log_request("xai", "WS", "/v1/realtime text turn", 0, 0.005, "test_voice speech-to-speech text turn")83    for must in ("session.created", "conversation.created", "session.updated", "response.created", "response.output_audio.delta",84                 "response.output_audio_transcript.done", "response.done"):85        assert must in types, types86    assert usage["billable_audio_seconds"] >= 1 and usage["output_token_details"]["audio_tokens"] > 087