"""OpenAI Audio API smoke tests (TTS, STT, translation, streaming). Gated by RUN_AUDIO_TESTS=true (≈ $0.002 per run). Last run: 2026-09-18 — all passed. Voice-consent listing is asserted loosely because our key got 404 'Endpoint not found.' """ from __future__ import annotations import json import uuid from pathlib import Path import pytest ROOT = Path(__file__).resolve().parents[2] MP3 = ROOT / "tmp-live" / "ok.mp3" pytestmark = pytest.mark.run_audio_tests def _multipart(fields: dict, data: bytes) -> tuple[bytes, str]: b = uuid.uuid4().hex out = b"" for k, v in fields.items(): for item in (v if isinstance(v, list) else [v]): out += f"--{b}\r\nContent-Disposition: form-data; name=\"{k + '[]' if isinstance(v, list) else k}\"\r\n\r\n{item}\r\n".encode() out += f"--{b}\r\nContent-Disposition: form-data; name=\"file\"; filename=\"ok.mp3\"\r\nContent-Type: audio/mpeg\r\n\r\n".encode() + data + b"\r\n" return out + f"--{b}--\r\n".encode(), f"multipart/form-data; boundary={b}" @pytest.fixture(scope="module") def mp3(openai) -> bytes: if MP3.exists() and MP3.stat().st_size > 1000: return MP3.read_bytes() st, data, hdrs = openai("POST", "/v1/audio/speech", {"model": "gpt-4o-mini-tts", "voice": "alloy", "input": "OK", "response_format": "mp3"}, est_cost_usd=0.0006, note="test_audio tts fixture") assert st == 200 and isinstance(data, bytes) and len(data) > 1000 MP3.parent.mkdir(exist_ok=True) MP3.write_bytes(data) return data def test_tts_mp3(openai): st, data, hdrs = openai("POST", "/v1/audio/speech", {"model": "gpt-4o-mini-tts", "voice": "marin", "input": "OK", "response_format": "mp3", "instructions": "Calm."}, est_cost_usd=0.0006, note="test_audio tts mp3") assert st == 200, data assert isinstance(data, bytes) and len(data) > 1000 assert "audio" in hdrs.get("Content-Type", "").lower() or "octet" in hdrs.get("Content-Type", "").lower() def test_tts_sse_events(openai): st, lines, _ = openai("POST", "/v1/audio/speech", {"model": "gpt-4o-mini-tts", "voice": "alloy", "input": "OK", "stream_format": "sse", "response_format": "pcm"}, stream=True, est_cost_usd=0.0006, note="test_audio tts sse") assert st == 200 types = [json.loads(l[5:])["type"] for l in lines if l.startswith("data:") and "[DONE]" not in l] assert types[0] == "speech.audio.delta" and types[-1] == "speech.audio.done" def test_stt_json_with_usage(openai, mp3): body, ct = _multipart({"model": "gpt-4o-mini-transcribe", "response_format": "json"}, mp3) st, data, _ = openai("POST", "/v1/audio/transcriptions", data=body, content_type=ct, est_cost_usd=0.0002, note="test_audio stt json") assert st == 200, data assert "ok" in data["text"].lower() assert data["usage"]["type"] == "tokens" and data["usage"]["input_token_details"]["audio_tokens"] > 0 def test_stt_whisper_verbose_word_timestamps(openai, mp3): body, ct = _multipart({"model": "whisper-1", "response_format": "verbose_json", "timestamp_granularities": ["word", "segment"]}, mp3) st, data, _ = openai("POST", "/v1/audio/transcriptions", data=body, content_type=ct, est_cost_usd=0.0002, note="test_audio whisper verbose") assert st == 200, data assert data["task"] == "transcribe" and data["language"] == "english" assert data["words"] and {"word", "start", "end"} <= set(data["words"][0]) assert data["segments"] and "avg_logprob" in data["segments"][0] assert data["usage"]["type"] == "duration" def test_stt_diarized(openai, mp3): body, ct = _multipart({"model": "gpt-4o-transcribe-diarize", "response_format": "diarized_json"}, mp3) st, data, _ = openai("POST", "/v1/audio/transcriptions", data=body, content_type=ct, est_cost_usd=0.0002, note="test_audio diarize") assert st == 200, data seg = data["segments"][0] assert seg["type"] == "transcript.text.segment" and seg["speaker"] == "A" def test_stt_stream_events(openai, mp3): body, ct = _multipart({"model": "gpt-4o-mini-transcribe", "stream": "true"}, mp3) st, lines, _ = openai("POST", "/v1/audio/transcriptions", data=body, content_type=ct, stream=True, est_cost_usd=0.0002, note="test_audio stt stream") assert st == 200 evs = [json.loads(l[5:]) for l in lines if l.startswith("data:") and "[DONE]" not in l] # terminator: `data: [DONE]` assert evs[0]["type"] == "transcript.text.delta" and evs[-1]["type"] == "transcript.text.done" assert evs[-1]["usage"]["type"] == "tokens" def test_translation_whisper(openai, mp3): body, ct = _multipart({"model": "whisper-1", "response_format": "json"}, mp3) st, data, _ = openai("POST", "/v1/audio/translations", data=body, content_type=ct, est_cost_usd=0.0002, note="test_audio translation") assert st == 200, data assert "text" in data def test_voice_consents_list_shape_or_restricted(openai): st, data, _ = openai("GET", "/v1/audio/voice_consents?limit=2", note="test_audio voice consents list") if st == 200: assert data["object"] == "list" and isinstance(data["data"], list) and "has_more" in data else: # 2026-09-18: 404 'Endpoint not found.' — custom voices are access-gated assert st in (403, 404), data assert data["error"]["type"] == "invalid_request_error" def test_chat_audio_model_requires_audio_modality(openai): st, data, _ = openai("POST", "/v1/chat/completions", {"model": "gpt-audio-mini", "modalities": ["text"], "max_completion_tokens": 8, "messages": [{"role": "user", "content": "Reply with OK."}]}, note="test_audio chat text-only") assert st == 400 and data["error"]["code"] == "invalid_value"