Python 88.3%
TypeScript 7.6%
Shell 4.1%
1"""OpenAI Audio API smoke tests (TTS, STT, translation, streaming). Gated by RUN_AUDIO_TESTS=true (≈ $0.002 per run).23Last run: 2026-09-18 — all passed. Voice-consent listing is asserted loosely because our key got 404 'Endpoint not found.'4"""5from __future__ import annotations67import json8import uuid9from pathlib import Path1011import pytest1213ROOT = Path(__file__).resolve().parents[2]14MP3 = ROOT / "tmp-live" / "ok.mp3"15pytestmark = pytest.mark.run_audio_tests161718def _multipart(fields: dict, data: bytes) -> tuple[bytes, str]:19 b = uuid.uuid4().hex20 out = b""21 for k, v in fields.items():22 for item in (v if isinstance(v, list) else [v]):23 out += f"--{b}\r\nContent-Disposition: form-data; name=\"{k + '[]' if isinstance(v, list) else k}\"\r\n\r\n{item}\r\n".encode()24 out += f"--{b}\r\nContent-Disposition: form-data; name=\"file\"; filename=\"ok.mp3\"\r\nContent-Type: audio/mpeg\r\n\r\n".encode() + data + b"\r\n"25 return out + f"--{b}--\r\n".encode(), f"multipart/form-data; boundary={b}"262728@pytest.fixture(scope="module")29def mp3(openai) -> bytes:30 if MP3.exists() and MP3.stat().st_size > 1000:31 return MP3.read_bytes()32 st, data, hdrs = openai("POST", "/v1/audio/speech", {"model": "gpt-4o-mini-tts", "voice": "alloy", "input": "OK", "response_format": "mp3"},33 est_cost_usd=0.0006, note="test_audio tts fixture")34 assert st == 200 and isinstance(data, bytes) and len(data) > 100035 MP3.parent.mkdir(exist_ok=True)36 MP3.write_bytes(data)37 return data383940def test_tts_mp3(openai):41 st, data, hdrs = openai("POST", "/v1/audio/speech", {"model": "gpt-4o-mini-tts", "voice": "marin", "input": "OK", "response_format": "mp3", "instructions": "Calm."},42 est_cost_usd=0.0006, note="test_audio tts mp3")43 assert st == 200, data44 assert isinstance(data, bytes) and len(data) > 100045 assert "audio" in hdrs.get("Content-Type", "").lower() or "octet" in hdrs.get("Content-Type", "").lower()464748def test_tts_sse_events(openai):49 st, lines, _ = openai("POST", "/v1/audio/speech", {"model": "gpt-4o-mini-tts", "voice": "alloy", "input": "OK", "stream_format": "sse", "response_format": "pcm"},50 stream=True, est_cost_usd=0.0006, note="test_audio tts sse")51 assert st == 20052 types = [json.loads(l[5:])["type"] for l in lines if l.startswith("data:") and "[DONE]" not in l]53 assert types[0] == "speech.audio.delta" and types[-1] == "speech.audio.done"545556def test_stt_json_with_usage(openai, mp3):57 body, ct = _multipart({"model": "gpt-4o-mini-transcribe", "response_format": "json"}, mp3)58 st, data, _ = openai("POST", "/v1/audio/transcriptions", data=body, content_type=ct, est_cost_usd=0.0002, note="test_audio stt json")59 assert st == 200, data60 assert "ok" in data["text"].lower()61 assert data["usage"]["type"] == "tokens" and data["usage"]["input_token_details"]["audio_tokens"] > 0626364def test_stt_whisper_verbose_word_timestamps(openai, mp3):65 body, ct = _multipart({"model": "whisper-1", "response_format": "verbose_json", "timestamp_granularities": ["word", "segment"]}, mp3)66 st, data, _ = openai("POST", "/v1/audio/transcriptions", data=body, content_type=ct, est_cost_usd=0.0002, note="test_audio whisper verbose")67 assert st == 200, data68 assert data["task"] == "transcribe" and data["language"] == "english"69 assert data["words"] and {"word", "start", "end"} <= set(data["words"][0])70 assert data["segments"] and "avg_logprob" in data["segments"][0]71 assert data["usage"]["type"] == "duration"727374def test_stt_diarized(openai, mp3):75 body, ct = _multipart({"model": "gpt-4o-transcribe-diarize", "response_format": "diarized_json"}, mp3)76 st, data, _ = openai("POST", "/v1/audio/transcriptions", data=body, content_type=ct, est_cost_usd=0.0002, note="test_audio diarize")77 assert st == 200, data78 seg = data["segments"][0]79 assert seg["type"] == "transcript.text.segment" and seg["speaker"] == "A"808182def test_stt_stream_events(openai, mp3):83 body, ct = _multipart({"model": "gpt-4o-mini-transcribe", "stream": "true"}, mp3)84 st, lines, _ = openai("POST", "/v1/audio/transcriptions", data=body, content_type=ct, stream=True, est_cost_usd=0.0002, note="test_audio stt stream")85 assert st == 20086 evs = [json.loads(l[5:]) for l in lines if l.startswith("data:") and "[DONE]" not in l] # terminator: `data: [DONE]`87 assert evs[0]["type"] == "transcript.text.delta" and evs[-1]["type"] == "transcript.text.done"88 assert evs[-1]["usage"]["type"] == "tokens"899091def test_translation_whisper(openai, mp3):92 body, ct = _multipart({"model": "whisper-1", "response_format": "json"}, mp3)93 st, data, _ = openai("POST", "/v1/audio/translations", data=body, content_type=ct, est_cost_usd=0.0002, note="test_audio translation")94 assert st == 200, data95 assert "text" in data969798def test_voice_consents_list_shape_or_restricted(openai):99 st, data, _ = openai("GET", "/v1/audio/voice_consents?limit=2", note="test_audio voice consents list")100 if st == 200:101 assert data["object"] == "list" and isinstance(data["data"], list) and "has_more" in data102 else: # 2026-09-18: 404 'Endpoint not found.' — custom voices are access-gated103 assert st in (403, 404), data104 assert data["error"]["type"] == "invalid_request_error"105106107def test_chat_audio_model_requires_audio_modality(openai):108 st, data, _ = openai("POST", "/v1/chat/completions", {"model": "gpt-audio-mini", "modalities": ["text"], "max_completion_tokens": 8,109 "messages": [{"role": "user", "content": "Reply with OK."}]}, note="test_audio chat text-only")110 assert st == 400 and data["error"]["code"] == "invalid_value"111