Python 88.3%
TypeScript 7.6%
Shell 4.1%
1"""Gemini TTS + transcription. Everything here ran free (free tier) on 2026-09-18, but audio calls are gated by2RUN_AUDIO_TESTS=true (marker run_audio_tests) because they bill on paid keys ($1/M text in, $20/M audio out for 3.1 TTS)."""3from __future__ import annotations45import base646import io7import json8import time9import wave10import pytest1112TTS = "gemini-3.1-flash-tts-preview"13STT = "gemini-3.5-transcribe"141516def _speech_cfg(voice="Kore"):17 return {"responseModalities": ["AUDIO"], "speechConfig": {"voiceConfig": {"prebuiltVoiceConfig": {"voiceName": voice}}}}181920def test_invalid_voice_rejected(gemini):21 st, body, _ = gemini("POST", f"/v1beta/models/{TTS}:generateContent", {"contents": [{"parts": [{"text": "Say: OK"}]}], "generationConfig": _speech_cfg("NotAVoice")}, note="test_speech invalid voice")22 if st == 429 and "PerDay" in json.dumps(body.get("error", {}).get("details", [])):23 pytest.skip("daily free-tier TTS quota exhausted (10 requests/day/model) — quota check precedes validation")24 assert st == 400 and body["error"]["status"] == "INVALID_ARGUMENT" and "voice" in body["error"]["message"].lower()252627def test_transcribe_live_model_rejects_unary(gemini):28 st, body, _ = gemini("POST", "/v1beta/models/gemini-3.5-transcribe-live:generateContent", {"contents": [{"parts": [{"text": "x"}]}]}, note="test_speech live unary")29 assert st == 400 and "bidiGenerateContent" in body["error"]["message"]303132def _retry_delay_s(body) -> int:33 """Free-tier TTS has a small per-minute quota (observed quotaValue 10): honour google.rpc.RetryInfo once."""34 for d in (body.get("error", {}).get("details") or []):35 if d.get("@type", "").endswith("RetryInfo"):36 return min(int(float(d.get("retryDelay", "60s").rstrip("s"))) + 1, 90)37 return 0383940@pytest.mark.run_audio_tests41def test_tts_returns_pcm_and_transcribe_reads_it_back(gemini):42 body = {"contents": [{"parts": [{"text": "Say: OK"}]}], "generationConfig": _speech_cfg()}43 st, res, _ = gemini("POST", f"/v1beta/models/{TTS}:generateContent", body, est_cost_usd=0.001, note="test_speech tts")44 if st == 429:45 quota = json.dumps(res.get("error", {}).get("details", []))46 if "PerDay" in quota: # free tier: GenerateRequestsPerDayPerProjectPerModel-FreeTier = 10 (observed 2026-09-18)47 pytest.xfail("daily free-tier TTS quota exhausted (10 requests/day/model): " + quota[:160])48 if _retry_delay_s(res):49 time.sleep(_retry_delay_s(res))50 st, res, _ = gemini("POST", f"/v1beta/models/{TTS}:generateContent", body, est_cost_usd=0.001, note="test_speech tts retry after RetryInfo")51 assert st == 200, res52 cand = res["candidates"][0]53 assert "content" in cand, f"empty candidate: {cand}" # observed finishReason OTHER without content once (multi-speaker)54 part = cand["content"]["parts"][0]["inlineData"]55 assert part["mimeType"].lower().startswith("audio/l16") and "24000" in part["mimeType"]56 pcm = base64.b64decode(part["data"])57 assert 10_000 < len(pcm) < 400_00058 assert any(d["modality"] == "AUDIO" for d in res["usageMetadata"]["candidatesTokensDetails"])59 buf = io.BytesIO()60 with wave.open(buf, "wb") as w:61 w.setnchannels(1); w.setsampwidth(2); w.setframerate(24000); w.writeframes(pcm)62 st, tr, _ = gemini("POST", f"/v1beta/models/{STT}:generateContent",63 {"contents": [{"parts": [{"inlineData": {"mimeType": "audio/wav", "data": base64.b64encode(buf.getvalue()).decode()}}]}],64 "generationConfig": {"audioTranscriptionConfig": {"languageCodes": ["en-US"], "wordTimestamp": True}}},65 est_cost_usd=0.0001, note="test_speech transcribe")66 assert st == 200, tr67 parts = tr["candidates"][0].get("content", {}).get("parts", [])68 assert parts and "audioTranscription" in parts[0]69 assert "ok" in parts[0]["audioTranscription"]["text"].lower()70 assert parts[0]["audioTranscription"]["words"][0]["startOffset"].endswith("s")71 assert tr["usageMetadata"]["promptTokensDetails"][0]["modality"] == "AUDIO"727374@pytest.mark.run_audio_tests75def test_transcribe_smart_incompatible_with_diarization(gemini):76 buf = io.BytesIO()77 with wave.open(buf, "wb") as w:78 w.setnchannels(1); w.setsampwidth(2); w.setframerate(16000); w.writeframes(b"\x00\x00" * 8000)79 body = {"contents": [{"parts": [{"inlineData": {"mimeType": "audio/wav", "data": base64.b64encode(buf.getvalue()).decode()}}]}],80 "generationConfig": {"audioTranscriptionConfig": {"mode": "SMART", "diarization": True}}}81 st, res, _ = gemini("POST", f"/v1beta/models/{STT}:generateContent", body, note="test_speech smart+diarization")82 assert st == 400 and "incompatible" in res["error"]["message"].lower()83