SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
4.7 KB · 83 lines python
Raw Blame History
1"""Gemini TTS + transcription. Everything here ran free (free tier) on 2026-09-18, but audio calls are gated by2RUN_AUDIO_TESTS=true (marker run_audio_tests) because they bill on paid keys ($1/M text in, $20/M audio out for 3.1 TTS)."""3from __future__ import annotations45import base646import io7import json8import time9import wave10import pytest1112TTS = "gemini-3.1-flash-tts-preview"13STT = "gemini-3.5-transcribe"141516def _speech_cfg(voice="Kore"):17    return {"responseModalities": ["AUDIO"], "speechConfig": {"voiceConfig": {"prebuiltVoiceConfig": {"voiceName": voice}}}}181920def test_invalid_voice_rejected(gemini):21    st, body, _ = gemini("POST", f"/v1beta/models/{TTS}:generateContent", {"contents": [{"parts": [{"text": "Say: OK"}]}], "generationConfig": _speech_cfg("NotAVoice")}, note="test_speech invalid voice")22    if st == 429 and "PerDay" in json.dumps(body.get("error", {}).get("details", [])):23        pytest.skip("daily free-tier TTS quota exhausted (10 requests/day/model) — quota check precedes validation")24    assert st == 400 and body["error"]["status"] == "INVALID_ARGUMENT" and "voice" in body["error"]["message"].lower()252627def test_transcribe_live_model_rejects_unary(gemini):28    st, body, _ = gemini("POST", "/v1beta/models/gemini-3.5-transcribe-live:generateContent", {"contents": [{"parts": [{"text": "x"}]}]}, note="test_speech live unary")29    assert st == 400 and "bidiGenerateContent" in body["error"]["message"]303132def _retry_delay_s(body) -> int:33    """Free-tier TTS has a small per-minute quota (observed quotaValue 10): honour google.rpc.RetryInfo once."""34    for d in (body.get("error", {}).get("details") or []):35        if d.get("@type", "").endswith("RetryInfo"):36            return min(int(float(d.get("retryDelay", "60s").rstrip("s"))) + 1, 90)37    return 0383940@pytest.mark.run_audio_tests41def test_tts_returns_pcm_and_transcribe_reads_it_back(gemini):42    body = {"contents": [{"parts": [{"text": "Say: OK"}]}], "generationConfig": _speech_cfg()}43    st, res, _ = gemini("POST", f"/v1beta/models/{TTS}:generateContent", body, est_cost_usd=0.001, note="test_speech tts")44    if st == 429:45        quota = json.dumps(res.get("error", {}).get("details", []))46        if "PerDay" in quota:  # free tier: GenerateRequestsPerDayPerProjectPerModel-FreeTier = 10 (observed 2026-09-18)47            pytest.xfail("daily free-tier TTS quota exhausted (10 requests/day/model): " + quota[:160])48        if _retry_delay_s(res):49            time.sleep(_retry_delay_s(res))50            st, res, _ = gemini("POST", f"/v1beta/models/{TTS}:generateContent", body, est_cost_usd=0.001, note="test_speech tts retry after RetryInfo")51    assert st == 200, res52    cand = res["candidates"][0]53    assert "content" in cand, f"empty candidate: {cand}"  # observed finishReason OTHER without content once (multi-speaker)54    part = cand["content"]["parts"][0]["inlineData"]55    assert part["mimeType"].lower().startswith("audio/l16") and "24000" in part["mimeType"]56    pcm = base64.b64decode(part["data"])57    assert 10_000 < len(pcm) < 400_00058    assert any(d["modality"] == "AUDIO" for d in res["usageMetadata"]["candidatesTokensDetails"])59    buf = io.BytesIO()60    with wave.open(buf, "wb") as w:61        w.setnchannels(1); w.setsampwidth(2); w.setframerate(24000); w.writeframes(pcm)62    st, tr, _ = gemini("POST", f"/v1beta/models/{STT}:generateContent",63                       {"contents": [{"parts": [{"inlineData": {"mimeType": "audio/wav", "data": base64.b64encode(buf.getvalue()).decode()}}]}],64                        "generationConfig": {"audioTranscriptionConfig": {"languageCodes": ["en-US"], "wordTimestamp": True}}},65                       est_cost_usd=0.0001, note="test_speech transcribe")66    assert st == 200, tr67    parts = tr["candidates"][0].get("content", {}).get("parts", [])68    assert parts and "audioTranscription" in parts[0]69    assert "ok" in parts[0]["audioTranscription"]["text"].lower()70    assert parts[0]["audioTranscription"]["words"][0]["startOffset"].endswith("s")71    assert tr["usageMetadata"]["promptTokensDetails"][0]["modality"] == "AUDIO"727374@pytest.mark.run_audio_tests75def test_transcribe_smart_incompatible_with_diarization(gemini):76    buf = io.BytesIO()77    with wave.open(buf, "wb") as w:78        w.setnchannels(1); w.setsampwidth(2); w.setframerate(16000); w.writeframes(b"\x00\x00" * 8000)79    body = {"contents": [{"parts": [{"inlineData": {"mimeType": "audio/wav", "data": base64.b64encode(buf.getvalue()).decode()}}]}],80            "generationConfig": {"audioTranscriptionConfig": {"mode": "SMART", "diarization": True}}}81    st, res, _ = gemini("POST", f"/v1beta/models/{STT}:generateContent", body, note="test_speech smart+diarization")82    assert st == 400 and "incompatible" in res["error"]["message"].lower()83