"""Thinking: thinkingBudget/includeThoughts (gemini-3.5-flash), thinkingLevel (gemini-3.8-flash), conflicts. Verified 2026-09-18. The two thinking-model calls cost ~ $0.0015 together; gated behind RUN_EXPENSIVE_TESTS like other thinking tests. """ from __future__ import annotations import pytest from tests.gemini._core_helpers import call, gc, user FLASH = "gemini-3.5-flash" G38 = "gemini-3.8-flash" @pytest.mark.run_expensive_tests def test_thinking_budget_include_thoughts(gemini): st, r, _ = call(gemini, "POST", gc(FLASH), {"contents": user("What is 17*23? Reply with the number only."), "generationConfig": {"maxOutputTokens": 600, "thinkingConfig": {"thinkingBudget": 256, "includeThoughts": True}}}, timeout=180, note="test_thinking budget+thoughts") assert st == 200, r parts = r["candidates"][0]["content"]["parts"] assert r["usageMetadata"]["thoughtsTokenCount"] > 0 assert "391" in "".join(p.get("text", "") for p in parts if not p.get("thought")) assert parts[-1].get("thoughtSignature") if any(p.get("thought") for p in parts): assert parts[0]["thought"] is True, "thought summary parts come first" @pytest.mark.run_expensive_tests def test_thinking_level_low_and_minimal_error(gemini): st, r, _ = call(gemini, "POST", gc(G38), {"contents": user("Reply with OK."), "generationConfig": {"maxOutputTokens": 64, "thinkingConfig": {"thinkingLevel": "low"}}}, timeout=180, note="test_thinking level low") assert st == 200 and "OK" in r["candidates"][0]["content"]["parts"][0]["text"] st, r, _ = call(gemini, "POST", gc(G38), {"contents": user("Reply with OK."), "generationConfig": {"maxOutputTokens": 16, "thinkingConfig": {"thinkingLevel": "minimal"}}}, note="test_thinking level minimal") assert st == 400 and "MINIMAL is not supported" in r["error"]["message"] def test_conflicts_and_ranges_cheap(gemini, models): st, r, _ = call(gemini, "POST", gc(models["gemini"]), {"contents": user("Reply with OK."), "generationConfig": {"maxOutputTokens": 16, "thinkingConfig": {"thinkingLevel": "low", "thinkingBudget": 100}}}, note="test_thinking level+budget") assert st == 400 and "only one of thinking budget and thinking level" in r["error"]["message"] st, r, _ = call(gemini, "POST", gc(models["gemini"]), {"contents": user("Reply with OK."), "generationConfig": {"maxOutputTokens": 16, "thinkingConfig": {"thinkingBudget": 100000}}}, note="test_thinking budget range") assert st == 400 and "[-1, 65535]" in r["error"]["message"] st, r, _ = call(gemini, "POST", gc(models["gemini"]), {"contents": user("Reply with OK."), "generationConfig": {"maxOutputTokens": 16, "thinkingConfig": {"thinkingBudget": 0}}}, note="test_thinking flash-lite budget 0") assert st == 400, "3.5 Flash-Lite cannot disable thinking via budget 0" st, r, _ = call(gemini, "POST", gc(models["gemini"]), {"contents": user("Reply with OK."), "generationConfig": {"maxOutputTokens": 16, "thinkingConfig": {"thinkingLevel": "minimal"}}}, note="test_thinking flash-lite minimal") assert st == 200 and "thoughtsTokenCount" not in r["usageMetadata"] def test_thought_signature_roundtrip_and_corruption(gemini, models): st, r, _ = call(gemini, "POST", gc(models["gemini"]), {"contents": user("Reply with OK."), "generationConfig": {"maxOutputTokens": 8}}, note="test_thinking sig get") assert st == 200 parts = r["candidates"][0]["content"]["parts"] hist = [{"role": "user", "parts": [{"text": "Reply with OK."}]}, {"role": "model", "parts": parts}, {"role": "user", "parts": [{"text": "Again, reply with OK."}]}] st, r, _ = call(gemini, "POST", gc(models["gemini"]), {"contents": hist, "generationConfig": {"maxOutputTokens": 8}}, note="test_thinking sig roundtrip") assert st == 200, r bad = [{"role": "user", "parts": [{"text": "Reply with OK."}]}, {"role": "model", "parts": [{"text": "OK.", "thoughtSignature": "bm90LWEtcmVhbC1zaWc="}]}, {"role": "user", "parts": [{"text": "Again."}]}] st, r, _ = call(gemini, "POST", gc(models["gemini"]), {"contents": bad, "generationConfig": {"maxOutputTokens": 8}}, note="test_thinking sig corrupted") assert st == 400 and "Corrupted thought signature" in r["error"]["message"]