#!/usr/bin/env python3 # STATUS: LIVE_VERIFIED — executed 2026-09-18 (exit 0 in 12.8s) # Thinking: thinkingBudget + includeThoughts on gemini-3.5-flash (thought summary part + thoughtsTokenCount), then thinkingLevel low on the same model. # Run: .venv/bin/python examples/gemini/thinking/thinking.py (GEMINI_API_KEY read from .env via scripts.live; google-genai 2.24) import pathlib import sys ROOT = pathlib.Path(__file__).resolve().parents[3] sys.path.insert(0, str(ROOT)) from scripts import live # noqa: E402 loads .env, provides log_request from google import genai # noqa: E402 from google.genai import types # noqa: E402,F401 client = genai.Client() MODEL = "gemini-3.5-flash" r = client.models.generate_content(model=MODEL, contents="What is 17*23? Reply with the number only.", config=types.GenerateContentConfig(max_output_tokens=600, thinking_config=types.ThinkingConfig(thinking_budget=256, include_thoughts=True))) live.log_request("gemini", "POST", "/v1beta/models/{model}:generateContent", 200, 0.0012, "example thinking/thinking.py (budget 256)") for p in r.candidates[0].content.parts: print("THOUGHT:" if p.thought else "ANSWER:", (p.text or "")[:120].replace("\n", " ")) print("thoughtsTokenCount:", r.usage_metadata.thoughts_token_count, "candidates:", r.usage_metadata.candidates_token_count) assert "391" in r.text r2 = client.models.generate_content(model=MODEL, contents="Reply with OK.", config=types.GenerateContentConfig(max_output_tokens=32, thinking_config=types.ThinkingConfig(thinking_level="low"))) live.log_request("gemini", "POST", "/v1beta/models/{model}:generateContent", 200, 0.00003, "example thinking/thinking.py (level low)") print("level low:", r2.text, r2.usage_metadata.thoughts_token_count)