Python 88.3%
TypeScript 7.6%
Shell 4.1%
1#!/usr/bin/env python32"""Build the Gemini model catalogue + pricing fragments.34Inputs (offline):5 sources/gemini/models-api-raw.json live GET /v1beta/models (scripts/discover_gemini_models.py)6 sources/gemini/models-v1-raw.json live GET /v1/models7 sources/gemini/pages/gemini-api/docs/models/*.md official model pages (property tables)8 tmp-live/gemini-models-probe.json optional: per-model GET / generateContent probes (sanitized)9 hand-encoded tables below, transcribed from pricing.md, deprecations.md, thinking.md, rate-limits.md,10 interactions.md, flex-inference.md, computer-use.md (retrieved 2026-09-18).1112Outputs:13 generated/fragments/models/gemini-models.json14 generated/fragments/pricing/gemini-pricing.json1516Usage: python3 scripts/gen_gemini_models_fragment.py17"""18from __future__ import annotations1920import glob21import json22import os23import re24from pathlib import Path2526ROOT = Path(__file__).resolve().parents[1]27RAW = ROOT / "sources/gemini/models-api-raw.json"28RAW_V1 = ROOT / "sources/gemini/models-v1-raw.json"29PAGES = ROOT / "sources/gemini/pages/gemini-api/docs/models"30PROBE = ROOT / "tmp-live/gemini-models-probe.json"31OUT_MODELS = ROOT / "generated/fragments/models/gemini-models.json"32OUT_PRICING = ROOT / "generated/fragments/pricing/gemini-pricing.json"33RETRIEVED = "2026-09-18"34VERIFIED_AT = "2026-09-19"35D = "https://ai.google.dev/gemini-api/docs/"36SRC_MODELS = D + "models"37SRC_PRICING = D + "pricing"38SRC_DEPREC = D + "deprecations"39SRC_RATE = D + "rate-limits"40SRC_THINK = D + "thinking"41SRC_LIST = "https://ai.google.dev/api/models"424344def src(url: str, note: str | None = None) -> dict:45 d = {"url": url, "retrieved_at": RETRIEVED}46 if note:47 d["note"] = note48 return d495051# --------------------------------------------------------------------------------------------------52# Model pages -> structured53# --------------------------------------------------------------------------------------------------54def parse_pages() -> dict[str, dict]:55 out: dict[str, dict] = {}56 for f in sorted(glob.glob(str(PAGES / "*.md"))):57 slug = os.path.basename(f)[:-3]58 txt = open(f, encoding="utf-8").read()59 warn = re.search(r"> \[!WARNING\]\n> (.*)", txt)60 for s in re.split(r"\n## ", txt):61 lines = s.split("\n")62 head = lines[0].strip()63 if "| Property |" not in s:64 continue65 rows: dict[str, str] = {}66 for l in lines:67 m = re.match(r"\|\s*([^|]+?)\s*\|\s*(.*?)\s*\|\s*$", l)68 if m and m.group(1) not in ("Property", "---"):69 k = re.sub(r"\^.*?\^", "", m.group(1)).strip()70 v = re.sub(r"\]\(https?://[^)]+\)", "]", m.group(2)).replace("[", "").replace("]", "")71 rows[k] = v72 caps: dict[str, str] = {}73 for key in ("Capabilities", "Consumption options"):74 if key in rows:75 for m in re.finditer(r"\*\*(.+?)\*\*\s*([^*]+?)(?=\s*\*\*|$)", rows[key]):76 caps[m.group(1).strip()] = m.group(2).strip()77 lim = rows.get("Token limits") or rows.get("Limits") or ""78 mi = re.search(r"Input (?:token limit|context window)\*\*\s*([\d,]+)", lim)79 mo = re.search(r"Output token limit\*\*\s*([\d,]+)", lim)80 out[head] = {81 "page": slug, "url": D + "models/" + slug,82 "code": rows.get("Model code") or rows.get("Agent code"),83 "data_types": rows.get("Supported data types"), "limits_raw": lim,84 "input_limit": int(mi.group(1).replace(",", "")) if mi else None,85 "output_limit": int(mo.group(1).replace(",", "")) if mo else None,86 "caps": caps, "versions": rows.get("Versions"),87 "latest_update": rows.get("Latest update"), "knowledge_cutoff": rows.get("Knowledge cutoff"),88 "model_card": rows.get("Model card"), "warning": re.sub(r"\]\(https?://[^)]+\)", "]", warn.group(1)) if warn else None,89 }90 return out919293# --------------------------------------------------------------------------------------------------94# Hand-encoded documentation tables95# --------------------------------------------------------------------------------------------------96# deprecations.md (release / earliest shutdown / replacement). None = "No shutdown date announced".97DEPRECATIONS: dict[str, tuple[str | None, str | None, str | None]] = {98 "gemini-3.8-live": ("2026-09-15", None, None), "gemini-3.8-live-extended-thinking": ("2026-09-15", None, None),99 "gemini-3.8-flash": ("2026-09-02", None, None), "gemini-3.7-flash": ("2026-08-13", None, None),100 "gemini-3.6-flash": ("2026-07-21", None, None), "gemini-3.5-flash-lite": ("2026-07-21", None, None),101 "gemini-3.5-flash": ("2026-05-19", None, None), "gemini-3.1-flash-image": ("2026-05-28", None, None),102 "gemini-3-pro-image": ("2026-05-28", None, None), "gemini-3.1-flash-lite": ("2026-05-07", "2027-05-07", "gemini-3.5-flash-lite"),103 "gemini-3.1-flash-image-preview": ("2026-02-26", "2026-06-25", "gemini-3.1-flash-image"),104 "gemini-3.1-pro-preview": ("2026-02-19", None, None), "gemini-3-pro-image-preview": ("2025-11-20", "2026-06-25", "gemini-3-pro-image"),105 "gemini-3-flash-preview": ("2025-12-17", None, "gemini-3.6-flash"), "gemini-3-pro-preview": ("2025-11-18", "2026-03-09", "gemini-3.1-pro-preview"),106 "gemini-3.1-flash-lite-preview": ("2026-03-03", "2026-05-25", "gemini-3.1-flash-lite"),107 "gemini-2.5-pro": ("2025-06-17", None, None),108 "gemini-2.5-pro-preview-03-25": ("2025-03-03", "2025-12-02", "gemini-3.1-pro-preview"),109 "gemini-2.5-pro-preview-05-06": ("2025-05-06", "2025-12-02", "gemini-3.1-pro-preview"),110 "gemini-2.5-pro-preview-06-05": ("2025-06-05", "2025-12-02", "gemini-3.1-pro-preview"),111 "gemini-2.5-flash": ("2025-06-17", None, None), "gemini-2.5-flash-image": ("2025-10-02", "2026-10-02", "gemini-3.1-flash-image-preview"),112 "gemini-2.5-flash-lite": ("2025-07-22", None, None),113 "gemini-2.5-flash-lite-preview-09-2025": ("2025-09-25", "2026-03-31", "gemini-3.1-flash-lite"),114 "gemini-2.5-flash-preview-05-20": ("2025-05-20", "2025-11-18", "gemini-3.6-flash"),115 "gemini-2.5-flash-image-preview": ("2025-05-07", "2026-01-15", "gemini-2.5-flash-image"),116 "gemini-2.5-flash-preview-09-25": ("2025-09-25", "2026-02-17", "gemini-3.6-flash"),117 "gemini-2.0-flash": ("2025-02-05", "2026-06-01", "gemini-3.6-flash"), "gemini-2.0-flash-001": ("2025-02-05", "2026-06-01", "gemini-3.6-flash"),118 "gemini-2.0-flash-lite": ("2025-02-25", "2026-06-01", "gemini-3.1-flash-lite"), "gemini-2.0-flash-lite-001": ("2025-02-25", "2026-06-01", "gemini-3.1-flash-lite"),119 "gemini-2.0-flash-preview-image-generation": ("2025-05-07", "2025-11-14", "gemini-2.5-flash-image"),120 "gemini-2.0-flash-lite-preview": ("2025-02-05", "2025-12-09", "gemini-2.5-flash-lite"),121 "gemini-2.0-flash-lite-preview-02-05": ("2025-02-05", "2025-12-09", "gemini-2.5-flash-lite"),122 "gemini-3.5-transcribe-live": ("2026-08", None, None), "gemini-2.0-flash-live-001": ("2025-04-09", "2025-12-09", "gemini-3.8-live"),123 "gemini-3.5-live-translate-preview": ("2026-06", None, None), "gemini-3.1-flash-live-preview": ("2026-03-11", None, "gemini-3.8-live"),124 "gemini-2.5-flash-native-audio-preview-12-2025": ("2025-12-12", None, "gemini-3.8-live"),125 "gemini-live-2.5-flash-preview": ("2025-06-17", "2025-12-09", "gemini-3.8-live"),126 "gemini-3.5-transcribe": ("2026-08", None, None), "gemini-3.1-flash-tts-preview": ("2026-04-13", None, None),127 "gemini-2.5-flash-preview-tts": ("2025-05-20", None, "gemini-3.1-flash-tts-preview"), "gemini-2.5-pro-preview-tts": ("2025-05-20", None, "gemini-3.1-flash-tts-preview"),128 "gemini-embedding-2": ("2026-04-22", None, None), "gemini-embedding-001": ("2025-07-14", "2028-05-14", "gemini-embedding-2"),129 "text-embedding-004": ("2024-04-09", "2026-01-14", "gemini-embedding-2"), "embedding-2-preview": ("2026-03-10", "2026-08-10", "gemini-embedding-2"),130 "embedding-001": ("2024-04-09", "2025-10-30", "gemini-embedding-2"), "embedding-gecko-001": (None, "2025-10-30", "gemini-embedding-2"),131 "gemini-embedding-exp": (None, "2025-10-30", "gemini-embedding-2"), "gemini-embedding-exp-03-07": (None, "2025-10-30", "gemini-embedding-2"),132 "imagen-4.0-generate-001": ("2025-06-24", "2026-08-17", "gemini-3.1-flash-image"), "imagen-4.0-ultra-generate-001": ("2025-06-24", "2026-08-17", "gemini-3.1-flash-image"),133 "imagen-4.0-fast-generate-001": ("2025-06-24", "2026-08-17", "gemini-3.1-flash-image"), "imagen-3.0-generate-002": ("2025-02-06", "2025-11-10", "imagen-4.0-generate-001"),134 "imagen-4.0-generate-preview-06-06": ("2025-06-24", "2026-02-17", "imagen-4.0-generate-001"), "imagen-4.0-ultra-generate-preview-06-06": ("2025-06-24", "2026-02-17", "imagen-4.0-ultra-generate-001"),135 "veo-3.0-generate-001": ("2025-09-09", "2026-06-30", "veo-3.1-generate-preview"), "veo-3.0-fast-generate-001": ("2025-09-09", "2026-06-30", "veo-3.1-fast-generate-preview"),136 "veo-2.0-generate-001": ("2025-04-09", "2026-06-30", "veo-3.1-generate-preview"), "veo-3.1-lite-generate-preview": ("2026-03-31", None, None),137 "veo-3.1-generate-preview": ("2025-10-15", None, None), "veo-3.1-fast-generate-preview": ("2025-10-15", None, None),138 "veo-3.0-generate-preview": ("2025-07-31", "2025-11-12", "veo-3.1-generate-preview"), "veo-3.0-fast-generate-preview": ("2025-07-31", "2025-11-12", "veo-3.1-fast-generate-preview"),139 "gemini-omni-1.1-flash": ("2026-08-27", None, None), "gemini-omni-flash-preview": ("2026-06-30", "2026-09-30", "gemini-omni-1.1-flash"),140 "lyria-3.5": ("2026-09-03", None, None), "lyria-3-clip-preview": ("2026-03-25", None, None), "lyria-3-pro-preview": ("2026-03-25", None, "lyria-3.5"),141 "lyria-realtime-exp": ("2025-05-20", None, None),142 "gemini-robotics-er-1.6-preview": ("2026-04-14", "2026-08-31", "gemini-robotics-er-2-preview"), "gemini-robotics-er-1.5-preview": ("2025-09-25", "2026-04-30", "gemini-robotics-er-1.6-preview"),143 "antigravity-preview-09-2026": ("2026-09-17", None, None), "antigravity-preview-05-2026": ("2026-05-19", "2026-10-05", "antigravity-preview-09-2026"),144 # changelog-only release dates145 "gemini-robotics-er-2-preview": ("2026-07-30", None, None), "gemini-robotics-er-2-streaming-preview": ("2026-07-30", None, None),146 "gemma-4-26b-a4b-it": ("2026-04-02", None, None), "gemma-4-31b-it": ("2026-04-02", None, None),147 "gemini-embedding-2-preview": ("2026-03-10", None, "gemini-embedding-2"), "deep-research-preview-04-2026": ("2026-04-21", None, None),148 "deep-research-max-preview-04-2026": ("2026-04-21", None, None), "deep-research-pro-preview-12-2025": ("2025-12-12", None, "deep-research-preview-04-2026"),149 "gemini-3.1-flash-lite-image": ("2026-06-30", None, None), "gemini-2.5-computer-use-preview-10-2025": ("2025-10-07", None, "gemini-3.8-flash (built-in computer use)"),150 "gemini-3.1-pro-preview-customtools": ("2026-02-19", None, None), "nano-banana-pro-preview": ("2025-11-20", "2026-06-25", "gemini-3-pro-image"),151}152SHUTDOWN_PAST_CUTOFF = "2026-09-18"153154# thinking.md "Controlling thinking" table: default level / supported levels155THINKING: dict[str, tuple[str, list[str]]] = {156 "gemini-3.8-flash": ("medium", ["low", "medium", "high"]), "gemini-3.7-flash": ("medium", ["low", "medium", "high"]),157 "gemini-3.6-flash": ("medium", ["minimal", "low", "medium", "high"]), "gemini-3.5-flash-lite": ("minimal", ["minimal", "low", "medium", "high"]),158 "gemini-3.1-pro-preview": ("high", ["low", "medium", "high"]), "gemini-3.1-pro-preview-customtools": ("high", ["low", "medium", "high"]),159 "gemini-3.1-flash-lite-image": ("minimal", ["minimal", "high"]), "gemini-3-flash-preview": ("high", ["minimal", "low", "medium", "high"]),160 "gemini-3-pro-preview": ("high", ["low", "high"]), "gemini-3.5-flash": ("medium", ["minimal", "low", "medium", "high"]),161 "gemini-2.5-pro": ("on (dynamic budget)", ["low", "medium", "high"]), "gemini-2.5-flash": ("on (dynamic budget)", ["low", "medium", "high"]),162 "gemini-2.5-flash-lite": ("off", ["low", "medium", "high"]),163 # openai.md reasoning_effort mapping: 3.1 Flash-Lite supports minimal..high164 "gemini-3.1-flash-lite": ("unknown (docs table omits it; openai.md maps minimal/low/medium/high)", ["minimal", "low", "medium", "high"]),165}166167# rate-limits.md "Batch enqueued tokens" (Tier 1 / Tier 2 / Tier 3), keyed by pretty name -> ids168BATCH_ENQUEUED: dict[str, tuple[int, int, int]] = {169 "gemini-3.1-pro-preview": (5_000_000, 500_000_000, 1_000_000_000), "gemini-3.1-pro-preview-customtools": (5_000_000, 500_000_000, 1_000_000_000),170 "gemini-3.5-flash-lite": (10_000_000, 500_000_000, 1_000_000_000), "gemini-3.8-flash": (3_000_000, 400_000_000, 1_000_000_000),171 "gemini-3.7-flash": (3_000_000, 400_000_000, 1_000_000_000), "gemini-3.1-flash-lite": (10_000_000, 500_000_000, 1_000_000_000),172 "gemini-3.1-flash-lite-preview": (10_000_000, 500_000_000, 1_000_000_000), "gemini-3.6-flash": (3_000_000, 400_000_000, 1_000_000_000),173 "gemini-3.5-flash": (3_000_000, 400_000_000, 1_000_000_000), "gemini-2.5-pro": (5_000_000, 500_000_000, 1_000_000_000),174 "gemini-2.5-pro-preview-tts": (25_000, 100_000, 1_000_000), "gemini-2.5-flash": (3_000_000, 400_000_000, 1_000_000_000),175 "gemini-2.5-flash-image": (3_000_000, 400_000_000, 1_000_000_000), "gemini-2.5-flash-preview-tts": (100_000, 100_000, 4_000_000),176 "gemini-2.5-flash-lite": (10_000_000, 500_000_000, 1_000_000_000), "gemini-2.0-flash": (10_000_000, 1_000_000_000, 5_000_000_000),177 "gemini-2.0-flash-lite": (10_000_000, 1_000_000_000, 5_000_000_000), "gemini-3.1-flash-image-preview": (1_000_000, 250_000_000, 750_000_000),178 "gemini-3.1-flash-image": (1_000_000, 250_000_000, 750_000_000), "gemini-3.1-flash-lite-image": (2_000_000, 270_000_000, 1_000_000_000),179 "gemini-3-pro-image-preview": (2_000_000, 270_000_000, 1_000_000_000), "gemini-3-pro-image": (2_000_000, 270_000_000, 1_000_000_000),180 "nano-banana-pro-preview": (2_000_000, 270_000_000, 1_000_000_000),181 "gemini-embedding-001": (500_000, 5_000_000, 10_000_000), "gemini-embedding-2": (500_000, 5_000_000, 10_000_000), "gemini-embedding-2-preview": (500_000, 5_000_000, 10_000_000),182}183184INTERACTIONS_SUPPORTED = {"gemini-3.5-flash", "gemini-3.1-flash-lite", "gemini-3.1-flash-lite-preview", "gemini-3.1-pro-preview",185 "gemini-3-flash-preview", "gemini-2.5-pro", "gemini-2.5-flash", "gemini-2.5-flash-lite", "lyria-3-clip-preview",186 "lyria-3-pro-preview", "deep-research-pro-preview-12-2025", "deep-research-preview-04-2026", "deep-research-max-preview-04-2026"}187# Newer GA models are used with the Interactions API throughout the docs (latest-model.md, api-key.md, omni.md, agents.md) even though188# interactions.md's table has not been refreshed.189INTERACTIONS_DOCS_EXAMPLES = {"gemini-3.8-flash", "gemini-3.7-flash", "gemini-3.6-flash", "gemini-3.5-flash-lite", "gemini-omni-1.1-flash",190 "gemini-omni-flash-preview", "antigravity-preview-05-2026", "antigravity-preview-09-2026", "gemini-3.1-flash-tts-preview",191 "gemini-3-pro-image", "gemini-3.1-flash-image", "gemini-3.1-flash-lite-image"}192FLEX_SUPPORTED = {"gemini-3.8-flash", "gemini-3.7-flash", "gemini-3.6-flash", "gemini-3.5-flash-lite", "gemini-3.5-flash", "gemini-3.1-flash-lite",193 "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-2.5-pro", "gemini-2.5-flash"}194COMPUTER_USE_MODELS = {"gemini-3.8-flash", "gemini-3.7-flash", "gemini-3.5-flash-lite", "gemini-3.5-flash", "gemini-3-flash-preview",195 "gemini-2.5-computer-use-preview-10-2025"}196LATEST_ALIASES = {197 "gemini-flash-latest": {"resolves_to_live": "gemini-3.8-flash", "evidence": "generateContent modelVersion=gemini-3.8-flash (2026-09-19)",198 "history": ["2026-01-21 -> gemini-3-flash-preview", "2026-05-19 -> gemini-3.5-flash", "observed 2026-09-19 -> gemini-3.8-flash (not announced in changelog)"]},199 "gemini-pro-latest": {"resolves_to_live": "gemini-3.1-pro (quota dimension model=gemini-3.1-pro in the 429 body)", "evidence": "429 RESOURCE_EXHAUSTED free-tier limit 0, quotaDimensions.model=gemini-3.1-pro (2026-09-19)",200 "history": ["2026-01-21 -> gemini-3-pro-preview", "2026-03-09 gemini-3-pro-preview shut down -> points to gemini-3.1-pro-preview"]},201 "gemini-flash-lite-latest": {"resolves_to_live": "unknown (not probed)", "evidence": "GET 200; description 'Latest release of Gemini Flash-Lite'", "history": []},202 "gemini-2.5-flash-native-audio-latest": {"resolves_to_live": "unknown (Live API only, not probed)", "evidence": "GET /v1beta/models listing", "history": []},203}204205# ---- pricing (pricing.md, USD). Text-model families: (tier -> dims). "storage" = context-cache storage per 1M tokens per hour.206INTRO = "Introductory price through 2026-12-31; from 2027-01-01: {}"207TEXT_PRICING: dict[str, dict] = {208 "gemini-3.8-flash": {"standard": {"input": 0.75, "output": 3.75, "cached_input": 0.075, "cache_storage_hour": 0.50},209 "batch": {"input": 0.375, "output": 1.875, "cached_input": 0.0375, "cache_storage_hour": 0.50},210 "flex": {"input": 0.375, "output": 1.875, "cached_input": 0.0375, "cache_storage_hour": 0.50},211 "priority": {"input": 1.35, "output": 6.75, "cached_input": 0.135, "cache_storage_hour": 0.50},212 "_2027": {"standard": (1.50, 7.50, 0.15, 1.00), "batch": (0.75, 3.75, 0.075, 1.00), "flex": (0.75, 3.75, 0.075, 1.00), "priority": (2.70, 13.50, 0.27, 1.00)},213 "free_tier": "input/output/caching free of charge (standard & priority rows); Batch/Flex not available", "grounding": "gemini3"},214 "gemini-3.5-flash": {"standard": {"input": 1.50, "output": 9.00, "cached_input": 0.15, "cache_storage_hour": 1.00},215 "batch": {"input": 0.75, "output": 4.50, "cached_input": 0.075, "cache_storage_hour": 1.00},216 "flex": {"input": 0.75, "output": 4.50, "cached_input": 0.08, "cache_storage_hour": 1.00},217 "priority": {"input": 2.70, "output": 16.20, "cached_input": 0.27, "cache_storage_hour": 1.00},218 "free_tier": "input/output/caching free of charge; Batch/Flex not available", "grounding": "gemini3"},219 "gemini-3.5-flash-lite": {"standard": {"input": 0.30, "output": 2.50, "cached_input": 0.03, "cache_storage_hour": 1.00},220 "batch": {"input": 0.15, "output": 1.25, "cached_input": 0.02, "cache_storage_hour": 1.00},221 "flex": {"input": 0.15, "output": 1.25, "cached_input": 0.02, "cache_storage_hour": 1.00},222 "priority": {"input": 0.54, "output": 4.50, "cached_input": 0.05, "cache_storage_hour": 1.00},223 "free_tier": "input/output free of charge on all four rows (page lists Batch/Flex as 'Free of charge' too); caching not available on free tier", "grounding": "gemini3",224 "input_note": "text / image / video / audio same price"},225 "gemini-3.1-flash-lite": {"standard": {"input": 0.25, "audio_input": 0.50, "output": 1.50, "cached_input": 0.025, "cached_audio_input": 0.05, "cache_storage_hour": 1.00},226 "batch": {"input": 0.125, "audio_input": 0.25, "output": 0.75, "cached_input": 0.0125, "cached_audio_input": 0.025, "cache_storage_hour": 0.50},227 "flex": {"input": 0.125, "audio_input": 0.25, "output": 0.75, "cached_input": 0.0125, "cached_audio_input": 0.025, "cache_storage_hour": 0.50},228 "priority": {"input": 0.45, "audio_input": 0.90, "output": 2.70, "cached_input": 0.045, "cached_audio_input": 0.09, "cache_storage_hour": 1.80},229 "free_tier": "input/output free of charge; caching not available on free tier", "grounding": "gemini3"},230 "gemini-3-flash-preview": {"standard": {"input": 0.50, "audio_input": 1.00, "output": 3.00, "cached_input": 0.05, "cached_audio_input": 0.10, "cache_storage_hour": 1.00},231 "batch": {"input": 0.25, "audio_input": 0.50, "output": 1.50, "cached_input": 0.05, "cached_audio_input": 0.10, "cache_storage_hour": 1.00},232 "flex": {"input": 0.25, "audio_input": 0.50, "output": 1.50, "cached_input": 0.05, "cached_audio_input": 0.10, "cache_storage_hour": 1.00},233 "priority": {"input": 0.90, "audio_input": 1.80, "output": 5.40, "cached_input": 0.09, "cached_audio_input": 0.18, "cache_storage_hour": 1.80},234 "free_tier": "input/output/caching free of charge; Batch/Flex not available", "grounding": "gemini3"},235 "gemini-3.1-pro-preview": {"standard": {"input": 2.00, "input_over_200k": 4.00, "output": 12.00, "output_over_200k": 18.00, "cached_input": 0.20, "cached_input_over_200k": 0.40, "cache_storage_hour": 4.50},236 "batch": {"input": 1.00, "input_over_200k": 2.00, "output": 6.00, "output_over_200k": 9.00, "cached_input": 0.20, "cached_input_over_200k": 0.40, "cache_storage_hour": 4.50},237 "flex": {"input": 1.00, "input_over_200k": 2.00, "output": 6.00, "output_over_200k": 9.00, "cached_input": 0.20, "cached_input_over_200k": 0.40, "cache_storage_hour": 4.50},238 "priority": {"input": 3.60, "input_over_200k": 7.20, "output": 21.60, "output_over_200k": 32.40, "cached_input": 0.36, "cached_input_over_200k": 0.72, "cache_storage_hour": 8.10},239 "free_tier": "Not available (paid tier only)", "grounding": "gemini3", "long_context_threshold": 200_000},240 "gemini-2.5-pro": {"standard": {"input": 1.25, "input_over_200k": 2.50, "output": 10.00, "output_over_200k": 15.00, "cached_input": 0.125, "cached_input_over_200k": 0.25, "cache_storage_hour": 4.50},241 "batch": {"input": 0.625, "input_over_200k": 1.25, "output": 5.00, "output_over_200k": 7.50, "cached_input": 0.125, "cached_input_over_200k": 0.25, "cache_storage_hour": 4.50},242 "flex": {"input": 0.625, "input_over_200k": 1.25, "output": 5.00, "output_over_200k": 7.50, "cached_input": 0.125, "cached_input_over_200k": 0.25, "cache_storage_hour": 4.50},243 "priority": {"input": 2.25, "input_over_200k": 4.50, "output": 18.00, "output_over_200k": 27.00, "cached_input": 0.225, "cached_input_over_200k": 0.45, "cache_storage_hour": 8.10},244 "free_tier": "input/output free of charge (standard/priority rows); caching, Batch, Flex not available", "grounding": "gemini25_pro", "long_context_threshold": 200_000},245 "gemini-2.5-flash": {"standard": {"input": 0.30, "audio_input": 1.00, "output": 2.50, "cached_input": 0.03, "cached_audio_input": 0.10, "cache_storage_hour": 1.00},246 "batch": {"input": 0.15, "audio_input": 0.50, "output": 1.25, "cached_input": 0.03, "cached_audio_input": 0.10, "cache_storage_hour": 1.00},247 "flex": {"input": 0.15, "audio_input": 0.50, "output": 1.25, "cached_input": 0.03, "cached_audio_input": 0.10, "cache_storage_hour": 1.00},248 "priority": {"input": 0.54, "audio_input": 1.80, "output": 4.50, "cached_input": 0.054, "cached_audio_input": 0.18, "cache_storage_hour": 1.80},249 "free_tier": "input/output free of charge; caching not available; Google Search grounding free up to 500 RPD (shared with Flash-Lite)", "grounding": "gemini25_flash"},250 "gemini-2.5-flash-lite": {"standard": {"input": 0.10, "audio_input": 0.30, "output": 0.40, "cached_input": 0.01, "cached_audio_input": 0.03, "cache_storage_hour": 1.00},251 "batch": {"input": 0.05, "audio_input": 0.15, "output": 0.20, "cached_input": 0.01, "cached_audio_input": 0.03, "cache_storage_hour": 1.00},252 "flex": {"input": 0.05, "audio_input": 0.15, "output": 0.20, "cached_input": 0.01, "cached_audio_input": 0.03, "cache_storage_hour": 1.00},253 "priority": {"input": 0.18, "audio_input": 0.54, "output": 0.72, "cached_input": 0.018, "cached_audio_input": 0.054, "cache_storage_hour": 1.80},254 "free_tier": "input/output free of charge; caching not available; Google Search grounding free up to 500 RPD (shared with Flash)", "grounding": "gemini25_flash"},255 "gemini-robotics-er-2-preview": {"standard": {"input": 1.00, "output": 5.00, "cached_input": 0.10, "cache_storage_hour": 0.50},256 "batch": {"input": 0.50, "output": 2.50, "cached_input": 0.05, "cache_storage_hour": 0.50},257 "_2027": {"standard": (2.00, 10.00, 0.20, 1.00), "batch": (1.00, 5.00, 0.10, 1.00)},258 "free_tier": "input/output free of charge; caching not available", "grounding": "gemini3", "input_note": "text / image / video / audio same price"},259 "gemini-2.5-computer-use-preview-10-2025": {"standard": {"input": 1.00, "output": 5.00}, "_2027": {"standard": (2.00, 10.00, None, None)},260 "free_tier": "input/output free of charge", "grounding": "gemini3",261 "extra_note": "pricing.md also shows an unlabeled second table ($1.25/$2.50 input, $10/$15 output, <=200k / >200k) under this heading; likely stale legacy rates"},262}263TEXT_PRICING["gemini-3.7-flash"] = dict(TEXT_PRICING["gemini-3.8-flash"])264TEXT_PRICING["gemini-3.6-flash"] = dict(TEXT_PRICING["gemini-3.8-flash"])265TEXT_PRICING["gemini-3.1-pro-preview-customtools"] = dict(TEXT_PRICING["gemini-3.1-pro-preview"]) | {"same_as": "gemini-3.1-pro-preview (pricing.md: 'Custom Tools endpoint ... Same as Gemini 3.1 Pro Preview pricing')"}266267GROUNDING = {268 "gemini3": {"google_search": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests; billed per search query executed (a prompt may trigger several)",269 "google_maps": "5,000 prompts per month free (shared across Gemini 3), then $14 per 1,000 search queries"},270 "gemini25_pro": {"google_search": "1,500 RPD free, then $35 per 1,000 grounded prompts", "google_maps": "10,000 RPD free, then $25 per 1,000 grounded prompts"},271 "gemini25_flash": {"google_search": "1,500 RPD free (limit shared Flash/Flash-Lite), then $35 per 1,000 grounded prompts; free tier: 500 RPD", "google_maps": "1,500 RPD free, then $25 per 1,000 grounded prompts; free tier 500 RPD"},272}273274# Non-text pricing summaries attached to model records (details emitted as price records in build_pricing()).275MEDIA_PRICING: dict[str, dict] = {276 "gemini-3.8-live": {"input_text": 0.75, "input_audio": 3.00, "input_audio_per_min": 0.005, "input_image_video": 1.00, "input_image_video_per_min": 0.002, "output_text": 4.50, "output_audio": 12.00, "output_audio_per_min": 0.018, "unit": "per 1M tokens", "free_tier": "free of charge", "grounding": "gemini3 (Google Search supported on free tier for these models)"},277 "gemini-3.5-live-translate-preview": {"input_audio": 3.50, "input_audio_per_min": 0.0053, "output_audio": 21.00, "output_audio_per_min": 0.0315, "unit": "per 1M tokens", "note": "25 audio tokens/second; effective ~$0.0368 per minute", "free_tier": "free of charge"},278 "gemini-3.5-transcribe-live": {"input_audio": 3.50, "input_audio_per_min": 0.005, "output_text": 21.00, "output_text_per_min": 0.004, "unit": "per 1M tokens", "note": "25 audio tokens/s input, 175 text tokens/min output; blended ~$0.009/min", "free_tier": "free of charge"},279 "gemini-3.5-transcribe": {"input_audio": 2.00, "input_audio_per_min": 0.003, "output_text": 12.00, "output_text_per_min": 0.002, "unit": "per 1M tokens", "note": "blended ~$0.005/min", "free_tier": "free of charge"},280 "gemini-omni-1.1-flash": {"input": 1.50, "output_text": 9.00, "output_video": 17.50, "unit": "per 1M tokens", "note": "5,792 tokens per second of 720p video => ~$0.10 per second", "free_tier": "not available"},281 "gemini-3.1-flash-image": {"input": 0.50, "output_text": 3.00, "output_image": 60.00, "per_image": {"0.5K (747 tok)": 0.045, "1K (1120 tok)": 0.067, "2K (1680 tok)": 0.101, "4K (2520 tok)": 0.151}, "batch": {"input": 0.25, "output_text": 1.50, "output_image": 30.00}, "unit": "per 1M tokens", "free_tier": "not available", "grounding": "5,000 free search requests/month shared, then $14/1k (web + image search grounding)"},282 "gemini-3.1-flash-lite-image": {"input": 0.25, "output_text": 1.50, "output_image": 30.00, "per_image": {"1K (1120 tok)": 0.0336}, "batch": {"input": 0.125, "output_text": 0.75, "output_image": 15.00}, "unit": "per 1M tokens", "free_tier": "not available"},283 "gemini-3-pro-image": {"input": 2.00, "input_per_image": 0.0011, "output_text": 12.00, "output_image": 120.00, "per_image": {"1K/2K (1120 tok)": 0.134, "4K (2000 tok)": 0.24}, "batch": {"input_text": 1.00, "input_per_image": 0.0006, "output_text": 6.00, "per_image_1k_2k": 0.067, "per_image_4k": 0.12}, "flex": "same as batch", "priority": {"input": 3.60, "output_text": 21.60, "output_image": 216.00}, "unit": "per 1M tokens", "free_tier": "not available", "grounding": "gemini3 (Google Search)"},284 "gemini-2.5-flash-image": {"input": 0.30, "output_per_image": 0.039, "output_image": 30.00, "note": "1290 tokens per image up to 1024x1024", "batch": {"input": 0.15, "output_per_image": 0.0195}, "flex": "same as batch", "priority": {"input": 0.54, "output_per_image": 0.0702}, "unit": "per 1M tokens", "free_tier": "not available"},285 "gemini-3.1-flash-tts-preview": {"input_text": 1.00, "output_audio": 20.00, "batch": {"input_text": 0.50, "output_audio": 10.00}, "unit": "per 1M tokens", "note": "25 audio tokens per second", "free_tier": "free of charge (standard); batch not available"},286 "gemini-2.5-flash-preview-tts": {"input_text": 0.50, "output_audio": 10.00, "batch": {"input_text": 0.25, "output_audio": 5.00}, "unit": "per 1M tokens", "free_tier": "free of charge (standard)"},287 "gemini-2.5-pro-preview-tts": {"input_text": 1.00, "output_audio": 20.00, "batch": {"input_text": 0.50, "output_audio": 10.00}, "unit": "per 1M tokens", "free_tier": "not available"},288 "gemini-2.5-flash-native-audio-preview-12-2025": {"input_text": 0.50, "input_audio_video": 3.00, "output_text": 2.00, "output_audio": 12.00, "unit": "per 1M tokens", "free_tier": "free of charge"},289 "veo-3.1-generate-preview": {"per_second": {"720p": 0.40, "1080p": 0.40, "4k": 0.60}, "unit": "per second of generated video (with audio)", "free_tier": "not available", "note": "charged only if the video is successfully generated"},290 "veo-3.1-fast-generate-preview": {"per_second": {"720p": 0.10, "1080p": 0.12, "4k": 0.30}, "unit": "per second", "free_tier": "not available"},291 "veo-3.1-lite-generate-preview": {"per_second": {"720p": 0.05, "1080p": 0.08, "4k": "not supported"}, "unit": "per second", "free_tier": "not available"},292 "lyria-3.5": {"per_request": 0.08, "unit": "per song (full length)", "free_tier": "not available"},293 "lyria-3-clip-preview": {"per_request": 0.04, "unit": "per song (30 s clip)", "free_tier": "not available"},294 "lyria-3-pro-preview": {"per_request": 0.08, "unit": "per song (full length)", "free_tier": "not available"},295 "gemini-embedding-2": {"input_text": 0.20, "input_image": 0.45, "input_image_per_image": 0.00012, "input_audio": 6.50, "input_audio_per_second": 0.00016, "input_video": 12.00, "input_video_per_frame": 0.00079, "batch": {"input_text": 0.10, "input_image": 0.225, "input_audio": 3.25, "input_video": 6.00}, "unit": "per 1M tokens", "free_tier": "free of charge (standard); batch not available"},296 "gemma-4": {"free_tier": "free of charge (input, output, caching, storage)", "paid_tier": "not available", "tuning": "not available", "grounding": "not available"},297}298for a, b in (("gemini-3.8-live-extended-thinking", "gemini-3.8-live"), ("gemini-3.1-flash-live-preview", "gemini-3.8-live"), ("gemini-omni-flash-preview", "gemini-omni-1.1-flash"),299 ("gemini-3-pro-image-preview", "gemini-3-pro-image"), ("nano-banana-pro-preview", "gemini-3-pro-image"), ("gemini-3.1-flash-image-preview", "gemini-3.1-flash-image"),300 ("gemini-embedding-2-preview", "gemini-embedding-2"), ("gemma-4-26b-a4b-it", "gemma-4"), ("gemma-4-31b-it", "gemma-4")):301 MEDIA_PRICING[a] = dict(MEDIA_PRICING[b]) | {"same_as": b}302303# Documented ids that are NOT in the live listing (retired / never listed / page-only) -> minimal records.304NON_LIVE_DOCUMENTED = [305 "gemini-2.0-flash", "gemini-2.0-flash-001", "gemini-2.0-flash-lite", "gemini-2.0-flash-lite-001", "gemini-2.0-flash-exp",306 "gemini-2.0-flash-preview-image-generation", "gemini-2.0-flash-lite-preview", "gemini-2.0-flash-lite-preview-02-05", "gemini-2.0-flash-live-001",307 "gemini-2.5-pro-preview-03-25", "gemini-2.5-pro-preview-05-06", "gemini-2.5-pro-preview-06-05", "gemini-2.5-flash-preview-05-20",308 "gemini-2.5-flash-preview-09-2025", "gemini-2.5-flash-preview-09-25", "gemini-2.5-flash-lite-preview-09-2025", "gemini-2.5-flash-image-preview",309 "gemini-live-2.5-flash-preview", "text-embedding-004", "embedding-001", "embedding-gecko-001", "gemini-embedding-exp", "gemini-embedding-exp-03-07", "embedding-2-preview",310 "imagen-4.0-generate-001", "imagen-4.0-ultra-generate-001", "imagen-4.0-fast-generate-001", "imagen-3.0-generate-002",311 "imagen-4.0-generate-preview-06-06", "imagen-4.0-ultra-generate-preview-06-06",312 "veo-2.0-generate-001", "veo-3.0-generate-001", "veo-3.0-fast-generate-001", "veo-3.0-generate-preview", "veo-3.0-fast-generate-preview",313 "gemini-robotics-er-1.5-preview", "gemini-robotics-er-1.6-preview", "lyria-3.5-clip-preview", "lyria-3.5-pro-preview",314]315316317# --------------------------------------------------------------------------------------------------318def family_of(i: str) -> tuple[str, str | None]:319 if i.startswith("gemma"):320 return "Gemma 4", "4"321 if i.startswith("lyria"):322 return "Lyria", re.sub(r"^lyria-([\d.]+|realtime).*", r"\1", i)323 if i.startswith("veo"):324 return "Veo", re.sub(r"^veo-([\d.]+).*", r"\1", i)325 if i.startswith("imagen"):326 return "Imagen", re.sub(r"^imagen-([\d.]+).*", r"\1", i)327 if i.startswith("deep-research"):328 return "Deep Research (agent)", None329 if i.startswith("antigravity"):330 return "Antigravity (managed agent)", None331 if i == "aqa":332 return "AQA (Attributed Question Answering)", None333 if i.startswith("nano-banana"):334 return "Gemini 3 (image)", "3"335 if "embedding" in i:336 return "Gemini Embedding", re.sub(r"^gemini-embedding-([\d]+).*", r"\1", i) if i.startswith("gemini-embedding-") else None337 if i.startswith("gemini-omni"):338 return "Gemini Omni", "1.1" if "1.1" in i else "preview"339 if i.startswith("gemini-robotics"):340 return "Gemini Robotics-ER", re.sub(r"^gemini-robotics-er-([\d.]+).*", r"\1", i)341 if i.endswith("-latest"):342 return "Gemini (latest alias)", None343 m = re.match(r"gemini-(\d(?:\.\d)?)", i)344 gen = m.group(1) if m else None345 return f"Gemini {gen}" if gen else "Gemini", gen346347348def kind_of(i: str, live: dict | None) -> str:349 if i.endswith("-latest"):350 return "alias"351 if i.endswith("-exp") or "-exp-" in i:352 return "experimental"353 if i.startswith(("deep-research", "antigravity")):354 return "agent"355 if "preview" in i:356 return "preview"357 return "stable"358359360def modalities_for(i: str, page: dict | None, live: dict | None) -> dict:361 inp, out = set(), set()362 dt = (page or {}).get("data_types") or ""363 low = dt.lower()364 if "**input" in low:365 a, b = re.split(r"\*\*output\*\*", dt, flags=re.I) if re.search(r"\*\*output\*\*", dt, re.I) else (dt, "")366 al, bl = a.lower(), b.lower()367 for k, v in (("text", "text"), ("image", "image"), ("audio", "audio"), ("video", "video"), ("pdf", "pdf")):368 if k in al:369 inp.add(v)370 if k in bl:371 out.add(v)372 if "embedding" in bl:373 out = {"embedding"}374 if "lyrics" in bl or "mp3" in bl:375 out = {"audio (music)", "text (lyrics)"}376 if "video with audio" in bl:377 out = {"video (with native audio)"}378 methods = set((live or {}).get("supportedGenerationMethods", []))379 if not inp:380 inp.add("text")381 if "embedContent" in methods:382 out = {"embedding"}383 if "predictLongRunning" in methods and not out:384 out = {"video"}385 if "bidiGenerateMusic" in methods:386 out = {"audio (music, PCM stream)"}387 if not out:388 out.add("text")389 order = ["text", "image", "audio", "video", "pdf", "embedding", "video (with native audio)", "audio (music)", "text (lyrics)", "audio (music, PCM stream)"]390 return {"input": sorted(inp, key=lambda x: order.index(x) if x in order else 99), "output": sorted(out, key=lambda x: order.index(x) if x in order else 99)}391392393def tri(v: str | None):394 if v is None:395 return "unknown"396 l = v.lower()397 if l.startswith("not supported") or l.startswith("not supported"):398 return False399 if l.startswith("supported") and "(" in l:400 return v # e.g. "Supported (Preview)" / "Supported (Async only)"401 if l.startswith("supported"):402 return True403 return v404405406def build_capabilities(i: str, page: dict | None, live: dict | None, mods: dict) -> dict:407 caps = (page or {}).get("caps", {})408 methods = set((live or {}).get("supportedGenerationMethods", []))409 g = lambda *ks: next((caps[k] for k in ks if k in caps), None) # noqa: E731410 c: dict = {411 "text_input": "text" in mods["input"], "image_input": "image" in mods["input"], "audio_input": "audio" in mods["input"],412 "video_input": "video" in mods["input"], "pdf_input": "pdf" in mods["input"],413 "text_output": "text" in mods["output"], "image_output": "image" in mods["output"],414 "audio_output": any(o.startswith("audio") for o in mods["output"]), "video_output": any(o.startswith("video") for o in mods["output"]),415 "music_output": any("music" in o for o in mods["output"]), "embeddings": "embedding" in mods["output"],416 }417 th = THINKING.get(i)418 live_th = (live or {}).get("thinking")419 c["thinking"] = tri(g("Thinking")) if g("Thinking") else (True if live_th else ("unknown" if live is None else False))420 c["thinking_live_flag"] = live_th if live is not None else "n/a"421 c["thinking_default_level"] = th[0] if th else "unknown"422 c["thinking_levels"] = th[1] if th else "unknown"423 c["thinking_level_param"] = bool(th) and i.startswith("gemini-3") or i.startswith("gemma")424 c["thinking_budget_legacy_param"] = i.startswith("gemini-2.5") if live is not None else "unknown"425 c["thought_signatures"] = True if i.startswith("gemini-3") and c["thinking"] else ("unknown" if c["thinking"] == "unknown" else False)426 c["structured_output"] = tri(g("Structured outputs"))427 c["function_calling"] = tri(g("Function calling"))428 c["parallel_function_calling"] = c["function_calling"] if c["function_calling"] in (True, False) else "unknown"429 c["compositional_function_calling"] = c["parallel_function_calling"]430 c["google_search_grounding"] = tri(g("Search grounding"))431 c["google_maps_grounding"] = tri(g("Grounding with Google Maps"))432 c["url_context"] = tri(g("URL context", "URL Context"))433 c["code_execution"] = tri(g("Code execution"))434 c["computer_use"] = tri(g("Computer use")) if g("Computer use") else (True if i in COMPUTER_USE_MODELS else (False if page else "unknown"))435 c["file_search"] = tri(g("File search"))436 c["context_caching_explicit"] = ("createCachedContent" in methods) if live is not None else tri(g("Caching"))437 c["context_caching_docs"] = tri(g("Caching"))438 c["context_caching_implicit"] = True if (i.startswith(("gemini-2.5", "gemini-3")) and "generateContent" in methods and c["thinking"] is not False and "image" not in i and "tts" not in i and "live" not in i) else "unknown"439 c["batch_api"] = ("batchGenerateContent" in methods or "asyncBatchEmbedContent" in methods) if live is not None else tri(g("Batch API"))440 c["batch_api_docs"] = tri(g("Batch API"))441 c["flex_inference"] = tri(g("Flex inference")) if g("Flex inference") else (True if i in FLEX_SUPPORTED else "unknown")442 c["priority_inference"] = tri(g("Priority inference")) if g("Priority inference") else "unknown"443 c["live_api"] = ("bidiGenerateContent" in methods) if live is not None else tri(g("Live API"))444 c["tts"] = "tts" in i445 c["audio_generation_docs"] = tri(g("Audio generation"))446 c["image_generation"] = tri(g("Image generation")) if g("Image generation") else c["image_output"]447 c["video_generation"] = c["video_output"]448 c["music_generation"] = c["music_output"]449 c["transcription_dedicated"] = "transcribe" in i450 c["live_translation"] = tri(g("Live translation")) if g("Live translation") else False451 c["speaker_diarization"] = tri(g("Speaker diarization")) if g("Speaker diarization") else "n/a"452 c["tuning"] = False453 c["tuning_note"] = "Fine-tuning is no longer offered on the Gemini Developer API (model-tuning.md); use Gemini Enterprise Agent Platform supervised tuning"454 c["interactions_api"] = True if (i in INTERACTIONS_SUPPORTED or i in INTERACTIONS_DOCS_EXAMPLES) else ("unknown" if "generateContent" in methods else False)455 c["interactions_api_listed_in_docs_table"] = i in INTERACTIONS_SUPPORTED456 c["deep_research_agent"] = i.startswith("deep-research")457 c["managed_agent"] = i.startswith("antigravity")458 c["openai_compatible_chat"] = ("generateContent" in methods and not i.startswith(("deep-research", "antigravity", "lyria", "gemini-omni"))) if live is not None else "unknown"459 c["openai_compatible_embeddings"] = "embedContent" in methods460 c["openai_compatible_images_generations"] = i in ("gemini-2.5-flash-image", "gemini-3-pro-image-preview", "nano-banana-pro-preview")461 c["openai_compatible_videos"] = i == "veo-3.1-generate-preview"462 c["system_instructions"] = "generateContent" in methods and not c["tts"] and not c["embeddings"]463 c["sampling_params_temperature_top_p_top_k"] = ("deprecated since 2026-07-21 for Gemini 3.x (keep defaults; live Model.maxTemperature=" + str((live or {}).get("maxTemperature")) + ")") if i.startswith("gemini-3") else ("live defaults temperature=%s topP=%s topK=%s maxTemperature=%s" % tuple((live or {}).get(k) for k in ("temperature", "topP", "topK", "maxTemperature")) if live else "unknown")464 c["batch_enqueued_tokens_tier1_tier2_tier3"] = list(BATCH_ENQUEUED[i]) if i in BATCH_ENQUEUED else "not listed"465 if "embedding" in i:466 c["embedding_dimensions"] = "flexible 128-3072 (recommended 768, 1536, 3072; default 3072, MRL truncation; embedding-2 auto-renormalizes)"467 c["embedding_task_type_param"] = i == "gemini-embedding-001"468 c["embedding_input_token_limit"] = 2048 if i == "gemini-embedding-001" else 8192469 return c470471472def endpoints_for(live: dict | None, i: str) -> list[dict]:473 methods = (live or {}).get("supportedGenerationMethods", [])474 out = []475 for m in methods:476 route = {"generateContent": f"POST /v1beta/models/{i}:generateContent", "streamGenerateContent": f"POST /v1beta/models/{i}:streamGenerateContent?alt=sse",477 "countTokens": f"POST /v1beta/models/{i}:countTokens", "createCachedContent": "POST /v1beta/cachedContents (model=models/" + i + ")",478 "batchGenerateContent": f"POST /v1beta/models/{i}:batchGenerateContent", "embedContent": f"POST /v1beta/models/{i}:embedContent",479 "asyncBatchEmbedContent": f"POST /v1beta/models/{i}:asyncBatchEmbedContent", "countTextTokens": f"POST /v1beta/models/{i}:countTextTokens (legacy PaLM method)",480 "predictLongRunning": f"POST /v1beta/models/{i}:predictLongRunning (+ GET /v1beta/{{operation}})", "bidiGenerateContent": "WSS /ws/google.ai.generativelanguage.v1beta.GenerativeService.BidiGenerateContent (Live API)",481 "bidiGenerateMusic": "WSS ...BidiGenerateMusic (Live Music API)", "generateAnswer": f"POST /v1beta/models/{i}:generateAnswer"}.get(m, m)482 out.append({"name": m, "route": route, "source": "live supportedGenerationMethods"})483 if "generateContent" in methods:484 out.append({"name": "streamGenerateContent", "route": f"POST /v1beta/models/{i}:streamGenerateContent?alt=sse", "source": "implied by generateContent (not listed in supportedGenerationMethods)"})485 if "generateContent" in methods and not i.startswith(("lyria", "gemini-omni")):486 out.append({"name": "interactions", "route": "POST /v1beta/interactions (model=" + i + ")" if not i.startswith(("deep-research", "antigravity")) else "POST /v1beta/interactions (agent=" + i + ")", "source": "docs (Interactions API); GA in v1 for models"})487 if i.startswith(("lyria-3", "gemini-omni")):488 out.append({"name": "interactions", "route": "POST /v1beta/interactions (model=" + i + ")", "source": "docs (music-generation.md / omni.md use the Interactions API)"})489 return out490491492def tools_for(c: dict) -> list[dict]:493 t = []494 for key, typ in (("google_search_grounding", "google_search"), ("google_maps_grounding", "google_maps"), ("url_context", "url_context"),495 ("code_execution", "code_execution"), ("computer_use", "computer_use"), ("file_search", "file_search"), ("function_calling", "function_declarations")):496 if c.get(key) not in (False, "unknown", None):497 t.append({"type": typ, "category": "server" if typ not in ("function_declarations",) else "client", "support": c[key]})498 return t499500501def rate_limits_for(i: str) -> dict:502 return {"documented": "Per-model RPM/TPM/RPD are shown only in AI Studio (https://aistudio.google.com/rate-limit); the docs page defines tiers Free/1/2/3, spend-based limits and batch limits.",503 "usage_tiers": {"Free": "active project; Pro models not available (observed limit 0)", "Tier 1": "billing linked; spend cap $250; spend rate limit $10/10 min",504 "Tier 2": "$100 paid + 3 days; $2,000 cap; $50/10 min", "Tier 3": "$1,000 paid + 30 days; $20,000-$100,000+ cap; $200/10 min"},505 "priority_tier_multiplier": "0.3x the standard rate limit for the model/tier",506 "batch_enqueued_tokens": dict(zip(("Tier 1", "Tier 2", "Tier 3"), BATCH_ENQUEUED[i])) if i in BATCH_ENQUEUED else "not listed",507 "ref": "generated/fragments/rate-limits/gemini-rate-limits.json", "doc": SRC_RATE}508509510def pricing_for(i: str, live: dict | None) -> dict | str:511 if i in TEXT_PRICING:512 p = dict(TEXT_PRICING[i])513 out = {"currency": "USD", "unit": "per 1M tokens", "tiers": {k: v for k, v in p.items() if k in ("standard", "batch", "flex", "priority")},514 "free_tier": p.get("free_tier"), "grounding": GROUNDING.get(p.get("grounding"), p.get("grounding")), "output_includes_thinking_tokens": True,515 "batch_discount": 0.5, "flex_discount": 0.5, "priority_premium": "1.8x standard (docs: 75-100% more)"}516 if "_2027" in p:517 out["from_2027_01_01"] = {k: {"input": v[0], "output": v[1], "cached_input": v[2], "cache_storage_hour": v[3]} for k, v in p["_2027"].items()}518 out["intro_pricing_note"] = "Introductory prices through December 31, 2026"519 for k in ("long_context_threshold", "input_note", "same_as", "extra_note"):520 if k in p:521 out[k] = p[k]522 return out523 if i in MEDIA_PRICING:524 return {"currency": "USD"} | MEDIA_PRICING[i]525 if i in ("gemini-flash-latest", "gemini-pro-latest", "gemini-flash-lite-latest", "gemini-2.5-flash-native-audio-latest"):526 return "billed as the model the alias currently resolves to (see aliases)"527 if i == "gemini-embedding-001":528 return "not listed on pricing.md (2026-09-18); file-search.md bills indexing embeddings at $0.15 per 1M tokens"529 if i == "aqa":530 return "not listed on pricing.md"531 if i.startswith(("deep-research", "antigravity")):532 return "agent: all model inference at standard Gemini list rates (input, output, intermediate/reasoning tokens); tools per tool pricing; sandbox compute not billed during preview"533 if i == "lyria-realtime-exp":534 return "not listed on pricing.md (experimental)"535 if i.startswith("gemini-2.5-flash-native-audio-preview-09"):536 return "not listed on pricing.md; the 12-2025 preview is $0.50 text / $3.00 audio-video input, $2.00 text / $12.00 audio output per 1M tokens"537 if i == "gemini-3.5-transcribe-live":538 return {"currency": "USD"} | MEDIA_PRICING["gemini-3.5-transcribe-live"]539 if i == "gemini-robotics-er-2-streaming-preview":540 return "pricing.md section is empty ('Standard' heading without a table) as of 2026-09-18"541 return "not listed on pricing.md" if live else "retired / not priced"542543544def lifecycle(i: str, kind: str, live: dict | None, page: dict | None) -> tuple[list[str], str, dict | None]:545 status: list[str] = []546 dep = DEPRECATIONS.get(i)547 rel, shut, repl = dep if dep else (None, None, None)548 deprecation = None549 if page or i in DEPRECATIONS or i in ("gemini-flash-latest", "gemini-pro-latest", "gemini-flash-lite-latest", "gemini-3.1-pro-preview-customtools", "veo-3.1-fast-generate-preview", "gemini-2.0-flash-exp") or i.startswith("gemma-4"):550 status.append("DOCUMENTED") # gemma-4: pricing.md + changelog only; gemini-2.0-flash-exp: listed as shut down on the 2.0 Flash page551 if live is not None:552 status.append("LIVE_DISCOVERED")553 if kind == "preview" or kind == "agent":554 status.append("PREVIEW")555 if kind == "experimental":556 status.append("BETA")557 docs_stage = "Stable (GA)" if kind == "stable" else kind.capitalize()558 if shut:559 deprecation = {"announced_release": rel, "earliest_shutdown": shut, "replacement": repl, "source": SRC_DEPREC}560 if shut <= SHUTDOWN_PAST_CUTOFF:561 status.append("RETIRED")562 docs_stage = "Shut down"563 else:564 status.append("DEPRECATED")565 docs_stage = "Deprecated (shutdown scheduled)"566 elif live is None and kind != "alias":567 # documented but absent from the live listing and no shutdown date: treat as documentation-only568 status.append("UNVERIFIED")569 return status, docs_stage, deprecation570571572def apply_probe(rec: dict, probe: dict) -> None:573 i = rec["id"]574 g = probe.get(f"GET models/{i}")575 gen = probe.get(f"POST models/{i}:generateContent")576 ver: dict = {"method": "live_api" if (g or gen) else "docs_only", "verified_at": VERIFIED_AT + "T00:00:00Z", "result": "not_tested", "http_status": None, "request_note": None}577 if g:578 ver["http_status"] = g["status"]579 ver["request_note"] = f"GET /v1beta/models/{i} -> {g['status']}"580 ver["result"] = "success" if g["status"] == 200 else "failure"581 if g["status"] != 200:582 ver["error_body"] = g["body"]583 if gen:584 st = gen["status"]585 b = gen["body"]586 d = {"http_status": st, "request": "generateContent 'Reply with OK.' maxOutputTokens=8"}587 if st == 200:588 cand = (b.get("candidates") or [{}])[0]589 parts = cand.get("content", {}).get("parts", [])590 d |= {"modelVersion": b.get("modelVersion"), "responseId_present": "responseId" in b, "finishReason": cand.get("finishReason"),591 "usageMetadata": b.get("usageMetadata"), "thoughtSignature_present": any("thoughtSignature" in p for p in parts),592 "text": "".join(p.get("text", "") for p in parts)[:40], "response_headers": gen.get("headers"), "header_names": gen.get("all_header_names")}593 ver["result"] = "success"594 ver["request_note"] = (ver["request_note"] + "; " if ver["request_note"] else "") + f"POST :generateContent -> 200 (modelVersion={b.get('modelVersion')})"595 else:596 d |= {"error": b}597 msg = json.dumps(b)598 if st == 404 and "no longer available to new users" in msg:599 ver["result"] = "restricted"600 rec["status"].append("ACCOUNT_RESTRICTED")601 rec["restrictions"] = {"access": b["error"]["message"], "note": "GET models/{id} still returns 200; only generation is refused for new users"}602 elif st == 429 and "free_tier" in msg:603 ver["result"] = "restricted"604 rec["status"].append("ACCOUNT_RESTRICTED")605 rec["restrictions"] = {"access": "429 RESOURCE_EXHAUSTED on the Free tier: quota metric generate_content_free_tier_requests / _input_token_count has limit 0 for this model (Pro models are paid-tier only)", "quota_dimensions_model": next((v["quotaDimensions"].get("model") for det in b["error"].get("details", []) for v in det.get("violations", [])), None)}606 else:607 ver["result"] = "failure"608 rec["status"].append("FAILED_VERIFICATION")609 ver["request_note"] = (ver["request_note"] + "; " if ver["request_note"] else "") + f"POST :generateContent -> {st} {b.get('error', {}).get('status')}"610 rec["generate_content_probe"] = d611 if ver["result"] == "success" and "LIVE_VERIFIED" not in rec["status"]:612 rec["status"].append("LIVE_VERIFIED")613 rec["verification"] = ver614615616def build_models() -> list[dict]:617 live_models = {m["name"].removeprefix("models/"): m for m in json.load(open(RAW))["models"]}618 v1_ids = {m["name"].removeprefix("models/") for m in json.load(open(RAW_V1))["models"]} if RAW_V1.exists() else set()619 pages = parse_pages()620 probe = json.load(open(PROBE)) if PROBE.exists() else {}621 page_by_id: dict[str, dict] = {}622 for head, pg in pages.items():623 page_by_id[head] = pg624 for m in re.finditer(r"`([a-z0-9][a-z0-9.\-]+)`", (pg.get("code") or "")):625 page_by_id.setdefault(m.group(1), pg)626 # nano-banana-pro-preview is the same model as gemini-3-pro-image-preview627 page_by_id.setdefault("nano-banana-pro-preview", pages.get("gemini-3-pro-image-preview"))628 page_by_id.setdefault("gemini-3.1-pro-preview-customtools", pages.get("gemini-3.1-pro-preview"))629 page_by_id.setdefault("gemini-omni-flash-preview", pages.get("gemini-omni-1.1-flash"))630 page_by_id.setdefault("gemini-3.5-transcribe-live", pages.get("gemini-3.5-transcribe"))631 page_by_id.setdefault("veo-3.1-fast-generate-preview", pages.get("veo-3.1-generate-preview"))632 for k in ("imagen-4.0-ultra-generate-001", "imagen-4.0-fast-generate-001"):633 page_by_id.setdefault(k, pages.get("imagen-4.0-generate-001"))634635 ids = list(live_models) + [i for i in NON_LIVE_DOCUMENTED if i not in live_models]636 records = []637 for i in ids:638 live = live_models.get(i)639 page = page_by_id.get(i)640 kind = kind_of(i, live)641 fam, gen = family_of(i)642 mods = modalities_for(i, page, live)643 caps = build_capabilities(i, page, live, mods)644 status, stage, deprecation = lifecycle(i, kind, live, page)645 rel, shut, repl = DEPRECATIONS.get(i, (None, None, None))646 rec: dict = {647 "provider": "gemini", "id": i, "display_name": (live or {}).get("displayName") or (page or {}).get("page", i).replace("-", " ").title(),648 "kind": kind, "aliases": [], "snapshots": [], "family": fam, "generation": gen,649 "description": (live or {}).get("description") or ((page or {}).get("warning") or "documented model (see deprecations)")[:300],650 "status": status, "lifecycle_docs": stage, "release_date": rel,651 "knowledge_cutoff": (page or {}).get("knowledge_cutoff") or ("January 2025 (Gemini 3 model cards; not stated on the API model pages)" if i.startswith("gemini-3") and page and "image" not in i else None),652 "context_window": (live or {}).get("inputTokenLimit") or (page or {}).get("input_limit"),653 "max_output": (live or {}).get("outputTokenLimit") or (page or {}).get("output_limit"),654 "docs_input_token_limit": (page or {}).get("input_limit"), "docs_output_token_limit": (page or {}).get("output_limit"),655 "modalities": mods, "thinking": caps["thinking_default_level"] if caps["thinking"] not in (False, "unknown") else ("not supported" if caps["thinking"] is False else "unknown"),656 "capabilities": caps,657 "live_model_metadata": ({k: live.get(k) for k in ("name", "version", "displayName", "description", "inputTokenLimit", "outputTokenLimit", "supportedGenerationMethods", "thinking", "temperature", "topP", "topK", "maxTemperature")} if live else None),658 "supportedGenerationMethods": (live or {}).get("supportedGenerationMethods"),659 "endpoints": endpoints_for(live, i), "tools": tools_for(caps),660 "pricing": pricing_for(i, live), "rate_limits": rate_limits_for(i) if live else {"ref": "generated/fragments/rate-limits/gemini-rate-limits.json"},661 "beta_headers": [], "restrictions": None,662 "availability": {663 "account": "listed for our key (GET /v1beta/models)" if live else "not listed for our key",664 "api_versions": {"v1beta": live is not None, "v1": i in v1_ids, "note": "docs (api-versions.md) say all models are in both versions; live GET /v1/models lists only 22 GA ids and GET /v1/models/gemini-3.1-pro-preview -> 404 'for api version v1'"},665 "free_tier": (TEXT_PRICING.get(i, {}).get("free_tier") or (MEDIA_PRICING.get(i, {}) or {}).get("free_tier") or "see pricing"),666 "regions": "Gemini API / AI Studio available in ~190 countries and territories (available-regions.md); EEA/UK/CH: free & paid tiers available, but API clients offered to EEA/CH/UK end users must use Paid Services (terms)",667 "platforms": ["Gemini Developer API (generativelanguage.googleapis.com)", "Google AI Studio", "Gemini Enterprise Agent Platform (Vertex AI) for most Gemini/Veo/Imagen ids - not verified here"],668 },669 "deprecation": deprecation,670 "discrepancies_live_vs_docs": [],671 "last_verified": VERIFIED_AT, "verification": {"method": "docs_only" if live is None else "live_api", "verified_at": VERIFIED_AT + "T00:00:00Z",672 "result": "n/a" if live is None else "success", "http_status": 200 if live else None,673 "request_note": "listed by GET /v1beta/models (paginated pageSize=10, 6 pages)" if live else "not present in GET /v1beta/models"},674 "sources": [src(SRC_MODELS), src(SRC_LIST, "GET /v1beta/models live listing 2026-09-19")] + ([src(page["url"])] if page else []) + [src(SRC_PRICING), src(SRC_DEPREC), src(SRC_RATE)],675 }676 if caps["thinking"] not in (False, "unknown"):677 rec["sources"].append(src(SRC_THINK))678 # aliases / snapshots679 if i in LATEST_ALIASES:680 rec["aliases"] = [dict(LATEST_ALIASES[i], alias=i)]681 rec["kind"] = "alias"682 rec["alias_policy"] = "hot-swapped with every new release of the variation; 2-week e-mail notice before breaking changes (models.md)"683 if i == "gemini-3-pro-image-preview":684 rec["aliases"] = ["nano-banana-pro-preview (same live metadata: version 3.0, Nano Banana Pro)"]685 if i == "nano-banana-pro-preview":686 rec["aliases"] = ["gemini-3-pro-image-preview"]687 if i == "gemini-3.1-pro-preview":688 rec["snapshots"] = ["gemini-3.1-pro-preview-customtools (variant endpoint, same version 3.1-pro-preview-01-2026)"]689 if i == "gemini-2.5-flash-image":690 rec["snapshots"] = ["gemini-2.5-flash-image-preview (shut down 2026-01-15)"]691 if i in ("gemini-3.5-flash",):692 rec["aliases"] = ["gemini-flash-latest pointed here 2026-05-19 -> 2026-09 (now gemini-3.8-flash)"]693 if i == "gemini-3.8-flash":694 rec["aliases"] = ["gemini-flash-latest (observed 2026-09-19 via modelVersion)"]695 if page and page.get("versions"):696 rec["docs_versions"] = page["versions"]697 if page and page.get("model_card"):698 rec["model_card"] = page["model_card"]699 if page and page.get("latest_update"):700 rec["docs_latest_update"] = page["latest_update"]701 # discrepancies702 disc = rec["discrepancies_live_vs_docs"]703 if live and page:704 if page.get("input_limit") and live.get("inputTokenLimit") != page["input_limit"]:705 disc.append(f"inputTokenLimit live={live.get('inputTokenLimit')} vs docs={page['input_limit']}")706 if page.get("output_limit") and live.get("outputTokenLimit") != page["output_limit"]:707 disc.append(f"outputTokenLimit live={live.get('outputTokenLimit')} vs docs={page['output_limit']}")708 th_doc = page["caps"].get("Thinking")709 if th_doc and (th_doc.lower().startswith("supported") != bool(live.get("thinking"))):710 disc.append(f"thinking: docs '{th_doc}' vs live thinking={live.get('thinking')}")711 if page["caps"].get("Batch API", "").lower().startswith("supported") and "batchGenerateContent" not in live.get("supportedGenerationMethods", []) and "asyncBatchEmbedContent" not in live.get("supportedGenerationMethods", []):712 disc.append("docs: Batch API supported, but batchGenerateContent absent from live supportedGenerationMethods")713 if page["caps"].get("Caching", "").lower().startswith("supported") and "createCachedContent" not in live.get("supportedGenerationMethods", []):714 disc.append("docs: Caching supported, but createCachedContent absent from live supportedGenerationMethods")715 if "Live API" in page["caps"] and page["caps"]["Live API"].lower().startswith("supported") and "bidiGenerateContent" not in live.get("supportedGenerationMethods", []):716 disc.append("docs: Live API supported, but bidiGenerateContent absent from live supportedGenerationMethods")717 if live and i in ("gemini-3-pro-preview",):718 disc.append("changelog 2026-03-09: 'gemini-3-pro-preview now points to gemini-3.1-pro-preview', yet live GET returns version '3-pro-preview-11-2025' (deprecations.md lists it as shut down 2026-03-09)")719 if live and i == "gemini-3.8-flash":720 disc.append("live version field is '3.0' (other 3.x GA models carry dated versions like 3.7-flash-08-2026)")721 if live and i in ("gemini-3.8-live", "gemini-3.8-live-extended-thinking"):722 disc.append("live version field is '3.1-flash-live-03-2026' (same as gemini-3.1-flash-live-preview)")723 if live and i == "gemini-3.8-live" and not live.get("thinking"):724 disc.append("docs: 'Thinking Supported (interleaved reasoning)' vs live thinking flag absent")725 if live and i.startswith("gemma-4"):726 disc.append("Gemma 4 not documented in models.md (only pricing.md + changelog); live thinking=true and probe returned thoughtsTokenCount=5")727 if live and i == "gemini-3.5-transcribe" and live.get("thinking"):728 disc.append("docs: Thinking not supported vs live thinking=true")729 if live and "-latest" in i and live.get("version", "").startswith("Gemini"):730 disc.append("live version field holds the display name ('%s') instead of a version" % live.get("version"))731 if live and i in ("nano-banana-pro-preview", "gemini-3.1-pro-preview-customtools", "gemini-2.5-flash-native-audio-latest", "gemini-2.5-flash-native-audio-preview-09-2025", "aqa", "antigravity-preview-09-2026"):732 disc.append("live-listed id without a dedicated docs model page" + (" (aqa: legacy generateAnswer-only model, undocumented)" if i == "aqa" else ""))733 if live and i.startswith("deep-research"):734 disc.append("live version 'deepthink-exp-05-20' and inputTokenLimit 131,072 vs docs 'Input context window 1,048,576'")735 if live and i == "gemini-2.5-flash-image":736 disc.append("docs: input limit 65,536 vs live 32,768; live description still says 'Gemini 2.5 Flash Preview Image'")737 apply_probe(rec, probe)738 # normalise statuses (unique, ordered)739 seen = set()740 rec["status"] = [s for s in rec["status"] if not (s in seen or seen.add(s))]741 if "LIVE_VERIFIED" in rec["status"] and "UNVERIFIED" in rec["status"]:742 rec["status"].remove("UNVERIFIED")743 records.append(rec)744 return records745746747# --------------------------------------------------------------------------------------------------748def build_pricing() -> list[dict]:749 recs: list[dict] = []750751 def p(model, dim, price, unit="per 1M tokens", tier="standard", notes=None, section=None):752 recs.append({"provider": "gemini", "model_or_service": model, "dimension": dim, "price": price, "currency": "USD", "unit": unit, "tier": tier,753 "effective_notes": notes, "source": SRC_PRICING + (f"#{section}" if section else ""), "retrieved_at": RETRIEVED})754755 for mid, t in TEXT_PRICING.items():756 sec = mid757 intro = "_2027" in t758 for tier in ("standard", "batch", "flex", "priority"):759 if tier not in t:760 continue761 for dim, price in t[tier].items():762 unit = "per 1M tokens per hour" if dim == "cache_storage_hour" else "per 1M tokens"763 note = None764 if intro:765 idx = {"input": 0, "output": 1, "cached_input": 2, "cache_storage_hour": 3}.get(dim)766 fut = t["_2027"].get(tier)767 if fut and idx is not None and fut[idx] is not None:768 note = INTRO.format(f"${fut[idx]}")769 if dim.endswith("over_200k"):770 note = (note + "; " if note else "") + "prompts > 200k tokens (long-context tier)"771 elif t.get("long_context_threshold") and dim in ("input", "output", "cached_input"):772 note = (note + "; " if note else "") + "prompts <= 200k tokens"773 if dim == "output":774 note = (note + "; " if note else "") + "includes thinking tokens"775 if t.get("input_note") and dim == "input":776 note = (note + "; " if note else "") + t["input_note"]777 p(mid, dim, price, unit, tier, note, sec)778 p(mid, "free_tier", 0, "per 1M tokens", "free", t.get("free_tier"), sec)779 g = GROUNDING.get(t.get("grounding"))780 if g:781 p(mid, "grounding_google_search", 14.0 if t["grounding"] == "gemini3" else 35.0, "per 1K requests" if t["grounding"] == "gemini3" else "per 1K grounded prompts", "standard", g["google_search"], sec)782 p(mid, "grounding_google_maps", 14.0 if t["grounding"] == "gemini3" else 25.0, "per 1K search queries" if t["grounding"] == "gemini3" else "per 1K grounded prompts", "standard", g["google_maps"], sec)783 # Live / audio784 live = MEDIA_PRICING["gemini-3.8-live"]785 for mid in ("gemini-3.8-live", "gemini-3.8-live-extended-thinking", "gemini-3.1-flash-live-preview"):786 sec = "live-api-models"787 p(mid, "input_text", live["input_text"], notes="Live API", section=sec)788 p(mid, "audio_input", live["input_audio"], notes="= $0.005 per minute", section=sec)789 p(mid, "image_video_input", live["input_image_video"], notes="= $0.002 per minute", section=sec)790 p(mid, "output_text", live["output_text"], notes="includes thinking tokens", section=sec)791 p(mid, "audio_output", live["output_audio"], notes="= $0.018 per minute", section=sec)792 p(mid, "free_tier", 0, notes="free of charge; Google Search grounding supported on free tier for these models", tier="free", section=sec)793 p("gemini-3.5-live-translate-preview", "audio_input", 3.50, notes="= $0.0053/min; 25 tokens per second of audio; effective ~$0.0368/min in+out")794 p("gemini-3.5-live-translate-preview", "audio_output", 21.00, notes="= $0.0315/min")795 p("gemini-3.5-transcribe-live", "audio_input", 3.50, notes="= $0.005/min (25 audio tokens/s)")796 p("gemini-3.5-transcribe-live", "output_text", 21.00, notes="= $0.004/min (175 text tokens/min); blended ~$0.009/min")797 p("gemini-3.5-transcribe", "audio_input", 2.00, notes="= $0.003/min")798 p("gemini-3.5-transcribe", "output_text", 12.00, notes="= $0.002/min; blended ~$0.005/min")799 for mid in ("gemini-3.5-live-translate-preview", "gemini-3.5-transcribe-live", "gemini-3.5-transcribe"):800 p(mid, "free_tier", 0, tier="free", notes="free of charge")801 p("gemini-2.5-flash-native-audio-preview-12-2025", "input_text", 0.50)802 p("gemini-2.5-flash-native-audio-preview-12-2025", "audio_video_input", 3.00)803 p("gemini-2.5-flash-native-audio-preview-12-2025", "output_text", 2.00)804 p("gemini-2.5-flash-native-audio-preview-12-2025", "audio_output", 12.00)805 # Omni806 for mid in ("gemini-omni-1.1-flash", "gemini-omni-flash-preview"):807 p(mid, "input", 1.50, notes="text / image / video / audio")808 p(mid, "output_text", 9.00, notes="includes thinking tokens")809 p(mid, "video_output", 17.50, notes="5,792 tokens per second of 720p video => ~$0.10 per second")810 p(mid, "video_output_per_second", 0.10, "per second", notes="effective, 720p, standard")811 # Image models812 for mid in ("gemini-3.1-flash-image", "gemini-3.1-flash-image-preview"):813 p(mid, "input", 0.50, notes="text/image"); p(mid, "output_text", 3.00, notes="text and thinking"); p(mid, "image_output", 60.00, notes="per 1M image tokens")814 for k, v, tok in (("0.5K", 0.045, 747), ("1K", 0.067, 1120), ("2K", 0.101, 1680), ("4K", 0.151, 2520)):815 p(mid, f"image_output_{k}", v, "per image", notes=f"{tok} tokens per image")816 p(mid, "input", 0.25, tier="batch"); p(mid, "output_text", 1.50, tier="batch"); p(mid, "image_output", 30.00, tier="batch")817 for k, v in (("0.5K", 0.022), ("1K", 0.034), ("2K", 0.050), ("4K", 0.076)):818 p(mid, f"image_output_{k}", v, "per image", tier="batch")819 p(mid, "grounding_google_search", 14.0, "per 1K requests", notes="5,000 free/month shared across Gemini 3.x; text and image search grounding; retrieved context not charged as input")820 p("gemini-3.1-flash-lite-image", "input", 0.25, notes="text/image/video"); p("gemini-3.1-flash-lite-image", "output_text", 1.50); p("gemini-3.1-flash-lite-image", "image_output", 30.00)821 p("gemini-3.1-flash-lite-image", "image_output_1K", 0.0336, "per image", notes="1120 tokens; only 1K supported")822 p("gemini-3.1-flash-lite-image", "input", 0.125, tier="batch"); p("gemini-3.1-flash-lite-image", "output_text", 0.75, tier="batch"); p("gemini-3.1-flash-lite-image", "image_output", 15.00, tier="batch"); p("gemini-3.1-flash-lite-image", "image_output_1K", 0.0168, "per image", tier="batch")823 for mid in ("gemini-3-pro-image", "gemini-3-pro-image-preview", "nano-banana-pro-preview"):824 p(mid, "input", 2.00, notes="text/image; image input = 560 tokens = $0.0011 per image"); p(mid, "image_input", 0.0011, "per image")825 p(mid, "output_text", 12.00, notes="text and thinking"); p(mid, "image_output", 120.00, notes="per 1M image tokens")826 p(mid, "image_output_1K_2K", 0.134, "per image", notes="1120 tokens"); p(mid, "image_output_4K", 0.24, "per image", notes="2000 tokens")827 for tier in ("batch", "flex"):828 p(mid, "input", 1.00, tier=tier, notes="text"); p(mid, "image_input", 0.0006, "per image", tier=tier); p(mid, "output_text", 6.00, tier=tier)829 p(mid, "image_output_1K_2K", 0.067, "per image", tier=tier); p(mid, "image_output_4K", 0.12, "per image", tier=tier)830 p(mid, "input", 3.60, tier="priority"); p(mid, "output_text", 21.60, tier="priority"); p(mid, "image_output", 216.00, tier="priority")831 p(mid, "grounding_google_search", 14.0, "per 1K requests", notes="5,000 free/month shared across Gemini 3.x")832 p("gemini-2.5-flash-image", "input", 0.30, notes="text/image"); p("gemini-2.5-flash-image", "image_output", 0.039, "per image", notes="$30 per 1M image tokens; 1290 tokens per image up to 1024x1024; deprecated, shutdown 2026-10-02")833 for tier in ("batch", "flex"):834 p("gemini-2.5-flash-image", "input", 0.15, tier=tier); p("gemini-2.5-flash-image", "image_output", 0.0195, "per image", tier=tier)835 p("gemini-2.5-flash-image", "input", 0.54, tier="priority"); p("gemini-2.5-flash-image", "image_output", 0.0702, "per image", tier="priority")836 # TTS837 for mid, a, b in (("gemini-3.1-flash-tts-preview", 1.00, 20.00), ("gemini-2.5-flash-preview-tts", 0.50, 10.00), ("gemini-2.5-pro-preview-tts", 1.00, 20.00)):838 p(mid, "input_text", a); p(mid, "audio_output", b, notes="25 audio tokens per second")839 p(mid, "input_text", a / 2, tier="batch"); p(mid, "audio_output", b / 2, tier="batch")840 # Veo841 for mid, d in (("veo-3.1-generate-preview", {"720p": 0.40, "1080p": 0.40, "4k": 0.60}), ("veo-3.1-fast-generate-preview", {"720p": 0.10, "1080p": 0.12, "4k": 0.30}), ("veo-3.1-lite-generate-preview", {"720p": 0.05, "1080p": 0.08})):842 for res, v in d.items():843 p(mid, f"video_output_{res}", v, "per second", notes="video with audio (default); charged only when the video is successfully generated")844 if mid.endswith("lite-generate-preview"):845 p(mid, "video_output_4k", None, "per second", notes="4K output not supported")846 # Lyria847 p("lyria-3.5", "per_song", 0.08, "per request", notes="full song"); p("lyria-3-clip-preview", "per_song", 0.04, "per request", notes="30 s clip"); p("lyria-3-pro-preview", "per_song", 0.08, "per request", notes="full song; legacy")848 # Embeddings849 for mid in ("gemini-embedding-2", "gemini-embedding-2-preview"):850 p(mid, "input_text", 0.20); p(mid, "image_input", 0.45, notes="= $0.00012 per image"); p(mid, "audio_input", 6.50, notes="= $0.00016 per second"); p(mid, "video_input", 12.00, notes="= $0.00079 per frame")851 p(mid, "input_text", 0.10, tier="batch"); p(mid, "image_input", 0.225, tier="batch", notes="= $0.00006 per image"); p(mid, "audio_input", 3.25, tier="batch", notes="= $0.00008 per second"); p(mid, "video_input", 6.00, tier="batch", notes="= $0.000395 per frame")852 p(mid, "free_tier", 0, tier="free", notes="standard free of charge; batch not available on free tier")853 p("gemini-embedding-001", "input", None, notes="not listed on pricing.md as of 2026-09-18; File Search indexing embeddings billed at $0.15 per 1M tokens (file-search tool row)")854 # Gemma855 for mid in ("gemma-4-26b-a4b-it", "gemma-4-31b-it"):856 p(mid, "input", 0, tier="free", notes="Gemma 4: free tier free of charge; paid tier 'Not available'"); p(mid, "output", 0, tier="free"); p(mid, "cached_input", 0, tier="free"); p(mid, "cache_storage_hour", 0, "per 1M tokens per hour", tier="free")857 # Tools & services858 p("tool:google_search", "grounding_google_search", 14.0, "per 1K requests", notes="Gemini 3.x: 5,000 free search requests per month shared across all Gemini 3 models; billed per executed search query; retrieved context not charged as input")859 p("tool:google_search", "grounding_google_search", 35.0, "per 1K grounded prompts", notes="Gemini 2.5 models: 1,500 RPD free (shared Flash/Flash-Lite); billed per grounded prompt")860 p("tool:google_search", "grounding_google_search", 0, "per 1K grounded prompts", tier="free", notes="Free tier: 500 RPD free (shared Flash/Flash-Lite); not available for Pro")861 p("tool:google_maps", "grounding_google_maps", 14.0, "per 1K search queries", notes="Gemini 3.x: 5,000 prompts/month free shared across Gemini 3")862 p("tool:google_maps", "grounding_google_maps", 25.0, "per 1K grounded prompts", notes="Gemini 2.5: 1,500 RPD free (Flash/Flash-Lite), 10,000 RPD free for Pro; free tier 500 RPD, not for Pro")863 p("tool:code_execution", "tool_usage", 0, "per call", notes="no per-call fee; generated code + results billed as output tokens when created and as input tokens when re-consumed; no charge for session runtime; free tier: free")864 p("tool:url_context", "tool_usage", 0, "per call", notes="retrieved page content charged as input tokens at model rates")865 p("tool:computer_use", "tool_usage", 0, "per call", notes="charged as regular tokens per model pricing; not available on free tier")866 p("tool:file_search", "indexing_embeddings", 0.15, "per 1M tokens", notes="charged once at indexing; storage and query-time embeddings free; retrieved document tokens billed as input tokens")867 p("tool:custom_tools_endpoint", "tokens", None, notes="gemini-3.1-pro-preview-customtools: same as Gemini 3.1 Pro Preview")868 p("agent:deep-research", "tokens", None, notes="all model inference at standard Gemini list rates incl. intermediate/reasoning tokens; tool fees per tool pricing (search retrieved tokens excluded; url_context/file search retrieved tokens included)")869 p("agent:managed-agents", "environment_compute", 0, "per hour", notes="sandbox CPU/memory/execution not billed during preview; inference at list rates")870 p("agent:antigravity", "environment_compute", 0, "per hour", notes="same as managed agents; inference at list rates")871 p("service:batch_api", "discount", 0.5, "multiplier", tier="batch", notes="50% of standard interactive price; target turnaround 24h")872 p("service:flex_inference", "discount", 0.5, "multiplier", tier="flex", notes="50% discount; best-effort, sheddable; 1-15 min target latency")873 p("service:priority_inference", "premium", "1.8x", "multiplier", tier="priority", notes="75-100% more than standard (tables show 1.8x); graceful downgrade to standard when limits exceeded")874 p("service:context_cache_storage", "cache_storage_hour", "0.50-8.10", "per 1M tokens per hour", notes="model dependent: $0.50 (3.6-3.8 Flash intro) / $1.00 (most Flash) / $1.80 (Flash priority) / $4.50 (Pro) / $8.10 (Pro priority)")875 p("service:document_tokens", "pdf_page", None, "per page", notes="DOCUMENT modality (PDF) billed at the image token rate; appears under promptTokensDetails modality DOCUMENT")876 p("service:google_ai_studio", "usage", 0, "per call", tier="free", notes="AI Studio usage is free of charge in all available regions")877 return recs878879880OUT_DOC = ROOT / "docs/models/gemini-models.md"881CAP_COLS = [("thinking", "Think"), ("structured_output", "Struct"), ("function_calling", "FC"), ("google_search_grounding", "Search"), ("google_maps_grounding", "Maps"),882 ("url_context", "URL"), ("code_execution", "Code"), ("computer_use", "CU"), ("file_search", "FS"), ("context_caching_explicit", "Cache"),883 ("batch_api", "Batch"), ("flex_inference", "Flex"), ("priority_inference", "Prio"), ("live_api", "Live"), ("interactions_api", "Inter."), ("openai_compatible_chat", "OAI")]884EP_COLS = ["generateContent", "countTokens", "createCachedContent", "batchGenerateContent", "embedContent", "asyncBatchEmbedContent", "countTextTokens", "predictLongRunning", "bidiGenerateContent", "bidiGenerateMusic", "generateAnswer"]885886887def _sym(v) -> str:888 if v is True:889 return "Y"890 if v is False:891 return "-"892 if v in ("unknown", None, "n/a"):893 return "?"894 s = str(v)895 return "Y*" if s.lower().startswith("supported") else s[:12]896897898def write_doc(models: list[dict]) -> None:899 live = [m for m in models if "LIVE_DISCOVERED" in m["status"]]900 retired = [m for m in models if "LIVE_DISCOVERED" not in m["status"]]901 L: list[str] = []902 L.append("# Gemini models — catalogue (Gemini Developer API)\n")903 L.append("**Status:** DOCUMENTED + LIVE_DISCOVERED (58 ids via paginated `GET /v1beta/models`, 22 via `GET /v1/models`) + LIVE_VERIFIED for the ids probed on 2026-09-19 (see `verification` per record). "904 "Machine-readable twin: `generated/fragments/models/gemini-models.json` (97 records: 58 live + 39 documented-only/retired). This page is **generated** by `scripts/gen_gemini_models_fragment.py` — edit the script, not this file.\n")905 L.append("**Sources:** https://ai.google.dev/gemini-api/docs/models · model pages `…/docs/models/<id>` · https://ai.google.dev/api/models · https://ai.google.dev/gemini-api/docs/pricing · https://ai.google.dev/gemini-api/docs/deprecations · https://ai.google.dev/gemini-api/docs/rate-limits · https://ai.google.dev/gemini-api/docs/thinking · https://ai.google.dev/gemini-api/docs/api-versions · discovery document revision 20260918.\n")906 L.append("**Last verified:** 2026-09-18 (docs) / 2026-09-19 UTC (live). Related: [pricing](../gemini/pricing.md) · [rate limits](../gemini/rate-limits.md) · [deprecations & changelog](../gemini/deprecations-and-changelog.md) · [auth / headers / versions](../gemini/authentication-headers-versions.md) · [errors](../errors/gemini.md) · [SDKs](../gemini/sdks.md) · [OpenAI compatibility](../gemini/openai-compatibility.md) · [Gemini index](../gemini/index.md)\n")907 L.append("## 1. Families and naming\n")908 L.append("| Family | Live ids (2026-09-19) | Notes |\n|---|---|---|")909 fams: dict[str, list[str]] = {}910 for m in live:911 fams.setdefault(m["family"], []).append(m["id"])912 for f, ids in fams.items():913 L.append(f"| {f} | {', '.join('`'+i+'`' for i in ids)} | {len(ids)} ids |")914 L.append("\nNaming patterns (models.md, convention since Sept 2025): **stable** `gemini-3.6-flash`; **preview** `gemini-3.1-pro-preview` / dated `…-preview-09-2025` (billing on, tighter limits, ≥2 weeks deprecation notice); "915 "**latest** aliases `gemini-flash-latest`, `gemini-pro-latest`, `gemini-flash-lite-latest`, `gemini-2.5-flash-native-audio-latest` (hot-swapped, 2-week e-mail notice for breaking changes); **experimental** `lyria-realtime-exp` (no production use). "916 "Live `version` strings are inconsistent: dated (`3.7-flash-08-2026`), bare (`3.0` for gemini-3.8-flash and all image models), or the display name (`Gemini Flash Latest`).\n")917 L.append("### Alias resolution observed\n\n| Alias | Resolves to (live) | Evidence |\n|---|---|---|")918 for a, d in LATEST_ALIASES.items():919 L.append(f"| `{a}` | {d['resolves_to_live']} | {d['evidence']} |")920 L.append("\n`gemini-3-pro-preview` was shut down 2026-03-09 and per the changelog 'now points to gemini-3.1-pro-preview'; live `GET` still returns `version: 3-pro-preview-11-2025`. `nano-banana-pro-preview` is the same live object as `gemini-3-pro-image-preview` (version 3.0, display name Nano Banana Pro). `gemini-3.1-pro-preview-customtools` is a variant endpoint of `gemini-3.1-pro-preview` (same version `3.1-pro-preview-01-2026`) tuned to prefer custom tools.\n")921 L.append("## 2. Live catalogue (58 ids) — limits, lifecycle, verification\n")922 L.append("| id | Display name | Kind | Status | In / Out tokens (live) | Thinking (live flag / docs default) | Release | Shutdown | Verified |\n|---|---|---|---|---|---|---|---|---|")923 for m in live:924 lm = m["live_model_metadata"] or {}925 dep = m.get("deprecation") or {}926 L.append(f"| `{m['id']}` | {m['display_name']} | {m['kind']} | {', '.join(m['status'])} | {lm.get('inputTokenLimit'):,} / {lm.get('outputTokenLimit'):,} | {lm.get('thinking')} / {m['capabilities']['thinking_default_level']} | {m.get('release_date') or '?'} | {dep.get('earliest_shutdown') or '-'} | {m['verification'].get('request_note') or ''} |")927 L.append("\nShut-down ids still returned by the live listing: `gemini-3-pro-preview`, `gemini-3.1-flash-lite-preview`, `gemini-3.1-flash-image-preview`, `gemini-3-pro-image-preview`/`nano-banana-pro-preview` (deprecations.md dates in the past). Conversely `imagen-4.0-*` (shut down 2026-08-17) is gone: `GET` → 404 `Model is not found … for api version v1beta`.\n")928 L.append("## 3. Model × Capability matrix (live ids)\n")929 L.append("Legend: Y = supported (docs and/or live), Y* = supported with qualifier (see JSON, e.g. `Supported (Preview)`), - = not supported, ? = unknown/not documented. Columns: Think=thinking, Struct=structured output (responseSchema/responseJsonSchema), FC=function calling, Search/Maps=grounding, URL=url_context, Code=code execution, CU=computer use, FS=file search, Cache=explicit context caching (live `createCachedContent`), Batch=batchGenerateContent/asyncBatchEmbedContent, Flex/Prio=service tiers, Live=bidiGenerateContent, Inter.=Interactions API, OAI=OpenAI-compatible chat.\n")930 L.append("| id | Modalities in → out | " + " | ".join(c[1] for c in CAP_COLS) + " |\n|---|---|" + "---|" * len(CAP_COLS))931 for m in live:932 c = m["capabilities"]933 L.append(f"| `{m['id']}` | {'+'.join(m['modalities']['input'])} → {'+'.join(m['modalities']['output'])} | " + " | ".join(_sym(c.get(k)) for k, _ in CAP_COLS) + " |")934 L.append("\nThinking levels (thinking.md): " + "; ".join(f"`{k}` default **{v[0]}**, levels {'/'.join(v[1])}" for k, v in THINKING.items() if not k.startswith('gemini-2.5')) + ". Gemini 2.5: `thinking_budget` (2.5 Flash-Lite off by default; 2.5 Pro cannot be disabled). "935 "Gemini 3.x: `thinking_level` replaces `thinking_budget` (400 if both are sent); `minimal` is rejected by 3.7/3.8 Flash; thought signatures must be echoed in multi-turn function calling (`MISSING_THOUGHT_SIGNATURE`).\n")936 L.append("## 4. Model × Endpoint matrix (`supportedGenerationMethods`, verbatim from live)\n")937 L.append("`streamGenerateContent` is never listed but is implied by `generateContent`. `interactions` (POST /v1beta/interactions) is not reported by the Models API; see column Inter. above.\n")938 L.append("| id | " + " | ".join(EP_COLS) + " |\n|---|" + "---|" * len(EP_COLS))939 for m in live:940 ms = set(m["supportedGenerationMethods"] or [])941 L.append(f"| `{m['id']}` | " + " | ".join("Y" if e in ms else "-" for e in EP_COLS) + " |")942 L.append("\n## 5. Model × Tool matrix (built-in tools; docs support tables)\n")943 tool_cols = ["google_search", "google_maps", "url_context", "code_execution", "computer_use", "file_search", "function_declarations"]944 L.append("| id | " + " | ".join(tool_cols) + " |\n|---|" + "---|" * len(tool_cols))945 for m in live:946 if not m["tools"] and not m["capabilities"].get("function_calling") in (True,):947 continue948 tmap = {t["type"]: t["support"] for t in m["tools"]}949 L.append(f"| `{m['id']}` | " + " | ".join(_sym(tmap.get(t, False)) for t in tool_cols) + " |")950 L.append("\nTool pricing: Google Search $14/1k requests after 5,000 free/month (Gemini 3; $35/1k grounded prompts after 1,500 RPD on 2.5); Maps $14/1k queries (Gemini 3) or $25/1k prompts (2.5); code execution / URL context / computer use billed as tokens; File Search indexing $0.15/1M tokens. Built-in tools can be combined with function calling since 2026-03-18 (tool-combination.md).\n")951 L.append("## 6. Live vs docs discrepancies\n")952 L.append("| id | Discrepancy |\n|---|---|")953 for m in models:954 for d in m["discrepancies_live_vs_docs"]:955 L.append(f"| `{m['id']}` | {d} |")956 L.append("\nGlobal: api-versions.md says all models exist in both `v1` and `v1beta`, but `GET /v1/models` lists only 22 GA ids and `GET /v1/models/gemini-3.1-pro-preview` → 404. The Gemini 2.5 family returns 404 `no longer available to new users` on generation for new keys (undocumented). Our key is Free tier: Pro models answer 429 `limit: 0`.\n")957 L.append("## 7. Documented-only / retired ids (not in the live listing)\n")958 L.append("| id | Family | Status | Released | Shutdown | Replacement |\n|---|---|---|---|---|---|")959 for m in retired:960 dep = m.get("deprecation") or {}961 L.append(f"| `{m['id']}` | {m['family']} | {', '.join(m['status'])} | {m.get('release_date') or '?'} | {dep.get('earliest_shutdown') or '-'} | {dep.get('replacement') or '-'} |")962 L.append("\n## 8. Preview / availability policy\n")963 L.append("- Preview models: production use allowed, billing enabled, more restrictive rate limits, ≥2 weeks deprecation notice; `-latest` aliases: 2-week e-mail notice before a breaking swap.\n- Free tier: many text models are free of charge (content used to improve Google products — Unpaid Services terms); Pro, image, video, music, Omni and Deep Research/Antigravity agents need a paid key (observed `limit: 0` on Pro).\n- Regions: ~190 countries/territories (available-regions.md). EEA/UK/CH: free and paid tiers available to developers, but API clients serving end users there must use Paid Services (terms §Use Restrictions).\n- Gemini Enterprise Agent Platform (Vertex AI) offers most Gemini/Veo ids with regional endpoints and enterprise controls — see [vertex-vs-gemini-api](../gemini/vertex-vs-gemini-api.md).\n")964 L.append("## 9. Live probe summary (2026-09-19)\n")965 L.append("| Call | Result |\n|---|---|")966 for m in models:967 p = m.get("generate_content_probe")968 if p:969 L.append(f"| `POST models/{m['id']}:generateContent` (maxOutputTokens 8) | HTTP {p['http_status']}; " + (f"modelVersion `{p.get('modelVersion')}`, finishReason {p.get('finishReason')}, usage {json.dumps(p.get('usageMetadata'))}, thoughtSignature={p.get('thoughtSignature_present')}" if p['http_status'] == 200 else json.dumps(p.get('error'))[:220].replace('|', '/')) + " |")970 L.append("| `POST /v1beta/openai/chat/completions` gemini-3.5-flash-lite | 200 with `Authorization: Bearer`; 400 INVALID_ARGUMENT with only `x-goog-api-key` |")971 L.append("| `GET /v1beta/openai/models` | 200, 58 ids prefixed `models/` |")972 L.append("| `GET /v1/models` | 200, 22 ids (3 pages at pageSize=10) |")973 L.append("\nEstimated cost of all probes: < $0.001 (a few hundred tokens on Free-tier models).\n")974 OUT_DOC.parent.mkdir(parents=True, exist_ok=True)975 OUT_DOC.write_text("\n".join(L) + "\n")976977978def main() -> None:979 models = build_models()980 prices = build_pricing()981 write_doc(models)982 OUT_MODELS.parent.mkdir(parents=True, exist_ok=True)983 OUT_PRICING.parent.mkdir(parents=True, exist_ok=True)984 OUT_MODELS.write_text(json.dumps({"record_type": "model", "provider": "gemini", "generated_at": RETRIEVED, "generator": "scripts/gen_gemini_models_fragment.py",985 "live_listing": {"v1beta_count": 58, "v1_count": 22, "retrieved_at": VERIFIED_AT}, "count": len(models), "records": models}, indent=1, ensure_ascii=False) + "\n")986 OUT_PRICING.write_text(json.dumps({"record_type": "price", "provider": "gemini", "generated_at": RETRIEVED, "generator": "scripts/gen_gemini_models_fragment.py",987 "count": len(prices), "records": prices}, indent=1, ensure_ascii=False) + "\n")988 print(f"models={len(models)} prices={len(prices)} -> {OUT_MODELS.relative_to(ROOT)}, {OUT_PRICING.relative_to(ROOT)}")989990991if __name__ == "__main__":992 main()993