Python 88.3%
TypeScript 7.6%
Shell 4.1%
1#!/usr/bin/env python32"""docs/comparisons/pricing.md — side-by-side prices + cost models for OpenAI, Anthropic, xAI and Gemini,3computed from generated/models.json (pricing blocks) and generated/pricing.json (tool/service rows). Re-runnable."""4import json5ROOT = "/Users/simon-pierreboucher/Desktop/doc-api"6TODAY = "2026-09-18"7M = {r["id"]: r for r in json.load(open(f"{ROOT}/generated/models.json"))}8P = json.load(open(f"{ROOT}/generated/pricing.json"))910OAI = ["gpt-6-astra","gpt-5.6-sol","gpt-5.6-terra","gpt-5.6-luna","gpt-5.5","gpt-5.5-pro","gpt-5.4","gpt-5.4-pro","gpt-5.4-mini","gpt-5.4-nano","gpt-5.3-codex","gpt-5.2","gpt-5.2-pro","gpt-5.1","gpt-5","gpt-5-mini","gpt-5-nano","gpt-5-pro","o3","o3-pro","o4-mini","o3-mini","gpt-4.1","gpt-4.1-mini","gpt-4.1-nano","gpt-4o","gpt-4o-mini","chat-latest","gpt-5.6-cyber"]11ANT = ["claude-fable-5-1","claude-fable-5","claude-mythos-5-1","claude-mythos-5","claude-opus-5","claude-opus-4-8","claude-opus-4-7","claude-opus-4-6","claude-opus-4-5-20251101","claude-sonnet-5","claude-sonnet-4-6","claude-sonnet-4-5-20250929","claude-haiku-4-5-20251001"]12XAI = ["grok-4.6","grok-4.5","grok-4.3","grok-4.20-0309-reasoning","grok-4.20-0309-non-reasoning","grok-4.20-multi-agent-0309","grok-build-0.1"]13GEM = ["gemini-3.8-flash","gemini-3.7-flash","gemini-3.6-flash","gemini-3.5-flash","gemini-3.5-flash-lite","gemini-3.1-pro-preview","gemini-3.1-pro-preview-customtools","gemini-3.1-flash-lite","gemini-3-flash-preview","gemini-2.5-pro","gemini-2.5-flash","gemini-2.5-flash-lite","gemini-robotics-er-2-preview","gemini-2.5-computer-use-preview-10-2025","gemma-4-31b-it","gemma-4-26b-a4b-it"]1415def f(x, nd=4):16 if x is None: return "—"17 if isinstance(x, str): return x18 s = f"{x:.{nd}f}".rstrip("0").rstrip(".")19 return s if s else "0"20def usd(x):21 if x is None: return "—"22 return f"${x:,.2f}" if x >= 0.1 else f"${x:.4f}"23def st(r): return " · ".join(f"`{s}`" for s in r["status"])24def cm(inp, out, ki=1.0, ko=0.1):25 if inp is None or out is None or isinstance(inp, str) or isinstance(out, str): return None26 return ki*inp + ko*out27def rows(prov, prefix):28 return [r for r in P if r["provider"] == prov and str(r["model_or_service"]).startswith(prefix)]2930L = []31L.append("# Pricing — side-by-side and cost models (OpenAI · Anthropic · xAI · Gemini)\n")32L.append(f"**Status:** every number below is read from `generated/models.json` (`pricing` block of each model, sourced from the vendor pricing/model pages on {TODAY}) and `generated/pricing.json` (tool/service/rule rows). Figures are USD per 1M tokens unless stated. Statuses next to model ids are the model record statuses. No live billing was reconciled; treat as list prices. Gemini 3.6–3.8 Flash rows are **introductory prices through 2026-12-31** (the records carry a `from_2027_01_01` block at 2×); xAI `x_search` switches from per-call to per-post/per-profile billing on **2026-09-21**.")33L.append("**Sources:** https://developers.openai.com/api/docs/pricing · https://platform.claude.com/docs/en/about-claude/pricing · https://docs.x.ai/developers/pricing (+ live `GET /v1/language-models` price ticks) · https://ai.google.dev/gemini-api/docs/pricing · docs/openai/pricing.md · docs/anthropic/pricing.md · docs/xai/pricing.md · docs/gemini/pricing.md")34L.append(f"**Last verified:** {TODAY}\n")3536L.append("## 1. Pricing dimensions — how the four price lists are structured\n")37L.append("| Dimension | OpenAI | Anthropic | xAI | Gemini |\n|---|---|---|---|---|")38L.append("| Base tokens | `input`, `cached_input`, `output` per model; separate `*_long_context` table for >272k-token prompts (2× input, 1.5× output) on GPT-5.4/5.5/5.6/6 Astra | `input`, `output`; 1M context at **standard** price on Claude 4.6+ (no long-context premium) | `input`, `cached_input`, `image_input` (= text input price), `output`; **long-context tier** at 2× for prompts **≥ 200,000 tokens — billed on ALL tokens of that request** (`live_raw_ticks.long_context_threshold`); reasoning tokens billed as output on every Grok call | `input`, `cached_input`, `output` per model; modality-specific input rows (`audio_input`, `image_input`, `video_input`); **>200k tier** (`input_over_200k` 2×, `output_over_200k` 1.5×) on Pro models only; thinking tokens billed as output (`output_includes_thinking_tokens`) |")39L.append("| Cache write | free on models before GPT-5.6; **1.25× input** on GPT-5.6+ (`cache_write` row) | **1.25× input** for 5-minute TTL, **2× input** for 1-hour TTL | none — automatic prefix cache, no write charge | implicit cache: none; **explicit `cachedContents`: storage per 1M tokens per hour** (`cache_storage_hour`: $0.50 Flash 3.6–3.8, $1.00 most Flash, $4.50 Pro; 1.8× on priority) |")40L.append("| Cache read | model-specific `cached_input` (0.1× on GPT-5.6+, 0.25× on o3, 0.5× on gpt-4o…) | **0.1× input** (0.025× on Fable 5.1 / Mythos 5.1) | `cached_input` per model: 0.25× (grok-4.6 $0.50), 0.15× (grok-4.5 $0.30), 0.16× (grok-4.3 / 4.20 $0.20), 0.2× (grok-build $0.20) | `cached_input` = **0.1× input** on every priced model (e.g. gemini-3.8-flash $0.075, gemini-3.1-pro-preview $0.20) |")41L.append("| Batch | `batch` tier = 50 % of standard (own `cached_input`) | `batch_input`/`batch_output` = 50 % of standard; cache multipliers stack | **20 % off** (`batch_discount: 0.2`) on grok-4.3 / 4.20 / 4.20-multi-agent only; **not supported** on grok-4.6, grok-4.5, grok-build-0.1 (`batch_discount: null`); Imagine models accepted in Batch but billed at standard rates | `batch` tier = **50 %** of standard (`service:batch_api` multiplier 0.5) on text, image, TTS and embedding models; not on the free tier |")42L.append("| Cheaper async / best-effort tier | `flex` = batch rates on synchronous calls (slower, may 429) | none | none (Batch API is the only discount) | **`flex`** = 50 % of standard (`service:flex_inference`), 1–15 min target latency, sheddable (429 when capacity is shed); ten text models listed |")43L.append("| Faster / priority tier | `fast` = 2× standard (renamed from priority 2026-07-30; `fast_long_context` 4×/3×) | `speed: fast` (Opus 5 / 4.8 only) = 2× standard ($10 / $50) | **`service_tier: priority`** = **2×** on all token types (caching discount applied before the multiplier; billed only when the response echoes `service_tier: priority`) | **`priority`** tier = **1.8×** standard (`service:priority_inference`, docs: 75–100 % more); 0.3× the standard rate limit; graceful downgrade to standard when exceeded |")44L.append("| Reasoning tokens | billed as output (`usage.output_tokens_details.reasoning_tokens`) | billed as output (`usage.output_tokens_details.thinking_tokens`) | billed as output (`usage.completion_tokens_details.reasoning_tokens`); every Grok reasoning model spends them even on `reasoning_effort: low` | billed as output (`usageMetadata.thoughtsTokenCount`); Gemini 3.x thinking cannot be fully disabled on Flash/Pro (`thinkingLevel: minimal` where supported) |")45L.append("| Regional inference | +10 % on regional hosts for models released ≥ 2026-03-05 | `inference_geo: us` ×1.1 on all token dimensions (Claude 4.6+); Bedrock/Vertex regional +10 % | **`https://us.api.x.ai/v1`** = **1.1×** global token rates (grok-4.6 only in the live catalogue: $2.20 / $0.55 / $6.60); `eu-west-1.api.x.ai` LIVE_DISCOVERED, no separate price row | no regional price row on the Gemini Developer API (data residency is a Vertex AI feature); AI Studio usage free |")46L.append("| Free tier | none (Free usage tier = rate-limit tier, still billed) | none | none (prepaid credits; Tier 0 = $0 spend) | **yes** — `free_tier` rows on Flash / Flash-Lite / Live / TTS / transcribe / embedding / Gemma models: input & output $0, restrictive limits, content may be used to improve Google products; **Pro models and paid-only media models are not available on the free tier** (observed `limit: 0`) |")47L.append("| Tool-definition overhead | none published (definitions are input tokens) | published per model: tool-use system prompt 286–804 tokens, toolsets 4,500–6,600 tokens | none published (definitions are input tokens; server-side tool results count as input) | none published; `toolUsePromptTokenCount` reports tokens consumed by URL context / grounding results |")48L.append("| Per-image / per-second media | image models per token (≈ per image), Sora per second | — | Imagine **per image** ($0.02 / $0.04–$0.08 / $0.05) and **per second** of video ($0.05 / $0.08); voice per minute; TTS per 1M characters; STT per hour | image models per token with a **per-image** equivalent ($0.034–$0.24), Veo **per second** ($0.05–$0.60), Lyria **per song** ($0.04 / $0.08), Live/TTS/transcribe per 1M audio tokens (≈ per minute) |")4950# ---------------- OpenAI51L.append("\n## 2. OpenAI current text models (USD / 1M tokens)\n")52L.append("| Model | Status | Input | Cached in | Cache write | Output | Batch in / out | Flex in / out | Fast in / out | Long-context in / out (>272k) |")53L.append("|---|---|---|---|---|---|---|---|---|---|")54for mid in OAI:55 r = M.get(mid)56 if not r: continue57 p = r.get("pricing") or {}58 s, b, fl, fa, lc = (p.get("standard") or {}), (p.get("batch") or {}), (p.get("flex") or {}), (p.get("fast") or {}), (p.get("standard_long_context") or {})59 L.append(f"| `{mid}` | {st(r)} | {f(s.get('input'))} | {f(s.get('cached_input'))} | {f(s.get('cache_write'))} | {f(s.get('output'))} | {f(b.get('input'))} / {f(b.get('output'))} | {f(fl.get('input'))} / {f(fl.get('output'))} | {f(fa.get('input'))} / {f(fa.get('output'))} | {f(lc.get('input'))} / {f(lc.get('output'))} |")60L.append("\nNotes from the records: `gpt-5.6` is an alias of `gpt-5.6-sol` (promotional $4/$20 at least through 2026-11-21); pro models have no cached-input price (no caching); `chat-latest` has no batch/flex/fast tiers; `gpt-5.6-cyber` is ACCOUNT_RESTRICTED and priced at short-context rates. The `o3` record carries a model-page table ($1 / $0.25 / $4) that disagrees with its pricing-page standard row ($2 / $0.5 / $8) — flagged as a data inconsistency.\n")6162# ---------------- Anthropic63L.append("## 3. Anthropic current models (USD / 1M tokens)\n")64L.append("| Model | Status | Input | Cache write 5m | Cache write 1h | Cache read | Output | Batch in / out | Fast in / out | Min cacheable tokens |")65L.append("|---|---|---|---|---|---|---|---|---|---|")66for mid in ANT:67 r = M.get(mid)68 if not r: continue69 p = r.get("pricing") or {}70 fm = p.get("fast_mode") or {}71 c = r.get("capabilities") or {}72 L.append(f"| `{mid}` | {st(r)} | {f(p.get('input'))} | {f(p.get('cache_write_5m'))} | {f(p.get('cache_write_1h'))} | {f(p.get('cache_read'))} | {f(p.get('output'))} | {f(p.get('batch_input'))} / {f(p.get('batch_output'))} | {f(fm.get('input'))} / {f(fm.get('output'))} | {c.get('min_cacheable_tokens','—')} |")73L.append("\nNotes: Sonnet 5's introductory $2 / $10 was made permanent on 2026-08-10. Fable 5.1 / Mythos 5.1 cache reads are 0.025× ($0.25). Retired models (Opus 4.1 $15/$75, Sonnet 4 $3/$15, Haiku 3.5 $0.80/$4) remain priced on Bedrock/Vertex only. Priority Tier is no longer sold; Claude Platform on AWS and Foundry convert the same USD rates to CCUs at $0.01.\n")7475# ---------------- xAI76L.append("## 4. xAI current Grok text models (USD / 1M tokens)\n")77L.append("Standard = prompt < 200,000 tokens. **Long context = prompt ≥ 200,000 tokens: the higher rate applies to every token of that request** (input, cached, output). `image_input` equals the text input price on every model. Batch = 20 % off standard where supported. Priority = `service_tier: priority`, 2×. US regional = `https://us.api.x.ai/v1`, 1.1×.\n")78L.append("| Model | Status | Input | Cached in | Output | Long-ctx in / cached / out (≥200k) | Batch in / cached / out | Priority × | US regional × | Context |")79L.append("|---|---|---|---|---|---|---|---|---|---|")80for mid in XAI:81 r = M.get(mid)82 if not r: continue83 p = r.get("pricing") or {}84 s, lc = p.get("standard") or {}, p.get("long_context") or {}85 bd = p.get("batch_discount")86 if isinstance(bd, (int, float)) and bd:87 bcell = f"{f(s['input']*(1-bd))} / {f(s['cached_input']*(1-bd))} / {f(s['output']*(1-bd))} (−{int(bd*100)} %)"88 else:89 bcell = "not supported"90 pm = p.get("priority_multiplier"); pm = f"{pm}×" if isinstance(pm, (int, float)) else (pm or "—")91 us = p.get("us_regional_multiplier"); us = f"{us}×" if isinstance(us, (int, float)) else "—"92 L.append(f"| `{mid}` | {st(r)} | {f(s.get('input'))} | {f(s.get('cached_input'))} | {f(s.get('output'))} | {f(lc.get('input'))} / {f(lc.get('cached_input'))} / {f(lc.get('output'))} | {bcell} | {pm} | {us} | {r.get('context_window') or '—'} |")93L.append("\nNotes from the records: `grok-4.20-multi-agent-0309` is BETA and only reachable on `/v1/responses`; its `reasoning.effort` selects the agent count (low/medium = 4, high/xhigh = 16) and all agents' tokens are billed. `grok-build-0.1` (PREVIEW) is the coding model behind the `grok-code-fast-1` redirect. The retired ids `grok-3`, `grok-4-0709`, `grok-4-fast-*`, `grok-4-1-fast-*` **redirect to `grok-4.3` and are billed at grok-4.3 rates**. `grok-embedding-small` has no published price (ACCOUNT_RESTRICTED). Max output is not documented per model (`max_output: null` in every Grok record; Responses `max_output_tokens` defaults to 128,000).\n")9495L.append("### 4b. xAI media and voice prices\n")96L.append("| Model / service | Status | Price | Unit / notes |\n|---|---|---|---|")97for mid in ["grok-imagine-image","grok-imagine-image-2.0","grok-imagine-image-quality","grok-imagine-video","grok-imagine-video-1.5","grok-voice-think-fast-2.0","grok-voice-transcribe-2.0"]:98 r = M.get(mid); p = (r or {}).get("pricing") or {}99 if not r: continue100 if isinstance(p, str): L.append(f"| `{mid}` | {st(r)} | — | {p} |"); continue101 if "per_image_default" in p:102 mat = p.get("matrix")103 m = "; ".join(f"{x['quality']}/{x['resolution']} ${x['price']}" for x in mat) if mat else "single price"104 L.append(f"| `{mid}` | {st(r)} | ${f(p['per_image_default'])} per image | {m}; batch discount {p.get('batch_discount')} (standard rates in Batch) |")105 elif "per_second" in p:106 L.append(f"| `{mid}` | {st(r)} | ${f(p['per_second'])} per second of generated video | {p.get('note','')} |")107 elif "audio_per_minute" in p:108 L.append(f"| `{mid}` (speech-to-speech) | {st(r)} | ${f(p['audio_per_minute'])} per minute (${f(p['audio_per_hour'])} / h) + ${f(p['text_input_per_message'])} per text item | {p.get('note','')} |")109 elif "rest_per_hour" in p:110 L.append(f"| `{mid}` (speech-to-text) | {st(r)} | ${f(p['rest_per_hour'])} / hour REST (`POST /v1/stt`), ${f(p['streaming_per_hour'])} / hour streaming (`wss://api.x.ai/v1/stt`) | per hour of audio |")111tts = next((r for r in P if r["provider"]=="xai" and r["model_or_service"]=="text-to-speech"), None)112if tts: L.append(f"| text-to-speech (`POST /v1/tts`, `wss://api.x.ai/v1/tts`) | `DOCUMENTED` | ${f(tts['price'])} {tts['unit']} | {tts.get('effective_notes','')} |")113114# ---------------- Gemini115L.append("\n## 5. Gemini current text models (USD / 1M tokens)\n")116L.append("Standard tier for prompts ≤ 200k tokens; **>200k** columns only exist on Pro models (Flash models have one price for any prompt length). `Cached in` = implicit or explicit cache hit (0.1×). Storage = explicit `cachedContents` storage per 1M tokens per hour. Batch = 50 %, Flex = 50 %, Priority = 1.8×. Free = free-tier availability recorded on the pricing page.\n")117L.append("| Model | Status | Input | Cached in | Output | >200k in / cached / out | Batch in / out | Flex in / out | Priority in / out | Cache storage $/1M/h | Free tier |")118L.append("|---|---|---|---|---|---|---|---|---|---|---|")119for mid in GEM:120 r = M.get(mid)121 if not r: continue122 p = r.get("pricing") or {}123 if isinstance(p, str) or "tiers" not in p:124 ft = p.get("free_tier") if isinstance(p, dict) else p125 L.append(f"| `{mid}` | {st(r)} | free | — | free | — | — | — | — | — | {ft} |"); continue126 t = p["tiers"]; s = t.get("standard") or {}; b = t.get("batch") or {}; fl = t.get("flex") or {}; pr = t.get("priority") or {}127 lc = f"{f(s.get('input_over_200k'))} / {f(s.get('cached_input_over_200k'))} / {f(s.get('output_over_200k'))}" if s.get("input_over_200k") else "— (flat)"128 ft = str(p.get("free_tier") or "—")129 ft = ft if len(ft) <= 70 else ft[:67] + "…"130 L.append(f"| `{mid}` | {st(r)} | {f(s.get('input'))} | {f(s.get('cached_input'))} | {f(s.get('output'))} | {lc} | {f(b.get('input'))} / {f(b.get('output'))} | {f(fl.get('input'))} / {f(fl.get('output'))} | {f(pr.get('input'))} / {f(pr.get('output'))} | {f(s.get('cache_storage_hour'))} | {ft} |")131L.append("\nNotes from the records: gemini-3.8 / 3.7 / 3.6 Flash share one introductory price list ($0.75 / $0.075 / $3.75) **through 2026-12-31**, doubling on 2027-01-01 ($1.50 / $0.15 / $7.50, storage $1.00); gemini-3.5-flash is the higher-priced Flash ($1.50 / $9.00). Pro models (`gemini-3.1-pro-preview`, `gemini-2.5-pro`) are **paid-tier only** (ACCOUNT_RESTRICTED on this free-tier key). Gemini 2.5 models are 'no longer available to new users' (404 on generateContent). Gemma 4 models are free of charge with no paid tier. `gemini-2.5-flash` / `-pro` have separate `audio_input` rows ($1.00 / 1M); Gemini 3.x prices text, image, video and audio input alike. Google Search grounding: Gemini 3.x 5,000 free requests/month then $14 / 1k **search queries**; Gemini 2.5: 1,500 RPD free then $35 / 1k **grounded prompts**.\n")132133L.append("### 5b. Gemini media, voice and embedding prices\n")134L.append("| Model | Status | Price | Notes |\n|---|---|---|---|")135def gm(mid, how):136 r = M.get(mid)137 if not r: return138 p = r.get("pricing") or {}139 L.append(f"| `{mid}` | {st(r)} | {how(p)} | {(p.get('note') or p.get('free_tier') or '') if isinstance(p, dict) else p} |")140gm("gemini-3.1-flash-image", lambda p: f"text in ${f(p['input'])}, text out ${f(p['output_text'])}, image out ${f(p['output_image'])} / 1M → per image " + ", ".join(f"{k} ${v}" for k, v in p['per_image'].items()) + f"; batch image out ${f(p['batch']['output_image'])}")141gm("gemini-3.1-flash-lite-image", lambda p: f"text in ${f(p['input'])}, image out ${f(p['output_image'])} / 1M → " + ", ".join(f"{k} ${v}" for k, v in p['per_image'].items()))142gm("gemini-3-pro-image", lambda p: f"text in ${f(p['input'])}, image in ${f(p['input_per_image'])} per image, image out ${f(p['output_image'])} / 1M → " + ", ".join(f"{k} ${v}" for k, v in p['per_image'].items()) + f"; priority image out ${f(p['priority']['output_image'])}")143gm("gemini-2.5-flash-image", lambda p: f"${f(p['output_per_image'])} per image (${f(p['output_image'])} / 1M, 1,290 tok/image); batch ${f(p['batch']['output_per_image'])}")144gm("veo-3.1-generate-preview", lambda p: "per second: " + ", ".join(f"{k} ${v}" for k, v in p['per_second'].items()))145gm("veo-3.1-fast-generate-preview", lambda p: "per second: " + ", ".join(f"{k} ${v}" for k, v in p['per_second'].items()))146gm("veo-3.1-lite-generate-preview", lambda p: "per second: " + ", ".join(f"{k} {('$'+str(v)) if not isinstance(v,str) else v}" for k, v in p['per_second'].items()))147gm("gemini-omni-1.1-flash", lambda p: f"text in ${f(p['input'])}, text out ${f(p['output_text'])}, video out ${f(p['output_video'])} / 1M")148gm("lyria-3.5", lambda p: f"${f(p['per_request'])} {p['unit']}")149gm("lyria-3-clip-preview", lambda p: f"${f(p['per_request'])} {p['unit']}")150gm("gemini-3.1-flash-tts-preview", lambda p: f"text in ${f(p['input_text'])}, audio out ${f(p['output_audio'])} / 1M (25 audio tok/s ≈ ${f(p['output_audio']*25*60/1e6)} / min); batch ${f(p['batch']['input_text'])} / ${f(p['batch']['output_audio'])}")151gm("gemini-2.5-flash-preview-tts", lambda p: f"text in ${f(p['input_text'])}, audio out ${f(p['output_audio'])} / 1M")152gm("gemini-3.8-live", lambda p: f"text in ${f(p['input_text'])}, audio in ${f(p['input_audio'])} (≈ ${f(p['input_audio_per_min'])} / min), image/video in ${f(p['input_image_video'])}, text out ${f(p['output_text'])}, audio out ${f(p['output_audio'])} (≈ ${f(p['output_audio_per_min'])} / min)")153gm("gemini-2.5-flash-native-audio-preview-12-2025", lambda p: f"text in ${f(p['input_text'])}, audio/video in ${f(p['input_audio_video'])}, text out ${f(p['output_text'])}, audio out ${f(p['output_audio'])}")154gm("gemini-3.5-transcribe", lambda p: f"audio in ${f(p['input_audio'])} (≈ ${f(p['input_audio_per_min'])} / min), text out ${f(p['output_text'])} (≈ ${f(p['output_text_per_min'])} / min)")155gm("gemini-3.5-transcribe-live", lambda p: f"audio in ${f(p['input_audio'])} (≈ ${f(p['input_audio_per_min'])} / min), text out ${f(p['output_text'])}")156gm("gemini-3.5-live-translate-preview", lambda p: f"audio in ${f(p['input_audio'])}, audio out ${f(p['output_audio'])} / 1M")157gm("gemini-embedding-2", lambda p: f"text ${f(p['input_text'])}, image ${f(p['input_image'])} (${f(p['input_image_per_image'])} per image), audio ${f(p['input_audio'])} (${f(p['input_audio_per_second'])} / s), video ${f(p['input_video'])} (${f(p['input_video_per_frame'])} / frame) / 1M; batch text ${f(p['batch']['input_text'])}")158gm("gemini-embedding-001", lambda p: "—")159160# ---------------- cost models161L.append("\n## 6. Cost model A — 1M input tokens + 100k output tokens, no cache\n")162L.append("Formula: `cost = 1.0 × input_price + 0.1 × output_price`, treating the 1M input tokens as **aggregate volume at the standard (short-prompt) rate**. The last column shows what a **single request whose prompt crosses the long-context threshold** costs instead (OpenAI >272k, xAI ≥200k applied to all tokens, Gemini Pro >200k; Anthropic and Gemini Flash have no threshold). Tiers shown where the model record lists them.\n")163L.append("| Provider | Model | Standard | Batch | Flex | Fast / Priority | Long-context request |")164L.append("|---|---|---|---|---|---|---|")165for mid in OAI:166 r = M.get(mid); p = (r or {}).get("pricing") or {}167 if not r or not p.get("standard"): continue168 s, b, fl, fa, lc = p.get("standard") or {}, p.get("batch") or {}, p.get("flex") or {}, p.get("fast") or {}, p.get("standard_long_context") or {}169 L.append(f"| openai | `{mid}` | {usd(cm(s.get('input'), s.get('output')))} | {usd(cm(b.get('input'), b.get('output')))} | {usd(cm(fl.get('input'), fl.get('output')))} | {usd(cm(fa.get('input'), fa.get('output')))} | {usd(cm(lc.get('input'), lc.get('output')))} |")170for mid in ANT:171 r = M.get(mid); p = (r or {}).get("pricing") or {}172 if not r or not isinstance(p, dict) or p.get("input") is None: continue173 fm = p.get("fast_mode") or {}174 L.append(f"| anthropic | `{mid}` | {usd(cm(p.get('input'), p.get('output')))} | {usd(cm(p.get('batch_input'), p.get('batch_output')))} | — | {usd(cm(fm.get('input'), fm.get('output')))} | same as standard (no premium) |")175for mid in XAI:176 r = M.get(mid); p = (r or {}).get("pricing") or {}177 if not r or not p.get("standard"): continue178 s, lc = p["standard"], p.get("long_context") or {}179 bd = p.get("batch_discount")180 bcost = cm(s["input"]*(1-bd), s["output"]*(1-bd)) if isinstance(bd, (int, float)) and bd else None181 pm = p.get("priority_multiplier"); pm = pm if isinstance(pm, (int, float)) else 2.0182 L.append(f"| xai | `{mid}` | {usd(cm(s['input'], s['output']))} | {usd(bcost) if bcost else 'not supported'} | — | {usd(cm(s['input']*pm, s['output']*pm))} (priority) | {usd(cm(lc.get('input'), lc.get('output')))} |")183for mid in GEM:184 r = M.get(mid); p = (r or {}).get("pricing") or {}185 if not r or not isinstance(p, dict) or "tiers" not in p: continue186 t = p["tiers"]; s = t.get("standard") or {}; b = t.get("batch") or {}; fl = t.get("flex") or {}; pr = t.get("priority") or {}187 lcv = cm(s.get("input_over_200k"), s.get("output_over_200k")) if s.get("input_over_200k") else None188 L.append(f"| gemini | `{mid}` | {usd(cm(s.get('input'), s.get('output')))} | {usd(cm(b.get('input'), b.get('output')))} | {usd(cm(fl.get('input'), fl.get('output')))} | {usd(cm(pr.get('input'), pr.get('output')))} (priority) | {usd(lcv) if lcv else 'same as standard (flat)'} |")189L.append("\nGemma 4 (`gemma-4-31b-it`, `gemma-4-26b-a4b-it`) and every Gemini free-tier row cost $0 at list price but carry free-tier limits and data-use terms; they are omitted from the arithmetic.\n")190191L.append("## 7. Cost model B — the same workload with a 900k-token cached prefix (steady state)\n")192L.append("Assumptions: each request = 900k tokens **read** from cache + 100k fresh input + 100k output; the initial cache **write** is amortised over N = 10 requests. Formula per request: `0.9 × cache_read + 0.1 × input + 0.1 × output + write_cost / N`. Write cost: OpenAI pre-5.6 free, GPT-5.6+ `cache_write` × 0.9; Anthropic 5-minute write (1.25×) × 0.9; xAI none (automatic cache); Gemini implicit caching none, **explicit `cachedContents` = storage of 0.9M tokens for one hour** (`cache_storage_hour` × 0.9) amortised over N. A 900k prefix exceeds xAI's 200k threshold, so xAI is shown at **long-context** rates (the only rates that apply to such a request); Gemini Pro at its >200k rates, Gemini Flash flat.\n")193L.append("| Provider | Model | Uncached (A, same rates) | Cached steady-state (N=10) | Saving | Write / storage cost used |")194L.append("|---|---|---|---|---|---|")195N = 10196for mid in OAI:197 r = M.get(mid); p = (r or {}).get("pricing") or {}198 s = p.get("standard") or {}199 if not r or s.get("input") is None or s.get("cached_input") is None: continue200 w = s.get("cache_write") or 0.0201 a = cm(s["input"], s["output"])202 bcost = 0.9*s["cached_input"] + 0.1*s["input"] + 0.1*s["output"] + 0.9*w/N203 L.append(f"| openai | `{mid}` | {usd(a)} | {usd(bcost)} | {100*(1-bcost/a):.0f} % | {f(w) if w else 'free'} |")204for mid in ANT:205 r = M.get(mid); p = (r or {}).get("pricing") or {}206 if not r or not isinstance(p, dict) or p.get("input") is None or p.get("cache_read") is None: continue207 a = cm(p["input"], p["output"])208 bcost = 0.9*p["cache_read"] + 0.1*p["input"] + 0.1*p["output"] + 0.9*p["cache_write_5m"]/N209 L.append(f"| anthropic | `{mid}` | {usd(a)} | {usd(bcost)} | {100*(1-bcost/a):.0f} % | {f(p['cache_write_5m'])} (5m write) |")210for mid in XAI:211 r = M.get(mid); p = (r or {}).get("pricing") or {}212 lc = (p or {}).get("long_context") or {}213 if not r or lc.get("input") is None: continue214 a = cm(lc["input"], lc["output"])215 bcost = 0.9*lc["cached_input"] + 0.1*lc["input"] + 0.1*lc["output"]216 L.append(f"| xai | `{mid}` | {usd(a)} (long-context rates) | {usd(bcost)} | {100*(1-bcost/a):.0f} % | none (automatic; cached rate {f(lc['cached_input'])}) |")217for mid in GEM:218 r = M.get(mid); p = (r or {}).get("pricing") or {}219 if not r or not isinstance(p, dict) or "tiers" not in p: continue220 s = p["tiers"].get("standard") or {}221 if s.get("cached_input") is None: continue222 inp = s.get("input_over_200k") or s["input"]; out = s.get("output_over_200k") or s["output"]; cr = s.get("cached_input_over_200k") or s["cached_input"]223 a = cm(inp, out)224 storage = (s.get("cache_storage_hour") or 0.0) * 0.9225 bcost = 0.9*cr + 0.1*inp + 0.1*out + storage/N226 L.append(f"| gemini | `{mid}` | {usd(a)}{' (>200k rates)' if s.get('input_over_200k') else ''} | {usd(bcost)} | {100*(1-bcost/a):.0f} % | implicit: none; explicit: storage {f(s.get('cache_storage_hour'))} /1M/h × 0.9 |")227228L.append("\n### Caching break-even by provider\n")229L.append("| Provider | Write cost | Read multiplier | Pays for itself on… | Notes |\n|---|---|---|---|---|")230L.append("| OpenAI (pre-5.6) | free | model-specific 0.25×–0.5× | first hit | implicit; `prompt_cache_key` routing; 5–10 min in-memory, `24h` retention option |")231L.append("| OpenAI (GPT-5.6+, GPT-6) | 1.25× | 0.1× | second use (1.25 + 0.1 = 1.35 < 2.0) | explicit `prompt_cache_breakpoint` optional; `ttl: 30m` |")232L.append("| Anthropic | 1.25× (5 m) / 2× (1 h) | 0.1× (0.025× Fable 5.1 / Mythos 5.1) | second use (5 m); 1 h only if the gap between calls exceeds 5 min | explicit `cache_control`; hits refresh TTL for free |")233L.append("| xAI | none | 0.15×–0.25× | first hit | automatic; `prompt_cache_key` / `x-grok-conv-id` for sticky routing; no TTL or minimum documented; cached tokens still count toward TPM |")234L.append("| Gemini | implicit: none; explicit: storage per token-hour | 0.1× | implicit: first hit (≥ 4,096-token prompts on 3.x, 2,048 on 2.5); explicit: when `0.9 × read + storage_hour × hours / N < 0.9 × input` — for gemini-3.8-flash one hour of storage ($0.45 per 0.9M) is recovered after a single hit ($0.675 − $0.0675 saved) | explicit cache TTL default 1 h (`ttl` / `expireTime`), min 1,024 tokens (live), free tier `limit: 0` |")235236L.append("\n## 8. Batch vs synchronous — the same 100k-request job\n")237L.append("| | OpenAI Batch API | Anthropic Message Batches | xAI Batch API | Gemini Batch API |\n|---|---|---|---|---|")238L.append("| Discount | 50 % on input, cached input and output (`batch` tier) | 50 % on input, output, cache writes and cache reads (stacks with caching) | **20 %** on input, output, cached and reasoning tokens — grok-4.3 / 4.20 / 4.20-multi-agent only; grok-4.6, grok-4.5, grok-build: not supported; Imagine models: standard rates | **50 %** (`service:batch_api`) on text, image, TTS, embedding models; Flex gives the same 50 % synchronously (best-effort) |")239L.append("| Window | 24 h (`completion_window: \"24h\"`, only value); output retained 30 days | 24 h expiry (`expires_at`); results 29 days | no completion window documented; results listed via `GET /v1/batches/{id}/results`; batch requests do not count toward rate limits; video URLs in results expire after 1 h | target 24 h ('usually much faster'); poll `GET /v1beta/batches/{id}`; concurrent batch jobs 100; enqueued-token caps per model and tier (e.g. Tier 1: 3M gemini-3.8-flash, 5M gemini-3.1-pro-preview) |")240L.append("| How requests are supplied | JSONL file (`custom_id`, `method`, `url`, `body`) → `POST /v1/batches {input_file_id, endpoint}`; 50,000 requests / 200 MB; one endpoint + one model per file | inline `requests[] {custom_id, params}` → `POST /v1/messages/batches`; 100,000 requests / 256 MB; models mixed freely | `POST /v1/batches` then `POST /v1/batches/{id}/requests` (inline chat/responses/image/video bodies) or JSONL upload; `:cancel`; no DELETE (405) | `POST /v1beta/models/{model}:batchGenerateContent` with inline `requests[]` or a JSONL file from the Files API (2 GB / 20 GB storage); `:asyncBatchEmbedContent` for embeddings; `/v1beta/openai/batches` compatibility route |")241L.append("| Example: 100k requests × (2k in + 300 out) on the flagship mid-tier | gpt-5.6-sol batch: 200M × $2 + 30M × $10 = **$700** (vs $1,400 standard) | claude-sonnet-5 batch: 200M × $1 + 30M × $5 = **$350** (vs $700 standard) | grok-4.3 batch: 200M × $1.00 + 30M × $2.00 = **$260** (vs $325 standard); grok-4.6 has no batch: **$430** standard | gemini-3.8-flash batch: 200M × $0.375 + 30M × $1.875 = **$131.25** (vs $262.50 standard); gemini-3.1-pro-preview batch: 200M × $1 + 30M × $6 = **$380** (vs $760) |")242L.append("| Example: same job on the small models | gpt-5.4-mini batch: 200M × $0.375 + 30M × $2.25 = **$142.50** | claude-haiku-4-5 batch: 200M × $0.5 + 30M × $2.5 = **$175** | grok-build-0.1: no batch, **$260** standard | gemini-3.5-flash-lite batch: 200M × $0.15 + 30M × $1.25 = **$67.50** (vs $135) |")243244L.append("\n## 9. Tool and platform prices (from `generated/pricing.json`)\n")245L.append("| Item | OpenAI | Anthropic | xAI | Gemini |\n|---|---|---|---|---|")246L.append("| Web search | $10 / 1k calls (`web_search`, image search); preview $25 / 1k on non-reasoning models | $10 / 1k searches (errors not billed) | **$5 / 1k successful calls** (`web_search`, image search included; `view_image` results billed as image tokens) | Google Search grounding: Gemini 3.x **5,000 free queries / month** (shared) then **$14 / 1k search queries**; Gemini 2.5: 1,500 RPD free then **$35 / 1k grounded prompts**; free tier 500 RPD (not Pro) |")247L.append("| Social / vertical search | — | — | **`x_search`** $5 / 1k calls until 2026-09-21 12:00 PT, then **$5 / 1k posts fetched + $10 / 1k user profiles fetched** | **Google Maps grounding**: Gemini 3.x 5,000 free prompts / month then $14 / 1k queries; Gemini 2.5: $25 / 1k grounded prompts (1,500 RPD free; 10,000 for Pro) |")248L.append("| Web fetch / URL context | — | $0 per fetch (content tokens only) | — (web search `open_page` action) | `urlContext` free of charge; retrieved content billed as input tokens (`toolUsePromptTokenCount`); ≤ 20 URLs / request |")249L.append("| Code execution | container session: 1 GB $0.03 · 4 GB $0.12 · 16 GB $0.48 · 64 GB $1.92 per 20 min (per-minute, 5-min minimum since 2026-06-02); shared with hosted shell | 1,550 free container-hours / org / month, then $0.05 / container-hour (5-min minimum); free when `web_search`/`web_fetch` 20260209+ is in the request | **$5 / 1k calls** (`code_interpreter` / `code_execution`) + tokens | **no per-call fee** (`codeExecution`); generated code/results billed as output then as input when re-read; 30 s per execution |")250L.append("| File search / RAG | $2.50 / 1k calls + $0.10 / GB / day storage (1 GB free) | — | **`file_search` / `collections_search` $2.50 / 1k calls** + tokens; collections storage **$0.10 / GiB / day**, downloads $0.20 / GiB; `attachment_search` (implicit, files attached to messages) **$10 / 1k calls** | **`fileSearch`**: indexing embeddings **$0.15 / 1M tokens** once; storage and query-time embeddings free; retrieved chunks billed as input tokens; store size by tier (Free 1 GB … Tier 3 1 TB) |")251L.append("| MCP | tokens only (`mcp` tool) | tokens only (`mcp_servers`, beta) | tokens only (`mcp` tool; tool outputs count as input) | not documented (`mcpServers` UNVERIFIED on generateContent; Interactions `mcp_server` type) |")252L.append("| Computer use | tokens (`computer` tool) | tokens (toolsets ≈ 4,500 tokens of definitions) | — | tokens only (`computerUse`, PREVIEW; screenshots are image input tokens); not on the free tier |")253L.append("| Files API storage | not priced separately (`purpose=batch` files expire 30 d) | not priced on the pricing page | **$0.025 / GiB / day** storage, $0.20 / GiB downloads; TTL via `expires_after` | free (2 GB / file, 20 GB / project, 48 h TTL) |")254L.append("| Managed agents runtime | model tokens at Responses rates + container rates (no session fee) | $0.08 per session-hour (`usage.active_seconds`) + model tokens + $10 / 1k web searches | xAI Responses agentic loop: tokens + per-call tool fees; `usage-guideline-violation` fee **$0.05 / request** when a violation is caught before generation | Interactions agents (Deep Research, Antigravity, custom managed agents): model inference at list rates incl. intermediate/reasoning tokens + tool fees; **sandbox compute not billed during preview** |")255L.append("| Realtime / voice | gpt-realtime-2.1: audio $32 in / $64 out, text $4 / $24, image $5 per 1M; mini $10 / $20 audio; gpt-live-1 $0.05 / min | — | **grok-voice-think-fast-2.0 $0.08 / min** ($4.80 / h) audio sent or received + $0.004 per text item; STT $0.10 / h REST, $0.20 / h streaming; TTS **$15 / 1M characters** | **gemini-3.8-live**: audio in $3 / 1M (≈ $0.005 / min), audio out $12 / 1M (≈ $0.018 / min), text $0.75 / $4.50; TTS `gemini-3.1-flash-tts-preview` $1 text in / $20 audio out per 1M (≈ $0.03 / min); `gemini-3.5-transcribe` ≈ $0.005 / min blended; free tier available |")256L.append("| Image generation | gpt-image-2 / 2.5: text in $5, image in $8, image out $30 per 1M tokens (≈ $0.006–$0.21 per image); batch 50 % | — | **grok-imagine-image $0.02 / image**, **grok-imagine-image-2.0 $0.04–$0.08** (quality × resolution), grok-imagine-image-quality $0.05 (DEPRECATED → 2026-11-02); `image_generation` tool billed at these rates | **gemini-3.1-flash-image** $0.045 (0.5K) – $0.151 (4K) per image (image out $60 / 1M), **gemini-3.1-flash-lite-image** $0.034 (1K), **gemini-3-pro-image** $0.134 (1K/2K) / $0.24 (4K); batch 50 % |")257L.append("| Video | sora-2 $0.10 / s (720p), sora-2-pro $0.30–$0.70 / s — shutting down 2026-09-24 | — | **grok-imagine-video $0.05 / s**, **grok-imagine-video-1.5 $0.08 / s** (1080p, reference-to-video) | **Veo 3.1** $0.40 / s (720p/1080p), $0.60 / s (4K); **Veo 3.1 Fast** $0.10 / $0.12 / $0.30; **Veo 3.1 Lite** $0.05 / $0.08 (no 4K); `gemini-omni-1.1-flash` video out $17.50 / 1M tokens (≈ $0.10 / s) |")258L.append("| Music | — | — | — | **lyria-3.5 $0.08 / song**, lyria-3-clip-preview $0.04 / 30-s clip, lyria-3-pro-preview $0.08; Lyria RealTime (experimental) unpriced |")259L.append("| Embeddings | text-embedding-3-small $0.02, -large $0.13 per 1M | — | `grok-embedding-small`: no published price (ACCOUNT_RESTRICTED) | **gemini-embedding-2** text $0.20 / 1M (batch $0.10), image $0.45, audio $6.50, video $12 per 1M; free tier available; gemini-embedding-001 unpriced (DEPRECATED → 2028-05-14) |")260L.append("| Moderation | free (`omni-moderation-latest`) | — (built-in refusals) | — (built-in; usage-guideline violations still billed + $0.05 fee) | — (`safetySettings` thresholds, free) |")261L.append("| Token counting | free (`POST /v1/responses/input_tokens`) | free (`POST /v1/messages/count_tokens`, own RPM bucket) | free (`POST /v1/tokenize-text`; no `usage`/cost ticks returned) | free (`POST /v1beta/models/{model}:countTokens`) |")262L.append("| Data residency | +10 % regional hosts (models ≥ 2026-03-05) | ×1.1 `inference_geo: us`; +10 % Bedrock/Vertex regional | ×1.1 on `us.api.x.ai` (grok-4.6) | not priced on the Developer API (Vertex AI feature) |")263L.append("| Per-request overhead | none published | tool-use system prompt: e.g. Opus 5 406 tokens (any/tool) / 286 (auto); `bash_20250124` +244–325; `text_editor` +700; `computer_toolset_20260801` ≈4,500; `browser_toolset_20260801` ≈6,600 | none published; `cost_in_usd_ticks` in every `usage` object (1 tick = $1e-10) makes the effective cost observable per call | none published; `usageMetadata` breaks tokens down by modality (`promptTokensDetails[]`), `thoughtsTokenCount`, `toolUsePromptTokenCount`, `cachedContentTokenCount` |")264265L.append("\n## 10. Reading prices programmatically\n")266L.append("```bash\n# OpenAI standard row for one model\njq '.[] | select(.id==\"gpt-5.6-sol\") | .pricing.standard' generated/models.json\n# Anthropic per-dimension rows\njq '[.[] | select(.provider==\"anthropic\" and .model_or_service==\"claude-sonnet-5\")]' generated/pricing.json\n# xAI standard + long-context blocks and the live price ticks (1 tick = USD 1e-10 per token)\njq '.[] | select(.id==\"grok-4.6\") | .pricing | {standard, long_context, batch_discount, priority_multiplier, us_regional_multiplier}' generated/models.json\n# Gemini tiers (standard/batch/flex/priority) and free-tier note\njq '.[] | select(.id==\"gemini-3.8-flash\") | .pricing | {tiers, free_tier, from_2027_01_01}' generated/models.json\n# every tool / service price row across the four providers\njq '[.[] | select(.model_or_service|test(\"^(tool:|service:|agent:|rule:)\"))]' generated/pricing.json\n```\n")267L.append("Related: [models](models.md) · [caching and reasoning](caching-and-reasoning.md) · [features](features.md) · [realtime and media](realtime-and-media.md) · [FAQ](../faq.md).\n")268open(f"{ROOT}/docs/comparisons/pricing.md", "w").write("\n".join(L))269print("pricing.md written", len(L), "lines")270