SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
11.8 KB · 87 lines json
Raw Blame History
1{2 "record_type": "rate_limits",3 "provider": "gemini",4 "generated_at": "2026-09-18",5 "status": ["DOCUMENTED", "LIVE_DISCOVERED"],6 "sources": [7  {"url": "https://ai.google.dev/gemini-api/docs/rate-limits", "retrieved_at": "2026-09-18"},8  {"url": "https://ai.google.dev/gemini-api/docs/billing", "retrieved_at": "2026-09-18"},9  {"url": "https://ai.google.dev/gemini-api/docs/priority-inference", "retrieved_at": "2026-09-18"},10  {"url": "https://ai.google.dev/gemini-api/docs/flex-inference", "retrieved_at": "2026-09-18"},11  {"url": "https://ai.google.dev/gemini-api/docs/batch-api", "retrieved_at": "2026-09-18"},12  {"url": "https://ai.google.dev/gemini-api/docs/live-api/session", "retrieved_at": "2026-09-18"},13  {"url": "https://ai.google.dev/gemini-api/docs/troubleshooting", "retrieved_at": "2026-09-18"},14  {"url": "https://aistudio.google.com/rate-limit", "retrieved_at": null, "note": "per-model RPM/TPM/RPD tables live here (account-specific UI), not in the docs page"}15 ],16 "documented": {17  "principles": [18   "Three dimensions: requests per minute (RPM), input tokens per minute (TPM), requests per day (RPD); exceeding any one of them returns 429 RESOURCE_EXHAUSTED.",19   "Limits are applied per Google Cloud project, not per API key. RPD quotas reset at midnight Pacific time.",20   "Model-specific extra dimensions exist: images per minute (IPM) for image-generation models, tokens per day (TPD) for some models.",21   "Experimental and preview models have more restrictive limits.",22   "Since 2026 the docs page no longer publishes the per-model RPM/TPM/RPD matrix; it points to the AI Studio rate-limit page. 'Specified rate limits are not guaranteed and actual capacity may vary.'",23   "Tier upgrades Free -> Tier 1 are instant once billing is linked; later upgrades take effect within ~10 minutes; upgrade can be denied 'based on other factors'."24  ],25  "usage_tiers": [26   {"tier": "Free", "qualification": "Active project or free trial", "billing_tier_spend_cap": null, "spend_rate_limit_per_10_min": null,27    "notes": "Free of charge on many models; content may be used to improve Google products (terms: Unpaid Services). Pro models (gemini-3.1-pro-preview, gemini-2.5-pro) and paid-only media models are not available (observed limit 0)."},28   {"tier": "Tier 1", "qualification": "Set up and link an active Cloud Billing account", "billing_tier_spend_cap": "$250", "spend_rate_limit_per_10_min": "$10"},29   {"tier": "Tier 2", "qualification": "Paid $100 cumulative on Google Cloud + 3 days since first successful payment", "billing_tier_spend_cap": "$2,000", "spend_rate_limit_per_10_min": "$50"},30   {"tier": "Tier 3", "qualification": "Paid $1,000 cumulative + 30 days since first successful payment", "billing_tier_spend_cap": "$20,000 - $100,000+", "spend_rate_limit_per_10_min": "$200"}31  ],32  "spend_based_rate_limits": "Rolling 10-minute window per tier ($10 / $50 / $200); exceeding it returns 429 RESOURCE_EXHAUSTED; whether it applies depends on billing history and account standing.",33  "priority_inference": {"rate_limit": "0.3x the standard rate limit for each model and tier", "overflow": "graceful server-side downgrade to Standard processing instead of failing", "pricing": "75-100% more than Standard (tables show 1.8x)"},34  "flex_inference": {"rate_limit": "own limits; best-effort, sheddable capacity", "latency": "minutes (1-15 min target)", "errors": "429 RESOURCE_EXHAUSTED when capacity is shed; clients must retry (docs: 'Implement retries', per-request timeouts)", "pricing": "50% discount",35                     "supported_models": ["gemini-3.8-flash", "gemini-3.7-flash", "gemini-3.6-flash", "gemini-3.5-flash-lite", "gemini-3.5-flash", "gemini-3.1-flash-lite", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-2.5-pro", "gemini-2.5-flash"]},36  "batch_api": {37   "concurrent_batch_requests": 100, "input_file_size_limit": "2 GB", "file_storage_limit": "20 GB", "turnaround": "target 24 hours (usually much faster)", "pricing": "50% of standard",38   "enqueued_tokens_per_model": {39    "note": "Maximum tokens enqueued across all active batch jobs for a given model.",40    "tier_1": {"gemini-3.1-pro-preview": 5000000, "gemini-3.5-flash-lite": 10000000, "gemini-3.8-flash": 3000000, "gemini-3.7-flash": 3000000, "gemini-3.1-flash-lite": 10000000, "gemini-3.1-flash-lite-preview": 10000000, "gemini-3.6-flash": 3000000, "gemini-3.5-flash": 3000000, "gemini-2.5-pro": 5000000, "gemini-2.5-pro-preview-tts": 25000, "gemini-2.5-flash": 3000000, "gemini-2.5-flash-preview": 3000000, "gemini-2.5-flash-image-preview": 3000000, "gemini-2.5-flash-preview-tts": 100000, "gemini-2.5-flash-lite": 10000000, "gemini-2.5-flash-lite-preview": 10000000, "gemini-2.0-flash": 10000000, "gemini-2.0-flash-image": 3000000, "gemini-2.0-flash-lite": 10000000, "gemini-3.1-flash-image-preview": 1000000, "gemini-3.1-flash-lite-image": 2000000, "gemini-3-pro-image-preview": 2000000, "gemini-embedding": 500000},41    "tier_2": {"gemini-3.1-pro-preview": 500000000, "gemini-3.5-flash-lite": 500000000, "gemini-3.1-flash-lite": 500000000, "gemini-3.1-flash-lite-preview": 500000000, "gemini-3.8-flash": 400000000, "gemini-3.7-flash": 400000000, "gemini-3.6-flash": 400000000, "gemini-3.5-flash": 400000000, "gemini-2.5-pro": 500000000, "gemini-2.5-pro-preview-tts": 100000, "gemini-2.5-flash": 400000000, "gemini-2.5-flash-preview": 400000000, "gemini-2.5-flash-image-preview": 400000000, "gemini-2.5-flash-preview-tts": 100000, "gemini-2.5-flash-lite": 500000000, "gemini-2.5-flash-lite-preview": 500000000, "gemini-2.0-flash": 1000000000, "gemini-2.0-flash-image": 400000000, "gemini-2.0-flash-lite": 1000000000, "gemini-3.1-flash-image-preview": 250000000, "gemini-3.1-flash-lite-image": 270000000, "gemini-3-pro-image-preview": 270000000, "gemini-embedding": 5000000},42    "tier_3": {"gemini-3.1-pro-preview": 1000000000, "gemini-3.5-flash-lite": 1000000000, "gemini-3.1-flash-lite": 1000000000, "gemini-3.1-flash-lite-preview": 1000000000, "gemini-3.8-flash": 1000000000, "gemini-3.7-flash": 1000000000, "gemini-3.6-flash": 1000000000, "gemini-3.5-flash": 1000000000, "gemini-2.5-pro": 1000000000, "gemini-2.5-pro-preview-tts": 1000000, "gemini-2.5-flash": 1000000000, "gemini-2.5-flash-preview": 1000000000, "gemini-2.5-flash-image-preview": 1000000000, "gemini-2.5-flash-preview-tts": 4000000, "gemini-2.5-flash-lite": 1000000000, "gemini-2.5-flash-lite-preview": 1000000000, "gemini-2.0-flash": 5000000000, "gemini-2.0-flash-image": 1000000000, "gemini-2.0-flash-lite": 5000000000, "gemini-3.1-flash-image-preview": 750000000, "gemini-3.1-flash-lite-image": 1000000000, "gemini-3-pro-image-preview": 1000000000, "gemini-embedding": 10000000},43    "docs_note": "The batch table still names shut-down models (2.0 Flash, 2.5 previews) and preview image ids; the GA image ids (gemini-3.1-flash-image, gemini-3-pro-image) are not listed - assume the preview row applies."44   }45  },46  "live_api_sessions": {47   "session_lifetime": "audio-only 15 minutes, audio+video 2 minutes without context-window compression; with contextWindowCompression sessions can run longer",48   "connection_lifetime": "~10 minutes per WebSocket connection; use sessionResumption handles (valid 2 h after the last session termination) to continue",49   "ephemeral_tokens": "short-lived tokens for client-to-server WebSocket connections (POST /v1beta/auth_tokens); Live API only",50   "note": "Live API concurrency/session-count limits are not published on the rate-limits page."51  },52  "grounding_free_quotas": {53   "google_search_gemini_3": "5,000 free search requests per month shared across all Gemini 3.x models, then $14 per 1,000 requests",54   "google_search_gemini_2_5": "1,500 RPD free (shared Flash/Flash-Lite; free tier 500 RPD), then $35 per 1,000 grounded prompts",55   "google_maps_gemini_3": "5,000 prompts per month free shared, then $14 per 1,000 search queries",56   "google_maps_gemini_2_5": "1,500 RPD free (10,000 RPD for Pro), then $25 per 1,000 grounded prompts"57  },58  "model_parameter_ranges_troubleshooting_page": {"candidateCount": "1-8", "temperature": "0.0-1.0 (troubleshooting page; live Model.maxTemperature=2 for most Gemini models, 1 for image models)", "topP": "0.0-1.0"},59  "increase_requests": "Paid-tier increase form https://forms.gle/ETzX94k8jf7iSotH9 ('no guarantees')."60 },61 "error_semantics_429": {62  "http_status": 429, "status": "RESOURCE_EXHAUSTED",63  "message_pattern": "You exceeded your current quota, please check your plan and billing details. ... * Quota exceeded for metric: generativelanguage.googleapis.com/<metric>, limit: <n>, model: <model> ... Please retry in <seconds>s.",64  "details": [65   {"@type": "type.googleapis.com/google.rpc.Help", "links": [{"description": "Learn more about Gemini API quotas", "url": "https://ai.google.dev/gemini-api/docs/rate-limits"}]},66   {"@type": "type.googleapis.com/google.rpc.QuotaFailure", "violations": [{"quotaMetric": "generativelanguage.googleapis.com/generate_content_free_tier_input_token_count", "quotaId": "GenerateContentInputTokensPerModelPerDay-FreeTier", "quotaDimensions": {"location": "global", "model": "gemini-3.1-pro"}}, {"quotaMetric": "generativelanguage.googleapis.com/generate_content_free_tier_requests", "quotaId": "GenerateRequestsPerDayPerProjectPerModel-FreeTier"}, {"quotaMetric": "generativelanguage.googleapis.com/generate_content_free_tier_requests", "quotaId": "GenerateRequestsPerMinutePerProjectPerModel-FreeTier"}]}67  ],68  "quota_metrics_observed": ["generativelanguage.googleapis.com/generate_content_free_tier_input_token_count", "generativelanguage.googleapis.com/generate_content_free_tier_requests"],69  "quota_ids_observed": ["GenerateContentInputTokensPerModelPerDay-FreeTier", "GenerateRequestsPerDayPerProjectPerModel-FreeTier", "GenerateRequestsPerMinutePerProjectPerModel-FreeTier"],70  "retry_hint": "The retry delay is only in the message text ('Please retry in 54.22098241s.'); no Retry-After header was returned.",71  "retryable": "yes with exponential backoff + jitter (docs); Python SDK retries 429/5xx up to 4 attempts, initial ~1 s, max 60 s"72 },73 "headers": {74  "documented_rate_limit_headers": "none - the Gemini docs do not document any x-ratelimit-* response headers",75  "observed_on_generateContent_2026_09_19": ["Server-Timing: gfet4t7; dur=<ms>", "X-Gemini-Service-Tier: standard (also present on 429 responses)", "Alt-Svc", "Vary", "X-Content-Type-Options", "X-Frame-Options", "X-XSS-Protection", "Accept-Ranges", "Transfer-Encoding", "Content-Type: application/json; charset=UTF-8", "Date", "Server: scaffolding on HTTPServer2"],76  "observed_on_openai_compat": ["Server-Timing", "Set-Cookie", "Cache-Control", "Expires", "P3P", "Vary", "X-Content-Type-Options", "X-Frame-Options", "X-XSS-Protection"],77  "conclusion": "Rate-limit state (remaining RPM/TPM/RPD) is NOT exposed in response headers; only the 429 body carries quota metric/limit/retry seconds. Usage is visible in AI Studio (Projects / rate-limit pages)."78 },79 "observed_for_this_key": {80  "note": "Account-specific observations from 2026-09-19 UTC probes; not documentation.",81  "tier": "Free tier (no billing): 429 with 'limit: 0' on generate_content_free_tier_* metrics for model gemini-3.1-pro (requested as gemini-3.1-pro-preview and via gemini-pro-latest).",82  "successful_calls": ["gemini-3.5-flash-lite", "gemini-3.5-flash", "gemini-3.8-flash", "gemma-4-26b-a4b-it", "gemini-flash-latest (-> gemini-3.8-flash)", "OpenAI-compat chat.completions gemini-3.5-flash-lite"],83  "restricted_calls": ["gemini-2.5-flash-lite -> 404 NOT_FOUND 'no longer available to new users'", "gemini-3.1-pro-preview / gemini-pro-latest -> 429 RESOURCE_EXHAUSTED free-tier limit 0"],84  "usage_metadata_shape": {"promptTokenCount": 5, "candidatesTokenCount": 2, "totalTokenCount": 7, "promptTokensDetails": [{"modality": "TEXT", "tokenCount": 5}], "thoughtsTokenCount": "present when the model thought (5 on 3.5/3.8 Flash and Gemma 4 with maxOutputTokens=8 -> finishReason MAX_TOKENS, empty text)", "serviceTier": "standard"}85 }86}87