[
 {
  "record_type": "rate_limits",
  "provider": "anthropic",
  "generated_at": "2026-09-18",
  "status": [
   "DOCUMENTED",
   "LIVE_DISCOVERED"
  ],
  "sources": [
   {
    "url": "https://platform.claude.com/docs/en/api/rate-limits",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://platform.claude.com/docs/en/api/service-tiers",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://platform.claude.com/docs/en/build-with-claude/fast-mode",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://platform.claude.com/docs/en/build-with-claude/token-counting",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://platform.claude.com/docs/en/build-with-claude/files",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://platform.claude.com/docs/en/api/overview",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://platform.claude.com/docs/en/release-notes/overview",
    "retrieved_at": "2026-09-18"
   }
  ],
  "documented": {
   "principles": [
    "Two limit kinds: spend limits (monthly USD cap per tier) and rate limits (RPM, ITPM, OTPM per model class).",
    "Limits are per organization; tier assigned automatically from usage history and account standing; new orgs may start in an Evaluation tier below the published numbers.",
    "Token-bucket algorithm: capacity replenishes continuously (60 RPM may be enforced as 1 request/second; bursts can 429).",
    "Rate limits are applied separately per model class; different models can be used up to their own limits simultaneously.",
    "Rate limits are shared across inference_geo values (us and global draw from the same pool).",
    "Acceleration limits: a sharp usage increase can trigger 429 rate_limit_error (since 2025-08-11; previously 529 overloaded_error). Ramp traffic gradually.",
    "All limits are maxima, not guaranteed minimums.",
    "History: tokens-per-minute replaced by ITPM + OTPM on 2024-11-20; dedicated 1M-context rate limits removed 2026-03-13; usage tiers (formerly Tier 1-4) consolidated into Start/Build/Scale on 2026-06-26 with Sonnet/Haiku raised to Opus levels."
   ],
   "tiers": {
    "Start": {
     "monthly_spend_cap_usd": 500
    },
    "Build": {
     "monthly_spend_cap_usd": 1000
    },
    "Scale": {
     "monthly_spend_cap_usd": 200000
    },
    "Custom": {
     "monthly_spend_cap_usd": null,
     "note": "no cap; arranged with account team; contact sales via Console Rate limits page"
    },
    "Evaluation": {
     "note": "starting tier for new/low-history orgs, limits below Start; increases automatically"
    },
    "Claude Platform on AWS": {
     "note": "same limits; placed on Start, moves up with paid AWS Marketplace invoices; no self-serve increase flow; no per-workspace limits; no fast mode"
    }
   },
   "messages_api_per_model_class": {
    "unit": {
     "rpm": "requests/min",
     "itpm": "uncached input tokens/min",
     "otpm": "output tokens/min"
    },
    "classes": {
     "Claude Fable 5.x (Fable 5.1 + Fable 5 combined; Mythos 5.1 + Mythos 5 share a separate combined limit on the same terms)": {
      "Start": [
       1000,
       500000,
       100000
      ],
      "Build": [
       2000,
       1500000,
       300000
      ],
      "Scale": [
       4000,
       4000000,
       800000
      ]
     },
     "Claude Opus 5 (own bucket)": {
      "Start": [
       1000,
       2000000,
       400000
      ],
      "Build": [
       5000,
       5000000,
       1000000
      ],
      "Scale": [
       10000,
       10000000,
       2000000
      ]
     },
     "Claude Opus 4.x (Opus 4.8 + 4.7 + 4.6 + 4.5 combined)": {
      "Start": [
       1000,
       2000000,
       400000
      ],
      "Build": [
       5000,
       5000000,
       1000000
      ],
      "Scale": [
       10000,
       10000000,
       2000000
      ]
     },
     "Claude Sonnet 5 (own bucket)": {
      "Start": [
       1000,
       2000000,
       400000
      ],
      "Build": [
       5000,
       5000000,
       1000000
      ],
      "Scale": [
       10000,
       10000000,
       2000000
      ]
     },
     "Claude Sonnet 4.x (Sonnet 4.6 + 4.5 combined)": {
      "Start": [
       1000,
       2000000,
       400000
      ],
      "Build": [
       5000,
       5000000,
       1000000
      ],
      "Scale": [
       10000,
       10000000,
       2000000
      ]
     },
     "Claude Haiku 4.5": {
      "Start": [
       1000,
       2000000,
       400000
      ],
      "Build": [
       5000,
       5000000,
       1000000
      ],
      "Scale": [
       10000,
       10000000,
       2000000
      ]
     },
     "Claude Haiku 3.5 (retired on Claude API; counts cache_read_input_tokens toward ITPM)": {
      "Start": [
       1000,
       100000,
       20000
      ],
      "Build": [
       2000,
       200000,
       40000
      ],
      "Scale": [
       4000,
       400000,
       80000
      ]
     }
    },
    "values_order": [
     "rpm",
     "itpm",
     "otpm"
    ]
   },
   "cache_aware_itpm": {
    "counts_toward_itpm": [
     "input_tokens (tokens after the last cache breakpoint)",
     "cache_creation_input_tokens"
    ],
    "does_not_count": [
     "cache_read_input_tokens (all models except Claude Haiku 3.5)"
    ],
    "estimation": "ITPM estimated at request start, adjusted to actual input tokens during the request",
    "otpm": "evaluated in real time on actual generated tokens; max_tokens does not count",
    "total_input_formula": "total_input_tokens = cache_read_input_tokens + cache_creation_input_tokens + input_tokens",
    "example": "2,000,000 ITPM with 80% cache hit rate ~= 10,000,000 total input tokens/min"
   },
   "message_batches_api": {
    "note": "shared across all models; RPM applies to all Batches endpoints; queue limit counts individual batch requests not yet processed",
    "Start": {
     "rpm": 1000,
     "max_requests_in_processing_queue": 200000,
     "max_requests_per_batch": 100000
    },
    "Build": {
     "rpm": 2000,
     "max_requests_in_processing_queue": 300000,
     "max_requests_per_batch": 100000
    },
    "Scale": {
     "rpm": 4000,
     "max_requests_in_processing_queue": 500000,
     "max_requests_per_batch": 100000
    },
    "Custom": "contact sales"
   },
   "token_counting_api": {
    "Start": {
     "rpm": 5000
    },
    "Build": {
     "rpm": 10000
    },
    "Scale": {
     "rpm": 20000
    },
    "note": "free; separate from Messages limits"
   },
   "files_api": {
    "rpm_approx": 500,
    "note": "per organization, shared across upload/list/retrieve/download/delete; contact sales to raise"
   },
   "managed_agents": {
    "create_endpoints_rpm": 300,
    "read_endpoints_rpm": 1200,
    "note": "per organization, separate from Messages API"
   },
   "fast_mode": {
    "note": "dedicated limits separate from standard Opus limits when speed: fast on Opus 5 / Opus 4.8; 429 + retry-after when exceeded; Opus 4.7 -> error, Opus 4.6 -> standard speed",
    "headers": [
     "anthropic-fast-input-tokens-limit",
     "anthropic-fast-input-tokens-remaining",
     "anthropic-fast-input-tokens-reset",
     "anthropic-fast-output-tokens-limit",
     "anthropic-fast-output-tokens-remaining",
     "anthropic-fast-output-tokens-reset"
    ]
   },
   "programmatic_tool_calling": "each tool call from code execution counts as a separate invocation against the same limits as regular tool calls",
   "priority_tier": {
    "status": "no longer sold; existing commitments honoured to contract end",
    "burndown": {
     "cache_read": 0.1,
     "cache_write_5m": 1.25,
     "cache_write_1h": 2.0,
     "inference_geo_us_4_6_plus": 1.1,
     "other_tokens": 1.0
    },
    "assignment": "service_tier auto (default) uses priority capacity when ITPM and OTPM capacity are both sufficient, else standard; requests still draw from regular rate limits; standard_only opts out",
    "headers": [
     "anthropic-priority-input-tokens-limit",
     "anthropic-priority-input-tokens-remaining",
     "anthropic-priority-input-tokens-reset",
     "anthropic-priority-output-tokens-limit",
     "anthropic-priority-output-tokens-remaining",
     "anthropic-priority-output-tokens-reset"
    ],
    "unsupported_models": [
     "claude-fable-5-1",
     "claude-mythos-5-1",
     "claude-mythos-5",
     "claude-mythos-preview",
     "claude-opus-5",
     "claude-sonnet-5"
    ]
   },
   "workspaces": {
    "note": "per-workspace spend and rate limits can be set below the org limit (not on the default workspace); per limiter type; org limits always apply; anthropic-workspace-id response header identifies the workspace",
    "rate_limits_api": "GET org/workspace configured limits programmatically (manage-claude/rate-limits-api; Admin API rate_limits list endpoints)"
   },
   "response_headers": {
    "retry-after": "seconds to wait; earlier retries fail; NOT sent on the spend-cap 429",
    "anthropic-ratelimit-requests-limit|remaining|reset": "requests per period; reset in RFC 3339",
    "anthropic-ratelimit-tokens-limit|remaining|reset": "most restrictive token limit in effect (workspace limit if exceeded, else total input+output); remaining rounded to nearest thousand",
    "anthropic-ratelimit-input-tokens-limit|remaining|reset": "input tokens per period",
    "anthropic-ratelimit-output-tokens-limit|remaining|reset": "output tokens per period",
    "anthropic-priority-*": "Priority Tier only",
    "anthropic-fast-*": "fast mode only",
    "request-id": "unique request id (also in error bodies since 2025-08-19)",
    "anthropic-organization-id": "org id of the credential (since 2025-02-10)",
    "anthropic-workspace-id": "wrkspc_ id the credential resolved to (since 2026-08-11); absent on Admin API requests"
   },
   "http_429_semantics": {
    "rate_limit_exceeded": {
     "status": 429,
     "error_type": "rate_limit_error",
     "retry_after": true,
     "message": "describes which limit was exceeded"
    },
    "acceleration_limit": {
     "status": 429,
     "error_type": "rate_limit_error",
     "note": "sharp usage increase"
    },
    "tier_spend_cap_reached": {
     "status": 429,
     "error_type": "rate_limit_error",
     "details.error_code": "enforced_spend_limit_reached",
     "retry_after": false,
     "resumes": "00:00 UTC first day of next month or on tier increase"
    },
    "self_set_spend_limit_reached": {
     "status": 400,
     "error_type": "invalid_request_error",
     "message_prefix": "You have reached your specified API usage limits (or ...workspace API usage limits)"
    },
    "claude_code_workspace_limit": {
     "status": 429,
     "retry_after": true,
     "note": "Claude Code workspace limits are checked separately"
    },
    "overloaded": {
     "status": 529,
     "error_type": "overloaded_error",
     "note": "capacity; Priority Tier reduces these (see api/errors)"
    }
   },
   "console": "Usage page shows Rate Limit - Input Tokens / Output Tokens charts (hourly max per minute vs limit, cache rate); Rate limits page shows tier and Request rate limit increase"
  },
  "observed_for_this_key": {
   "note": "Values returned to OUR key on 2026-09-18/19 UTC; account-specific, not general limits. Do not treat as documentation.",
   "endpoint": "POST /v1/messages (claude-opus-5, claude-sonnet-5, claude-fable-5-1, claude-opus-4-8, claude-sonnet-4-6 all returned identical limit values)",
   "headers": {
    "anthropic-ratelimit-input-tokens-limit": "10000000",
    "anthropic-ratelimit-output-tokens-limit": "2000000",
    "anthropic-ratelimit-requests-limit": "10000",
    "anthropic-ratelimit-tokens-limit": "12000000",
    "anthropic-ratelimit-*-remaining": "present (e.g. requests-remaining 9999 after one call; tokens-remaining 12000000)",
    "anthropic-ratelimit-*-reset": "present, RFC 3339 (e.g. 2026-09-19T01:44:39Z)",
    "request-id": "present",
    "anthropic-organization-id": "present",
    "CF-RAY": "present (Cloudflare edge, e.g. ...-YUL)"
   },
   "interpretation": "matches the documented Scale-tier Opus/Sonnet numbers (10,000 RPM / 10M ITPM / 2M OTPM); tokens-limit 12,000,000 = ITPM + OTPM as documented for the total-tokens header",
   "fable_5_1_note": "claude-fable-5-1 returned the same 10M/2M/10k headers as Opus 5 although the documented Fable Scale limits are 4,000 RPM / 4M ITPM / 800k OTPM - account-specific; recorded, not generalized",
   "GET /v1/models and GET /v1/models/{id}": "no anthropic-ratelimit-* headers returned",
   "POST /v1/messages/count_tokens": "no anthropic-ratelimit-* headers returned",
   "GET /v1/organizations/me (Admin API, observed earlier in this run by another agent)": {
    "anthropic-ratelimit-requests-limit": "100",
    "note": "only the requests-limit family observed on the Admin API"
   },
   "retry-after": "not observed (no 429 triggered)"
  },
  "_fragment": "generated/fragments/rate-limits/anthropic-rate-limits.json"
 },
 {
  "record_type": "rate_limits",
  "provider": "gemini",
  "generated_at": "2026-09-18",
  "status": [
   "DOCUMENTED",
   "LIVE_DISCOVERED"
  ],
  "sources": [
   {
    "url": "https://ai.google.dev/gemini-api/docs/rate-limits",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://ai.google.dev/gemini-api/docs/billing",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://ai.google.dev/gemini-api/docs/priority-inference",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://ai.google.dev/gemini-api/docs/flex-inference",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://ai.google.dev/gemini-api/docs/batch-api",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://ai.google.dev/gemini-api/docs/live-api/session",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://ai.google.dev/gemini-api/docs/troubleshooting",
    "retrieved_at": "2026-09-18"
   },
   {
    "url": "https://aistudio.google.com/rate-limit",
    "retrieved_at": null,
    "note": "per-model RPM/TPM/RPD tables live here (account-specific UI), not in the docs page"
   }
  ],
  "documented": {
   "principles": [
    "Three dimensions: requests per minute (RPM), input tokens per minute (TPM), requests per day (RPD); exceeding any one of them returns 429 RESOURCE_EXHAUSTED.",
    "Limits are applied per Google Cloud project, not per API key. RPD quotas reset at midnight Pacific time.",
    "Model-specific extra dimensions exist: images per minute (IPM) for image-generation models, tokens per day (TPD) for some models.",
    "Experimental and preview models have more restrictive limits.",
    "Since 2026 the docs page no longer publishes the per-model RPM/TPM/RPD matrix; it points to the AI Studio rate-limit page. 'Specified rate limits are not guaranteed and actual capacity may vary.'",
    "Tier upgrades Free -> Tier 1 are instant once billing is linked; later upgrades take effect within ~10 minutes; upgrade can be denied 'based on other factors'."
   ],
   "usage_tiers": [
    {
     "tier": "Free",
     "qualification": "Active project or free trial",
     "billing_tier_spend_cap": null,
     "spend_rate_limit_per_10_min": null,
     "notes": "Free of charge on many models; content may be used to improve Google products (terms: Unpaid Services). Pro models (gemini-3.1-pro-preview, gemini-2.5-pro) and paid-only media models are not available (observed limit 0)."
    },
    {
     "tier": "Tier 1",
     "qualification": "Set up and link an active Cloud Billing account",
     "billing_tier_spend_cap": "$250",
     "spend_rate_limit_per_10_min": "$10"
    },
    {
     "tier": "Tier 2",
     "qualification": "Paid $100 cumulative on Google Cloud + 3 days since first successful payment",
     "billing_tier_spend_cap": "$2,000",
     "spend_rate_limit_per_10_min": "$50"
    },
    {
     "tier": "Tier 3",
     "qualification": "Paid $1,000 cumulative + 30 days since first successful payment",
     "billing_tier_spend_cap": "$20,000 - $100,000+",
     "spend_rate_limit_per_10_min": "$200"
    }
   ],
   "spend_based_rate_limits": "Rolling 10-minute window per tier ($10 / $50 / $200); exceeding it returns 429 RESOURCE_EXHAUSTED; whether it applies depends on billing history and account standing.",
   "priority_inference": {
    "rate_limit": "0.3x the standard rate limit for each model and tier",
    "overflow": "graceful server-side downgrade to Standard processing instead of failing",
    "pricing": "75-100% more than Standard (tables show 1.8x)"
   },
   "flex_inference": {
    "rate_limit": "own limits; best-effort, sheddable capacity",
    "latency": "minutes (1-15 min target)",
    "errors": "429 RESOURCE_EXHAUSTED when capacity is shed; clients must retry (docs: 'Implement retries', per-request timeouts)",
    "pricing": "50% discount",
    "supported_models": [
     "gemini-3.8-flash",
     "gemini-3.7-flash",
     "gemini-3.6-flash",
     "gemini-3.5-flash-lite",
     "gemini-3.5-flash",
     "gemini-3.1-flash-lite",
     "gemini-3.1-pro-preview",
     "gemini-3-flash-preview",
     "gemini-2.5-pro",
     "gemini-2.5-flash"
    ]
   },
   "batch_api": {
    "concurrent_batch_requests": 100,
    "input_file_size_limit": "2 GB",
    "file_storage_limit": "20 GB",
    "turnaround": "target 24 hours (usually much faster)",
    "pricing": "50% of standard",
    "enqueued_tokens_per_model": {
     "note": "Maximum tokens enqueued across all active batch jobs for a given model.",
     "tier_1": {
      "gemini-3.1-pro-preview": 5000000,
      "gemini-3.5-flash-lite": 10000000,
      "gemini-3.8-flash": 3000000,
      "gemini-3.7-flash": 3000000,
      "gemini-3.1-flash-lite": 10000000,
      "gemini-3.1-flash-lite-preview": 10000000,
      "gemini-3.6-flash": 3000000,
      "gemini-3.5-flash": 3000000,
      "gemini-2.5-pro": 5000000,
      "gemini-2.5-pro-preview-tts": 25000,
      "gemini-2.5-flash": 3000000,
      "gemini-2.5-flash-preview": 3000000,
      "gemini-2.5-flash-image-preview": 3000000,
      "gemini-2.5-flash-preview-tts": 100000,
      "gemini-2.5-flash-lite": 10000000,
      "gemini-2.5-flash-lite-preview": 10000000,
      "gemini-2.0-flash": 10000000,
      "gemini-2.0-flash-image": 3000000,
      "gemini-2.0-flash-lite": 10000000,
      "gemini-3.1-flash-image-preview": 1000000,
      "gemini-3.1-flash-lite-image": 2000000,
      "gemini-3-pro-image-preview": 2000000,
      "gemini-embedding": 500000
     },
     "tier_2": {
      "gemini-3.1-pro-preview": 500000000,
      "gemini-3.5-flash-lite": 500000000,
      "gemini-3.1-flash-lite": 500000000,
      "gemini-3.1-flash-lite-preview": 500000000,
      "gemini-3.8-flash": 400000000,
      "gemini-3.7-flash": 400000000,
      "gemini-3.6-flash": 400000000,
      "gemini-3.5-flash": 400000000,
      "gemini-2.5-pro": 500000000,
      "gemini-2.5-pro-preview-tts": 100000,
      "gemini-2.5-flash": 400000000,
      "gemini-2.5-flash-preview": 400000000,
      "gemini-2.5-flash-image-preview": 400000000,
      "gemini-2.5-flash-preview-tts": 100000,
      "gemini-2.5-flash-lite": 500000000,
      "gemini-2.5-flash-lite-preview": 500000000,
      "gemini-2.0-flash": 1000000000,
      "gemini-2.0-flash-image": 400000000,
      "gemini-2.0-flash-lite": 1000000000,
      "gemini-3.1-flash-image-preview": 250000000,
      "gemini-3.1-flash-lite-image": 270000000,
      "gemini-3-pro-image-preview": 270000000,
      "gemini-embedding": 5000000
     },
     "tier_3": {
      "gemini-3.1-pro-preview": 1000000000,
      "gemini-3.5-flash-lite": 1000000000,
      "gemini-3.1-flash-lite": 1000000000,
      "gemini-3.1-flash-lite-preview": 1000000000,
      "gemini-3.8-flash": 1000000000,
      "gemini-3.7-flash": 1000000000,
      "gemini-3.6-flash": 1000000000,
      "gemini-3.5-flash": 1000000000,
      "gemini-2.5-pro": 1000000000,
      "gemini-2.5-pro-preview-tts": 1000000,
      "gemini-2.5-flash": 1000000000,
      "gemini-2.5-flash-preview": 1000000000,
      "gemini-2.5-flash-image-preview": 1000000000,
      "gemini-2.5-flash-preview-tts": 4000000,
      "gemini-2.5-flash-lite": 1000000000,
      "gemini-2.5-flash-lite-preview": 1000000000,
      "gemini-2.0-flash": 5000000000,
      "gemini-2.0-flash-image": 1000000000,
      "gemini-2.0-flash-lite": 5000000000,
      "gemini-3.1-flash-image-preview": 750000000,
      "gemini-3.1-flash-lite-image": 1000000000,
      "gemini-3-pro-image-preview": 1000000000,
      "gemini-embedding": 10000000
     },
     "docs_note": "The batch table still names shut-down models (2.0 Flash, 2.5 previews) and preview image ids; the GA image ids (gemini-3.1-flash-image, gemini-3-pro-image) are not listed - assume the preview row applies."
    }
   },
   "live_api_sessions": {
    "session_lifetime": "audio-only 15 minutes, audio+video 2 minutes without context-window compression; with contextWindowCompression sessions can run longer",
    "connection_lifetime": "~10 minutes per WebSocket connection; use sessionResumption handles (valid 2 h after the last session termination) to continue",
    "ephemeral_tokens": "short-lived tokens for client-to-server WebSocket connections (POST /v1beta/auth_tokens); Live API only",
    "note": "Live API concurrency/session-count limits are not published on the rate-limits page."
   },
   "grounding_free_quotas": {
    "google_search_gemini_3": "5,000 free search requests per month shared across all Gemini 3.x models, then $14 per 1,000 requests",
    "google_search_gemini_2_5": "1,500 RPD free (shared Flash/Flash-Lite; free tier 500 RPD), then $35 per 1,000 grounded prompts",
    "google_maps_gemini_3": "5,000 prompts per month free shared, then $14 per 1,000 search queries",
    "google_maps_gemini_2_5": "1,500 RPD free (10,000 RPD for Pro), then $25 per 1,000 grounded prompts"
   },
   "model_parameter_ranges_troubleshooting_page": {
    "candidateCount": "1-8",
    "temperature": "0.0-1.0 (troubleshooting page; live Model.maxTemperature=2 for most Gemini models, 1 for image models)",
    "topP": "0.0-1.0"
   },
   "increase_requests": "Paid-tier increase form https://forms.gle/ETzX94k8jf7iSotH9 ('no guarantees')."
  },
  "error_semantics_429": {
   "http_status": 429,
   "status": "RESOURCE_EXHAUSTED",
   "message_pattern": "You exceeded your current quota, please check your plan and billing details. ... * Quota exceeded for metric: generativelanguage.googleapis.com/<metric>, limit: <n>, model: <model> ... Please retry in <seconds>s.",
   "details": [
    {
     "@type": "type.googleapis.com/google.rpc.Help",
     "links": [
      {
       "description": "Learn more about Gemini API quotas",
       "url": "https://ai.google.dev/gemini-api/docs/rate-limits"
      }
     ]
    },
    {
     "@type": "type.googleapis.com/google.rpc.QuotaFailure",
     "violations": [
      {
       "quotaMetric": "generativelanguage.googleapis.com/generate_content_free_tier_input_token_count",
       "quotaId": "GenerateContentInputTokensPerModelPerDay-FreeTier",
       "quotaDimensions": {
        "location": "global",
        "model": "gemini-3.1-pro"
       }
      },
      {
       "quotaMetric": "generativelanguage.googleapis.com/generate_content_free_tier_requests",
       "quotaId": "GenerateRequestsPerDayPerProjectPerModel-FreeTier"
      },
      {
       "quotaMetric": "generativelanguage.googleapis.com/generate_content_free_tier_requests",
       "quotaId": "GenerateRequestsPerMinutePerProjectPerModel-FreeTier"
      }
     ]
    }
   ],
   "quota_metrics_observed": [
    "generativelanguage.googleapis.com/generate_content_free_tier_input_token_count",
    "generativelanguage.googleapis.com/generate_content_free_tier_requests"
   ],
   "quota_ids_observed": [
    "GenerateContentInputTokensPerModelPerDay-FreeTier",
    "GenerateRequestsPerDayPerProjectPerModel-FreeTier",
    "GenerateRequestsPerMinutePerProjectPerModel-FreeTier"
   ],
   "retry_hint": "The retry delay is only in the message text ('Please retry in 54.22098241s.'); no Retry-After header was returned.",
   "retryable": "yes with exponential backoff + jitter (docs); Python SDK retries 429/5xx up to 4 attempts, initial ~1 s, max 60 s"
  },
  "headers": {
   "documented_rate_limit_headers": "none - the Gemini docs do not document any x-ratelimit-* response headers",
   "observed_on_generateContent_2026_09_19": [
    "Server-Timing: gfet4t7; dur=<ms>",
    "X-Gemini-Service-Tier: standard (also present on 429 responses)",
    "Alt-Svc",
    "Vary",
    "X-Content-Type-Options",
    "X-Frame-Options",
    "X-XSS-Protection",
    "Accept-Ranges",
    "Transfer-Encoding",
    "Content-Type: application/json; charset=UTF-8",
    "Date",
    "Server: scaffolding on HTTPServer2"
   ],
   "observed_on_openai_compat": [
    "Server-Timing",
    "Set-Cookie",
    "Cache-Control",
    "Expires",
    "P3P",
    "Vary",
    "X-Content-Type-Options",
    "X-Frame-Options",
    "X-XSS-Protection"
   ],
   "conclusion": "Rate-limit state (remaining RPM/TPM/RPD) is NOT exposed in response headers; only the 429 body carries quota metric/limit/retry seconds. Usage is visible in AI Studio (Projects / rate-limit pages)."
  },
  "observed_for_this_key": {
   "note": "Account-specific observations from 2026-09-19 UTC probes; not documentation.",
   "tier": "Free tier (no billing): 429 with 'limit: 0' on generate_content_free_tier_* metrics for model gemini-3.1-pro (requested as gemini-3.1-pro-preview and via gemini-pro-latest).",
   "successful_calls": [
    "gemini-3.5-flash-lite",
    "gemini-3.5-flash",
    "gemini-3.8-flash",
    "gemma-4-26b-a4b-it",
    "gemini-flash-latest (-> gemini-3.8-flash)",
    "OpenAI-compat chat.completions gemini-3.5-flash-lite"
   ],
   "restricted_calls": [
    "gemini-2.5-flash-lite -> 404 NOT_FOUND 'no longer available to new users'",
    "gemini-3.1-pro-preview / gemini-pro-latest -> 429 RESOURCE_EXHAUSTED free-tier limit 0"
   ],
   "usage_metadata_shape": {
    "promptTokenCount": 5,
    "candidatesTokenCount": 2,
    "totalTokenCount": 7,
    "promptTokensDetails": [
     {
      "modality": "TEXT",
      "tokenCount": 5
     }
    ],
    "thoughtsTokenCount": "present when the model thought (5 on 3.5/3.8 Flash and Gemma 4 with maxOutputTokens=8 -> finishReason MAX_TOKENS, empty text)",
    "serviceTier": "standard"
   }
  },
  "_fragment": "generated/fragments/rate-limits/gemini-rate-limits.json"
 },
 {
  "provider": "openai",
  "retrieved_at": "2026-09-18",
  "sources": [
   "https://developers.openai.com/api/docs/guides/rate-limits",
   "https://developers.openai.com/api/docs/models/<id> (Rate limits section)",
   "https://developers.openai.com/api/docs/guides/batch",
   "https://developers.openai.com/api/docs/guides/fast-mode"
  ],
  "concepts": [
   {
    "metric": "RPM",
    "meaning": "requests per minute",
    "scope": "organization and project, per model (or shared-limit group)"
   },
   {
    "metric": "RPD",
    "meaning": "requests per day",
    "scope": "some models / free tier"
   },
   {
    "metric": "TPM",
    "meaning": "tokens per minute (max(max_tokens, estimated prompt tokens) counted per request)",
    "scope": "per model"
   },
   {
    "metric": "TPD",
    "meaning": "tokens per day",
    "scope": "some models"
   },
   {
    "metric": "IPM",
    "meaning": "images per minute",
    "scope": "image models (gpt-image-*)"
   },
   {
    "metric": "minutes-of-audio per minute",
    "meaning": "audio duration admitted per minute",
    "scope": "streaming audio models (gpt-realtime-whisper, gpt-realtime-translate)"
   },
   {
    "metric": "concurrent sessions",
    "meaning": "simultaneous live sessions",
    "scope": "gpt-live-1 (v1/live/sessions)"
   },
   {
    "metric": "batch queue limit",
    "meaning": "total input tokens queued across pending batch jobs for a model; released when the batch completes",
    "scope": "Batch API, per model"
   },
   {
    "metric": "long-context limit",
    "meaning": "separate RPM/TPM/batch-queue table for requests >272K input tokens (GPT-5.4/5.5/5.6/6 Astra)",
    "scope": "long-context models; visible in the console"
   },
   {
    "metric": "shared limits",
    "meaning": "some model families share one limit pool (listed under 'shared limit' in the console)",
    "scope": "organization"
   },
   {
    "metric": "project-scoped token limit",
    "meaning": "optional per-project token limit exposed via x-ratelimit-*-project-tokens headers",
    "scope": "project"
   },
   {
    "metric": "monthly usage limit",
    "meaning": "approved monthly spend ceiling per organization, separate from configurable spend limits",
    "scope": "organization"
   },
   {
    "metric": "vector store ingestion",
    "meaning": "/vector_stores/{id}/files and /file_batches share 300 requests per minute per vector store",
    "scope": "per vector store"
   },
   {
    "metric": "ramp-rate (slow_down)",
    "meaning": "429 rate_limit_error/slow_down when traffic increases too quickly even below RPM/TPM; above ~1M TPM ramp at most +50% every 15 minutes",
    "scope": "per model"
   }
  ],
  "usage_tiers": [
   {
    "tier": "Free",
    "qualification": "allowed geography",
    "usage_limit_usd_per_month": 100
   },
   {
    "tier": "Tier 1",
    "qualification": "$5 paid",
    "usage_limit_usd_per_month": 100
   },
   {
    "tier": "Tier 2",
    "qualification": "$50 paid",
    "usage_limit_usd_per_month": 500
   },
   {
    "tier": "Tier 3",
    "qualification": "$100 paid",
    "usage_limit_usd_per_month": 1000
   },
   {
    "tier": "Tier 4",
    "qualification": "$250 paid",
    "usage_limit_usd_per_month": 5000
   },
   {
    "tier": "Tier 5",
    "qualification": "$1,000 paid",
    "usage_limit_usd_per_month": 200000
   }
  ],
  "headers": [
   {
    "name": "Retry-After",
    "sample": "56",
    "description": "minimum seconds to wait before retrying a temporary 429 (slow_down / rate limit) or 503 (server_is_overloaded)"
   },
   {
    "name": "x-ratelimit-limit-requests",
    "sample": "60",
    "description": "max requests before exhausting the rate limit"
   },
   {
    "name": "x-ratelimit-limit-tokens",
    "sample": "150000",
    "description": "max tokens before exhausting the rate limit"
   },
   {
    "name": "x-ratelimit-remaining-requests",
    "sample": "59",
    "description": "remaining requests"
   },
   {
    "name": "x-ratelimit-remaining-tokens",
    "sample": "149984",
    "description": "remaining tokens"
   },
   {
    "name": "x-ratelimit-reset-requests",
    "sample": "1s",
    "description": "time until request limit resets"
   },
   {
    "name": "x-ratelimit-reset-tokens",
    "sample": "6m0s",
    "description": "time until token limit resets"
   },
   {
    "name": "x-ratelimit-limit-project-tokens",
    "sample": "60000",
    "description": "project token limit (present when a project-scoped limit applies)"
   },
   {
    "name": "x-ratelimit-remaining-project-tokens",
    "sample": "57000",
    "description": "remaining project tokens"
   },
   {
    "name": "x-ratelimit-reset-project-tokens",
    "sample": "3s",
    "description": "time until project token limit resets"
   }
  ],
  "errors": [
   {
    "http_status": 429,
    "type": "rate_limit_error",
    "code": "slow_down",
    "meaning": "request rate increased too quickly",
    "action": "honour Retry-After, reduce rate, ramp gradually"
   },
   {
    "http_status": 429,
    "type": "rate_limit_error",
    "code": "rate_limit_exceeded",
    "meaning": "RPM/TPM/RPD/TPD/IPM exhausted",
    "action": "exponential backoff with jitter; batch requests; reduce max_tokens"
   },
   {
    "http_status": 429,
    "type": "insufficient_quota",
    "code": "insufficient_quota",
    "meaning": "monthly usage / spend limit reached (429 also returned when a hard spend limit is hit)",
    "action": "do not retry; raise limits"
   },
   {
    "http_status": 503,
    "type": "service_unavailable_error",
    "code": "server_is_overloaded",
    "meaning": "model temporarily overloaded",
    "action": "honour Retry-After, retry with increasing delay"
   }
  ],
  "fine_tuning_limits_endpoint": "GET /v1/fine_tuning/model_limits",
  "capacity_products": [
   "Scale Tier (pay-as-you-go traffic routinely hitting ramp limits)",
   "Reserved Tier (GPT-5.6 and later)",
   "Ultrafast mode (limited preview, GPT-5.6 Sol, announced 2026-08-13)"
  ],
  "per_model_documented_tiers": {
   "babbage-002": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "RPD": 10000,
       "TPM": 10000,
       "Batch queue limit": 100000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 40000,
       "Batch queue limit": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 80000,
       "Batch queue limit": 5000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 300000,
       "Batch queue limit": 30000000
      },
      "Tier 5": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 1000000,
       "Batch queue limit": 150000000
      }
     }
    }
   },
   "chat-latest": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 50000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "chatgpt-4o-latest": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "chatgpt-image-latest": {
    "default": {
     "metrics": [
      "TPM",
      "IPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "TPM": 100000,
       "IPM": 5
      },
      "Tier 2": {
       "TPM": 250000,
       "IPM": 20
      },
      "Tier 3": {
       "TPM": 800000,
       "IPM": 50
      },
      "Tier 4": {
       "TPM": 3000000,
       "IPM": 150
      },
      "Tier 5": {
       "TPM": 8000000,
       "IPM": 250
      }
     }
    }
   },
   "codex-mini-latest": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 1000,
       "TPM": 100000,
       "Batch queue limit": 1000000
      },
      "Tier 2": {
       "RPM": 2000,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 150000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "computer-use-preview": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 3": {
       "RPM": 3000,
       "TPM": 20000000,
       "Batch queue limit": 450000000
      },
      "Tier 4": {
       "RPM": 3000,
       "TPM": 20000000,
       "Batch queue limit": 450000000
      },
      "Tier 5": {
       "RPM": 3000,
       "TPM": 20000000,
       "Batch queue limit": 450000000
      }
     }
    }
   },
   "dall-e-3": {
    "default": {
     "metrics": [
      "RPM"
     ],
     "note": null,
     "tiers": {
      "Tier free": {
       "RPM": "1 img/min"
      },
      "Tier 1": {
       "RPM": "500 img/min"
      },
      "Tier 2": {
       "RPM": "2500 img/min"
      },
      "Tier 3": {
       "RPM": "5000 img/min"
      },
      "Tier 4": {
       "RPM": "7500 img/min"
      },
      "Tier 5": {
       "RPM": "10000 img/min"
      }
     }
    }
   },
   "davinci-002": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "RPD": 10000,
       "TPM": 10000,
       "Batch queue limit": 100000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 40000,
       "Batch queue limit": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 80000,
       "Batch queue limit": 5000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 300000,
       "Batch queue limit": 30000000
      },
      "Tier 5": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 1000000,
       "Batch queue limit": 150000000
      }
     }
    }
   },
   "gpt-3.5-turbo": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "RPD": 10000,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 2000000,
       "Batch queue limit": 5000000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 4000000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 50000000,
       "Batch queue limit": 10000000000
      }
     }
    }
   },
   "gpt-4": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "RPD": 10000,
       "TPM": 10000,
       "Batch queue limit": 100000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 40000,
       "Batch queue limit": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 80000,
       "Batch queue limit": 5000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 300000,
       "Batch queue limit": 30000000
      },
      "Tier 5": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 1000000,
       "Batch queue limit": 150000000
      }
     }
    }
   },
   "gpt-4-turbo": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 600000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 800000,
       "Batch queue limit": 80000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 300000000
      }
     }
    }
   },
   "gpt-4-turbo-preview": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 600000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 800000,
       "Batch queue limit": 80000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 300000000
      }
     }
    }
   },
   "gpt-4.1": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    },
    "Long Context": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": "128k input tokens",
     "tiers": {
      "Tier 1": {
       "RPM": 100,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 250,
       "TPM": 500000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 500,
       "TPM": 1000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 1000,
       "TPM": 5000000,
       "Batch queue limit": 100000000
      },
      "Tier 5": {
       "RPM": 4000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      }
     }
    }
   },
   "gpt-4.1-mini": {
    "Standard": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "RPD": 10000,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "RPD": null,
       "TPM": 150000000,
       "Batch queue limit": 15000000000
      }
     }
    },
    "Long Context": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": "128k input tokens",
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "TPM": 400000,
       "Batch queue limit": 5000000
      },
      "Tier 2": {
       "RPM": 500,
       "TPM": 1000000,
       "Batch queue limit": 40000000
      },
      "Tier 3": {
       "RPM": 1000,
       "TPM": 2000000,
       "Batch queue limit": 80000000
      },
      "Tier 4": {
       "RPM": 2000,
       "TPM": 10000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 8000,
       "TPM": 20000000,
       "Batch queue limit": 2000000000
      }
     }
    }
   },
   "gpt-4.1-nano": {
    "Standard": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "RPD": 10000,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "RPD": null,
       "TPM": 150000000,
       "Batch queue limit": 15000000000
      }
     }
    },
    "Long Context": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": "128k input tokens",
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "TPM": 400000,
       "Batch queue limit": 5000000
      },
      "Tier 2": {
       "RPM": 500,
       "TPM": 1000000,
       "Batch queue limit": 40000000
      },
      "Tier 3": {
       "RPM": 1000,
       "TPM": 2000000,
       "Batch queue limit": 80000000
      },
      "Tier 4": {
       "RPM": 2000,
       "TPM": 10000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 8000,
       "TPM": 20000000,
       "Batch queue limit": 2000000000
      }
     }
    }
   },
   "gpt-4.5-preview": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 1000,
       "TPM": 125000,
       "Batch queue limit": 50000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 250000,
       "Batch queue limit": 500000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 500000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 1000000,
       "Batch queue limit": 100000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "gpt-4o": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "gpt-4o-audio-preview": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 2000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "gpt-4o-mini": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "RPD": 10000,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "RPD": null,
       "TPM": 150000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-4o-mini-audio-preview": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "free": {
       "RPM": 3,
       "RPD": 200,
       "TPM": 40000,
       "Batch queue limit": null
      },
      "Tier 1": {
       "RPM": 500,
       "RPD": 10000,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "RPD": null,
       "TPM": 150000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-4o-mini-realtime-preview": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "RPD": 1000,
       "TPM": 40000
      },
      "Tier 2": {
       "RPM": 400,
       "RPD": null,
       "TPM": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 800000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 4000000
      },
      "Tier 5": {
       "RPM": 20000,
       "RPD": null,
       "TPM": 15000000
      }
     }
    }
   },
   "gpt-4o-mini-search-preview": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "free": {
       "RPM": 3,
       "RPD": 200,
       "TPM": 40000,
       "Batch queue limit": null
      },
      "Tier 1": {
       "RPM": 500,
       "RPD": 10000,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "RPD": null,
       "TPM": 150000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-4o-mini-transcribe": {
    "default": {
     "metrics": [
      "RPM",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 50000
      },
      "Tier 2": {
       "RPM": 2000,
       "TPM": 150000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 600000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 8000000
      }
     }
    }
   },
   "gpt-4o-mini-tts": {
    "default": {
     "metrics": [
      "RPM",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 50000
      },
      "Tier 2": {
       "RPM": 2000,
       "TPM": 150000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 600000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 8000000
      }
     }
    }
   },
   "gpt-4o-realtime-preview": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "RPD": 1000,
       "TPM": 40000
      },
      "Tier 2": {
       "RPM": 400,
       "RPD": null,
       "TPM": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 800000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 4000000
      },
      "Tier 5": {
       "RPM": 20000,
       "RPD": null,
       "TPM": 15000000
      }
     }
    }
   },
   "gpt-4o-search-preview": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 100,
       "TPM": 30000,
       "Batch queue limit": 0
      },
      "Tier 2": {
       "RPM": 500,
       "TPM": 45000,
       "Batch queue limit": 0
      },
      "Tier 3": {
       "RPM": 500,
       "TPM": 80000,
       "Batch queue limit": 0
      },
      "Tier 4": {
       "RPM": 1000,
       "TPM": 200000,
       "Batch queue limit": 0
      },
      "Tier 5": {
       "RPM": 1000,
       "TPM": 3000000,
       "Batch queue limit": 0
      }
     }
    }
   },
   "gpt-4o-transcribe": {
    "default": {
     "metrics": [
      "RPM",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 10000
      },
      "Tier 2": {
       "RPM": 2000,
       "TPM": 100000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 400000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 6000000
      }
     }
    }
   },
   "gpt-4o-transcribe-diarize": {
    "default": {
     "metrics": [
      "RPM",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 10000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 100000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 400000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 6000000
      }
     }
    }
   },
   "gpt-5": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5-chat-latest": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 50000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5-codex": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 10000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5-mini": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 5000000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 180000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5-nano": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 180000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5-pro": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "gpt-5.1": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.1-chat-latest": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 50000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.1-codex": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.1-codex-max": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.1-codex-mini": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 5000000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 180000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.2": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.2-chat-latest": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 50000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.2-codex": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.2-pro": {
    "Standard": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    },
    "Long Context": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": "128k input tokens",
     "tiers": {
      "Tier 1": {
       "RPM": 100,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 250,
       "TPM": 500000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 500,
       "TPM": 1000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 1000,
       "TPM": 5000000,
       "Batch queue limit": 100000000
      },
      "Tier 5": {
       "RPM": 4000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      }
     }
    }
   },
   "gpt-5.3-chat-latest": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 50000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.3-codex": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.4": {
    "Standard": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    },
    "Long Context": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": "272K input tokens",
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "TPM": 400000,
       "Batch queue limit": 5000000
      },
      "Tier 2": {
       "RPM": 500,
       "TPM": 1000000,
       "Batch queue limit": 40000000
      },
      "Tier 3": {
       "RPM": 1000,
       "TPM": 2000000,
       "Batch queue limit": 80000000
      },
      "Tier 4": {
       "RPM": 2000,
       "TPM": 10000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 8000,
       "TPM": 20000000,
       "Batch queue limit": 2000000000
      }
     }
    }
   },
   "gpt-5.4-mini": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 5000000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 180000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.4-nano": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 180000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.4-pro": {
    "Standard": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 50,
       "TPM": 50000,
       "Batch queue limit": 900000
      },
      "Tier 2": {
       "RPM": 500,
       "TPM": 100000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 500,
       "TPM": 200000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 1000,
       "TPM": 400000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 1500,
       "TPM": 4000000,
       "Batch queue limit": 15000000000
      }
     }
    },
    "Long Context": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": "272K input tokens",
     "tiers": {
      "Tier 1": {
       "RPM": 20,
       "TPM": 40000,
       "Batch queue limit": 2000000
      },
      "Tier 2": {
       "RPM": 50,
       "TPM": 100000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 100,
       "TPM": 200000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 200,
       "TPM": 1000000,
       "Batch queue limit": 100000000
      },
      "Tier 5": {
       "RPM": 800,
       "TPM": 2000000,
       "Batch queue limit": 1000000000
      }
     }
    }
   },
   "gpt-5.5": {
    "Standard": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    },
    "Long Context": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": "272K input tokens",
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "TPM": 400000,
       "Batch queue limit": 5000000
      },
      "Tier 2": {
       "RPM": 500,
       "TPM": 1000000,
       "Batch queue limit": 40000000
      },
      "Tier 3": {
       "RPM": 1000,
       "TPM": 2000000,
       "Batch queue limit": 80000000
      },
      "Tier 4": {
       "RPM": 2000,
       "TPM": 10000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 8000,
       "TPM": 20000000,
       "Batch queue limit": 2000000000
      }
     }
    }
   },
   "gpt-5.5-pro": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 50,
       "TPM": 50000,
       "Batch queue limit": 500000
      },
      "Tier 2": {
       "RPM": 500,
       "TPM": 200000,
       "Batch queue limit": 1000000
      },
      "Tier 3": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 10000000
      },
      "Tier 4": {
       "RPM": 1000,
       "TPM": 1000000,
       "Batch queue limit": 20000000
      },
      "Tier 5": {
       "RPM": 2000,
       "TPM": 4000000,
       "Batch queue limit": 1500000000
      }
     }
    }
   },
   "gpt-5.6-cyber": {
    "Standard": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.6-luna": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 5000000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 180000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.6-sol": {
    "Standard": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-5.6-terra": {
    "Standard": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-6-astra": {
    "Standard": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-audio": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 2000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "gpt-audio-1.5": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 2000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "gpt-audio-mini": {
    "default": {
     "metrics": [
      "RPM",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000
      }
     }
    }
   },
   "gpt-daybreak-blue-latest": {
    "Standard": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-daybreak-red-latest": {
    "Standard": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 500000,
       "Batch queue limit": 1500000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 15000,
       "TPM": 40000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "gpt-image-1": {
    "default": {
     "metrics": [
      "TPM",
      "IPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "TPM": 100000,
       "IPM": 5
      },
      "Tier 2": {
       "TPM": 250000,
       "IPM": 20
      },
      "Tier 3": {
       "TPM": 800000,
       "IPM": 50
      },
      "Tier 4": {
       "TPM": 3000000,
       "IPM": 150
      },
      "Tier 5": {
       "TPM": 8000000,
       "IPM": 250
      }
     }
    }
   },
   "gpt-image-1-mini": {
    "default": {
     "metrics": [
      "IPM",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "IPM": 5,
       "TPM": 100000
      },
      "Tier 2": {
       "IPM": 20,
       "TPM": 250000
      },
      "Tier 3": {
       "IPM": 50,
       "TPM": 800000
      },
      "Tier 4": {
       "IPM": 150,
       "TPM": 3000000
      },
      "Tier 5": {
       "IPM": 250,
       "TPM": 8000000
      }
     }
    }
   },
   "gpt-image-1.5": {
    "default": {
     "metrics": [
      "TPM",
      "IPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "TPM": 100000,
       "IPM": 5
      },
      "Tier 2": {
       "TPM": 250000,
       "IPM": 20
      },
      "Tier 3": {
       "TPM": 800000,
       "IPM": 50
      },
      "Tier 4": {
       "TPM": 3000000,
       "IPM": 150
      },
      "Tier 5": {
       "TPM": 8000000,
       "IPM": 250
      }
     }
    }
   },
   "gpt-image-2": {
    "default": {
     "metrics": [
      "TPM",
      "IPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "TPM": 100000,
       "IPM": 5
      },
      "Tier 2": {
       "TPM": 250000,
       "IPM": 20
      },
      "Tier 3": {
       "TPM": 800000,
       "IPM": 50
      },
      "Tier 4": {
       "TPM": 3000000,
       "IPM": 150
      },
      "Tier 5": {
       "TPM": 8000000,
       "IPM": 250
      }
     }
    }
   },
   "gpt-image-2.5-flare": {
    "default": {
     "metrics": [
      "TPM",
      "IPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "TPM": 100000,
       "IPM": 5
      },
      "Tier 2": {
       "TPM": 250000,
       "IPM": 20
      },
      "Tier 3": {
       "TPM": 800000,
       "IPM": 50
      },
      "Tier 4": {
       "TPM": 3000000,
       "IPM": 150
      },
      "Tier 5": {
       "TPM": 8000000,
       "IPM": 250
      }
     }
    }
   },
   "gpt-image-2.5-sunburst": {
    "default": {
     "metrics": [
      "TPM",
      "IPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "TPM": 100000,
       "IPM": 5
      },
      "Tier 2": {
       "TPM": 250000,
       "IPM": 20
      },
      "Tier 3": {
       "TPM": 800000,
       "IPM": 50
      },
      "Tier 4": {
       "TPM": 3000000,
       "IPM": 150
      },
      "Tier 5": {
       "TPM": 8000000,
       "IPM": 250
      }
     }
    }
   },
   "gpt-live-1": {
    "Concurrent sessions": {
     "metrics": [
      "Concurrent sessions"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "Concurrent sessions": 25
      },
      "Tier 2": {
       "Concurrent sessions": 50
      },
      "Tier 3": {
       "Concurrent sessions": 200
      },
      "Tier 4": {
       "Concurrent sessions": 300
      },
      "Tier 5": {
       "Concurrent sessions": 500
      }
     }
    }
   },
   "gpt-live-transcribe": {
    "default": {
     "metrics": [
      "RPM",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 60000
      },
      "Tier 2": {
       "RPM": 2000,
       "TPM": 210000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 390000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 600000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 780000
      }
     }
    }
   },
   "gpt-oss-120b": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 0,
       "TPM": 0,
       "Batch queue limit": 0
      },
      "Tier 2": {
       "RPM": 0,
       "TPM": 0,
       "Batch queue limit": 0
      },
      "Tier 3": {
       "RPM": 0,
       "TPM": 0,
       "Batch queue limit": 0
      },
      "Tier 4": {
       "RPM": 0,
       "TPM": 0,
       "Batch queue limit": 0
      },
      "Tier 5": {
       "RPM": 0,
       "TPM": 0,
       "Batch queue limit": 0
      }
     }
    }
   },
   "gpt-oss-20b": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 0,
       "TPM": 0,
       "Batch queue limit": 0
      },
      "Tier 2": {
       "RPM": 0,
       "TPM": 0,
       "Batch queue limit": 0
      },
      "Tier 3": {
       "RPM": 0,
       "TPM": 0,
       "Batch queue limit": 0
      },
      "Tier 4": {
       "RPM": 0,
       "TPM": 0,
       "Batch queue limit": 0
      },
      "Tier 5": {
       "RPM": 0,
       "TPM": 0,
       "Batch queue limit": 0
      }
     }
    }
   },
   "gpt-realtime": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "RPD": 1000,
       "TPM": 40000
      },
      "Tier 2": {
       "RPM": 400,
       "RPD": null,
       "TPM": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 800000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 4000000
      },
      "Tier 5": {
       "RPM": 20000,
       "RPD": null,
       "TPM": 15000000
      }
     }
    }
   },
   "gpt-realtime-1.5": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "RPD": 1000,
       "TPM": 40000
      },
      "Tier 2": {
       "RPM": 400,
       "RPD": null,
       "TPM": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 800000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 4000000
      },
      "Tier 5": {
       "RPM": 20000,
       "RPD": null,
       "TPM": 15000000
      }
     }
    }
   },
   "gpt-realtime-2": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "RPD": 1000,
       "TPM": 40000
      },
      "Tier 2": {
       "RPM": 400,
       "RPD": null,
       "TPM": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 800000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 4000000
      },
      "Tier 5": {
       "RPM": 20000,
       "RPD": null,
       "TPM": 15000000
      }
     }
    }
   },
   "gpt-realtime-2.1": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "RPD": 1000,
       "TPM": 40000
      },
      "Tier 2": {
       "RPM": 400,
       "RPD": null,
       "TPM": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 800000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 4000000
      },
      "Tier 5": {
       "RPM": 20000,
       "RPD": null,
       "TPM": 15000000
      }
     }
    }
   },
   "gpt-realtime-2.1-mini": {
    "default": {
     "metrics": [
      "RPM",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "TPM": 40000
      },
      "Tier 2": {
       "RPM": 400,
       "TPM": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000
      },
      "Tier 5": {
       "RPM": 20000,
       "TPM": 15000000
      }
     }
    }
   },
   "gpt-realtime-mini": {
    "default": {
     "metrics": [
      "RPM",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 200,
       "TPM": 40000
      },
      "Tier 2": {
       "RPM": 400,
       "TPM": 200000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 4000000
      },
      "Tier 5": {
       "RPM": 20000,
       "TPM": 15000000
      }
     }
    }
   },
   "gpt-realtime-translate": {
    "default": {
     "metrics": [
      "Minutes-of-audio per minute"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "Minutes-of-audio per minute": 50
      },
      "Tier 2": {
       "Minutes-of-audio per minute": 200
      },
      "Tier 3": {
       "Minutes-of-audio per minute": 400
      },
      "Tier 4": {
       "Minutes-of-audio per minute": 650
      },
      "Tier 5": {
       "Minutes-of-audio per minute": 850
      }
     }
    }
   },
   "gpt-realtime-whisper": {
    "default": {
     "metrics": [
      "Minutes-of-audio per minute"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "Minutes-of-audio per minute": 100
      },
      "Tier 2": {
       "Minutes-of-audio per minute": 350
      },
      "Tier 3": {
       "Minutes-of-audio per minute": 650
      },
      "Tier 4": {
       "Minutes-of-audio per minute": 1000
      },
      "Tier 5": {
       "Minutes-of-audio per minute": 1300
      }
     }
    }
   },
   "gpt-transcribe": {
    "default": {
     "metrics": [
      "RPM",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 200000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 2000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 150000000
      }
     }
    }
   },
   "o1": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "o1-mini": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 200000,
       "Batch queue limit": null
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 2000000,
       "Batch queue limit": null
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 150000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "o1-preview": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": null
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": null
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "o1-pro": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "o3": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "o3-deep-research": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 200000,
       "Batch queue limit": 200000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 300000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 500000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 2000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 10000000
      }
     }
    }
   },
   "o3-mini": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 1000,
       "TPM": 100000,
       "Batch queue limit": 1000000
      },
      "Tier 2": {
       "RPM": 2000,
       "TPM": 200000,
       "Batch queue limit": 2000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 150000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "o3-pro": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 500,
       "TPM": 30000,
       "Batch queue limit": 90000
      },
      "Tier 2": {
       "RPM": 5000,
       "TPM": 450000,
       "Batch queue limit": 1350000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 800000,
       "Batch queue limit": 50000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 2000000,
       "Batch queue limit": 200000000
      },
      "Tier 5": {
       "RPM": 10000,
       "TPM": 30000000,
       "Batch queue limit": 5000000000
      }
     }
    }
   },
   "o4-mini": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 1000,
       "TPM": 100000,
       "Batch queue limit": 1000000
      },
      "Tier 2": {
       "RPM": 2000,
       "TPM": 2000000,
       "Batch queue limit": 2000000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 40000000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 1000000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 150000000,
       "Batch queue limit": 15000000000
      }
     }
    }
   },
   "o4-mini-deep-research": {
    "default": {
     "metrics": [
      "RPM",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 1000,
       "TPM": 200000,
       "Batch queue limit": 200000
      },
      "Tier 2": {
       "RPM": 2000,
       "TPM": 2000000,
       "Batch queue limit": 300000
      },
      "Tier 3": {
       "RPM": 5000,
       "TPM": 4000000,
       "Batch queue limit": 500000
      },
      "Tier 4": {
       "RPM": 10000,
       "TPM": 10000000,
       "Batch queue limit": 2000000
      },
      "Tier 5": {
       "RPM": 30000,
       "TPM": 150000000,
       "Batch queue limit": 10000000
      }
     }
    }
   },
   "omni-moderation-latest": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM"
     ],
     "note": null,
     "tiers": {
      "free": {
       "RPM": 250,
       "RPD": 5000,
       "TPM": 10000
      },
      "Tier 1": {
       "RPM": 500,
       "RPD": 10000,
       "TPM": 10000
      },
      "Tier 2": {
       "RPM": 500,
       "RPD": null,
       "TPM": 20000
      },
      "Tier 3": {
       "RPM": 1000,
       "RPD": null,
       "TPM": 50000
      },
      "Tier 4": {
       "RPM": 2000,
       "RPD": null,
       "TPM": 250000
      },
      "Tier 5": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 500000
      }
     }
    }
   },
   "sora-2": {
    "Standard RPM": {
     "metrics": [
      "RPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 25
      },
      "Tier 2": {
       "RPM": 50
      },
      "Tier 3": {
       "RPM": 125
      },
      "Tier 4": {
       "RPM": 200
      },
      "Tier 5": {
       "RPM": 375
      }
     }
    }
   },
   "sora-2-pro": {
    "Standard RPM": {
     "metrics": [
      "RPM"
     ],
     "note": null,
     "tiers": {
      "Tier 1": {
       "RPM": 10
      },
      "Tier 2": {
       "RPM": 25
      },
      "Tier 3": {
       "RPM": 50
      },
      "Tier 4": {
       "RPM": 75
      },
      "Tier 5": {
       "RPM": 150
      }
     }
    }
   },
   "text-embedding-3-large": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "free": {
       "RPM": 100,
       "RPD": 2000,
       "TPM": 40000,
       "Batch queue limit": null
      },
      "Tier 1": {
       "RPM": 3000,
       "RPD": null,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 1000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 5000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 5000000,
       "Batch queue limit": 500000000
      },
      "Tier 5": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 10000000,
       "Batch queue limit": 4000000000
      }
     }
    }
   },
   "text-embedding-3-small": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "free": {
       "RPM": 100,
       "RPD": 2000,
       "TPM": 40000,
       "Batch queue limit": null
      },
      "Tier 1": {
       "RPM": 3000,
       "RPD": null,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 1000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 5000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 5000000,
       "Batch queue limit": 500000000
      },
      "Tier 5": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 10000000,
       "Batch queue limit": 4000000000
      }
     }
    }
   },
   "text-embedding-ada-002": {
    "default": {
     "metrics": [
      "RPM",
      "RPD",
      "TPM",
      "Batch queue limit"
     ],
     "note": null,
     "tiers": {
      "free": {
       "RPM": 100,
       "RPD": 2000,
       "TPM": 40000,
       "Batch queue limit": null
      },
      "Tier 1": {
       "RPM": 3000,
       "RPD": null,
       "TPM": 1000000,
       "Batch queue limit": 3000000
      },
      "Tier 2": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 1000000,
       "Batch queue limit": 20000000
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null,
       "TPM": 5000000,
       "Batch queue limit": 100000000
      },
      "Tier 4": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 5000000,
       "Batch queue limit": 500000000
      },
      "Tier 5": {
       "RPM": 10000,
       "RPD": null,
       "TPM": 10000000,
       "Batch queue limit": 4000000000
      }
     }
    }
   },
   "tts-1": {
    "default": {
     "metrics": [
      "RPM",
      "RPD"
     ],
     "note": null,
     "tiers": {
      "free": {
       "RPM": 3,
       "RPD": 200
      },
      "Tier 1": {
       "RPM": 500,
       "RPD": null
      },
      "Tier 2": {
       "RPM": 2500,
       "RPD": null
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null
      },
      "Tier 4": {
       "RPM": 7500,
       "RPD": null
      },
      "Tier 5": {
       "RPM": 10000,
       "RPD": null
      }
     }
    }
   },
   "tts-1-hd": {
    "default": {
     "metrics": [
      "RPM"
     ],
     "note": null,
     "tiers": {
      "free": {
       "RPM": null
      },
      "Tier 1": {
       "RPM": 500
      },
      "Tier 2": {
       "RPM": 2500
      },
      "Tier 3": {
       "RPM": 5000
      },
      "Tier 4": {
       "RPM": 7500
      },
      "Tier 5": {
       "RPM": 10000
      }
     }
    }
   },
   "whisper-1": {
    "default": {
     "metrics": [
      "RPM",
      "RPD"
     ],
     "note": null,
     "tiers": {
      "free": {
       "RPM": 3,
       "RPD": 200
      },
      "Tier 1": {
       "RPM": 500,
       "RPD": null
      },
      "Tier 2": {
       "RPM": 2500,
       "RPD": null
      },
      "Tier 3": {
       "RPM": 5000,
       "RPD": null
      },
      "Tier 4": {
       "RPM": 7500,
       "RPD": null
      },
      "Tier 5": {
       "RPM": 10000,
       "RPD": null
      }
     }
    }
   }
  },
  "per_model_notes": {
   "gpt-live-1": [
    "Rate limits are measured in concurrent sessions.",
    "Unsupported usage tiers: Free."
   ]
  },
  "observed": [
   {
    "date": "2026-09-18",
    "request": "POST /v1/responses (this atlas key)",
    "headers": {
     "x-ratelimit-limit-requests": "30000",
     "x-ratelimit-limit-tokens": "180000000"
    },
    "note": "observed for OUR key/org/project on 2026-09-18 — account-specific, do not generalize"
   },
   {
    "date": "2026-09-19",
    "request": "POST /v1/responses model=gpt-5.6-luna",
    "headers": {
     "x-ratelimit-limit-requests": "30000",
     "x-ratelimit-limit-tokens": "180000000",
     "x-ratelimit-remaining-requests": "29999",
     "x-ratelimit-remaining-tokens": "180000000",
     "x-ratelimit-reset-requests": "2ms",
     "x-ratelimit-reset-tokens": "0s"
    },
    "note": "observed for OUR key — account-specific, do not generalize"
   },
   {
    "date": "2026-09-19",
    "request": "POST /v1/responses model=gpt-5.6-sol",
    "headers": {
     "x-ratelimit-limit-requests": "15000",
     "x-ratelimit-limit-tokens": "40000000",
     "x-ratelimit-remaining-requests": "14999",
     "x-ratelimit-remaining-tokens": "40000000",
     "x-ratelimit-reset-requests": "4ms",
     "x-ratelimit-reset-tokens": "0s"
    },
    "note": "observed for OUR key — account-specific, do not generalize"
   },
   {
    "date": "2026-09-19",
    "request": "POST /v1/responses model=gpt-5.6-terra",
    "headers": {
     "x-ratelimit-limit-requests": "15000",
     "x-ratelimit-limit-tokens": "40000000",
     "x-ratelimit-remaining-requests": "14999",
     "x-ratelimit-remaining-tokens": "40000000",
     "x-ratelimit-reset-requests": "4ms",
     "x-ratelimit-reset-tokens": "0s"
    },
    "note": "observed for OUR key — account-specific, do not generalize"
   },
   {
    "date": "2026-09-19",
    "request": "POST /v1/responses model=gpt-6-astra",
    "headers": {
     "x-ratelimit-limit-requests": "15000",
     "x-ratelimit-limit-tokens": "40000000",
     "x-ratelimit-remaining-requests": "14999",
     "x-ratelimit-remaining-tokens": "40000000",
     "x-ratelimit-reset-requests": "4ms",
     "x-ratelimit-reset-tokens": "0s"
    },
    "note": "observed for OUR key — account-specific, do not generalize"
   },
   {
    "date": "2026-09-19",
    "request": "POST /v1/responses model=gpt-5.5",
    "headers": {
     "x-ratelimit-limit-requests": "15000",
     "x-ratelimit-limit-tokens": "40000000",
     "x-ratelimit-remaining-requests": "14999",
     "x-ratelimit-remaining-tokens": "40000000",
     "x-ratelimit-reset-requests": "4ms",
     "x-ratelimit-reset-tokens": "0s"
    },
    "note": "observed for OUR key — account-specific, do not generalize"
   },
   {
    "date": "2026-09-19",
    "request": "POST /v1/responses model=gpt-5.4-mini",
    "headers": {
     "x-ratelimit-limit-requests": "30000",
     "x-ratelimit-limit-tokens": "180000000",
     "x-ratelimit-remaining-requests": "29999",
     "x-ratelimit-remaining-tokens": "180000000",
     "x-ratelimit-reset-requests": "2ms",
     "x-ratelimit-reset-tokens": "0s"
    },
    "note": "observed for OUR key — account-specific, do not generalize"
   }
  ],
  "_fragment": "generated/fragments/rate-limits/openai-rate-limits.json"
 },
 {
  "record_type": "rate_limits",
  "provider": "xai",
  "generated_at": "2026-09-18",
  "status": [
   "DOCUMENTED",
   "LIVE_DISCOVERED"
  ],
  "sources": [
   {
    "url": "https://docs.x.ai/developers/rate-limits",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/management-api-guide",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/debugging",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/pricing",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/advanced-api-usage/batch-api",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-4.6",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-4.5",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-4.3",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-4.20-0309-reasoning",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-4.20-0309-non-reasoning",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-build-0.1",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-4.20-multi-agent-0309",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-imagine-image",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-imagine-image-quality",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-imagine-image-2.0",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-imagine-video",
    "retrieved_at": "2026-09-19"
   },
   {
    "url": "https://docs.x.ai/developers/models/grok-imagine-video-1.5",
    "retrieved_at": "2026-09-19"
   }
  ],
  "documented": {
   "principles": [
    "Every team has per-model limits on two dimensions: requests per second (RPS) and tokens per minute (TPM). RPS is derived from the per-minute request budget (RPM/60): a full minute of requests cannot be spent in one second.",
    "Limits scale with the team's tier, determined by cumulative spend on the xAI API since 2026-01-01 (prepaid purchases or fulfilled invoices). Tiers unlock automatically and never downgrade.",
    "Tiers apply to text, embedding and voice models. Imagine (image/video) limits are RPS-only and are raised via sales@x.ai.",
    "TPM counts prompt tokens (text, image, audio), completion tokens, reasoning tokens and cached prompt tokens (cached tokens still count although billed less).",
    "Exceeding any limit -> HTTP 429 Too Many Requests (gRPC RESOURCE_EXHAUSTED). Use exponential backoff.",
    "Batch API requests do not count towards rate limits.",
    "Per-API-key caps (qps, qpm, tpm) can additionally be set via the Management API (POST/PUT api-keys); the tpm limiter engages when strictly exceeded and does not abort in-flight requests.",
    "Team-level personalized limits are shown on the console Models page (https://console.x.ai/team/default/models)."
   ],
   "tiers": {
    "Tier 0": {
     "spend_threshold_usd": 0,
     "note": "default"
    },
    "Tier 1": {
     "spend_threshold_usd": 50
    },
    "Tier 2": {
     "spend_threshold_usd": 250
    },
    "Tier 3": {
     "spend_threshold_usd": 1000
    },
    "Tier 4": {
     "spend_threshold_usd": 5000
    },
    "Enterprise": {
     "note": "on request (sales@x.ai)"
    }
   },
   "per_model": {
    "grok-4.6": {
     "rps_T0_T4": [
      150,
      172,
      208,
      312,
      500
     ],
     "tpm_T0_T4": [
      "50M",
      "53M",
      "60M",
      "74M",
      "100M"
     ]
    },
    "grok-4.5": {
     "rps_T0_T4": [
      150,
      172,
      208,
      312,
      500
     ],
     "tpm_T0_T4": [
      "50M",
      "53M",
      "60M",
      "74M",
      "100M"
     ]
    },
    "grok-4.3": {
     "rps_T0_T4": [
      37,
      50,
      75,
      125,
      208
     ],
     "tpm_T0_T4": [
      "10M",
      "15M",
      "25M",
      "45M",
      "85M"
     ]
    },
    "grok-4.20-0309-reasoning": {
     "rps_T0_T4": [
      37,
      50,
      75,
      125,
      208
     ],
     "tpm_T0_T4": [
      "10M",
      "15M",
      "25M",
      "45M",
      "85M"
     ]
    },
    "grok-4.20-0309-non-reasoning": {
     "rps_T0_T4": [
      37,
      50,
      75,
      125,
      208
     ],
     "tpm_T0_T4": [
      "10M",
      "15M",
      "25M",
      "45M",
      "85M"
     ]
    },
    "grok-build-0.1": {
     "rps_T0_T4": [
      37,
      50,
      75,
      125,
      208
     ],
     "tpm_T0_T4": [
      "10M",
      "15M",
      "25M",
      "45M",
      "85M"
     ]
    },
    "grok-4.20-multi-agent-0309": {
     "rps_T0_T4": [
      9,
      12,
      18,
      31,
      56
     ],
     "tpm_T0_T4": [
      "2.5M",
      "3.7M",
      "6.2M",
      "11M",
      "21M"
     ]
    },
    "grok-imagine-image": {
     "rps_T0_T4": [
      6,
      12,
      25,
      50,
      100
     ],
     "tpm_T0_T4": null
    },
    "grok-imagine-image-quality": {
     "rps_T0_T4": [
      6,
      12,
      25,
      50,
      100
     ],
     "tpm_T0_T4": null
    },
    "grok-imagine-image-2.0": {
     "rps_T0_T4": [
      6,
      12,
      25,
      50,
      100
     ],
     "tpm_T0_T4": null
    },
    "grok-imagine-video": {
     "rps_T0_T4": [
      10,
      20,
      39,
      79,
      158
     ],
     "tpm_T0_T4": null
    },
    "grok-imagine-video-1.5": {
     "rps_T0_T4": [
      10,
      20,
      39,
      79,
      158
     ],
     "tpm_T0_T4": null
    }
   },
   "voice_and_audio": {
    "grok-voice-think-fast-2.0": {
     "rps": null,
     "concurrent_sessions_T0_T4": [
      10,
      20,
      50,
      100,
      200
     ],
     "max_session_minutes": 120
    },
    "text-to-speech": {
     "rps_T0_T4": [
      50,
      50,
      100,
      250,
      500
     ],
     "concurrent_sessions_T0_T4": [
      100,
      200,
      200,
      300,
      500
     ]
    },
    "speech-to-text": {
     "rps_T0_T4": [
      10,
      10,
      20,
      30,
      40
     ],
     "concurrent_sessions_T0_T4": [
      100,
      200,
      200,
      300,
      500
     ],
     "note": "concurrent sessions apply to streaming"
    }
   },
   "increasing_limits": [
    "spend more (automatic)",
    "request an increase in the console Models page",
    "sales@x.ai for enterprise capacity"
   ],
   "http_429_semantics": "429 on inference endpoints when RPS or TPM is exceeded; no documented Retry-After header (none observed in this run since no 429 was triggered)"
  },
  "observed_for_this_key": {
   "note": "Headers returned to OUR key on 2026-09-19 (Tier 0 team, spend < $50). Account-specific; not documentation.",
   "headers_seen": [
    "x-ratelimit-limit-requests",
    "x-ratelimit-remaining-requests",
    "x-ratelimit-limit-tokens",
    "x-ratelimit-remaining-tokens",
    "x-request-id",
    "Server-Timing (cfEdge/cfOrigin)",
    "CF-RAY (Cloudflare edge, e.g. ...-YUL)"
   ],
   "headers_not_seen": [
    "x-ratelimit-reset-*",
    "retry-after (no 429 triggered)"
   ],
   "per_model": {
    "grok-4.6": {
     "x-ratelimit-limit-requests": "7200",
     "x-ratelimit-limit-tokens": "50000000"
    },
    "grok-4.5": {
     "x-ratelimit-limit-requests": "7200",
     "x-ratelimit-limit-tokens": "50000000"
    },
    "grok-4.3": {
     "x-ratelimit-limit-requests": "1800",
     "x-ratelimit-limit-tokens": "10000000"
    },
    "grok-4.20-0309-non-reasoning": {
     "x-ratelimit-limit-requests": "1800",
     "x-ratelimit-limit-tokens": "10000000"
    },
    "grok-build-0.1": {
     "x-ratelimit-limit-requests": "1800",
     "x-ratelimit-limit-tokens": "10000000"
    }
   },
   "tokenize_text": {
    "x-ratelimit-limit-requests": "1800",
    "x-ratelimit-limit-tokens": "absent"
   },
   "list_endpoints": "GET /v1/models, /v1/language-models, /v1/api-key, /v1/me return no x-ratelimit-* headers",
   "interpretation": "x-ratelimit-limit-tokens matches the documented Tier 0 TPM (50M for grok-4.6/4.5, 10M for grok-4.3/4.20/build). x-ratelimit-limit-requests looks like a per-MINUTE request budget (7200 -> 120 RPS; 1800 -> 30 RPS), lower than the documented Tier 0 RPS (150 / 37) - the docs say RPS = RPM/60, so the header is probably the RPM budget and our team sits slightly below the published T0 figures. Not verified further.",
   "grok-4.20-multi-agent-0309": "POST /v1/responses returned no x-ratelimit-* headers in our probe"
  },
  "_fragment": "generated/fragments/rate-limits/xai-rate-limits.json"
 }
]