SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
87.4 KB · 1,432 lines python
Raw Blame History
1#!/usr/bin/env python32"""Build the OpenAI models / pricing / rate-limits / deprecations fragments (+ table-heavy docs).34Inputs (offline, immutable):5  sources/openai/models-api-raw.json            live GET /v1/models (sanitized) — 2026-09-186  sources/openai/live-model-probes.json         sanitized results of GET /v1/models/{id} + minimal POST /v1/responses probes7  sources/openai/pages/api/docs/models/*.md     one page per documented model8  sources/openai/pages/api/docs/pricing.md      pricing tables (standard / batch / flex / fast / tools / fine-tuning …)9  sources/openai/pages/api/docs/deprecations.md10  sources/openai/openapi/openapi-master.yaml    model id enums (list items)1112Outputs:13  generated/fragments/models/openai-models.json14  generated/fragments/pricing/openai-pricing.json15  generated/fragments/rate-limits/openai-rate-limits.json16  generated/fragments/deprecations/openai-deprecations.json17  docs/models/openai-models.md, docs/openai/pricing.md, docs/openai/deprecations.md1819Re-runnable: python3 scripts/build_openai_models.py20"""21from __future__ import annotations2223import json24import re25from collections import OrderedDict, defaultdict26from datetime import datetime, timezone27from pathlib import Path2829ROOT = Path(__file__).resolve().parent.parent30DOCS_SRC = ROOT / "sources/openai/pages/api/docs"31MODEL_PAGES = DOCS_SRC / "models"32RETRIEVED_AT = "2026-09-18"33TODAY = "2026-09-18"34BASE_URL = "https://developers.openai.com/api/docs"35OPENAPI_URL = "https://github.com/openai/openai-openapi (openapi-master.yaml, downloaded 2026-09-18)"3637# --------------------------------------------------------------------------------------38# Release dates known from the changelog (sources/openai/pages/api/docs/changelog.md).39# Anything not listed falls back to the `created` timestamp of GET /v1/models (labelled).40# --------------------------------------------------------------------------------------41RELEASE_DATES: dict[str, str] = {42    "gpt-6-astra": "2026-09-03",43    "gpt-5.6-sol": "2026-07-09", "gpt-5.6-terra": "2026-07-09", "gpt-5.6-luna": "2026-07-09",44    "gpt-5.6-cyber": "2026-08-07", "gpt-daybreak-blue-latest": "2026-08-07", "gpt-daybreak-red-latest": "2026-08-07",45    "gpt-image-2.5-flare": "2026-09-08", "gpt-image-2.5-sunburst": "2026-09-08",46    "gpt-live-1": "2026-09-10", "gpt-rosalind-research": "2026-09-08",47    "gpt-transcribe": "2026-07-28", "gpt-live-transcribe": "2026-07-28",48    "gpt-realtime-2.1": "2026-07-06", "gpt-realtime-2.1-mini": "2026-07-06",49    "chat-latest": "2026-05-05",50    "gpt-realtime-2": "2026-05-07", "gpt-realtime-translate": "2026-05-07", "gpt-realtime-whisper": "2026-05-07",51    "gpt-5.5": "2026-04-24", "gpt-5.5-pro": "2026-04-24",52    "gpt-image-2": "2026-04-21",53    "gpt-5.4-mini": "2026-03-17", "gpt-5.4-nano": "2026-03-17",54    "gpt-5.4": "2026-03-05", "gpt-5.4-pro": "2026-03-05",55    "gpt-5.3-chat-latest": "2026-03-03", "gpt-5.3-codex": "2026-02-24",56    "gpt-realtime-1.5": "2026-02-23", "gpt-audio-1.5": "2026-02-23",57    "gpt-5.2-codex": "2026-01-14",58    "gpt-image-1.5": "2025-12-16", "chatgpt-image-latest": "2025-12-16",59    "gpt-5.2": "2025-12-11", "gpt-5.2-chat-latest": "2025-12-11", "gpt-5.2-pro": "2025-12-11",60    "gpt-5.1-codex-max": "2025-12-04",61    "gpt-5.1": "2025-11-13", "gpt-5.1-codex": "2025-11-13", "gpt-5.1-chat-latest": "2025-11-13", "gpt-5.1-codex-mini": "2025-11-13",62    "gpt-5-pro": "2025-10-06", "gpt-realtime-mini": "2025-10-06", "gpt-audio-mini": "2025-10-06",63    "gpt-image-1-mini": "2025-10-06", "sora-2": "2025-10-06", "sora-2-pro": "2025-10-06",64    "gpt-5-codex": "2025-09-23",65    "gpt-realtime": "2025-08-28", "gpt-audio": "2025-08-28",66    "gpt-5": "2025-08-07", "gpt-5-mini": "2025-08-07", "gpt-5-nano": "2025-08-07", "gpt-5-chat-latest": "2025-08-07",67    "o3-deep-research": "2025-06-26", "o4-mini-deep-research": "2025-06-26",68    "o3-pro": "2025-06-10", "codex-mini-latest": "2025-05-15",69    "o3": "2025-04-16", "o4-mini": "2025-04-16",70    "gpt-4.1": "2025-04-14", "gpt-4.1-mini": "2025-04-14", "gpt-4.1-nano": "2025-04-14",71    "gpt-4o-transcribe": "2025-03-20", "gpt-4o-mini-transcribe": "2025-03-20", "gpt-4o-mini-tts": "2025-03-20",72    "o1-pro": "2025-03-19",73    "gpt-4o-search-preview": "2025-03-11", "gpt-4o-mini-search-preview": "2025-03-11", "computer-use-preview": "2025-03-11",74    "gpt-4.5-preview": "2025-02-27", "o3-mini": "2025-01-31", "o1": "2024-12-17",75    "gpt-4o-mini-realtime-preview": "2024-12-17", "gpt-4o-mini-audio-preview": "2024-12-17",76    "gpt-4o-audio-preview": "2024-10-17", "gpt-4o-realtime-preview": "2024-10-01",77    "omni-moderation-latest": "2024-09-26", "o1-preview": "2024-09-12", "o1-mini": "2024-09-12",78    "chatgpt-4o-latest": "2024-08-15", "gpt-4o-mini": "2024-07-18", "gpt-4o": "2024-05-13",79    "gpt-4-turbo": "2024-04-09", "gpt-3.5-turbo-0125": "2024-01-25",80    "text-embedding-3-small": "2024-01-25", "text-embedding-3-large": "2024-01-25",81    "gpt-4-turbo-preview": "2023-11-06", "dall-e-3": "2023-11-06", "tts-1": "2023-11-06", "tts-1-hd": "2023-11-06",82    "gpt-3.5-turbo-1106": "2023-11-06",83}84# Not in the downloaded changelog; public OpenAI announcement date, flagged as such in the record.85RELEASE_DATES_EXTERNAL = {"gpt-oss-120b": "2025-08-05", "gpt-oss-20b": "2025-08-05"}8687LEGACY_IDS = {88    "gpt-3.5-turbo", "gpt-3.5-turbo-0125", "gpt-3.5-turbo-1106", "gpt-3.5-turbo-16k", "gpt-3.5-turbo-instruct",89    "gpt-3.5-turbo-instruct-0914", "gpt-4", "gpt-4-0613", "gpt-4-turbo", "gpt-4-turbo-2024-04-09", "gpt-4-turbo-preview",90    "babbage-002", "davinci-002", "text-embedding-ada-002", "text-moderation-latest", "text-moderation-stable",91    "tts-1", "tts-1-hd", "tts-1-1106", "tts-1-hd-1106", "whisper-1", "chatgpt-4o-latest", "gpt-4o-2024-05-13",92}9394# Models that require separate approval / provisioning per docs.95RESTRICTED_ACCESS = {96    "gpt-5.6-cyber": "Daybreak program (separate approval and provisioning) — https://openai.com/daybreak/",97    "gpt-daybreak-blue-latest": "Daybreak program (separate approval and provisioning) — https://openai.com/daybreak/",98    "gpt-daybreak-red-latest": "Daybreak program (separate approval and provisioning) — https://openai.com/daybreak/",99    "gpt-5.5-cyber": "Daybreak program (pricing page only)",100    "gpt-5.4-cyber": "Daybreak program (deprecations page only; shutdown 2026-10-01)",101    "gpt-rosalind-research": "Trusted-access program for approved life-sciences research (billing starts 2026-10-05)",102}103104# Capabilities documented outside the model pages (guides / changelog).105COMPACTION_MODELS = {"gpt-5.4", "gpt-5.4-2026-03-05", "gpt-5.4-pro", "gpt-5.4-pro-2026-03-05", "gpt-5.4-mini",106                     "gpt-5.4-mini-2026-03-17", "gpt-5.4-nano", "gpt-5.4-nano-2026-03-17"}107PRO_MODE_MODELS = {"gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"}108EXTENDED_CACHE_RETENTION = {"gpt-5.5", "gpt-5.5-pro", "gpt-5.4", "gpt-5.2", "gpt-5.1-codex-max", "gpt-5.1", "gpt-5.1-codex",109                            "gpt-5.1-codex-mini", "gpt-5.1-chat-latest", "gpt-5", "gpt-5-codex", "gpt-4.1"}110111MONTHS = {m: i for i, m in enumerate(112    ["jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec"], 1)}113114115def norm_date(s: str) -> str | None:116    """'Oct 1, 2026' | 'October 23, 2026' | '2026-09-24' | '2026‑03‑26' | 'June 3, 2026' -> ISO."""117    if not s:118        return None119    s = s.replace("‑", "-").replace("–", "-").strip().strip("`")120    m = re.search(r"(\d{4})-(\d{2})-(\d{2})", s)121    if m:122        return m.group(0)123    m = re.search(r"([A-Za-z]{3,9})\.? (\d{1,2})(?:st|nd|rd|th)?,? (\d{4})", s)124    if m and m.group(1)[:3].lower() in MONTHS:125        return f"{int(m.group(3)):04d}-{MONTHS[m.group(1)[:3].lower()]:02d}-{int(m.group(2)):02d}"126    return None127128129def parse_money(cell: str):130    """'$0.40' -> (0.4, None); '$15.00 / 1M characters' -> (15.0, 'per 1M characters'); 'Free' -> (0, None); '-' -> None."""131    c = cell.strip()132    if c in {"-", "—", ""}:133        return None134    if c.lower() == "free":135        return (0.0, None)136    m = re.search(r"\$([0-9][0-9,]*\.?[0-9]*)", c)137    if not m:138        return None139    price = float(m.group(1).replace(",", ""))140    unit = None141    m2 = re.search(r"/\s*([A-Za-z0-9 ]+)", c)142    if m2:143        unit = "per " + m2.group(1).strip()144    return (price, unit)145146147def parse_int(cell: str):148    c = cell.strip().replace(",", "")149    m = re.match(r"^(\d+)", c)150    return int(m.group(1)) if m else None151152153def md_table(lines: list[str], start: int):154    """Parse a markdown table beginning at lines[start] ('| a | b |'). Returns (header, rows, next_index)."""155    def cells(line: str) -> list[str]:156        # escaped pipes (`a \| b`) are cell content, not separators157        return [c.replace("\x00", "|").strip() for c in line.replace("\\|", "\x00").strip().strip("|").split("|")]158159    header = cells(lines[start])160    i = start + 1161    if i < len(lines) and re.match(r"^\|\s*-", lines[i]):162        i += 1163    rows = []164    while i < len(lines) and lines[i].startswith("|"):165        rows.append(cells(lines[i]))166        i += 1167    return header, rows, i168169170# --------------------------------------------------------------------------------------171# Model page parser172# --------------------------------------------------------------------------------------173DATED_RE = re.compile(r"-(\d{4}-\d{2}-\d{2}|\d{4})$")174175176def is_dated(mid: str) -> bool:177    return bool(re.search(r"-\d{4}-\d{2}-\d{2}$", mid) or re.search(r"-(0613|0314|1106|0125|0301|0914)$", mid))178179180def parse_model_page(path: Path) -> dict:181    text = path.read_text(encoding="utf-8")182    lines = text.splitlines()183    page: dict = {"slug": path.stem, "file": str(path.relative_to(ROOT))}184    page["display_name"] = lines[0].lstrip("# ").strip()185    quotes = [l[2:].strip() for l in lines if l.startswith("> ") and "llms.txt" not in l]186    page["description"] = quotes[0] if quotes else ""187    m = re.search(r"Model ID: `([^`]+)`", text)188    page["id"] = m.group(1) if m else path.stem189190    # intro = text between 'Model ID' line and '## Model details'191    intro = []192    seen_id = False193    for l in lines:194        if l.startswith("Model ID:"):195            seen_id = True196            continue197        if l.startswith("## "):198            if seen_id:199                break200            continue201        if seen_id and l.strip():202            intro.append(l.strip())203    page["intro"] = " ".join(intro)204205    # sections206    sections: dict[str, list[str]] = OrderedDict()207    cur = None208    for l in lines:209        if l.startswith("## "):210            cur = l[3:].strip()211            sections[cur] = []212        elif cur is not None:213            sections[cur].append(l)214215    det = sections.get("Model details", [])216    page["default_snapshot"] = None217    page["input_modalities"], page["output_modalities"], page["unsupported_modalities"] = [], [], []218    page["context_window"] = page["max_output"] = page["max_input"] = None219    page["knowledge_cutoff"] = None220    page["reasoning"] = False221    for l in det:222        l = l.strip()223        if not l.startswith("- "):224            continue225        b = l[2:]226        if b.startswith("Default snapshot:"):227            page["default_snapshot"] = re.search(r"`([^`]+)`", b).group(1)228        elif b.startswith("Input modalities:"):229            page["input_modalities"] = [x.strip() for x in b.split(":", 1)[1].split(",")]230        elif b.startswith("Output modalities:"):231            page["output_modalities"] = [x.strip() for x in b.split(":", 1)[1].split(",")]232        elif b.startswith("Unsupported modalities:"):233            page["unsupported_modalities"] = [x.strip() for x in b.split(":", 1)[1].split(",")]234        elif b.endswith("context window"):235            page["context_window"] = parse_int(b)236        elif b.endswith("max output tokens"):237            page["max_output"] = parse_int(b)238        elif b.startswith("Maximum input tokens:"):239            page["max_input"] = parse_int(b.split(":", 1)[1])240        elif b.endswith("knowledge cutoff"):241            page["knowledge_cutoff"] = norm_date(b)242        elif b.startswith("Reasoning token support"):243            page["reasoning"] = True244245    # pricing tables (### subsection -> rows) + bullet notes246    pricing_tables: dict[str, list[dict]] = OrderedDict()247    notes = []248    pl = sections.get("Pricing", [])249    sub = None250    i = 0251    while i < len(pl):252        l = pl[i]253        if l.startswith("### "):254            sub = l[4:].strip()255            i += 1256            continue257        if l.startswith("| Metric") and sub:258            _, rows, i = md_table(pl, i)259            # multi-line cells (sora) are already split; each row = [metric, price, unit]260            out = []261            for r in rows:262                if len(r) >= 3:263                    pm = parse_money(r[1])264                    out.append({"metric": r[0].strip(), "price": pm[0] if pm else None, "unit": r[2].strip()})265                elif len(r) == 1:  # continuation line of a multi-line metric cell266                    pass267            pricing_tables[sub] = out268            continue269        if l.startswith("- "):270            notes.append(l[2:].strip())271        elif l.strip() and not l.startswith("Pricing is based") and not l.startswith("|") and sub is None:272            notes.append(l.strip())273        i += 1274    page["pricing_tables"] = pricing_tables275    page["pricing_notes"] = notes276277    # endpoints278    eps = []279    el = sections.get("Endpoints", [])280    for idx, l in enumerate(el):281        if l.startswith("| Endpoint"):282            _, rows, _ = md_table(el, idx)283            for r in rows:284                if len(r) >= 3:285                    eps.append({"name": r[0], "route": r[1].strip("`"), "supported": r[2].strip().lower() == "supported"})286            break287    page["endpoints"] = eps288289    def bullets(name):290        return [l.strip()[2:].strip() for l in sections.get(name, []) if l.strip().startswith("- ")]291292    page["features"] = bullets("Supported features")293    page["has_features_section"] = "Supported features" in sections294    page["unsupported_features"] = bullets("Unsupported features")295    page["tools"] = bullets("Supported tools")296    page["has_tools_section"] = "Supported tools" in sections297    page["snapshots"] = [s.strip("`") for s in bullets("Snapshots")]298299    # rate limits: ### section (+ optional '> note') -> table300    rl_sections: dict[str, dict] = OrderedDict()301    rl = sections.get("Rate limits", [])302    rl_notes = []303    sub, note = "default", None304    i = 0305    while i < len(rl):306        l = rl[i]307        if l.startswith("### "):308            sub, note = l[4:].strip(), None309        elif l.startswith("> "):310            note = l[2:].strip()311        elif l.startswith("| Tier"):312            header, rows, i = md_table(rl, i)313            tiers = OrderedDict()314            for r in rows:315                tier = r[0].strip()316                tiers[tier] = OrderedDict()317                for h, v in zip(header[1:], r[1:]):318                    v = v.strip()319                    if v == "-":320                        tiers[tier][h] = None321                    elif re.match(r"^[\d,]+$", v):322                        tiers[tier][h] = int(v.replace(",", ""))323                    else:324                        tiers[tier][h] = v325            rl_sections[sub] = {"metrics": header[1:], "note": note, "tiers": tiers}326            continue327        elif l.strip() and not l.startswith("Rate limits ensure") and not l.startswith("|"):328            rl_notes.append(l.strip())329        i += 1330    page["rate_limits"] = rl_sections331    page["rate_limit_notes"] = rl_notes332333    # reasoning effort values from intro334    eff = None335    m = re.search(r"[Rr]easoning\.effort` supports `([^.]+)`\.", text) or \336        re.search(r"Reasoning\.effort supports: ([^.\n]+)\.", text) or \337        re.search(r"supports `?([a-z`, ]+(?:and|,) `?[a-z]+`?) reasoning effort settings", text) or \338        re.search(r"reasoning effort \(([a-z, ]+)\)", text)339    if m:340        raw = m.group(1)341        eff = [x.strip(" `") for x in re.split(r",| and ", raw) if x.strip(" `")]342        eff = [re.sub(r"\s*\(default\)", "", e).strip() for e in eff]343        page["reasoning_effort_default"] = None344        md = re.search(r"([a-z]+) \(default\)", raw)345        if md:346            page["reasoning_effort_default"] = md.group(1)347    if page["id"] == "gpt-5-pro":348        eff = ["high"]349        page["reasoning_effort_default"] = "high"350    page["reasoning_effort"] = eff351    return page352353354# --------------------------------------------------------------------------------------355# Family assignment356# --------------------------------------------------------------------------------------357def family_of(mid: str) -> str:358    m = mid359    if m.startswith("gpt-6"):360        return "gpt-6"361    if "cyber" in m or m.startswith("gpt-daybreak"):362        return "cyber-daybreak"363    if m == "gpt-rosalind-research":364        return "life-sciences"365    if m.startswith("gpt-oss"):366        return "open-weight"367    if m in ("chat-latest", "chatgpt-4o-latest") or m.endswith("-chat-latest"):368        return "chatgpt-latest"369    if "codex" in m:370        return "codex"371    if "embedding" in m or m.startswith("text-similarity") or m.startswith("text-search") or m.startswith("code-search"):372        return "embeddings"373    if "search" in m:374        return "search"375    if m == "computer-use-preview" or m.startswith("computer-use-preview-"):376        return "computer-use"377    if m.startswith("gpt-live-1"):378        return "live"379    if "transcribe" in m or m.startswith("whisper") or m in ("gpt-realtime-whisper", "gpt-realtime-translate"):380        return "speech-to-text"381    if "tts" in m:382        return "text-to-speech"383    if "realtime" in m:384        return "realtime"385    if "audio" in m:386        return "audio-chat"387    if "image" in m or m.startswith("dall-e"):388        return "image"389    if m.startswith("sora"):390        return "video"391    if "embedding" in m or m.startswith("text-similarity") or m.startswith("text-search") or m.startswith("code-search"):392        return "embeddings"393    if "moderation" in m:394        return "moderation"395    if re.match(r"^o[134]", m):396        return "o-series"397    if m.startswith("gpt-5.6"):398        return "gpt-5.6"399    if m.startswith("gpt-5.5"):400        return "gpt-5.5"401    if m.startswith("gpt-5.4"):402        return "gpt-5.4"403    if m.startswith("gpt-5.2"):404        return "gpt-5.2"405    if m.startswith("gpt-5.1"):406        return "gpt-5.1"407    if m.startswith("gpt-5"):408        return "gpt-5"409    if m.startswith("gpt-4.5"):410        return "gpt-4.5"411    if m.startswith("gpt-4.1"):412        return "gpt-4.1"413    if m.startswith("gpt-4o"):414        return "gpt-4o"415    if m.startswith("gpt-4"):416        return "gpt-4"417    if m.startswith("gpt-3.5"):418        return "gpt-3.5"419    if m in ("babbage-002", "davinci-002"):420        return "base-legacy"421    return "legacy-retired"422423424# --------------------------------------------------------------------------------------425# Deprecations parser426# --------------------------------------------------------------------------------------427NON_MODEL_HINTS = ("API", "OpenAI-Beta", "/v1/", "New fine-tuning training", "Videos API", "Assistants API")428429430def split_ids(cell: str) -> list[str]:431    ids = re.findall(r"`([^`]+)`", cell)432    return [i.strip() for i in ids if i.strip()]433434435def parse_deprecations() -> list[dict]:436    text = (DOCS_SRC / "deprecations.md").read_text(encoding="utf-8")437    lines = text.splitlines()438    entries = []439    phase, section_title, section_anchor, announced = None, None, None, None440    section_text = []441    i = 0442    while i < len(lines):443        l = lines[i]444        if l.startswith("## Upcoming deprecations"):445            phase = "upcoming"446        elif l.startswith("## Past deprecations"):447            phase = "past"448        elif l.startswith("## ") and phase:449            phase = None450        if l.startswith("### ") and phase:451            section_title = l[4:].strip()452            section_anchor = re.sub(r"[^a-z0-9]+", "-", section_title.lower()).strip("-")453            m = re.match(r"(\d{4}-\d{2}-\d{2}):\s*(.*)", section_title)454            announced = m.group(1) if m else None455            section_text = []456        elif l.startswith("#### ") and phase:457            section_text = []458        if phase and section_title and l.startswith("|") and not l.startswith("| ---"):459            header, rows, nxt = md_table(lines, i)460            hl = [h.lower() for h in header]461            if announced is None and section_text:462                for st in section_text:463                    m = re.search(r"On ([A-Z][a-z]+ \d{1,2}(?:st|nd|rd|th)?,? \d{4})", st)464                    if m:465                        announced = norm_date(m.group(1))466                        break467            if hl[0] == "date":  # milestone table (features)468                for r in rows:469                    entries.append({470                        "provider": "openai", "kind": "feature", "subject": section_title.split(":", 1)[-1].strip(),471                        "ids": [], "announced": announced, "milestone_date": norm_date(r[0]), "shutdown_date": None,472                        "update": r[1], "replacement": None, "phase": phase, "section": section_title,473                        "source": f"{BASE_URL}/deprecations#{section_anchor}", "retrieved_at": RETRIEVED_AT,474                    })475            else:476                subj_idx = 1477                repl_idx = next((k for k, h in enumerate(hl) if "replacement" in h or "substitute" in h), None)478                price_idx = next((k for k, h in enumerate(hl) if "price" in h), None)479                for r in rows:480                    if len(r) < 2:481                        continue482                    subject_cell = r[subj_idx]483                    ids = split_ids(subject_cell)484                    plain = re.sub(r"[`*]", "", subject_cell)485                    kind = "model"486                    if any(h in plain for h in NON_MODEL_HINTS) or subject_cell.startswith("/") or plain.startswith("OpenAI-Beta"):487                        kind = "endpoint_or_system"488                    if plain.startswith("ft-") or "fine-tuned" in section_title.lower() and plain.startswith("ft-"):489                        kind = "fine_tuned_model"490                    if any(x.startswith("ft-") for x in ids):491                        kind = "fine_tuned_model"492                    if kind == "model" and not ids:493                        ids = [plain.strip()] if plain.strip() and " " not in plain.strip() else []494                    entries.append({495                        "provider": "openai", "kind": kind,496                        "subject": plain.strip(),497                        "ids": ids, "primary_id": ids[0] if ids else None,498                        "announced": announced,499                        "shutdown_date": norm_date(r[0]), "shutdown_raw": r[0].strip(),500                        "replacement": re.sub(r"\s+", " ", r[repl_idx]).strip() if repl_idx is not None and repl_idx < len(r) else None,501                        "legacy_price": r[price_idx] if price_idx is not None and price_idx < len(r) else None,502                        "phase": phase, "section": section_title,503                        "source": f"{BASE_URL}/deprecations#{section_anchor}", "retrieved_at": RETRIEVED_AT,504                    })505            i = nxt506            continue507        if phase and section_title and l.strip() and not l.startswith("#"):508            section_text.append(l)509        i += 1510    return entries511512513# --------------------------------------------------------------------------------------514# pricing.md parser515# --------------------------------------------------------------------------------------516TIER_WORDS = {"Standard": "standard", "Batch": "batch", "Flex": "flex", "Fast mode": "fast"}517GROUP_WORDS = ["Flagship models", "Cyber models", "Multimodal models", "GPT-Live sessions",518               "Realtime and audio generation models", "Image generation models", "Video generation models",519               "Transcription models", "Tools", "Specialized models", "Finetuning"]520LONG_COLS = ["Short context input", "Short context cached input", "Short context cache writes", "Short context output",521             "Long context input", "Long context cached input", "Long context cache writes", "Long context output"]522LONG_DIM = {523    "Short context input": ("input", "short"), "Short context cached input": ("cached_input", "short"),524    "Short context cache writes": ("cache_write", "short"), "Short context output": ("output", "short"),525    "Long context input": ("input", "long"), "Long context cached input": ("cached_input", "long"),526    "Long context cache writes": ("cache_write", "long"), "Long context output": ("output", "long"),527}528529530def clean_model_cell(cell: str) -> tuple[str, str | None]:531    c = cell.strip().strip("`")532    note = None533    m = re.match(r"^(.*?)\s*\((.*)\)\s*$", c)534    if m:535        c, note = m.group(1).strip(), m.group(2).strip()536    if c == "Whisper":537        c = "whisper-1"538    return c, note539540541def parse_pricing_page() -> tuple[list[dict], list[str]]:542    lines = (DOCS_SRC / "pricing.md").read_text(encoding="utf-8").splitlines()543    prices: list[dict] = []544    footnotes: list[str] = []545    group, tier = None, "standard"546    src = f"{BASE_URL}/pricing"547    i = 0548549    def rec(model, dimension, price, unit, **kw):550        d = {"provider": "openai", "model_or_service": model, "dimension": dimension, "price": price,551             "currency": "USD", "unit": unit, "tier": kw.pop("tier", tier), "group": group,552             "effective_notes": kw.pop("effective_notes", None), "source": src, "retrieved_at": RETRIEVED_AT}553        d.update(kw)554        prices.append(d)555556    while i < len(lines):557        l = lines[i]558        s = l.strip()559        if s in GROUP_WORDS:560            group = s561            tier = "standard"562        elif s in TIER_WORDS:563            tier = TIER_WORDS[s]564        elif s.startswith("|") and not s.startswith("| ---"):565            header, rows, nxt = md_table(lines, i)566            h = header567            if h[0] == "Model" and len(h) == 9 and h[1] == "Short context input":568                for r in rows:569                    model, note = clean_model_cell(r[0])570                    for col, cell in zip(h[1:], r[1:]):571                        pm = parse_money(cell)572                        if pm is None:573                            continue574                        dim, ctx = LONG_DIM[col]575                        rec(model, dim, pm[0], "per 1M tokens", context=ctx,576                            effective_notes=(note + "; " if note else "") + ("long context (>272K input tokens) rate" if ctx == "long" else "short context rate"))577            elif h[:2] == ["Model", "Modality"]:578                for r in rows:579                    model, note = clean_model_cell(r[0])580                    modality = r[1].strip().lower()581                    for col, cell in zip(h[2:], r[2:]):582                        pm = parse_money(cell)583                        if pm is None:584                            continue585                        dim = {"Input": "input", "Cached input": "cached_input", "Output": "output", "Output / cost": "output"}[col]586                        rec(model, f"{modality}_{dim}", pm[0], pm[1] or "per 1M tokens", modality=modality, effective_notes=note)587            elif h[:2] == ["Model", "Size"]:588                for r in rows:589                    model, _ = clean_model_cell(r[0])590                    pm = parse_money(r[4])591                    rec(model, "video_output", pm[0], "per second", size=r[1], portrait=r[2], landscape=r[3])592            elif h[:2] == ["Model", "Use case"]:593                for r in rows:594                    model, _ = clean_model_cell(r[0])595                    for col, cell in zip(h[2:], r[2:]):596                        pm = parse_money(cell)597                        if pm is None:598                            continue599                        if col == "Input":600                            rec(model, "audio_input", pm[0], "per 1M tokens", use_case=r[1])601                        elif col == "Output":602                            rec(model, "text_output", pm[0], "per 1M tokens", use_case=r[1])603                        else:604                            rec(model, "audio_duration", pm[0], pm[1] or "per minute", use_case=r[1],605                                effective_notes="estimated cost per minute of audio")606            elif h[:2] == ["Tool", "Details"]:607                for r in rows:608                    tool, details, pricing = r[0], r[1], r[2]609                    for pm_txt in re.findall(r"\$[0-9.,]+(?: / [A-Za-z0-9 ]+)?(?: per [^.;,]+)?", pricing):610                        pass611                    # keep the full pricing string as effective_notes and extract first $ amount612                    pm = parse_money(pricing)613                    unit = pm[1] if pm and pm[1] else None614                    if "GB" in pricing and "session" in pricing:615                        unit = "per 20-minute session per container (by size)"616                    elif "GB-day" in pricing or "GB per day" in pricing:617                        unit = "per GB per day"618                    elif "1k calls" in pricing:619                        unit = "per 1k calls"620                    rec(f"tool:{tool}", details, pm[0] if pm else None, unit or "see notes", tier="standard",621                        effective_notes=pricing)622            elif h[:2] == ["Category", "Model"]:623                for r in rows:624                    model, note = clean_model_cell(r[1])625                    for col, cell in zip(h[2:], r[2:]):626                        pm = parse_money(cell)627                        if pm is None:628                            continue629                        dim = {"Input": "input", "Cached input": "cached_input", "Output": "output"}[col]630                        rec(model, dim, pm[0], "per 1M tokens", category=r[0], effective_notes=note)631            elif h[:2] == ["Model", "Training"]:632                for r in rows:633                    model, note = clean_model_cell(r[0])634                    for col, cell in zip(h[1:], r[1:]):635                        pm = parse_money(cell)636                        if pm is None:637                            continue638                        if col == "Training":639                            rec(model, "fine_tuning_training", pm[0], pm[1] or "per 1M training tokens", group="Finetuning",640                                effective_notes=note)641                        else:642                            dim = {"Input": "input", "Cached input": "cached_input", "Output": "output"}[col]643                            rec(model, f"fine_tuned_{dim}", pm[0], "per 1M tokens", effective_notes=note)644            elif h[:2] == ["Model", "Price per minute"]:645                for r in rows:646                    model, _ = clean_model_cell(r[0])647                    pm = parse_money(r[1])648                    rec(model, "session_duration", pm[0], "per minute (billed per second)")649            i = nxt650            continue651        elif s and not s.startswith("#") and len(s) > 60 and ("uplift" in s or "alias" in s or "billed" in s or652                                                              "Billing" in s or "winding down" in s or "Tokens used" in s):653            footnotes.append(s)654        i += 1655    return prices, footnotes656657658# --------------------------------------------------------------------------------------659# OpenAPI enum mention scan660# --------------------------------------------------------------------------------------661MODEL_LIKE = re.compile(r"^(gpt-|o[134]\b|o[134]-|chatgpt-|codex-|dall-e-|tts-1|whisper-1|text-embedding-|omni-moderation|text-moderation|sora-|babbage-002|davinci-002)")662663664def openapi_model_mentions() -> set[str]:665    ids = set()666    for l in (ROOT / "sources/openai/openapi/openapi-master.yaml").read_text(encoding="utf-8", errors="replace").splitlines():667        m = re.match(r"^\s+- ([A-Za-z0-9.\-]+)\s*$", l)668        if m and MODEL_LIKE.match(m.group(1)):669            ids.add(m.group(1))670    return ids671672673# --------------------------------------------------------------------------------------674# Build675# --------------------------------------------------------------------------------------676def main() -> None:677    live_raw = json.loads((ROOT / "sources/openai/models-api-raw.json").read_text())678    live = {m["id"]: m for m in live_raw["data"]}679    probes_path = ROOT / "sources/openai/live-model-probes.json"680    probes = json.loads(probes_path.read_text()) if probes_path.exists() else {"get_models": {}, "responses": {}}681682    pages = {}683    for p in sorted(MODEL_PAGES.glob("*.md")):684        if p.stem in ("all", "compare"):685            continue686        pg = parse_model_page(p)687        pages[pg["id"]] = pg688689    deps = parse_deprecations()690    dep_by_id: dict[str, list[dict]] = defaultdict(list)691    for e in deps:692        for mid in e["ids"]:693            dep_by_id[mid].append(e)694695    prices, footnotes = parse_pricing_page()696    price_models = {p["model_or_service"] for p in prices}697    openapi_ids = openapi_model_mentions()698699    # snapshot -> owning page700    snap_owner: dict[str, str] = {}701    for mid, pg in pages.items():702        for s in pg["snapshots"]:703            snap_owner.setdefault(s, mid)704        if pg["default_snapshot"]:705            snap_owner.setdefault(pg["default_snapshot"], mid)706    # extra aliases documented in prose707    ALIAS_TARGET = {"gpt-5.6": "gpt-5.6-sol"}708    for a, t in ALIAS_TARGET.items():709        snap_owner.setdefault(a, t)710711    # id universe712    universe = set(live) | set(pages) | set(snap_owner) | {i for e in deps if e["kind"] in ("model",) for i in e["ids"]}713    universe |= {m for m in price_models if not m.startswith("tool:")}714    universe |= {"gpt-rosalind-research", "gpt-5.5-cyber", "gpt-5.4-cyber"}715    universe |= openapi_ids716    universe = {u for u in universe if u and " " not in u and u not in ("gpt-4o-tts",)}717718    live_only = sorted(u for u in live if u not in pages and u not in snap_owner)719    docs_only = sorted(u for u in pages if u not in live)720721    records = []722    for mid in sorted(universe):723        owner = mid if mid in pages else snap_owner.get(mid)724        pg = pages.get(owner) if owner else None725        is_snapshot = pg is not None and mid != owner and is_dated(mid)726        is_alias = pg is not None and mid != owner and not is_dated(mid)727        lv = live.get(mid)728        deps_for = dep_by_id.get(mid, [])729730        status: list[str] = []731        flags: list[str] = []732        if pg:733            status.append("DOCUMENTED")734        elif mid in price_models or mid in ("gpt-rosalind-research", "gpt-5.5-cyber"):735            status.append("DOCUMENTED")736            flags.append("NO_MODEL_PAGE")737        elif mid in openapi_ids and not lv and not deps_for:738            status.append("DOCUMENTED")739            flags.append("OPENAPI_ENUM_ONLY")740        if lv:741            status.append("LIVE_VERIFIED")742            if not pg:743                status.append("LIVE_DISCOVERED")744                flags.append("DOCUMENTATION_INCOMPLETE")745        # deprecation / retirement746        shutdown_dates = [e["shutdown_date"] for e in deps_for if e.get("shutdown_date")]747        live_sd = lv.get("shutdown_date") if lv else None748        if live_sd:749            shutdown_dates.append(live_sd)750        sd = min(shutdown_dates) if shutdown_dates else None751        latest_sd = max(shutdown_dates) if shutdown_dates else None752        if deps_for or live_sd:753            if latest_sd and latest_sd <= TODAY:754                status.append("RETIRED")755                if lv:756                    flags.append("STILL_LISTED_AFTER_DOCUMENTED_SHUTDOWN")757            else:758                status.append("DEPRECATED")759        if "preview" in mid:760            status.append("PREVIEW")761        if mid in LEGACY_IDS:762            status.append("LEGACY")763        if pg and pg["intro"] and re.search(r"has been deprecated and removed", pg["intro"]) and "RETIRED" not in status:764            status.append("RETIRED")765        # probes766        verification = {"method": "docs_only", "verified_at": RETRIEVED_AT, "result": None, "http_status": None, "request_note": None}767        if lv:768            verification = {"method": "live_api", "verified_at": RETRIEVED_AT, "result": "success", "http_status": 200,769                            "request_note": "id present in GET /v1/models listing (2026-09-18)"}770        gp = probes.get("get_models", {}).get(mid)771        if gp:772            verification["get_model"] = gp773            if gp["status"] == 200:774                if "LIVE_VERIFIED" not in status:775                    status.append("LIVE_VERIFIED")776                verification.update({"method": "live_api", "result": "success", "http_status": 200,777                                     "request_note": f"GET /v1/models/{mid} -> 200"})778            elif gp["status"] in (401, 403) or (gp["status"] == 404 and mid in RESTRICTED_ACCESS):779                status.append("ACCOUNT_RESTRICTED")780                verification.update({"method": "live_api", "result": "restricted", "http_status": gp["status"],781                                     "request_note": f"GET /v1/models/{mid} -> {gp['status']} {gp.get('error_code') or ''}".strip()})782            elif gp["status"] == 404:783                verification.update({"method": "live_api", "result": "failure", "http_status": 404,784                                     "request_note": f"GET /v1/models/{mid} -> 404 (a 404 with our key never means the model does not exist)"})785                if "RETIRED" not in status and "DEPRECATED" not in status:786                    flags.append("NOT_RESOLVABLE_WITH_OUR_KEY")787        rp = probes.get("responses", {}).get(mid)788        if rp:789            verification["post_responses"] = rp790            if rp["status"] == 200:791                if "LIVE_VERIFIED" not in status:792                    status.append("LIVE_VERIFIED")793                if gp and gp["status"] == 404:794                    flags = [f for f in flags if f != "NOT_RESOLVABLE_WITH_OUR_KEY"] + ["ALIAS_WORKS_ON_POST_BUT_404_ON_GET_MODELS"]795                verification.update({"method": "live_api", "result": "success", "http_status": 200,796                                     "request_note": f"POST /v1/responses ('Reply with OK.', max_output_tokens 16) -> 200, model echoed: {rp.get('model_echo')}"})797            elif rp["status"] in (401, 403):798                status.append("ACCOUNT_RESTRICTED")799            else:800                status.append("FAILED_VERIFICATION")801                verification.update({"result": "failure", "http_status": rp["status"], "request_note": rp.get("error_message")})802        # de-dup keep order803        status = list(OrderedDict.fromkeys(status))804        if not status:805            status = ["UNVERIFIED"]806807        # ---- fields from page808        fam = family_of(mid)809        display = pg["display_name"] if pg and mid == owner else (f"{pg['display_name']} ({'snapshot' if is_snapshot else 'alias'} {mid})" if pg else mid)810        description = pg["description"] if pg else None811        if mid == "gpt-rosalind-research":812            description = "Life sciences reasoning for approved organizations (trusted-access program)."813        if mid == "gpt-5.5-cyber":814            description = "Cybersecurity model listed on the pricing page (Daybreak)."815        if mid == "gpt-5.4-cyber":816            description = "Cybersecurity model, deprecated 2026-09-11, shutdown 2026-10-01 (replacement gpt-5.6-cyber)."817818        snapshots = [s for s in (pg["snapshots"] if pg else []) if is_dated(s)]819        aliases = [s for s in (pg["snapshots"] if pg else []) if not is_dated(s) and s != mid]820        for a, t in ALIAS_TARGET.items():821            if t == mid:822                aliases.append(a)823        if mid == "gpt-5.6-sol":824            aliases.append("gpt-daybreak-blue-latest")825        if mid == "gpt-5.6-cyber":826            aliases.append("gpt-daybreak-red-latest")827        aliases = sorted(set(aliases))828829        # release date830        rd, rd_src = None, None831        if mid in RELEASE_DATES:832            rd, rd_src = RELEASE_DATES[mid], "changelog"833        elif owner in RELEASE_DATES and is_alias:834            rd, rd_src = RELEASE_DATES[owner], "changelog (parent model)"835        elif mid in RELEASE_DATES_EXTERNAL:836            rd, rd_src = RELEASE_DATES_EXTERNAL[mid], "public OpenAI announcement (not in downloaded docs)"837        elif is_snapshot and re.search(r"(\d{4}-\d{2}-\d{2})$", mid):838            rd, rd_src = re.search(r"(\d{4}-\d{2}-\d{2})$", mid).group(1), "snapshot date in id"839        elif lv:840            rd, rd_src = datetime.fromtimestamp(lv["created"], tz=timezone.utc).strftime("%Y-%m-%d"), "GET /v1/models created timestamp (approximation)"841842        in_mod = pg["input_modalities"] if pg else []843        out_mod = pg["output_modalities"] if pg else []844        feats = set(pg["features"]) if pg else set()845        unsup = set(pg["unsupported_features"]) if pg else set()846        tools = set(pg["tools"]) if pg else set()847        eps_sup = {e["name"] for e in pg["endpoints"] if e["supported"]} if pg else set()848        has_feats = bool(pg and pg["has_features_section"])849        has_tools = bool(pg and pg["has_tools_section"])850        supports_responses = "Responses" in eps_sup851852        def feat(name):853            if not pg:854                return "unknown"855            if name in feats:856                return True857            if name in unsup:858                return False859            return False if has_feats else "unknown"860861        def tool(name):862            if not pg:863                return "unknown"864            if name in tools:865                return True866            if has_tools:867                return False868            return False if not supports_responses else "unknown"869870        caps = OrderedDict()871        if pg:872            caps["text_in"] = "text" in in_mod873            caps["text_out"] = "text" in out_mod874            caps["image_in"] = ("image" in in_mod) or ("image_input" in feats)875            caps["image_out"] = "image" in out_mod876            caps["audio_in"] = "audio" in in_mod877            caps["audio_out"] = "audio" in out_mod878            caps["video_in"] = "video" in in_mod879            caps["video_out"] = "video" in out_mod880            caps["reasoning"] = pg["reasoning"]881            caps["reasoning_effort_values"] = pg["reasoning_effort"] if pg["reasoning"] else None882            caps["reasoning_effort_default"] = pg.get("reasoning_effort_default")883            caps["reasoning_mode_pro"] = True if owner in PRO_MODE_MODELS else ("unknown" if pg["reasoning"] and fam in ("gpt-6",) else False)884            caps["streaming"] = feat("streaming")885            caps["structured_outputs"] = feat("structured_outputs")886            caps["function_calling"] = feat("function_calling") if "function_calling" in feats or has_feats else ("function_calling" in tools or "unknown")887            caps["prompt_caching"] = feat("prompt_caching")888            caps["extended_prompt_cache_retention_24h"] = True if (owner in EXTENDED_CACHE_RETENTION or fam in ("gpt-5.6", "gpt-6", "cyber-daybreak")) else ("unknown" if caps["prompt_caching"] is True else False)889            caps["explicit_cache_breakpoints"] = True if fam in ("gpt-5.6", "gpt-6", "cyber-daybreak") else False890            caps["predicted_outputs"] = feat("predicted_outputs")891            caps["file_uploads"] = feat("file_uploads")892            caps["evals"] = feat("evals")893            caps["stored_completions"] = feat("stored_completions")894            caps["distillation"] = feat("stored_completions")895            caps["inpainting"] = feat("inpainting") if fam == "image" else False896            caps["fine_tuning"] = True if ("fine_tuning" in feats or "Fine-tuning" in eps_sup) else (False if "fine_tuning" in unsup or has_feats or pg["endpoints"] else "unknown")897            caps["batch"] = "Batch" in eps_sup898            caps["embeddings"] = "Embeddings" in eps_sup899            caps["realtime"] = "Realtime" in eps_sup900            caps["live_sessions"] = "Live" in eps_sup901            caps["moderation"] = "Moderation" in eps_sup902            caps["image_generation_api"] = "Image generation" in eps_sup903            caps["image_edit_api"] = "Image edit" in eps_sup904            caps["video_generation"] = "Videos" in eps_sup905            caps["speech_generation"] = "Speech generation" in eps_sup906            caps["transcription"] = "Transcription" in eps_sup or "Realtime transcription" in eps_sup907            caps["translation"] = "Translation" in eps_sup or "Realtime translation" in eps_sup908            caps["completions_legacy"] = "Completions (legacy)" in eps_sup909            caps["compaction"] = True if mid in COMPACTION_MODELS or owner in COMPACTION_MODELS else "unknown"910            caps["long_context_tier_272k"] = bool(pg["context_window"] and pg["context_window"] > 400000)911            # tools (Responses API)912            caps["tool_web_search"] = tool("web_search") if tool("web_search") != "unknown" else ("web_search" in feats or "unknown")913            caps["tool_file_search"] = tool("file_search") if tool("file_search") != "unknown" else ("file_search" in feats or "unknown")914            caps["tool_code_interpreter"] = tool("code_interpreter")915            caps["tool_image_generation"] = tool("image_generation") if tool("image_generation") != "unknown" else ("image_generation" in feats or "unknown")916            caps["tool_mcp"] = tool("mcp") if tool("mcp") != "unknown" else ("mcp" in feats or "unknown")917            caps["tool_computer_use"] = tool("computer_use") if mid != "computer-use-preview" and owner != "computer-use-preview" else True918            caps["tool_hosted_shell"] = tool("hosted_shell")919            caps["tool_apply_patch"] = tool("apply_patch")920            caps["tool_skills"] = tool("skills")921            caps["tool_tool_search"] = tool("tool_search")922            caps["tool_function_calling"] = tool("function_calling") if has_tools else caps["function_calling"]923            caps["web_search_feature"] = "web_search" in feats924        else:925            caps["note"] = "no model page; capabilities unknown"926927        service_tiers = {t: any(p["model_or_service"] == (owner or mid) and p["tier"] == t for p in prices)928                         for t in ("standard", "batch", "flex", "fast")}929        if pg and "Batch" in eps_sup:930            service_tiers["batch"] = True931932        # pricing object933        pricing = OrderedDict()934        for p in prices:935            if p["model_or_service"] != (owner or mid):936                continue937            tier = p["tier"]938            ctx = p.get("context")939            key = tier if not ctx or ctx == "short" else f"{tier}_long_context"940            pricing.setdefault(key, OrderedDict())941            dim = p["dimension"]942            if p.get("modality") and not dim.startswith(p["modality"]):943                dim = f"{p['modality']}_{dim}"944            if p.get("size"):945                dim = f"video_output_{p['size']}"946            pricing[key][dim] = p["price"]947            pricing[key].setdefault("_unit", p["unit"])948        if pg:949            for sub, rows in pg["pricing_tables"].items():950                key = "model_page:" + sub951                pricing[key] = OrderedDict((r["metric"], {"price": r["price"], "unit": r["unit"]}) for r in rows if r["price"] is not None)952            if pg["pricing_notes"]:953                pricing["notes"] = pg["pricing_notes"]954        if not pricing:955            pricing["note"] = "no price found in pricing.md or model page (retired / not billed / open-weight)"956957        endpoints = [{"name": e["name"], "route": "/" + e["route"]} for e in pg["endpoints"] if e["supported"]] if pg else []958959        rate_limits = OrderedDict()960        if pg:961            rate_limits["documented_tiers"] = pg["rate_limits"]962            if pg["rate_limit_notes"]:963                rate_limits["notes"] = pg["rate_limit_notes"]964            rate_limits["source"] = f"{BASE_URL}/models/{owner}"965        rate_limits["observed"] = "see generated/fragments/rate-limits/openai-rate-limits.json#observed (never generalize)"966967        # deprecation summary968        dep_summary = []969        for e in deps_for:970            dep_summary.append({"announced": e["announced"], "shutdown_date": e["shutdown_date"],971                                "replacement": e["replacement"], "phase": e["phase"], "section": e["section"], "source": e["source"]})972        if live_sd:973            dep_summary.append({"announced": None, "shutdown_date": live_sd, "replacement": None, "phase": "live",974                                "section": "GET /v1/models shutdown_date field", "source": "https://api.openai.com/v1/models"})975        doc_sd = [e["shutdown_date"] for e in deps_for if e.get("shutdown_date")]976        if live_sd and doc_sd and live_sd not in doc_sd:977            flags.append("SHUTDOWN_DATE_MISMATCH_DOCS_VS_LIVE")978        if live_sd and not doc_sd:979            flags.append("SHUTDOWN_DATE_ONLY_IN_LIVE_API")980981        sources = []982        if pg:983            sources.append({"url": f"{BASE_URL}/models/{owner}", "retrieved_at": RETRIEVED_AT})984        if lv:985            sources.append({"url": "https://api.openai.com/v1/models", "retrieved_at": RETRIEVED_AT, "note": "GET /v1/models (live listing)"})986        if (owner or mid) in price_models:987            sources.append({"url": f"{BASE_URL}/pricing", "retrieved_at": RETRIEVED_AT})988        for e in deps_for:989            if not any(s["url"] == e["source"] for s in sources):990                sources.append({"url": e["source"], "retrieved_at": RETRIEVED_AT})991        if mid in openapi_ids:992            sources.append({"url": OPENAPI_URL, "retrieved_at": RETRIEVED_AT, "note": "id appears in an OpenAPI enum"})993        if mid in RELEASE_DATES:994            sources.append({"url": f"{BASE_URL}/changelog", "retrieved_at": RETRIEVED_AT})995996        availability = OrderedDict()997        availability["account"] = RESTRICTED_ACCESS.get(mid) or RESTRICTED_ACCESS.get(owner or "") or "general (usage-tier based)"998        availability["service_tiers"] = service_tiers999        availability["listed_in_get_models"] = bool(lv)1000        availability["owned_by"] = lv["owned_by"] if lv else None1001        availability["created_ts"] = lv["created"] if lv else None1002        availability["openapi_enum"] = mid in openapi_ids1003        if fam == "open-weight":1004            availability["distribution"] = "open weights on Hugging Face (Apache 2.0); documented Responses/Batch endpoint table, rate limits 0 for all tiers"10051006        rec = OrderedDict([1007            ("provider", "openai"),1008            ("id", mid),1009            ("record_kind", "snapshot" if is_snapshot else ("alias" if is_alias else ("model" if pg else "id_only"))),1010            ("canonical_model", owner or mid),1011            ("display_name", display),1012            ("description", description),1013            ("aliases", aliases if mid == owner else []),1014            ("snapshots", snapshots if mid == owner else []),1015            ("default_snapshot", pg["default_snapshot"] if pg else None),1016            ("family", fam),1017            ("status", status),1018            ("flags", flags),1019            ("release_date", rd),1020            ("release_date_source", rd_src),1021            ("knowledge_cutoff", pg["knowledge_cutoff"] if pg else None),1022            ("context_window", pg["context_window"] if pg else None),1023            ("max_input", pg["max_input"] if pg else None),1024            ("max_output", pg["max_output"] if pg else None),1025            ("modalities", {"input": in_mod, "output": out_mod, "unsupported": pg["unsupported_modalities"] if pg else []}),1026            ("capabilities", caps),1027            ("endpoints", endpoints),1028            ("tools", sorted(tools)),1029            ("features_documented", sorted(feats)),1030            ("pricing", pricing),1031            ("rate_limits", rate_limits),1032            ("beta_headers", []),1033            ("restrictions", RESTRICTED_ACCESS.get(mid) or RESTRICTED_ACCESS.get(owner or "")),1034            ("availability", availability),1035            ("deprecation", dep_summary),1036            ("shutdown_date", latest_sd),1037            ("intro", pg["intro"][:600] if pg and mid == owner else None),1038            ("last_verified", RETRIEVED_AT),1039            ("verification", verification),1040            ("sources", sources),1041        ])1042        records.append(rec)10431044    # ---- additional price records from model pages not covered by pricing.md1045    page_prices = []1046    for mid, pg in pages.items():1047        if mid in price_models:1048            continue1049        for sub, rows in pg["pricing_tables"].items():1050            for r in rows:1051                if r["price"] is None:1052                    continue1053                dim_map = {"Input": "input", "Cached input": "cached_input", "Output": "output", "Cache writes": "cache_write",1054                           "Cost": "cost", "Price": "audio_duration", "Per minute": "session_duration"}1055                dim = dim_map.get(r["metric"], r["metric"].lower().replace(" ", "_"))1056                prefix = {"Audio tokens": "audio_", "Image tokens": "image_", "Video generation": "video_output_",1057                          "Embeddings": "embeddings_"}.get(sub, "")1058                page_prices.append({"provider": "openai", "model_or_service": mid, "dimension": prefix + dim if not dim.startswith(prefix) else dim,1059                                    "price": r["price"], "currency": "USD",1060                                    "unit": "per " + r["unit"] if not r["unit"].startswith("per") else r["unit"],1061                                    "tier": "standard", "group": "model page", "effective_notes": "; ".join(pg["pricing_notes"]) or None,1062                                    "source": f"{BASE_URL}/models/{mid}", "retrieved_at": RETRIEVED_AT})1063    all_prices = prices + page_prices1064    # derived tier records (documented rules)1065    rules = [1066        {"provider": "openai", "model_or_service": "rule:batch", "dimension": "discount", "price": -50, "currency": "USD", "unit": "percent vs standard",1067         "tier": "batch", "effective_notes": "Batch API: 50% lower cost, 24h completion window; per-model batch tables on the pricing page", "source": f"{BASE_URL}/guides/batch", "retrieved_at": RETRIEVED_AT},1068        {"provider": "openai", "model_or_service": "rule:flex", "dimension": "discount", "price": -50, "currency": "USD", "unit": "percent vs standard",1069         "tier": "flex", "effective_notes": "Flex processing (beta): tokens priced at Batch API rates; slower, may return 429 resource_unavailable", "source": f"{BASE_URL}/guides/flex-processing", "retrieved_at": RETRIEVED_AT},1070        {"provider": "openai", "model_or_service": "rule:fast", "dimension": "premium", "price": 100, "currency": "USD", "unit": "percent vs standard",1071         "tier": "fast", "effective_notes": "Fast mode (ex-Priority processing, renamed 2026-07-30): 2x standard token rates for GPT-5.6 Sol / GPT-6 Astra; service_tier 'fast' or 'priority'", "source": f"{BASE_URL}/guides/fast-mode", "retrieved_at": RETRIEVED_AT},1072        {"provider": "openai", "model_or_service": "rule:cache_write", "dimension": "cache_write", "price": 125, "currency": "USD", "unit": "percent of uncached input rate",1073         "tier": "standard", "effective_notes": "GPT-5.6 and later: cache writes billed at 1.25x uncached input; earlier models: no cache-write charge", "source": f"{BASE_URL}/guides/prompt-caching", "retrieved_at": RETRIEVED_AT},1074        {"provider": "openai", "model_or_service": "rule:cache_read_gpt-5.6+", "dimension": "cached_input", "price": 10, "currency": "USD", "unit": "percent of uncached input rate",1075         "tier": "standard", "effective_notes": "GPT-5.6 and later: cache reads at 0.1x uncached input rate", "source": f"{BASE_URL}/guides/prompt-caching", "retrieved_at": RETRIEVED_AT},1076        {"provider": "openai", "model_or_service": "rule:long_context", "dimension": "multiplier", "price": None, "currency": "USD", "unit": "x standard",1077         "tier": "standard", "effective_notes": "Prompts >272K input tokens: 2x input (and cache) rates, 1.5x output rate for the full request (GPT-5.4 / 5.5 / 5.6 / 6 Astra)", "source": f"{BASE_URL}/pricing", "retrieved_at": RETRIEVED_AT},1078        {"provider": "openai", "model_or_service": "rule:data_residency", "dimension": "uplift", "price": 10, "currency": "USD", "unit": "percent",1079         "tier": "standard", "effective_notes": "Regional processing endpoints (us./eu./ae. …api.openai.com): 10% uplift for models released on or after 2026-03-05 that are eligible for data residency", "source": f"{BASE_URL}/pricing", "retrieved_at": RETRIEVED_AT},1080        {"provider": "openai", "model_or_service": "rule:container_billing", "dimension": "session", "price": None, "currency": "USD", "unit": "per minute, 5-minute minimum",1081         "tier": "standard", "effective_notes": "Since 2026-06-02 eligible container sessions (Code Interpreter / Hosted Shell) are billed per minute with a 5-minute minimum instead of the full 20-minute rate", "source": f"{BASE_URL}/changelog", "retrieved_at": RETRIEVED_AT},1082        {"provider": "openai", "model_or_service": "rule:web_search_content_tokens_mini", "dimension": "input", "price": None, "currency": "USD", "unit": "8,000 input tokens per call",1083         "tier": "standard", "effective_notes": "gpt-4o-mini and gpt-4.1-mini with the non-preview web search tool: search content tokens billed as a fixed block of 8,000 input tokens per call", "source": f"{BASE_URL}/pricing", "retrieved_at": RETRIEVED_AT},1084        {"provider": "openai", "model_or_service": "rule:pro_mode", "dimension": "output", "price": None, "currency": "USD", "unit": "standard token rates",1085         "tier": "standard", "effective_notes": "reasoning.mode: pro (GPT-5.6) aggregates all model work and bills it at the model's standard token rates (more tokens than standard mode)", "source": f"{BASE_URL}/guides/reasoning", "retrieved_at": RETRIEVED_AT},1086        {"provider": "openai", "model_or_service": "gpt-rosalind-research", "dimension": "billing_start", "price": None, "currency": "USD", "unit": "date",1087         "tier": "standard", "effective_notes": "Billing begins 2026-10-05; cache-write pricing does not apply", "source": f"{BASE_URL}/pricing", "retrieved_at": RETRIEVED_AT},1088        {"provider": "openai", "model_or_service": "gpt-5.6-sol", "dimension": "promotion", "price": None, "currency": "USD", "unit": "note",1089         "tier": "standard", "effective_notes": "Promotional pricing ($4 in / $20 out) available at least through 2026-11-21", "source": f"{BASE_URL}/pricing", "retrieved_at": RETRIEVED_AT},1090    ]1091    for r in rules:1092        r.setdefault("group", "rules")1093    all_prices += rules10941095    # ---- rate limits fragment1096    rl_fragment = OrderedDict([1097        ("provider", "openai"),1098        ("retrieved_at", RETRIEVED_AT),1099        ("sources", [f"{BASE_URL}/guides/rate-limits", f"{BASE_URL}/models/<id> (Rate limits section)", f"{BASE_URL}/guides/batch", f"{BASE_URL}/guides/fast-mode"]),1100        ("concepts", [1101            {"metric": "RPM", "meaning": "requests per minute", "scope": "organization and project, per model (or shared-limit group)"},1102            {"metric": "RPD", "meaning": "requests per day", "scope": "some models / free tier"},1103            {"metric": "TPM", "meaning": "tokens per minute (max(max_tokens, estimated prompt tokens) counted per request)", "scope": "per model"},1104            {"metric": "TPD", "meaning": "tokens per day", "scope": "some models"},1105            {"metric": "IPM", "meaning": "images per minute", "scope": "image models (gpt-image-*)"},1106            {"metric": "minutes-of-audio per minute", "meaning": "audio duration admitted per minute", "scope": "streaming audio models (gpt-realtime-whisper, gpt-realtime-translate)"},1107            {"metric": "concurrent sessions", "meaning": "simultaneous live sessions", "scope": "gpt-live-1 (v1/live/sessions)"},1108            {"metric": "batch queue limit", "meaning": "total input tokens queued across pending batch jobs for a model; released when the batch completes", "scope": "Batch API, per model"},1109            {"metric": "long-context limit", "meaning": "separate RPM/TPM/batch-queue table for requests >272K input tokens (GPT-5.4/5.5/5.6/6 Astra)", "scope": "long-context models; visible in the console"},1110            {"metric": "shared limits", "meaning": "some model families share one limit pool (listed under 'shared limit' in the console)", "scope": "organization"},1111            {"metric": "project-scoped token limit", "meaning": "optional per-project token limit exposed via x-ratelimit-*-project-tokens headers", "scope": "project"},1112            {"metric": "monthly usage limit", "meaning": "approved monthly spend ceiling per organization, separate from configurable spend limits", "scope": "organization"},1113            {"metric": "vector store ingestion", "meaning": "/vector_stores/{id}/files and /file_batches share 300 requests per minute per vector store", "scope": "per vector store"},1114            {"metric": "ramp-rate (slow_down)", "meaning": "429 rate_limit_error/slow_down when traffic increases too quickly even below RPM/TPM; above ~1M TPM ramp at most +50% every 15 minutes", "scope": "per model"},1115        ]),1116        ("usage_tiers", [1117            {"tier": "Free", "qualification": "allowed geography", "usage_limit_usd_per_month": 100},1118            {"tier": "Tier 1", "qualification": "$5 paid", "usage_limit_usd_per_month": 100},1119            {"tier": "Tier 2", "qualification": "$50 paid", "usage_limit_usd_per_month": 500},1120            {"tier": "Tier 3", "qualification": "$100 paid", "usage_limit_usd_per_month": 1000},1121            {"tier": "Tier 4", "qualification": "$250 paid", "usage_limit_usd_per_month": 5000},1122            {"tier": "Tier 5", "qualification": "$1,000 paid", "usage_limit_usd_per_month": 200000},1123        ]),1124        ("headers", [1125            {"name": "Retry-After", "sample": "56", "description": "minimum seconds to wait before retrying a temporary 429 (slow_down / rate limit) or 503 (server_is_overloaded)"},1126            {"name": "x-ratelimit-limit-requests", "sample": "60", "description": "max requests before exhausting the rate limit"},1127            {"name": "x-ratelimit-limit-tokens", "sample": "150000", "description": "max tokens before exhausting the rate limit"},1128            {"name": "x-ratelimit-remaining-requests", "sample": "59", "description": "remaining requests"},1129            {"name": "x-ratelimit-remaining-tokens", "sample": "149984", "description": "remaining tokens"},1130            {"name": "x-ratelimit-reset-requests", "sample": "1s", "description": "time until request limit resets"},1131            {"name": "x-ratelimit-reset-tokens", "sample": "6m0s", "description": "time until token limit resets"},1132            {"name": "x-ratelimit-limit-project-tokens", "sample": "60000", "description": "project token limit (present when a project-scoped limit applies)"},1133            {"name": "x-ratelimit-remaining-project-tokens", "sample": "57000", "description": "remaining project tokens"},1134            {"name": "x-ratelimit-reset-project-tokens", "sample": "3s", "description": "time until project token limit resets"},1135        ]),1136        ("errors", [1137            {"http_status": 429, "type": "rate_limit_error", "code": "slow_down", "meaning": "request rate increased too quickly", "action": "honour Retry-After, reduce rate, ramp gradually"},1138            {"http_status": 429, "type": "rate_limit_error", "code": "rate_limit_exceeded", "meaning": "RPM/TPM/RPD/TPD/IPM exhausted", "action": "exponential backoff with jitter; batch requests; reduce max_tokens"},1139            {"http_status": 429, "type": "insufficient_quota", "code": "insufficient_quota", "meaning": "monthly usage / spend limit reached (429 also returned when a hard spend limit is hit)", "action": "do not retry; raise limits"},1140            {"http_status": 503, "type": "service_unavailable_error", "code": "server_is_overloaded", "meaning": "model temporarily overloaded", "action": "honour Retry-After, retry with increasing delay"},1141        ]),1142        ("fine_tuning_limits_endpoint", "GET /v1/fine_tuning/model_limits"),1143        ("capacity_products", ["Scale Tier (pay-as-you-go traffic routinely hitting ramp limits)", "Reserved Tier (GPT-5.6 and later)", "Ultrafast mode (limited preview, GPT-5.6 Sol, announced 2026-08-13)"]),1144        ("per_model_documented_tiers", OrderedDict((mid, pg["rate_limits"]) for mid, pg in sorted(pages.items()) if pg["rate_limits"])),1145        ("per_model_notes", OrderedDict((mid, pg["rate_limit_notes"]) for mid, pg in sorted(pages.items()) if pg["rate_limit_notes"])),1146        ("observed", [1147            {"date": RETRIEVED_AT, "request": "POST /v1/responses (this atlas key)", "headers": {"x-ratelimit-limit-requests": "30000", "x-ratelimit-limit-tokens": "180000000"},1148             "note": "observed for OUR key/org/project on 2026-09-18 — account-specific, do not generalize"},1149        ] + probes.get("observed_headers", [])),1150    ])11511152    # ---- deprecations fragment: add live shutdown dates not on the page1153    dep_fragment = list(deps)1154    doc_ids_with_sd = {i for e in deps for i in e["ids"] if e.get("shutdown_date")}1155    for mid, lv in sorted(live.items()):1156        if lv.get("shutdown_date"):1157            page_dates = {e["shutdown_date"] for e in dep_by_id.get(mid, []) if e.get("shutdown_date")}1158            dep_fragment.append({1159                "provider": "openai", "kind": "model", "subject": mid, "ids": [mid], "primary_id": mid, "announced": None,1160                "shutdown_date": lv["shutdown_date"], "shutdown_raw": lv["shutdown_date"], "replacement": None, "legacy_price": None,1161                "phase": "live_api_field", "section": "GET /v1/models shutdown_date",1162                "consistency": ("matches deprecations page" if lv["shutdown_date"] in page_dates else1163                                ("NOT on deprecations page" if not page_dates else f"differs from page dates {sorted(page_dates)}")),1164                "source": "https://api.openai.com/v1/models", "retrieved_at": RETRIEVED_AT})11651166    out = ROOT / "generated/fragments"1167    (out / "models").mkdir(parents=True, exist_ok=True)1168    (out / "pricing").mkdir(parents=True, exist_ok=True)1169    (out / "rate-limits").mkdir(parents=True, exist_ok=True)1170    (out / "deprecations").mkdir(parents=True, exist_ok=True)1171    meta = {"generated_by": "scripts/build_openai_models.py", "retrieved_at": RETRIEVED_AT,1172            "counts": {"records": len(records), "live_ids": len(live), "model_pages": len(pages),1173                       "live_only_ids": live_only, "docs_only_ids": docs_only}}1174    (out / "models/openai-models.json").write_text(json.dumps(records, indent=1, ensure_ascii=False) + "\n")1175    (out / "models/openai-models.meta.json").write_text(json.dumps(meta, indent=1) + "\n")1176    (out / "pricing/openai-pricing.json").write_text(json.dumps(all_prices, indent=1, ensure_ascii=False) + "\n")1177    (out / "rate-limits/openai-rate-limits.json").write_text(json.dumps(rl_fragment, indent=1, ensure_ascii=False) + "\n")1178    (out / "deprecations/openai-deprecations.json").write_text(json.dumps(dep_fragment, indent=1, ensure_ascii=False) + "\n")11791180    write_docs(records, pages, all_prices, footnotes, dep_fragment, live_only, docs_only, probes)1181    print(f"records={len(records)} prices={len(all_prices)} deprecations={len(dep_fragment)} live_only={len(live_only)} docs_only={len(docs_only)}")118211831184# --------------------------------------------------------------------------------------1185# Docs1186# --------------------------------------------------------------------------------------1187def fmt(v):1188    if v is True:1189        return "✅"1190    if v is False:1191        return "—"1192    if v is None:1193        return ""1194    if isinstance(v, list):1195        return ", ".join(map(str, v))1196    return str(v)119711981199def money(v):1200    return "" if v is None else (f"${v:g}")120112021203def write_docs(records, pages, prices, footnotes, deps, live_only, docs_only, probes):1204    by_id = {r["id"]: r for r in records}1205    canon = [r for r in records if r["record_kind"] in ("model", "id_only")]1206    fam_order = ["gpt-6", "gpt-5.6", "cyber-daybreak", "gpt-5.5", "gpt-5.4", "gpt-5.2", "gpt-5.1", "gpt-5", "codex", "o-series",1207                 "computer-use", "search", "chatgpt-latest", "life-sciences", "gpt-4.1", "gpt-4.5", "gpt-4o", "realtime", "live",1208                 "audio-chat", "speech-to-text", "text-to-speech", "image", "video", "embeddings", "moderation", "open-weight",1209                 "gpt-4", "gpt-3.5", "base-legacy", "legacy-retired"]1210    fams = defaultdict(list)1211    for r in canon:1212        fams[r["family"]].append(r)12131214    L = []1215    L.append("# OpenAI — Model catalogue (API Atlas)\n")1216    L.append("**Status:** DOCUMENTED + LIVE_VERIFIED (GET /v1/models listing on 2026-09-18, 136 ids; plus targeted GET /v1/models/{id} and minimal POST /v1/responses probes). Machine-readable twin: `generated/fragments/models/openai-models.json`.\n")1217    L.append("**Sources:** https://developers.openai.com/api/docs/models · https://developers.openai.com/api/docs/models/<id> (101 pages) · https://developers.openai.com/api/docs/pricing · https://developers.openai.com/api/docs/deprecations · https://developers.openai.com/api/docs/changelog · https://api.openai.com/v1/models · OpenAPI spec (openapi-master.yaml).\n")1218    L.append("**Last verified:** 2026-09-18\n")1219    live_n = sum(1 for r in records if "LIVE_VERIFIED" in r["status"])1220    L.append(f"Records: **{len(records)}** ids ({len(canon)} canonical models/ids, {sum(1 for r in records if r['record_kind']=='snapshot')} dated snapshots, "1221             f"{sum(1 for r in records if r['record_kind']=='alias')} aliases). LIVE_VERIFIED: {live_n}. Status vocabulary per CLAUDE.md; extra markers live in `flags[]` "1222             "(`DOCUMENTATION_INCOMPLETE`, `STILL_LISTED_AFTER_DOCUMENTED_SHUTDOWN`, `SHUTDOWN_DATE_ONLY_IN_LIVE_API`, `OPENAPI_ENUM_ONLY`, `NO_MODEL_PAGE`).\n")1223    L.append("## How to read\n\n- `LIVE_VERIFIED` = id returned by `GET /v1/models` with our key on 2026-09-18 (or a probe succeeded). `DOCUMENTED` = has an official model page (or pricing/deprecation entry). `DEPRECATED` = shutdown date announced and in the future; `RETIRED` = shutdown date passed (docs) — several such ids are **still listed live** (flag `STILL_LISTED_AFTER_DOCUMENTED_SHUTDOWN`).\n- Capability cells: ✅ documented supported · — documented absent (page has a features/tools section but does not list it, or endpoint table says Not supported) · `?` unknown (no page or no section).\n- Prices are USD per 1M tokens (standard tier) unless noted; see `docs/openai/pricing.md`.\n")12241225    # live-only / docs-only1226    L.append("## Discrepancies: live-only ids (in GET /v1/models, no model page)\n")1227    L.append("| id | family | status | created (live) | shutdown_date (live) | note |\n|---|---|---|---|---|---|")1228    for mid in live_only:1229        r = by_id[mid]1230        L.append(f"| `{mid}` | {r['family']} | {', '.join(r['status'])} | {r['release_date']} | {r['shutdown_date'] or ''} | {'; '.join(r['flags'])} |")1231    L.append("\n## Discrepancies: docs-only ids (model page exists, NOT in GET /v1/models)\n")1232    L.append("| id | family | status | GET /v1/models/{id} probe | note |\n|---|---|---|---|---|")1233    for mid in docs_only:1234        r = by_id[mid]1235        gp = probes.get("get_models", {}).get(mid)1236        probe = f"HTTP {gp['status']} {gp.get('error_code') or ''}".strip() if gp else "not probed"1237        note = r["restrictions"] or ("; ".join(d["section"] for d in r["deprecation"][:1]) if r["deprecation"] else "")1238        L.append(f"| `{mid}` | {r['family']} | {', '.join(r['status'])} | {probe} | {note} |")12391240    # live probe summary1241    L.append("\n## Live probes (2026-09-18)\n")1242    L.append("| id | GET /v1/models/{id} | POST /v1/responses | model echoed | error |\n|---|---|---|---|---|")1243    probe_ids = sorted(set(probes.get("get_models", {})) | set(probes.get("responses", {})))1244    for mid in probe_ids:1245        gp = probes.get("get_models", {}).get(mid)1246        rp = probes.get("responses", {}).get(mid)1247        L.append(f"| `{mid}` | {('HTTP ' + str(gp['status'])) if gp else ''} | {('HTTP ' + str(rp['status'])) if rp else ''} | {rp.get('model_echo','') if rp else ''} | "1248                 f"{(gp.get('error_code') or '') if gp else ''} {(rp.get('error_message') or '')[:120] if rp else ''} |")12491250    # per family sections1251    L.append("\n## Catalogue by family\n")1252    for fam in fam_order + sorted(set(fams) - set(fam_order)):1253        rs = fams.get(fam)1254        if not rs:1255            continue1256        L.append(f"\n### {fam}\n")1257        L.append("| id | display name | status | release | cutoff | context | max out | in → out | std price in/cached/out ($/1M) | snapshots | shutdown |\n|---|---|---|---|---|---|---|---|---|---|---|")1258        for r in sorted(rs, key=lambda x: (x["release_date"] or "0000"), reverse=True):1259            std = r["pricing"].get("standard", {})1260            pin = std.get("input"); pc = std.get("cached_input"); po = std.get("output")1261            if pin is None and "model_page:Text tokens" in r["pricing"]:1262                t = r["pricing"]["model_page:Text tokens"]1263                pin = t.get("Input", {}).get("price"); pc = t.get("Cached input", {}).get("price"); po = t.get("Output", {}).get("price")1264            price = f"{money(pin)} / {money(pc)} / {money(po)}" if (pin is not None or po is not None) else ""1265            mods = f"{'+'.join(r['modalities']['input'])} → {'+'.join(r['modalities']['output'])}" if r["modalities"]["input"] else ""1266            L.append(f"| `{r['id']}` | {r['display_name']} | {', '.join(r['status'])} | {r['release_date'] or ''} | {r['knowledge_cutoff'] or ''} | "1267                     f"{r['context_window'] or ''} | {r['max_output'] or ''} | {mods} | {price} | {', '.join('`'+s+'`' for s in r['snapshots'])} | {r['shutdown_date'] or ''} |")12681269    # capability matrix1270    cap_cols = ["reasoning", "reasoning_effort_values", "streaming", "structured_outputs", "function_calling", "prompt_caching",1271                "extended_prompt_cache_retention_24h", "predicted_outputs", "fine_tuning", "batch", "image_in", "audio_in", "audio_out",1272                "image_out", "video_out", "embeddings", "realtime", "compaction", "long_context_tier_272k",1273                "tool_web_search", "tool_file_search", "tool_code_interpreter", "tool_image_generation", "tool_mcp",1274                "tool_computer_use", "tool_hosted_shell", "tool_apply_patch", "tool_skills", "tool_tool_search"]1275    L.append("\n## Model × Capability (documented models with a page)\n")1276    L.append("| id | " + " | ".join(c.replace("tool_", "🛠") for c in cap_cols) + " |")1277    L.append("|---|" + "---|" * len(cap_cols))1278    for r in sorted(canon, key=lambda x: (fam_order.index(x["family"]) if x["family"] in fam_order else 99, x["id"])):1279        if r["record_kind"] != "model":1280            continue1281        c = r["capabilities"]1282        L.append(f"| `{r['id']}` | " + " | ".join(("?" if c.get(k) == "unknown" else fmt(c.get(k))) for k in cap_cols) + " |")12831284    # endpoint matrix1285    ep_names = ["Responses", "Chat Completions", "Batch", "Realtime", "Realtime transcription", "Realtime translation", "Live", "Assistants",1286                "Fine-tuning", "Embeddings", "Image generation", "Image edit", "Videos", "Speech generation", "Transcription", "Translation",1287                "Moderation", "Completions (legacy)"]1288    L.append("\n## Model × Endpoint (documented models with a page)\n")1289    L.append("| id | " + " | ".join(ep_names) + " |")1290    L.append("|---|" + "---|" * len(ep_names))1291    for r in sorted(canon, key=lambda x: (fam_order.index(x["family"]) if x["family"] in fam_order else 99, x["id"])):1292        if r["record_kind"] != "model":1293            continue1294        names = {e["name"] for e in r["endpoints"]}1295        L.append(f"| `{r['id']}` | " + " | ".join("✅" if n in names else "—" for n in ep_names) + " |")12961297    L.append("\n## Snapshot / alias index\n")1298    L.append("| id | kind | canonical model | status | shutdown |\n|---|---|---|---|---|")1299    for r in records:1300        if r["record_kind"] in ("snapshot", "alias"):1301            L.append(f"| `{r['id']}` | {r['record_kind']} | `{r['canonical_model']}` | {', '.join(r['status'])} | {r['shutdown_date'] or ''} |")13021303    L.append("\n## Decisions & caveats\n")1304    L.append("- `DOCUMENTATION_INCOMPLETE` is not part of the CLAUDE.md status vocabulary, so it is recorded in `flags[]` (status keeps `LIVE_VERIFIED`+`LIVE_DISCOVERED`).\n"1305             "- `LEGACY` is assigned editorially (models the docs call *older*, *legacy* or *previous generation*, plus their snapshots); OpenAI's own definition is \"no longer receives updates\".\n"1306             "- Release dates: changelog when available, else the `created` timestamp of GET /v1/models (labelled `release_date_source`). gpt-oss dates come from the public announcement (not in the downloaded docs).\n"1307             "- The live `GET /v1/models` payload exposes a `shutdown_date` field per model (not documented on the reference page). It is recorded in `deprecation[]` with `phase: live` and cross-checked with the deprecations page.\n"1308             "- Capability `false` means *the page has a features/tools section and does not list it*; it is not an experimental negative.\n"1309             "- Rate-limit tables are copied verbatim from the model pages (documented tiers). Our observed headers are account-specific and kept separate.\n")1310    (ROOT / "docs/models").mkdir(parents=True, exist_ok=True)1311    (ROOT / "docs/models/openai-models.md").write_text("\n".join(L) + "\n")13121313    # ---------------- pricing doc1314    P = []1315    P.append("# OpenAI — Pricing (API Atlas)\n")1316    P.append("**Status:** DOCUMENTED (prices copied from the official pricing page and model pages; not billed-verified beyond the minimal probes). Machine-readable twin: `generated/fragments/pricing/openai-pricing.json`.\n")1317    P.append("**Sources:** https://developers.openai.com/api/docs/pricing · https://developers.openai.com/api/docs/models/<id> · https://developers.openai.com/api/docs/guides/batch · https://developers.openai.com/api/docs/guides/flex-processing · https://developers.openai.com/api/docs/guides/fast-mode · https://developers.openai.com/api/docs/guides/prompt-caching · https://developers.openai.com/api/docs/guides/your-data\n")1318    P.append("**Last verified:** 2026-09-18\n")1319    P.append("All prices USD. Token prices are per 1M tokens. `cached_input` = cache read; `cache_write` (GPT-5.6+/GPT-6 only) = 1.25× uncached input. Long context = prompts >272K input tokens (2× input/cache, 1.5× output for the whole request).\n")1320    P.append("## Service tiers\n\n| Tier | `service_tier` | Price rule | Notes |\n|---|---|---|---|\n"1321             "| Standard | `default` / omitted | list price | |\n"1322             "| Batch | Batch API (`/v1/batches`, `completion_window: 24h`) | 50% of standard | separate, much higher queue limits; 24h turnaround |\n"1323             "| Flex | `flex` | Batch rates (50%) | beta, limited models; slower; may return 429 `resource_unavailable`; raise client timeout (15 min recommended) |\n"1324             "| Fast (ex-Priority) | `fast` or `priority` | 2× standard for GPT-5.6 Sol / GPT-6 Astra (per-model table) | renamed 2026-07-30; up to 2.5× faster; downgraded requests return `service_tier: default` and standard rates; no fine-tuned models/embeddings; unavailable for GPT-6 Astra with EU data residency |\n"1325             "| Ultrafast | — | — | limited preview for GPT-5.6 Sol (announced 2026-08-13), up to 14× faster |\n"1326             "| Regional processing | `us.`/`eu.`/`ae.` … prefixed domains | +10% uplift | models released on/after 2026-03-05 that are data-residency eligible |\n")13271328    def table(group_filter, cols, title, tier=None, ctx=None):1329        rows = defaultdict(dict)1330        for p in prices:1331            if p.get("group") != group_filter:1332                continue1333            if tier and p["tier"] != tier:1334                continue1335            if ctx and p.get("context") != ctx:1336                continue1337            if not ctx and p.get("context") == "long":1338                continue1339            key = p["model_or_service"] + (f" [{p['modality']}]" if p.get("modality") else "") + (f" [{p['size']}]" if p.get("size") else "")1340            rows[key][p["dimension"]] = p["price"]1341            rows[key]["_unit"] = p["unit"]1342            rows[key]["_note"] = p.get("effective_notes")1343        if not rows:1344            return1345        P.append(f"\n### {title}\n")1346        P.append("| model | " + " | ".join(cols) + " | unit | notes |")1347        P.append("|---|" + "---|" * (len(cols) + 2))1348        for k, d in rows.items():1349            P.append(f"| `{k}` | " + " | ".join(money(d.get(c)) for c in cols) + f" | {d.get('_unit','')} | {d.get('_note') or ''} |")13501351    P.append("## Text models (Flagship + legacy text)\n")1352    for tier in ("standard", "batch", "flex", "fast"):1353        table("Flagship models", ["input", "cached_input", "cache_write", "output"], f"{tier.capitalize()} — short context (≤272K)", tier=tier, ctx="short")1354        table("Flagship models", ["input", "cached_input", "cache_write", "output"], f"{tier.capitalize()} — long context (>272K)", tier=tier, ctx="long")1355    table("Cyber models", ["input", "cached_input", "cache_write", "output"], "Cyber / Daybreak models (standard)", ctx="short")1356    P.append("\n`gpt-daybreak-blue-latest` → `gpt-5.6-sol`, `gpt-daybreak-red-latest` → `gpt-5.6-cyber` (aliases, priced as the underlying model).\n")1357    table("GPT-Live sessions", ["session_duration"], "GPT-Live sessions (per minute, billed per second)")1358    table("Realtime and audio generation models", ["audio_input", "audio_cached_input", "audio_output", "text_input", "text_cached_input", "text_output", "image_input", "image_cached_input"], "Realtime & audio models")1359    table("Image generation models", ["image_input", "image_cached_input", "image_output", "text_input", "text_cached_input", "text_output"], "Image models — standard", tier="standard")1360    table("Image generation models", ["image_input", "image_cached_input", "image_output", "text_input", "text_cached_input", "text_output"], "Image models — batch", tier="batch")1361    table("Video generation models", ["video_output"], "Video (Sora 2) — standard, per second", tier="standard")1362    table("Video generation models", ["video_output"], "Video (Sora 2) — batch, per second", tier="batch")1363    table("Transcription models", ["audio_input", "text_output", "audio_duration"], "Transcription / translation models")1364    table("Specialized models", ["input", "cached_input", "output"], "Specialized models — standard", tier="standard")1365    table("Specialized models", ["input", "cached_input", "output"], "Specialized models — fast", tier="fast")1366    table("Finetuning", ["fine_tuning_training", "fine_tuned_input", "fine_tuned_cached_input", "fine_tuned_output"], "Fine-tuning — standard (platform winding down; see deprecations)", tier="standard")1367    table("Finetuning", ["fine_tuning_training", "fine_tuned_input", "fine_tuned_cached_input", "fine_tuned_output"], "Fine-tuning — batch inference", tier="batch")1368    P.append("\n### Built-in tools\n")1369    P.append("| tool | details | price | unit | full text |\n|---|---|---|---|---|")1370    for p in prices:1371        if p["model_or_service"].startswith("tool:"):1372            P.append(f"| {p['model_or_service'][5:]} | {p['dimension']} | {money(p['price'])} | {p['unit']} | {p['effective_notes']} |")1373    P.append("\n### Prices only on model pages (not in pricing.md)\n")1374    P.append("| model | dimension | price | unit | notes |\n|---|---|---|---|---|")1375    for p in prices:1376        if p.get("group") == "model page":1377            P.append(f"| `{p['model_or_service']}` | {p['dimension']} | {money(p['price'])} | {p['unit']} | {(p['effective_notes'] or '')[:160]} |")1378    P.append("\n### Pricing rules (derived records `rule:*`)\n")1379    P.append("| rule | dimension | value | unit | notes |\n|---|---|---|---|---|")1380    for p in prices:1381        if p["model_or_service"].startswith("rule:") or p["dimension"] in ("promotion", "billing_start"):1382            P.append(f"| {p['model_or_service']} | {p['dimension']} | {p['price'] if p['price'] is not None else ''} | {p['unit']} | {p['effective_notes']} |")1383    P.append("\n## Footnotes copied from the pricing page\n")1384    for f in footnotes:1385        P.append(f"- {f}")1386    P.append("\n## Caveats\n- Prices are documentation values as of 2026-09-18; the pricing page states promotional pricing for GPT-5.6 Sol through at least 2026-11-21.\n"1387             "- Fine-tuning: platform is winding down (no new orgs since 2026-05-07; job creation ends 2027-01-06); inference on fine-tuned models continues until the base model is deprecated.\n"1388             "- Realtime/Live sessions, image tokens and video seconds are billed on different units — check `unit` in every record.\n")1389    (ROOT / "docs/openai").mkdir(parents=True, exist_ok=True)1390    (ROOT / "docs/openai/pricing.md").write_text("\n".join(P) + "\n")13911392    # ---------------- deprecations doc1393    D = []1394    D.append("# OpenAI — Deprecations & retirements (API Atlas)\n")1395    D.append("**Status:** DOCUMENTED (deprecations page) + LIVE_VERIFIED cross-check (`shutdown_date` field of GET /v1/models, 2026-09-18). Machine-readable twin: `generated/fragments/deprecations/openai-deprecations.json`.\n")1396    D.append("**Sources:** https://developers.openai.com/api/docs/deprecations · https://developers.openai.com/api/docs/changelog · https://api.openai.com/v1/models\n")1397    D.append("**Last verified:** 2026-09-18\n")1398    D.append("## Policy (notice periods)\n\n| Category | Minimum notice | Examples |\n|---|---|---|\n| Generally available models | ≥ 6 months | gpt-5, o3 |\n| Specialized variants | ≥ 3 months | `gpt-5.1-chat-latest`, `gpt-5.3-codex`, `o3-deep-research` |\n| Preview models | may be ~2 weeks | `computer-use-preview`, `gpt-4o-audio-preview` |\n\n"1399             "*Deprecated* = retirement announced (immediately deprecated, always with a shutdown date). *Legacy* = no longer updated, will be deprecated later. *Sunset/shut down* = no longer accessible. Dedicated capacity may be negotiable after shutdown (sales).\n")1400    D.append("## Upcoming (shutdown after 2026-09-18)\n")1401    D.append("| shutdown | subject | kind | replacement | announced | section |\n|---|---|---|---|---|---|")1402    for e in sorted([e for e in deps if e.get("shutdown_date") and e["shutdown_date"] > TODAY and e["phase"] != "live_api_field"], key=lambda e: e["shutdown_date"]):1403        D.append(f"| {e['shutdown_date']} | `{e['subject']}` | {e['kind']} | {e['replacement'] or ''} | {e['announced'] or ''} | {e['section']} |")1404    D.append("\n## Feature / platform milestones\n")1405    D.append("| date | subject | update |\n|---|---|---|")1406    for e in [e for e in deps if e["kind"] == "feature"]:1407        D.append(f"| {e['milestone_date']} | {e['subject']} | {e['update']} |")1408    D.append("\n## Past (shutdown on or before 2026-09-18)\n")1409    D.append("| shutdown | subject | kind | replacement | announced | section |\n|---|---|---|---|---|---|")1410    for e in sorted([e for e in deps if e.get("shutdown_date") and e["shutdown_date"] <= TODAY and e["phase"] != "live_api_field"], key=lambda e: e["shutdown_date"], reverse=True):1411        D.append(f"| {e['shutdown_date']} | `{e['subject']}` | {e['kind']} | {e['replacement'] or ''} | {e['announced'] or ''} | {e['section']} |")1412    D.append("\n## Live cross-check: `shutdown_date` in GET /v1/models (2026-09-18)\n")1413    D.append("Every listed model carries a `shutdown_date` (null or ISO date). Rows below compare it with the deprecations page.\n")1414    D.append("| id | live shutdown_date | consistency with docs |\n|---|---|---|")1415    for e in [e for e in deps if e["phase"] == "live_api_field"]:1416        D.append(f"| `{e['primary_id']}` | {e['shutdown_date']} | {e['consistency']} |")1417    still = [r for r in records if "STILL_LISTED_AFTER_DOCUMENTED_SHUTDOWN" in r["flags"]]1418    D.append("\n## Notable findings\n")1419    D.append(f"- **{len(still)} ids are documented as shut down but were still returned by GET /v1/models on 2026-09-18**: " + ", ".join(f"`{r['id']}`" for r in still) + ". Listing ≠ usable: treat as RETIRED unless a call succeeds.")1420    only_live = [e for e in deps if e["phase"] == "live_api_field" and e["consistency"] == "NOT on deprecations page"]1421    D.append(f"- **{len(only_live)} live shutdown dates are not on the deprecations page**: " + ", ".join(f"`{e['primary_id']}` ({e['shutdown_date']})" for e in only_live) + ".")1422    diff = [e for e in deps if e["phase"] == "live_api_field" and e["consistency"].startswith("differs")]1423    if diff:1424        D.append("- Dates that differ between live field and docs: " + ", ".join(f"`{e['primary_id']}` live {e['shutdown_date']} vs {e['consistency']}" for e in diff) + ".")1425    D.append("- Assistants API shut down 2026-08-26 (Responses + Conversations replace it). Videos API and Sora 2 shut down 2026-09-24 with no replacement. Realtime beta and DALL·E shut down 2026-05-12. `v1/prompts`, Evals API and Agent Builder shut down 2026-11-30.\n"1426             "- Self-serve fine-tuning: closed to new orgs since 2026-05-07; last job creation 2027-01-06; inference persists until the base model is deprecated.\n")1427    (ROOT / "docs/openai/deprecations.md").write_text("\n".join(D) + "\n")142814291430if __name__ == "__main__":1431    main()1432