#!/usr/bin/env python3 """Re-runnable OpenAI model discovery: GET /v1/models -> sanitized snapshot + diff against the previous run. - Calls GET /v1/models (free) via scripts/live.py (auth injected, call logged to reports/live-requests.jsonl). - Diffs the returned ids against sources/openai/models-api-raw.json (previous run): prints NEW / REMOVED ids and any change of `shutdown_date`, and writes sources/openai/models-diff.json. - Saves the sanitized listing to sources/openai/models-api-raw.json and the id list to sources/openai/model-ids.txt (unless --dry-run). The previous listing is kept as sources/openai/models-api-raw.prev.json. Usage: python3 scripts/discover_openai_models.py [--dry-run] Exit code 0 always (a diff is information, not an error); prints a one-line summary. """ from __future__ import annotations import json import sys from datetime import datetime, timezone from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from scripts.live import openai_request, save_sanitized # noqa: E402 ROOT = Path(__file__).resolve().parent.parent RAW = ROOT / "sources/openai/models-api-raw.json" PREV = ROOT / "sources/openai/models-api-raw.prev.json" IDS = ROOT / "sources/openai/model-ids.txt" DIFF = ROOT / "sources/openai/models-diff.json" def main() -> int: dry = "--dry-run" in sys.argv previous = json.loads(RAW.read_text()) if RAW.exists() else {"data": []} prev_models = {m["id"]: m for m in previous.get("data", [])} st, body, hdrs = openai_request("GET", "/v1/models", note="discover_openai_models") if st != 200 or not isinstance(body, dict): print(f"GET /v1/models -> HTTP {st}; aborting without writing", file=sys.stderr) return 0 models = sorted(body.get("data", []), key=lambda m: (m.get("created", 0), m["id"])) cur_models = {m["id"]: m for m in models} new_ids = sorted(set(cur_models) - set(prev_models)) removed_ids = sorted(set(prev_models) - set(cur_models)) changed = [] for mid in sorted(set(cur_models) & set(prev_models)): a, b = prev_models[mid], cur_models[mid] for k in ("shutdown_date", "owned_by", "created"): if a.get(k) != b.get(k): changed.append({"id": mid, "field": k, "before": a.get(k), "after": b.get(k)}) now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") diff = { "checked_at": now, "previous_count": len(prev_models), "current_count": len(cur_models), "new": [cur_models[i] for i in new_ids], "removed": [prev_models[i] for i in removed_ids], "changed": changed, "dry_run": dry, } for m in diff["new"]: print(f"NEW {m['id']} (owned_by={m.get('owned_by')}, shutdown_date={m.get('shutdown_date')})") for m in diff["removed"]: print(f"REMOVED {m['id']} (was shutdown_date={m.get('shutdown_date')})") for c in changed: print(f"CHANGED {c['id']}.{c['field']}: {c['before']} -> {c['after']}") print(f"summary: {len(cur_models)} ids live, {len(new_ids)} new, {len(removed_ids)} removed, {len(changed)} changed") save_sanitized(diff, DIFF) if not dry: if RAW.exists(): PREV.write_text(RAW.read_text()) save_sanitized({"object": body.get("object", "list"), "retrieved_at": now, "data": models}, RAW) IDS.write_text("\n".join(sorted(cur_models)) + "\n") print(f"wrote {RAW.relative_to(ROOT)}, {IDS.relative_to(ROOT)}, {DIFF.relative_to(ROOT)}") else: print(f"dry run: wrote only {DIFF.relative_to(ROOT)}") return 0 if __name__ == "__main__": sys.exit(main())