Python 88.3%
TypeScript 7.6%
Shell 4.1%
1#!/usr/bin/env python32"""Eval with three graders (string_check, text_similarity, python) on a custom dataset, run twice: inline file_content3and via an uploaded purpose=evals JSONL file (file_id source). No model sampling => $0.4STATUS: LIVE_VERIFIED 2026-09-18. Evals API is DEPRECATED (shutdown 2026-11-30).5Run: python3 examples/openai/evals/eval-python-grader.py6"""7from __future__ import annotations8import json, sys, time, uuid9from pathlib import Path10sys.path.insert(0, str(Path(__file__).resolve().parents[3]))11from scripts.live import openai_request # noqa: E4021213NOTE = "example eval-python-grader"14PY = "from rapidfuzz import fuzz\ndef grade(sample, item):\n return fuzz.WRatio(sample['output_text'], item['expected']) / 100.0\n"151617def sample(text):18 return {"model": "my-app-v1", "output_text": text, "choices": [{"index": 0, "message": {"role": "assistant", "content": text}, "finish_reason": "stop"}]}192021rows = [{"item": {"q": "Capital of France?", "expected": "Paris"}, "sample": sample("Paris")},22 {"item": {"q": "Capital of Italy?", "expected": "Rome"}, "sample": sample("rome")},23 {"item": {"q": "Capital of Spain?", "expected": "Madrid"}, "sample": sample("Barcelona")}]24st, ev, _ = openai_request("POST", "/v1/evals", {25 "name": "atlas-platform-agent-py", "metadata": {"owner": "atlas-platform-agent"},26 "data_source_config": {"type": "custom", "include_sample_schema": True,27 "item_schema": {"type": "object", "properties": {"q": {"type": "string"}, "expected": {"type": "string"}}, "required": ["q", "expected"]}},28 "testing_criteria": [29 {"type": "string_check", "name": "exact", "input": "{{sample.output_text}}", "reference": "{{item.expected}}", "operation": "eq"},30 {"type": "text_similarity", "name": "fuzzy", "input": "{{sample.output_text}}", "reference": "{{item.expected}}", "evaluation_metric": "fuzzy_match", "pass_threshold": 0.8},31 {"type": "python", "name": "py", "source": PY, "pass_threshold": 0.8}]}, note=NOTE)32assert st == 201, ev33print("eval", ev["id"], "criteria", [c["id"] for c in ev["testing_criteria"]])34created_files = []353637def run_and_report(data_source, label):38 st, run, _ = openai_request("POST", f"/v1/evals/{ev['id']}/runs", {"name": label, "data_source": data_source}, note=NOTE + " " + label)39 assert st == 201, run40 t0 = time.time()41 while run["status"] not in ("completed", "failed", "canceled") and time.time() - t0 < 120:42 time.sleep(3)43 st, run, _ = openai_request("GET", f"/v1/evals/{ev['id']}/runs/{run['id']}", note=NOTE + " poll")44 print(f"{label}: {run['status']} in {time.time() - t0:.1f}s counts={run['result_counts']}")45 for c in run["per_testing_criteria_results"]:46 print(" ", c["testing_criteria"].split("-")[0], "passed", c["passed"], "failed", c["failed"])47 st, items, _ = openai_request("GET", f"/v1/evals/{ev['id']}/runs/{run['id']}/output_items?order=asc", note=NOTE)48 for it in items["data"]:49 print(" item", it["datasource_item_id"], it["status"], [(r["name"].split("-")[0], round(r["score"], 3), r["passed"]) for r in it["results"]])50 openai_request("DELETE", f"/v1/evals/{ev['id']}/runs/{run['id']}", note=NOTE + " cleanup")515253try:54 run_and_report({"type": "jsonl", "source": {"type": "file_content", "content": rows}}, "inline")55 b = f"----atlas{uuid.uuid4().hex}"56 body = (f"--{b}\r\nContent-Disposition: form-data; name=\"purpose\"\r\n\r\nevals\r\n--{b}\r\n"57 f"Content-Disposition: form-data; name=\"file\"; filename=\"rows.jsonl\"\r\nContent-Type: application/jsonl\r\n\r\n").encode() \58 + "".join(json.dumps(r) + "\n" for r in rows).encode() + f"\r\n--{b}--\r\n".encode()59 st, f, _ = openai_request("POST", "/v1/files", data=body, content_type=f"multipart/form-data; boundary={b}", note=NOTE + " upload evals jsonl")60 assert st == 200, f61 created_files.append(f["id"])62 run_and_report({"type": "jsonl", "source": {"type": "file_id", "id": f["id"]}}, "file_id")63finally:64 for fid in created_files:65 openai_request("DELETE", f"/v1/files/{fid}", note=NOTE + " cleanup")66 st, d, _ = openai_request("DELETE", f"/v1/evals/{ev['id']}", note=NOTE + " cleanup")67 print("eval deleted:", d.get("deleted"))68