#!/usr/bin/env python3 """Follow-up live probe: eval runs with a schema-compliant `sample` (no model sampling => free), file_id source, logs data_source_config, and a validation-only POST /v1/fine_tuning/jobs (bogus file id: nothing is created).""" from __future__ import annotations import json, sys, time, uuid from pathlib import Path ROOT = Path(__file__).resolve().parent.parent sys.path.insert(0, str(ROOT)) from scripts.live import openai_request, save_sanitized # noqa: E402 OUT = ROOT / "tmp-live" / "platform2" OUT.mkdir(parents=True, exist_ok=True) TAG = "atlas-platform-agent" RUN = uuid.uuid4().hex[:6] SUMMARY = [] CREATED = {"files": [], "evals": []} def rec(step, method, path, st, body, **extra): save_sanitized({"status": st, "body": body if not isinstance(body, bytes) else body.decode("utf-8", "replace")}, OUT / f"{len(SUMMARY):02d}-{step}.json") row = {"step": step, "method": method, "path": path, "status": st, **extra} SUMMARY.append(row); print(json.dumps(row, default=str), flush=True); return body def multipart(fields, files): b = f"----atlas{uuid.uuid4().hex}"; out = bytearray() for k, v in fields.items(): out += f"--{b}\r\nContent-Disposition: form-data; name=\"{k}\"\r\n\r\n{v}\r\n".encode() for k, (fn, data, ct) in files.items(): out += f"--{b}\r\nContent-Disposition: form-data; name=\"{k}\"; filename=\"{fn}\"\r\nContent-Type: {ct}\r\n\r\n".encode() + data + b"\r\n" out += f"--{b}--\r\n".encode(); return bytes(out), f"multipart/form-data; boundary={b}" def oai(method, path, json_body=None, *, data=None, content_type="application/json", note=""): st, body, _ = openai_request(method, path, json_body, data=data, content_type=content_type, note=f"{TAG} {note}".strip()) return st, body def sample(text, model="gpt-5.4-nano"): return {"model": model, "output_text": text, "choices": [{"index": 0, "message": {"role": "assistant", "content": text}, "finish_reason": "stop"}]} # ---- eval with custom config crit = [{"type": "string_check", "name": "exact", "input": "{{sample.output_text}}", "reference": "{{item.expected}}", "operation": "eq"}, {"type": "text_similarity", "name": "fuzzy", "input": "{{sample.output_text}}", "reference": "{{item.expected}}", "evaluation_metric": "fuzzy_match", "pass_threshold": 0.8}, {"type": "python", "name": "py", "source": "def grade(sample, item):\n return 1.0 if sample['output_text'].strip().upper() == item['expected'].upper() else 0.0\n", "pass_threshold": 0.5}] st, ev = oai("POST", "/v1/evals", {"name": f"{TAG}-{RUN}", "metadata": {"owner": TAG}, "data_source_config": {"type": "custom", "include_sample_schema": True, "item_schema": {"type": "object", "properties": {"q": {"type": "string"}, "expected": {"type": "string"}}, "required": ["q", "expected"]}}, "testing_criteria": crit}) rec("evals.create(3 criteria)", "POST", "/v1/evals", st, ev, id=ev.get("id"), error=ev.get("error")) EVAL_ID = ev.get("id") if EVAL_ID: CREATED["evals"].append(EVAL_ID) time.sleep(3) st, b = oai("GET", "/v1/evals?limit=5&order=desc", note="list 3s after create") rec("evals.list(3s later)", "GET", "/v1/evals", st, b, n=len(b.get("data", [])), ids=[e["id"] for e in b.get("data", [])]) content = [{"item": {"q": "ping", "expected": "OK"}, "sample": sample("OK")}, {"item": {"q": "ping2", "expected": "OK"}, "sample": sample("ok")}, {"item": {"q": "ping3", "expected": "OK"}, "sample": sample("Nope")}] st, run = oai("POST", f"/v1/evals/{EVAL_ID}/runs", {"name": f"{TAG}-{RUN}-run", "metadata": {"owner": TAG}, "data_source": {"type": "jsonl", "source": {"type": "file_content", "content": content}}}, note="jsonl file_content with sample, no sampling") rec("evals.runs.create(jsonl file_content)", "POST", f"/v1/evals/{EVAL_ID}/runs", st, run, id=run.get("id"), status_field=run.get("status"), model=run.get("model"), report_url=(run.get("report_url") or "")[:60], data_source_type=(run.get("data_source") or {}).get("type"), error=run.get("error")) RUN_ID = run.get("id") if RUN_ID: tl = []; t0 = time.time() for i in range(60): st, run = oai("GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", note="poll") tl.append({"t": round(time.time() - t0, 1), "status": run.get("status")}) if run.get("status") in ("completed", "failed", "canceled", "cancelled"): break time.sleep(3) rec("evals.runs.retrieve(poll)", "GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", st, run, timeline=tl, final_status=run.get("status"), result_counts=run.get("result_counts"), per_testing_criteria_results=run.get("per_testing_criteria_results"), per_model_usage=run.get("per_model_usage"), error=run.get("error")) st, b = oai("GET", f"/v1/evals/{EVAL_ID}/runs?limit=5&status=completed") rec("evals.runs.list(status=completed)", "GET", f"/v1/evals/{EVAL_ID}/runs", st, b, n=len(b.get("data", []))) st, b = oai("GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items?limit=10&order=asc") items = b.get("data", []) rec("evals.runs.output_items.list", "GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items", st, b, n=len(items), statuses=[x.get("status") for x in items], item_keys=sorted(items[0].keys()) if items else None, results=[[(r.get("name"), r.get("type"), r.get("score"), r.get("passed")) for r in x.get("results", [])] for x in items], sample_keys=sorted((items[0].get("sample") or {}).keys()) if items else None, sample=items[0].get("sample") if items else None) st, b = oai("GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items?status=fail") rec("evals.runs.output_items.list(status=fail)", "GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items", st, b, n=len(b.get("data", [])), ids=[x["id"] for x in b.get("data", [])]) if items: st, b = oai("GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items/{items[0]['id']}") rec("evals.runs.output_items.retrieve", "GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items/{{id}}", st, b, status_field=b.get("status"), datasource_item_id=b.get("datasource_item_id")) st, b = oai("POST", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", note="cancel completed run") rec("evals.runs.cancel(after completion)", "POST", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", st, b, status_field=b.get("status"), error=b.get("error")) st, b = oai("DELETE", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", note="cleanup") rec("evals.runs.delete", "DELETE", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", st, b, resp=b) # file_id source (purpose=evals) jsonl = "".join(json.dumps(x) + "\n" for x in content[:2]).encode() data, ct = multipart({"purpose": "evals"}, {"file": (f"{TAG}-{RUN}.jsonl", jsonl, "application/jsonl")}) st, f = oai("POST", "/v1/files", data=data, content_type=ct, note="upload evals jsonl") rec("files.create(evals)", "POST", "/v1/files", st, f, id=f.get("id"), purpose=f.get("purpose"), expires_at=f.get("expires_at")) if f.get("id"): CREATED["files"].append(f["id"]) st, run2 = oai("POST", f"/v1/evals/{EVAL_ID}/runs", {"name": f"{TAG}-{RUN}-run-file", "data_source": {"type": "jsonl", "source": {"type": "file_id", "id": f["id"]}}}, note="jsonl file_id") rec("evals.runs.create(jsonl file_id)", "POST", f"/v1/evals/{EVAL_ID}/runs", st, run2, id=run2.get("id"), status_field=run2.get("status"), error=run2.get("error")) if run2.get("id"): for i in range(40): st, run2 = oai("GET", f"/v1/evals/{EVAL_ID}/runs/{run2['id']}", note="poll") if run2.get("status") in ("completed", "failed", "canceled", "cancelled"): break time.sleep(3) rec("evals.runs.retrieve(file_id run)", "GET", f"/v1/evals/{EVAL_ID}/runs/{{run}}", st, run2, final_status=run2.get("status"), result_counts=run2.get("result_counts"), error=run2.get("error")) st, b = oai("DELETE", f"/v1/evals/{EVAL_ID}/runs/{run2['id']}", note="cleanup") rec("evals.runs.delete(file run)", "DELETE", f"/v1/evals/{EVAL_ID}/runs/{{run}}", st, b, deleted=b.get("deleted")) st, b = oai("DELETE", f"/v1/files/{f['id']}", note="cleanup") rec("files.delete", "DELETE", f"/v1/files/{f['id']}", st, b, deleted=b.get("deleted")) if b.get("deleted"): CREATED["files"].remove(f["id"]) # run creation errors: responses data_source with no model st, b = oai("POST", f"/v1/evals/{EVAL_ID}/runs", {"name": "bad", "data_source": {"type": "responses", "source": {"type": "file_content", "content": [{"item": {"q": "x", "expected": "y"}}]}}}, note="responses source without model (expect 400)") rec("evals.runs.create(responses, no model)", "POST", f"/v1/evals/{EVAL_ID}/runs", st, b, error=b.get("error"), status_field=b.get("status"), id=b.get("id")) if b.get("id"): # should not happen; cancel+delete if it did oai("POST", f"/v1/evals/{EVAL_ID}/runs/{b['id']}", note="cancel unexpected run") oai("DELETE", f"/v1/evals/{EVAL_ID}/runs/{b['id']}", note="cleanup unexpected run") st, b = oai("DELETE", f"/v1/evals/{EVAL_ID}", note="cleanup") rec("evals.delete", "DELETE", f"/v1/evals/{EVAL_ID}", st, b, resp=b) if b.get("deleted"): CREATED["evals"].remove(EVAL_ID) # ---- logs data_source_config + label_model criterion (create/delete only, no run) st, ev2 = oai("POST", "/v1/evals", {"name": f"{TAG}-{RUN}-logs", "data_source_config": {"type": "logs", "metadata": {"usecase": "atlas-test"}}, "testing_criteria": [{"type": "label_model", "name": "lbl", "model": "gpt-4.1-nano", "labels": ["good", "bad"], "passing_labels": ["good"], "input": [{"role": "developer", "content": "Label the answer good or bad."}, {"role": "user", "content": "{{sample.output_text}}"}]}]}, note="logs config + label_model (no run)") rec("evals.create(logs+label_model)", "POST", "/v1/evals", st, ev2, id=ev2.get("id"), schema_keys=sorted(((ev2.get("data_source_config") or {}).get("schema") or {}).get("properties", {}).keys()) if isinstance(ev2, dict) else None, error=ev2.get("error")) if ev2.get("id"): st, b = oai("DELETE", f"/v1/evals/{ev2['id']}", note="cleanup") rec("evals.delete(logs eval)", "DELETE", f"/v1/evals/{ev2['id']}", st, b, deleted=b.get("deleted")) # ---- fine-tuning: validation-only create (bogus training_file => nothing created) st, b = oai("POST", "/v1/fine_tuning/jobs", {"model": "gpt-4.1-nano-2025-04-14", "training_file": "file-atlasbogus000", "suffix": TAG}, note="validation-only, bogus training_file (expect 4xx, no job)") rec("fine_tuning.jobs.create(bogus file)", "POST", "/v1/fine_tuning/jobs", st, b, error=b.get("error"), id=b.get("id"), status_field=b.get("status")) if b.get("id"): st, c = oai("POST", f"/v1/fine_tuning/jobs/{b['id']}/cancel", note="cancel unexpected job") rec("fine_tuning.jobs.cancel(unexpected)", "POST", f"/v1/fine_tuning/jobs/{b['id']}/cancel", st, c, status_field=c.get("status")) st, b = oai("GET", "/v1/fine_tuning/jobs/ftjob-atlasbogus000/events", note="bogus id") rec("fine_tuning.jobs.events(bogus)", "GET", "/v1/fine_tuning/jobs/{id}/events", st, b, error=b.get("error")) st, b = oai("GET", "/v1/fine_tuning/jobs/ftjob-atlasbogus000/checkpoints", note="bogus id") rec("fine_tuning.jobs.checkpoints(bogus)", "GET", "/v1/fine_tuning/jobs/{id}/checkpoints", st, b, error=b.get("error")) st, b = oai("POST", "/v1/fine_tuning/jobs/ftjob-atlasbogus000/pause", note="bogus id") rec("fine_tuning.jobs.pause(bogus)", "POST", "/v1/fine_tuning/jobs/{id}/pause", st, b, error=b.get("error")) st, b = oai("GET", "/v1/fine_tuning/jobs?limit=2&metadata[owner]=atlas", note="metadata filter") rec("fine_tuning.jobs.list(metadata filter)", "GET", "/v1/fine_tuning/jobs", st, b, n=len(b.get("data", [])) if isinstance(b, dict) else None, error=b.get("error")) # batch: bogus file id create => error shape (nothing created) st, b = oai("POST", "/v1/batches", {"input_file_id": "file-atlasbogus000", "endpoint": "/v1/responses", "completion_window": "24h"}, note="bogus input file (expect 4xx)") rec("batches.create(bogus file)", "POST", "/v1/batches", st, b, error=b.get("error"), id=b.get("id")) st, b = oai("GET", "/v1/batches/batch_atlasbogus000", note="bogus id") rec("batches.retrieve(bogus)", "GET", "/v1/batches/{id}", st, b, error=b.get("error")) st, b = oai("GET", "/v1/vector_stores/vs_atlasbogus000", note="bogus id") rec("vector_stores.retrieve(bogus)", "GET", "/v1/vector_stores/{id}", st, b, error=b.get("error")) st, b = oai("POST", "/v1/uploads", {"filename": "x.txt", "purpose": "user_data", "bytes": 10, "mime_type": "text/plain", "expires_after": {"anchor": "created_at", "seconds": 3600}}, note="upload with expires_after then cancel") rec("uploads.create(expires_after)", "POST", "/v1/uploads", st, b, id=b.get("id"), expires_at=b.get("expires_at"), created_at=b.get("created_at"), error=b.get("error")) if b.get("id"): st, c = oai("POST", f"/v1/uploads/{b['id']}/cancel", note="cleanup") rec("uploads.cancel", "POST", "/v1/uploads/{id}/cancel", st, c, status_field=c.get("status")) st, files = oai("GET", "/v1/files?limit=100", note="leftover check") st2, evals = oai("GET", "/v1/evals?limit=100", note="leftover check") st3, stores = oai("GET", "/v1/vector_stores?limit=100", note="leftover check") left = {"files": [f["id"] for f in files.get("data", [])], "evals": [e["id"] for e in evals.get("data", [])], "vector_stores": [v["id"] for v in stores.get("data", [])]} rec("leftover_check", "GET", "/v1/{files,evals,vector_stores}", 200, left, **left) save_sanitized(SUMMARY, OUT / "summary.json") print("DONE", json.dumps(left))