Python 88.3%
TypeScript 7.6%
Shell 4.1%
1#!/usr/bin/env python32"""Follow-up live probe: eval runs with a schema-compliant `sample` (no model sampling => free), file_id source,3logs data_source_config, and a validation-only POST /v1/fine_tuning/jobs (bogus file id: nothing is created)."""4from __future__ import annotations5import json, sys, time, uuid6from pathlib import Path7ROOT = Path(__file__).resolve().parent.parent8sys.path.insert(0, str(ROOT))9from scripts.live import openai_request, save_sanitized # noqa: E4021011OUT = ROOT / "tmp-live" / "platform2"12OUT.mkdir(parents=True, exist_ok=True)13TAG = "atlas-platform-agent"14RUN = uuid.uuid4().hex[:6]15SUMMARY = []16CREATED = {"files": [], "evals": []}171819def rec(step, method, path, st, body, **extra):20 save_sanitized({"status": st, "body": body if not isinstance(body, bytes) else body.decode("utf-8", "replace")}, OUT / f"{len(SUMMARY):02d}-{step}.json")21 row = {"step": step, "method": method, "path": path, "status": st, **extra}22 SUMMARY.append(row); print(json.dumps(row, default=str), flush=True); return body232425def multipart(fields, files):26 b = f"----atlas{uuid.uuid4().hex}"; out = bytearray()27 for k, v in fields.items():28 out += f"--{b}\r\nContent-Disposition: form-data; name=\"{k}\"\r\n\r\n{v}\r\n".encode()29 for k, (fn, data, ct) in files.items():30 out += f"--{b}\r\nContent-Disposition: form-data; name=\"{k}\"; filename=\"{fn}\"\r\nContent-Type: {ct}\r\n\r\n".encode() + data + b"\r\n"31 out += f"--{b}--\r\n".encode(); return bytes(out), f"multipart/form-data; boundary={b}"323334def oai(method, path, json_body=None, *, data=None, content_type="application/json", note=""):35 st, body, _ = openai_request(method, path, json_body, data=data, content_type=content_type, note=f"{TAG} {note}".strip())36 return st, body373839def sample(text, model="gpt-5.4-nano"):40 return {"model": model, "output_text": text,41 "choices": [{"index": 0, "message": {"role": "assistant", "content": text}, "finish_reason": "stop"}]}424344# ---- eval with custom config45crit = [{"type": "string_check", "name": "exact", "input": "{{sample.output_text}}", "reference": "{{item.expected}}", "operation": "eq"},46 {"type": "text_similarity", "name": "fuzzy", "input": "{{sample.output_text}}", "reference": "{{item.expected}}", "evaluation_metric": "fuzzy_match", "pass_threshold": 0.8},47 {"type": "python", "name": "py", "source": "def grade(sample, item):\n return 1.0 if sample['output_text'].strip().upper() == item['expected'].upper() else 0.0\n", "pass_threshold": 0.5}]48st, ev = oai("POST", "/v1/evals", {"name": f"{TAG}-{RUN}", "metadata": {"owner": TAG},49 "data_source_config": {"type": "custom", "include_sample_schema": True,50 "item_schema": {"type": "object", "properties": {"q": {"type": "string"}, "expected": {"type": "string"}}, "required": ["q", "expected"]}},51 "testing_criteria": crit})52rec("evals.create(3 criteria)", "POST", "/v1/evals", st, ev, id=ev.get("id"), error=ev.get("error"))53EVAL_ID = ev.get("id")54if EVAL_ID:55 CREATED["evals"].append(EVAL_ID)56 time.sleep(3)57 st, b = oai("GET", "/v1/evals?limit=5&order=desc", note="list 3s after create")58 rec("evals.list(3s later)", "GET", "/v1/evals", st, b, n=len(b.get("data", [])), ids=[e["id"] for e in b.get("data", [])])59 content = [{"item": {"q": "ping", "expected": "OK"}, "sample": sample("OK")},60 {"item": {"q": "ping2", "expected": "OK"}, "sample": sample("ok")},61 {"item": {"q": "ping3", "expected": "OK"}, "sample": sample("Nope")}]62 st, run = oai("POST", f"/v1/evals/{EVAL_ID}/runs", {"name": f"{TAG}-{RUN}-run", "metadata": {"owner": TAG},63 "data_source": {"type": "jsonl", "source": {"type": "file_content", "content": content}}},64 note="jsonl file_content with sample, no sampling")65 rec("evals.runs.create(jsonl file_content)", "POST", f"/v1/evals/{EVAL_ID}/runs", st, run, id=run.get("id"), status_field=run.get("status"),66 model=run.get("model"), report_url=(run.get("report_url") or "")[:60], data_source_type=(run.get("data_source") or {}).get("type"), error=run.get("error"))67 RUN_ID = run.get("id")68 if RUN_ID:69 tl = []; t0 = time.time()70 for i in range(60):71 st, run = oai("GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", note="poll")72 tl.append({"t": round(time.time() - t0, 1), "status": run.get("status")})73 if run.get("status") in ("completed", "failed", "canceled", "cancelled"):74 break75 time.sleep(3)76 rec("evals.runs.retrieve(poll)", "GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", st, run, timeline=tl, final_status=run.get("status"),77 result_counts=run.get("result_counts"), per_testing_criteria_results=run.get("per_testing_criteria_results"), per_model_usage=run.get("per_model_usage"), error=run.get("error"))78 st, b = oai("GET", f"/v1/evals/{EVAL_ID}/runs?limit=5&status=completed")79 rec("evals.runs.list(status=completed)", "GET", f"/v1/evals/{EVAL_ID}/runs", st, b, n=len(b.get("data", [])))80 st, b = oai("GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items?limit=10&order=asc")81 items = b.get("data", [])82 rec("evals.runs.output_items.list", "GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items", st, b, n=len(items),83 statuses=[x.get("status") for x in items], item_keys=sorted(items[0].keys()) if items else None,84 results=[[(r.get("name"), r.get("type"), r.get("score"), r.get("passed")) for r in x.get("results", [])] for x in items],85 sample_keys=sorted((items[0].get("sample") or {}).keys()) if items else None, sample=items[0].get("sample") if items else None)86 st, b = oai("GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items?status=fail")87 rec("evals.runs.output_items.list(status=fail)", "GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items", st, b, n=len(b.get("data", [])), ids=[x["id"] for x in b.get("data", [])])88 if items:89 st, b = oai("GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items/{items[0]['id']}")90 rec("evals.runs.output_items.retrieve", "GET", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}/output_items/{{id}}", st, b, status_field=b.get("status"), datasource_item_id=b.get("datasource_item_id"))91 st, b = oai("POST", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", note="cancel completed run")92 rec("evals.runs.cancel(after completion)", "POST", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", st, b, status_field=b.get("status"), error=b.get("error"))93 st, b = oai("DELETE", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", note="cleanup")94 rec("evals.runs.delete", "DELETE", f"/v1/evals/{EVAL_ID}/runs/{RUN_ID}", st, b, resp=b)95 # file_id source (purpose=evals)96 jsonl = "".join(json.dumps(x) + "\n" for x in content[:2]).encode()97 data, ct = multipart({"purpose": "evals"}, {"file": (f"{TAG}-{RUN}.jsonl", jsonl, "application/jsonl")})98 st, f = oai("POST", "/v1/files", data=data, content_type=ct, note="upload evals jsonl")99 rec("files.create(evals)", "POST", "/v1/files", st, f, id=f.get("id"), purpose=f.get("purpose"), expires_at=f.get("expires_at"))100 if f.get("id"):101 CREATED["files"].append(f["id"])102 st, run2 = oai("POST", f"/v1/evals/{EVAL_ID}/runs", {"name": f"{TAG}-{RUN}-run-file", "data_source": {"type": "jsonl", "source": {"type": "file_id", "id": f["id"]}}}, note="jsonl file_id")103 rec("evals.runs.create(jsonl file_id)", "POST", f"/v1/evals/{EVAL_ID}/runs", st, run2, id=run2.get("id"), status_field=run2.get("status"), error=run2.get("error"))104 if run2.get("id"):105 for i in range(40):106 st, run2 = oai("GET", f"/v1/evals/{EVAL_ID}/runs/{run2['id']}", note="poll")107 if run2.get("status") in ("completed", "failed", "canceled", "cancelled"):108 break109 time.sleep(3)110 rec("evals.runs.retrieve(file_id run)", "GET", f"/v1/evals/{EVAL_ID}/runs/{{run}}", st, run2, final_status=run2.get("status"), result_counts=run2.get("result_counts"), error=run2.get("error"))111 st, b = oai("DELETE", f"/v1/evals/{EVAL_ID}/runs/{run2['id']}", note="cleanup")112 rec("evals.runs.delete(file run)", "DELETE", f"/v1/evals/{EVAL_ID}/runs/{{run}}", st, b, deleted=b.get("deleted"))113 st, b = oai("DELETE", f"/v1/files/{f['id']}", note="cleanup")114 rec("files.delete", "DELETE", f"/v1/files/{f['id']}", st, b, deleted=b.get("deleted"))115 if b.get("deleted"):116 CREATED["files"].remove(f["id"])117 # run creation errors: responses data_source with no model118 st, b = oai("POST", f"/v1/evals/{EVAL_ID}/runs", {"name": "bad", "data_source": {"type": "responses", "source": {"type": "file_content", "content": [{"item": {"q": "x", "expected": "y"}}]}}}, note="responses source without model (expect 400)")119 rec("evals.runs.create(responses, no model)", "POST", f"/v1/evals/{EVAL_ID}/runs", st, b, error=b.get("error"), status_field=b.get("status"), id=b.get("id"))120 if b.get("id"): # should not happen; cancel+delete if it did121 oai("POST", f"/v1/evals/{EVAL_ID}/runs/{b['id']}", note="cancel unexpected run")122 oai("DELETE", f"/v1/evals/{EVAL_ID}/runs/{b['id']}", note="cleanup unexpected run")123 st, b = oai("DELETE", f"/v1/evals/{EVAL_ID}", note="cleanup")124 rec("evals.delete", "DELETE", f"/v1/evals/{EVAL_ID}", st, b, resp=b)125 if b.get("deleted"):126 CREATED["evals"].remove(EVAL_ID)127128# ---- logs data_source_config + label_model criterion (create/delete only, no run)129st, ev2 = oai("POST", "/v1/evals", {"name": f"{TAG}-{RUN}-logs", "data_source_config": {"type": "logs", "metadata": {"usecase": "atlas-test"}},130 "testing_criteria": [{"type": "label_model", "name": "lbl", "model": "gpt-4.1-nano", "labels": ["good", "bad"], "passing_labels": ["good"],131 "input": [{"role": "developer", "content": "Label the answer good or bad."}, {"role": "user", "content": "{{sample.output_text}}"}]}]},132 note="logs config + label_model (no run)")133rec("evals.create(logs+label_model)", "POST", "/v1/evals", st, ev2, id=ev2.get("id"), schema_keys=sorted(((ev2.get("data_source_config") or {}).get("schema") or {}).get("properties", {}).keys()) if isinstance(ev2, dict) else None, error=ev2.get("error"))134if ev2.get("id"):135 st, b = oai("DELETE", f"/v1/evals/{ev2['id']}", note="cleanup")136 rec("evals.delete(logs eval)", "DELETE", f"/v1/evals/{ev2['id']}", st, b, deleted=b.get("deleted"))137138# ---- fine-tuning: validation-only create (bogus training_file => nothing created)139st, b = oai("POST", "/v1/fine_tuning/jobs", {"model": "gpt-4.1-nano-2025-04-14", "training_file": "file-atlasbogus000", "suffix": TAG}, note="validation-only, bogus training_file (expect 4xx, no job)")140rec("fine_tuning.jobs.create(bogus file)", "POST", "/v1/fine_tuning/jobs", st, b, error=b.get("error"), id=b.get("id"), status_field=b.get("status"))141if b.get("id"):142 st, c = oai("POST", f"/v1/fine_tuning/jobs/{b['id']}/cancel", note="cancel unexpected job")143 rec("fine_tuning.jobs.cancel(unexpected)", "POST", f"/v1/fine_tuning/jobs/{b['id']}/cancel", st, c, status_field=c.get("status"))144st, b = oai("GET", "/v1/fine_tuning/jobs/ftjob-atlasbogus000/events", note="bogus id")145rec("fine_tuning.jobs.events(bogus)", "GET", "/v1/fine_tuning/jobs/{id}/events", st, b, error=b.get("error"))146st, b = oai("GET", "/v1/fine_tuning/jobs/ftjob-atlasbogus000/checkpoints", note="bogus id")147rec("fine_tuning.jobs.checkpoints(bogus)", "GET", "/v1/fine_tuning/jobs/{id}/checkpoints", st, b, error=b.get("error"))148st, b = oai("POST", "/v1/fine_tuning/jobs/ftjob-atlasbogus000/pause", note="bogus id")149rec("fine_tuning.jobs.pause(bogus)", "POST", "/v1/fine_tuning/jobs/{id}/pause", st, b, error=b.get("error"))150st, b = oai("GET", "/v1/fine_tuning/jobs?limit=2&metadata[owner]=atlas", note="metadata filter")151rec("fine_tuning.jobs.list(metadata filter)", "GET", "/v1/fine_tuning/jobs", st, b, n=len(b.get("data", [])) if isinstance(b, dict) else None, error=b.get("error"))152# batch: bogus file id create => error shape (nothing created)153st, b = oai("POST", "/v1/batches", {"input_file_id": "file-atlasbogus000", "endpoint": "/v1/responses", "completion_window": "24h"}, note="bogus input file (expect 4xx)")154rec("batches.create(bogus file)", "POST", "/v1/batches", st, b, error=b.get("error"), id=b.get("id"))155st, b = oai("GET", "/v1/batches/batch_atlasbogus000", note="bogus id")156rec("batches.retrieve(bogus)", "GET", "/v1/batches/{id}", st, b, error=b.get("error"))157st, b = oai("GET", "/v1/vector_stores/vs_atlasbogus000", note="bogus id")158rec("vector_stores.retrieve(bogus)", "GET", "/v1/vector_stores/{id}", st, b, error=b.get("error"))159st, b = oai("POST", "/v1/uploads", {"filename": "x.txt", "purpose": "user_data", "bytes": 10, "mime_type": "text/plain", "expires_after": {"anchor": "created_at", "seconds": 3600}}, note="upload with expires_after then cancel")160rec("uploads.create(expires_after)", "POST", "/v1/uploads", st, b, id=b.get("id"), expires_at=b.get("expires_at"), created_at=b.get("created_at"), error=b.get("error"))161if b.get("id"):162 st, c = oai("POST", f"/v1/uploads/{b['id']}/cancel", note="cleanup")163 rec("uploads.cancel", "POST", "/v1/uploads/{id}/cancel", st, c, status_field=c.get("status"))164165st, files = oai("GET", "/v1/files?limit=100", note="leftover check")166st2, evals = oai("GET", "/v1/evals?limit=100", note="leftover check")167st3, stores = oai("GET", "/v1/vector_stores?limit=100", note="leftover check")168left = {"files": [f["id"] for f in files.get("data", [])], "evals": [e["id"] for e in evals.get("data", [])], "vector_stores": [v["id"] for v in stores.get("data", [])]}169rec("leftover_check", "GET", "/v1/{files,evals,vector_stores}", 200, left, **left)170save_sanitized(SUMMARY, OUT / "summary.json")171print("DONE", json.dumps(left))172