SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
4.4 KB · 70 lines python
Raw Blame History
1"""Evals smoke tests: create/retrieve/update/delete an eval and run it on inline JSONL with pre-computed samples ($0)."""2from __future__ import annotations3import time4import pytest56TAG = "atlas-platform-agent-test"789def sample(text):10    return {"model": "test-app", "output_text": text, "choices": [{"index": 0, "message": {"role": "assistant", "content": text}, "finish_reason": "stop"}]}111213@pytest.fixture(scope="module")14def eval_obj(openai):15    st, ev, _ = openai("POST", "/v1/evals", {"name": TAG, "metadata": {"owner": TAG},16                       "data_source_config": {"type": "custom", "include_sample_schema": True,17                                              "item_schema": {"type": "object", "properties": {"q": {"type": "string"}, "expected": {"type": "string"}}, "required": ["q", "expected"]}},18                       "testing_criteria": [{"type": "string_check", "name": "exact", "input": "{{sample.output_text}}", "reference": "{{item.expected}}", "operation": "eq"}]},19                       note="test_evals create")20    assert st == 201, ev21    yield ev22    st, d, _ = openai("DELETE", f"/v1/evals/{ev['id']}", note="test_evals cleanup")23    assert st == 200 and d["deleted"] and d["object"] == "eval.deleted"242526def test_eval_shape(eval_obj):27    ev = eval_obj28    assert ev["object"] == "eval" and ev["id"].startswith("eval_")29    schema = ev["data_source_config"]["schema"]30    assert set(schema["required"]) == {"item", "sample"} and schema["properties"]["sample"]["required"] == ["model", "choices"]31    assert ev["testing_criteria"][0]["id"].startswith("exact-")323334def test_retrieve_update(openai, eval_obj):35    st, got, _ = openai("GET", f"/v1/evals/{eval_obj['id']}", note="test_evals retrieve")36    assert st == 200 and got["id"] == eval_obj["id"]37    st, upd, _ = openai("POST", f"/v1/evals/{eval_obj['id']}", {"metadata": {"owner": TAG, "phase": "2"}}, note="test_evals update")38    assert st == 200 and upd["metadata"]["phase"] == "2"394041def test_run_inline_jsonl_and_output_items(openai, eval_obj):42    content = [{"item": {"q": "1", "expected": "OK"}, "sample": sample("OK")}, {"item": {"q": "2", "expected": "OK"}, "sample": sample("KO")}]43    st, run, _ = openai("POST", f"/v1/evals/{eval_obj['id']}/runs", {"name": "inline", "data_source": {"type": "jsonl", "source": {"type": "file_content", "content": content}}}, note="test_evals run create")44    assert st == 201 and run["object"] == "eval.run" and run["status"] in ("queued", "in_progress")45    try:46        t0 = time.time()47        while run["status"] not in ("completed", "failed", "canceled") and time.time() - t0 < 120:48            time.sleep(3)49            st, run, _ = openai("GET", f"/v1/evals/{eval_obj['id']}/runs/{run['id']}", note="test_evals poll")50        assert run["status"] == "completed", run51        assert run["result_counts"] == {"total": 2, "errored": 0, "failed": 1, "passed": 1}52        assert run["per_testing_criteria_results"][0]["passed"] == 1 and run["per_model_usage"] in ([], None)53        st, items, _ = openai("GET", f"/v1/evals/{eval_obj['id']}/runs/{run['id']}/output_items?order=asc", note="test_evals output items")54        assert st == 200 and sorted(i["status"] for i in items["data"]) == ["fail", "pass"]55        item = items["data"][0]56        assert item["object"] == "eval.run.output_item" and {"datasource_item", "results", "sample"} <= set(item)57        assert item["results"][0]["score"] in (0.0, 1.0) and isinstance(item["results"][0]["passed"], bool)58        st, one, _ = openai("GET", f"/v1/evals/{eval_obj['id']}/runs/{run['id']}/output_items/{item['id']}", note="test_evals output item")59        assert st == 200 and one["id"] == item["id"]60        st, fails, _ = openai("GET", f"/v1/evals/{eval_obj['id']}/runs/{run['id']}/output_items?status=fail", note="test_evals output items fail")61        assert st == 200 and len(fails["data"]) == 162    finally:63        st, d, _ = openai("DELETE", f"/v1/evals/{eval_obj['id']}/runs/{run['id']}", note="test_evals run cleanup")64        assert st == 200 and d["deleted"]656667def test_run_requires_sample_model_and_choices(openai, eval_obj):68    st, err, _ = openai("POST", f"/v1/evals/{eval_obj['id']}/runs", {"data_source": {"type": "jsonl", "source": {"type": "file_content", "content": [{"item": {"q": "1", "expected": "OK"}, "sample": {"output_text": "OK"}}]}}}, note="test_evals bad sample (expect 400)")69    assert st == 400 and "required property" in err["error"]["message"]70