SPB Git forge

spb/doc-api

Public
2commits 1branches 0releases
15.7 MBsize
maindefault branch
13 days agolast push
Python 88.3% TypeScript 7.6% Shell 4.1%
9.3 KB · 141 lines python
Raw Blame History
1"""Smoke tests for OpenAI tools (function calling, custom tools, shell/apply_patch proposals — cheap; hosted tools gated).23Cheap tests (default): forced function call, strict-schema validation error, custom grammar tool, shell (local) and apply_patch4proposals, image_generation invalid-size error. Expensive tests (RUN_EXPENSIVE_TESTS=true): web_search ($0.01), code_interpreter5($0.03 + container cleanup), MCP (DeepWiki), file_search ($0.0025 + vector store), tool_search (gpt-5.4-mini).6"""7from __future__ import annotations89import json10import time1112import pytest1314WEATHER = {"type": "function", "name": "get_weather", "description": "Current weather for a city.", "strict": True,15           "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"], "additionalProperties": False}}161718def _resp(openai, body, note, cost=0.0):19    body.setdefault("max_output_tokens", 64)20    body.setdefault("reasoning", {"effort": "low"})21    st, out, _ = openai("POST", "/v1/responses", body, est_cost_usd=cost, note=note)22    return st, out232425def _items(out, t):26    return [i for i in out.get("output", []) if i.get("type") == t]272829def test_function_call_forced_and_round_trip(openai, models):30    st, out = _resp(openai, {"model": models["openai"], "input": "Weather in Paris?", "tools": [WEATHER],31                             "tool_choice": {"type": "function", "name": "get_weather"}}, "test_tools forced function")32    assert st == 200, out33    call = _items(out, "function_call")[0]34    assert call["name"] == "get_weather" and json.loads(call["arguments"])["city"] and call["call_id"].startswith("call_")35    st, out2 = _resp(openai, {"model": models["openai"], "previous_response_id": out["id"], "tools": [WEATHER], "max_output_tokens": 32,36                              "input": [{"type": "function_call_output", "call_id": call["call_id"], "output": json.dumps({"temp_c": 18})}]},37                     "test_tools function output round trip")38    assert st == 200 and _items(out2, "message"), out2394041def test_strict_schema_rejects_missing_additional_properties(openai, models):42    bad = {**WEATHER, "parameters": {"type": "object", "properties": {"city": {"type": "string"}}}}43    st, out = _resp(openai, {"model": models["openai"], "input": "hi", "tools": [bad], "max_output_tokens": 16}, "test_tools strict invalid")44    assert st == 400 and out["error"]["code"] == "invalid_function_parameters", out454647def test_allowed_tools_choice(openai, models):48    st, out = _resp(openai, {"model": models["openai"], "input": "Weather in Oslo?", "tools": [WEATHER, {**WEATHER, "name": "get_time"}],49                             "tool_choice": {"type": "allowed_tools", "mode": "required", "tools": [{"type": "function", "name": "get_weather"}]},50                             "parallel_tool_calls": False}, "test_tools allowed_tools")51    assert st == 200 and [i["name"] for i in _items(out, "function_call")] == ["get_weather"], out525354def test_custom_tool_regex_grammar(openai, models):55    tool = {"type": "custom", "name": "answer_yes_no", "format": {"type": "grammar", "syntax": "regex", "definition": r"^(yes|no)$"}}56    st, out = _resp(openai, {"model": models["openai"], "input": "Is 2+2 equal to 4? Use the tool.", "tools": [tool],57                             "tool_choice": {"type": "custom", "name": "answer_yes_no"}, "max_output_tokens": 32}, "test_tools custom grammar")58    assert st == 200 and _items(out, "custom_tool_call")[0]["input"] in ("yes", "no"), out596061def test_shell_local_and_apply_patch_proposals(openai, models):62    st, out = _resp(openai, {"model": models["openai"], "input": "Propose the shell command `echo OK`.", "tools": [{"type": "shell", "environment": {"type": "local"}}],63                             "tool_choice": {"type": "shell"}}, "test_tools shell local")64    assert st == 200 and _items(out, "shell_call")[0]["action"]["commands"], out65    st, out = _resp(openai, {"model": models["openai"], "input": "Create hello.txt containing OK using apply_patch.", "tools": [{"type": "apply_patch"}],66                             "tool_choice": {"type": "apply_patch"}, "max_output_tokens": 128}, "test_tools apply_patch")67    op = _items(out, "apply_patch_call")[0]["operation"]68    assert st == 200 and op["type"] in ("create_file", "update_file", "delete_file") and op["path"], out697071def test_image_generation_invalid_size_error(openai, models):72    st, out = _resp(openai, {"model": models["openai"], "input": "Draw a dot.", "tools": [{"type": "image_generation", "size": "1x1"}], "max_output_tokens": 16},73                    "test_tools image invalid size")74    assert st == 400 and out["error"]["type"] == "image_generation_user_error", out757677def test_chat_completions_function_call(openai, models):78    st, out, _ = openai("POST", "/v1/chat/completions", {"model": models["openai_chat"], "messages": [{"role": "user", "content": "Weather in Paris?"}],79                                                          "tools": [{"type": "function", "function": {"name": "get_weather", "parameters": WEATHER["parameters"], "strict": True}}],80                                                          "tool_choice": {"type": "function", "function": {"name": "get_weather"}}, "max_tokens": 32}, note="test_tools chat fn")81    assert st == 200 and out["choices"][0]["message"]["tool_calls"][0]["function"]["name"] == "get_weather", out828384@pytest.mark.run_expensive_tests85def test_web_search_forced(openai, models):86    st, out = _resp(openai, {"model": models["openai"], "input": "What is today's date in Toronto? Reply in 5 words.",87                             "tools": [{"type": "web_search", "search_context_size": "low"}], "tool_choice": {"type": "web_search"},88                             "include": ["web_search_call.action.sources"], "max_output_tokens": 600}, "test_tools web_search", cost=0.01)89    assert st == 200 and _items(out, "web_search_call")[0]["action"]["type"] == "search", out909192@pytest.mark.run_expensive_tests93def test_code_interpreter_auto_container(openai, models):94    st, out = _resp(openai, {"model": models["openai"], "input": "Run print(2+2) in python and reply with just the number.",95                             "tools": [{"type": "code_interpreter", "container": {"type": "auto"}}], "include": ["code_interpreter_call.outputs"],96                             "max_output_tokens": 128}, "test_tools code_interpreter", cost=0.03)97    assert st == 20098    ci = _items(out, "code_interpreter_call")[0]99    assert ci["container_id"].startswith("cntr_") and any(o.get("type") == "logs" for o in ci["outputs"] or [])100    st, d, _ = openai("DELETE", f"/v1/containers/{ci['container_id']}", note="test_tools delete container")101    assert st == 200 and d["deleted"] is True102103104@pytest.mark.run_expensive_tests105def test_mcp_deepwiki(openai, models):106    st, out = _resp(openai, {"model": models["openai"], "input": "Use read_wiki_structure for openai/openai-python; reply with the first topic only.",107                             "tools": [{"type": "mcp", "server_label": "deepwiki", "server_url": "https://mcp.deepwiki.com/mcp", "require_approval": "never",108                                        "allowed_tools": ["read_wiki_structure"]}]}, "test_tools mcp")109    assert st == 200 and _items(out, "mcp_list_tools") and _items(out, "mcp_call")[0]["error"] is None, out110111112@pytest.mark.run_expensive_tests113def test_file_search_roundtrip(openai, models):114    st, vs, _ = openai("POST", "/v1/vector_stores", {"name": "atlas-tools-agent"}, note="test_tools vs create")115    assert st == 200116    body = (b'--b\r\nContent-Disposition: form-data; name="purpose"\r\n\r\nassistants\r\n--b\r\nContent-Disposition: form-data; name="file"; filename="atlas.txt"\r\n'117            b"Content-Type: text/plain\r\n\r\nThe secret codeword is PELICAN-42. Filler filler filler filler.\r\n--b--\r\n")118    st, f, _ = openai("POST", "/v1/files", data=body, content_type="multipart/form-data; boundary=b", note="test_tools upload")119    try:120        openai("POST", f"/v1/vector_stores/{vs['id']}/files", {"file_id": f["id"]}, note="test_tools attach")121        for _ in range(20):122            st, vf, _ = openai("GET", f"/v1/vector_stores/{vs['id']}/files/{f['id']}", note="poll")123            if vf.get("status") in ("completed", "failed"):124                break125            time.sleep(1.5)126        st, out = _resp(openai, {"model": models["openai"], "input": "What is the secret codeword? Reply with the codeword only.",127                                 "tools": [{"type": "file_search", "vector_store_ids": [vs["id"]], "max_num_results": 2}],128                                 "include": ["file_search_call.results"], "max_output_tokens": 300}, "test_tools file_search", cost=0.0025)129        assert st == 200 and _items(out, "file_search_call")[0]["results"], out130    finally:131        openai("DELETE", f"/v1/vector_stores/{vs['id']}", note="test_tools vs delete")132        openai("DELETE", f"/v1/files/{f['id']}", note="test_tools file delete")133134135@pytest.mark.run_expensive_tests136def test_tool_search_hosted(openai):137    st, out = _resp(openai, {"model": "gpt-5.4-mini", "input": "Weather in Paris? Find and use a tool.", "max_output_tokens": 96,138                             "tools": [{"type": "tool_search"}, {"type": "namespace", "name": "weather", "description": "Weather tools", "tools": [{**WEATHER, "defer_loading": True}]}]},139                    "test_tools tool_search", cost=0.001)140    assert st == 200 and _items(out, "tool_search_call") and _items(out, "tool_search_output"), out141