#!/usr/bin/env python3 """Live probes for the Anthropic tool-use ecosystem (API Atlas, 2026-09-18). Every call goes through scripts.live.anthropic_request (auto-logged + key masking). Raw sanitized responses are written to tmp-live/tools/.json. Budget target: <= $0.80 total. """ from __future__ import annotations import json, sys, time, base64 from pathlib import Path ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT)) from scripts import live # noqa: E402 OUT = ROOT / "tmp-live" / "tools" OUT.mkdir(parents=True, exist_ok=True) HAIKU = "claude-haiku-4-5-20251001" SONNET46 = "claude-sonnet-4-6" SONNET5 = "claude-sonnet-5" PRICE = {HAIKU: (1.0, 5.0), SONNET46: (3.0, 15.0), SONNET5: (2.0, 10.0)} # $/MTok in, out TOTAL = 0.0 SUMMARY: list[dict] = [] ONLY = set(sys.argv[1:]) def cost(model, usage, extra=0.0): if not isinstance(usage, dict): return extra pi, po = PRICE.get(model, (3.0, 15.0)) c = usage.get("input_tokens", 0) * pi / 1e6 + usage.get("output_tokens", 0) * po / 1e6 c += usage.get("cache_creation_input_tokens", 0) * pi * 1.25 / 1e6 + usage.get("cache_read_input_tokens", 0) * pi * 0.1 / 1e6 stu = usage.get("server_tool_use") or {} c += stu.get("web_search_requests", 0) * 0.01 return c + extra def call(step, body, beta=None, note="", stream=False, path="/v1/messages"): """POST and persist. Returns (status, body_json).""" global TOTAL if ONLY and step not in ONLY: return None, None t0 = time.time() st, out, hdrs = live.anthropic_request("POST", path, body, beta=beta, note=f"tools:{step} {note}", stream=stream) if stream: lines = list(out) out = {"_sse_lines": lines} # parse events for usage usage = {} for ln in lines: if ln.startswith("data: "): try: ev = json.loads(ln[6:]) except Exception: continue if ev.get("type") == "message_start": usage.update(ev["message"].get("usage", {})) if ev.get("type") == "message_delta": usage["output_tokens"] = ev.get("usage", {}).get("output_tokens", usage.get("output_tokens", 0)) out["_usage"] = usage else: usage = out.get("usage", {}) if isinstance(out, dict) else {} c = cost(body.get("model", ""), usage) if st == 200 else 0.0 TOTAL += c rec = {"step": step, "status": st, "model": body.get("model"), "beta": beta, "est_cost_usd": round(c, 5), "secs": round(time.time() - t0, 1), "request_id": hdrs.get("request-id"), "stop_reason": out.get("stop_reason") if isinstance(out, dict) else None} if st != 200 and isinstance(out, dict): rec["error"] = out.get("error") SUMMARY.append(rec) live.save_sanitized({"request": body, "beta": beta, "status": st, "response": out, "headers": live.interesting_headers(hdrs)}, OUT / f"{step}.json") print(json.dumps(rec), flush=True) return st, out def blocks(out, t): return [b for b in (out or {}).get("content", []) if b.get("type") == t] WEATHER = {"name": "get_weather", "description": "Get the current weather for a city. Returns a short text description.", "input_schema": {"type": "object", "properties": {"location": {"type": "string", "description": "City name, e.g. Paris"}}, "required": ["location"]}} TIME = {"name": "get_time", "description": "Get the current local time in a city.", "input_schema": {"type": "object", "properties": {"location": {"type": "string"}}, "required": ["location"]}} # ---------------- (a) custom tools ---------------- st, r = call("a1_forced_tool", {"model": HAIKU, "max_tokens": 200, "tools": [WEATHER], "tool_choice": {"type": "tool", "name": "get_weather"}, "messages": [{"role": "user", "content": "Weather in Paris?"}]}) if st == 200 and blocks(r, "tool_use"): tu = blocks(r, "tool_use")[0] call("a2_tool_result_final", {"model": HAIKU, "max_tokens": 60, "tools": [WEATHER], "messages": [{"role": "user", "content": "Weather in Paris?"}, {"role": "assistant", "content": r["content"]}, {"role": "user", "content": [{"type": "tool_result", "tool_use_id": tu["id"], "content": "18 C, light rain"}]}]}) call("a3_parallel_two_tools", {"model": HAIKU, "max_tokens": 300, "tools": [WEATHER, TIME], "messages": [{"role": "user", "content": "Give me the weather AND the local time in Tokyo. Call both tools at once."}]}) call("a4_disable_parallel", {"model": HAIKU, "max_tokens": 300, "tools": [WEATHER, TIME], "tool_choice": {"type": "auto", "disable_parallel_tool_use": True}, "messages": [{"role": "user", "content": "Give me the weather AND the local time in Tokyo. Call both tools at once."}]}) call("a5_tool_choice_any", {"model": HAIKU, "max_tokens": 150, "tools": [WEATHER, TIME], "tool_choice": {"type": "any"}, "messages": [{"role": "user", "content": "Say hello."}]}) call("a6_tool_choice_none", {"model": HAIKU, "max_tokens": 40, "tools": [WEATHER], "tool_choice": {"type": "none"}, "messages": [{"role": "user", "content": "Weather in Paris? Reply in 5 words."}]}) STRICT = dict(WEATHER, strict=True, input_schema={"type": "object", "properties": {"location": {"type": "string"}, "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}}, "required": ["location", "unit"], "additionalProperties": False}) call("a7_strict_any", {"model": HAIKU, "max_tokens": 150, "tools": [STRICT], "tool_choice": {"type": "any"}, "messages": [{"role": "user", "content": "Weather in Rome in celsius"}]}) call("a8_is_error_result", {"model": HAIKU, "max_tokens": 60, "tools": [WEATHER], "messages": [{"role": "user", "content": "Weather in Paris?"}, {"role": "assistant", "content": [{"type": "tool_use", "id": "toolu_atlas_fake_01", "name": "get_weather", "input": {"location": "Paris"}}]}, {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "toolu_atlas_fake_01", "is_error": True, "content": "Weather service unavailable (HTTP 503)"}]}]}) call("a9_text_before_tool_result_400", {"model": HAIKU, "max_tokens": 20, "tools": [WEATHER], "messages": [{"role": "user", "content": "Weather in Paris?"}, {"role": "assistant", "content": [{"type": "tool_use", "id": "toolu_atlas_fake_02", "name": "get_weather", "input": {"location": "Paris"}}]}, {"role": "user", "content": [{"type": "text", "text": "Here:"}, {"type": "tool_result", "tool_use_id": "toolu_atlas_fake_02", "content": "18 C"}]}]}) call("a10_input_examples", {"model": HAIKU, "max_tokens": 120, "tools": [dict(WEATHER, input_examples=[{"location": "Paris"}, {"location": "Tokyo"}])], "tool_choice": {"type": "any"}, "messages": [{"role": "user", "content": "Weather in Oslo"}]}) # ---------------- (b) streaming ---------------- call("b1_stream_input_json_delta", {"model": HAIKU, "max_tokens": 150, "stream": True, "tools": [WEATHER], "tool_choice": {"type": "any"}, "messages": [{"role": "user", "content": "Weather in San Francisco, CA?"}]}, stream=True) call("b2_stream_eager_input_streaming", {"model": HAIKU, "max_tokens": 150, "stream": True, "tools": [dict(WEATHER, eager_input_streaming=True)], "tool_choice": {"type": "any"}, "messages": [{"role": "user", "content": "Weather in San Francisco, CA?"}]}, stream=True) call("b3_stream_legacy_fgts_header", {"model": HAIKU, "max_tokens": 150, "stream": True, "tools": [WEATHER], "tool_choice": {"type": "any"}, "messages": [{"role": "user", "content": "Weather in San Francisco, CA?"}]}, beta="fine-grained-tool-streaming-2025-05-14", stream=True) # ---------------- (c) web search ---------------- call("c1_web_search_20250305", {"model": HAIKU, "max_tokens": 300, "tools": [{"type": "web_search_20250305", "name": "web_search", "max_uses": 1, "user_location": {"type": "approximate", "city": "Toronto", "region": "Ontario", "country": "CA", "timezone": "America/Toronto"}}], "messages": [{"role": "user", "content": "What is today's date in Toronto? Answer in 5 words."}]}) call("c2_web_search_20260209_haiku_no_allowed_callers", {"model": HAIKU, "max_tokens": 50, "tools": [{"type": "web_search_20260209", "name": "web_search", "max_uses": 1}], "messages": [{"role": "user", "content": "Reply with OK."}]}) call("c3_web_search_both_domain_lists_400", {"model": HAIKU, "max_tokens": 20, "tools": [{"type": "web_search_20250305", "name": "web_search", "allowed_domains": ["example.com"], "blocked_domains": ["foo.com"]}], "messages": [{"role": "user", "content": "Reply with OK."}]}) # ---------------- (d) web fetch ---------------- call("d1_web_fetch_20250910", {"model": HAIKU, "max_tokens": 200, "tools": [{"type": "web_fetch_20250910", "name": "web_fetch", "max_uses": 1, "citations": {"enabled": True}, "max_content_tokens": 2000}], "messages": [{"role": "user", "content": "Fetch https://example.com and tell me the page title in 3 words."}]}) # ---------------- (e) code execution ---------------- st, r = call("e1_code_execution_20250825", {"model": HAIKU, "max_tokens": 400, "tools": [{"type": "code_execution_20250825", "name": "code_execution"}], "messages": [{"role": "user", "content": "Use the code execution tool to run print(2+2) and reply with only the number."}]}) if st == 200 and isinstance(r, dict) and r.get("container"): cid = r["container"]["id"] call("e2_code_execution_container_reuse", {"model": HAIKU, "max_tokens": 300, "container": cid, "tools": [{"type": "code_execution_20250825", "name": "code_execution"}], "messages": [{"role": "user", "content": "Run the shell command `echo hi` with the code execution tool and reply with only its output."}]}) call("e3_code_execution_20260521_haiku", {"model": HAIKU, "max_tokens": 300, "tools": [{"type": "code_execution_20260521", "name": "code_execution"}], "messages": [{"role": "user", "content": "Use the code execution tool to run print(3*3) and reply with only the number."}]}) # ---------------- (f) tool search ---------------- DUMMY = [] for n, d in [("get_weather", "Get the current weather for a city"), ("get_stock_price", "Get the latest stock price for a ticker symbol"), ("send_email", "Send an email to a recipient"), ("create_calendar_event", "Create a calendar event"), ("translate_text", "Translate text into a target language"), ("lookup_zip_code", "Look up the city for a US zip code")]: DUMMY.append({"name": n, "description": d, "defer_loading": True, "input_schema": {"type": "object", "properties": {"q": {"type": "string", "description": "Main argument"}}, "required": ["q"]}}) call("f1_tool_search_regex", {"model": HAIKU, "max_tokens": 300, "tools": [{"type": "tool_search_tool_regex_20251119", "name": "tool_search_tool_regex"}] + DUMMY, "messages": [{"role": "user", "content": "What's the weather in Paris? Find and call the right tool."}]}) call("f2_tool_search_bm25", {"model": HAIKU, "max_tokens": 300, "tools": [{"type": "tool_search_tool_bm25_20251119", "name": "tool_search_tool_bm25"}] + DUMMY, "messages": [{"role": "user", "content": "What's the stock price of AAPL? Find and call the right tool."}]}) call("f3_tool_search_undated_alias", {"model": HAIKU, "max_tokens": 60, "tools": [{"type": "tool_search_tool_regex", "name": "tool_search_tool_regex"}] + DUMMY[:2], "messages": [{"role": "user", "content": "Reply with OK."}]}) call("f4_all_deferred_400", {"model": HAIKU, "max_tokens": 20, "tools": DUMMY[:2], "messages": [{"role": "user", "content": "Reply with OK."}]}) # ---------------- (g) programmatic tool calling ---------------- QDB = {"name": "query_database", "description": "Execute a SQL query. Returns a JSON array of row objects.", "input_schema": {"type": "object", "properties": {"sql": {"type": "string"}}, "required": ["sql"]}, "allowed_callers": ["code_execution_20260120"]} PTC_TOOLS = [{"type": "code_execution_20260120", "name": "code_execution"}, QDB] PTC_MSG = "Use code to call query_database with sql 'SELECT 1 AS one' and tell me the value of one. One sentence." st, r = call("g1_ptc_sonnet46", {"model": SONNET46, "max_tokens": 500, "tools": PTC_TOOLS, "messages": [{"role": "user", "content": PTC_MSG}]}) if st == 200 and isinstance(r, dict): pending = [b for b in blocks(r, "tool_use") if (b.get("caller") or {}).get("type", "").startswith("code_execution")] if pending and r.get("container"): results = [{"type": "tool_result", "tool_use_id": b["id"], "content": json.dumps([{"one": 1}])} for b in pending] call("g2_ptc_result", {"model": SONNET46, "max_tokens": 400, "container": r["container"]["id"], "tools": PTC_TOOLS, "messages": [{"role": "user", "content": PTC_MSG}, {"role": "assistant", "content": r["content"]}, {"role": "user", "content": results}]}) call("g3_ptc_haiku_unsupported", {"model": HAIKU, "max_tokens": 200, "tools": PTC_TOOLS, "messages": [{"role": "user", "content": PTC_MSG}]}) call("g4_ptc_tool_choice_400", {"model": SONNET46, "max_tokens": 20, "tools": PTC_TOOLS, "tool_choice": {"type": "tool", "name": "query_database"}, "messages": [{"role": "user", "content": PTC_MSG}]}) # ---------------- (h) computer use / text editor / bash / memory ---------------- PNG1x1 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg==" CU = {"type": "computer_20251124", "name": "computer", "display_width_px": 1024, "display_height_px": 768, "display_number": 1} st, r = call("h1_computer_20251124_sonnet46", {"model": SONNET46, "max_tokens": 100, "tools": [CU], "messages": [{"role": "user", "content": "Take a screenshot."}]}, beta="computer-use-2025-11-24") if st == 200 and blocks(r, "tool_use"): tu = blocks(r, "tool_use")[0] call("h2_computer_screenshot_result", {"model": SONNET46, "max_tokens": 50, "tools": [CU], "messages": [{"role": "user", "content": "Take a screenshot."}, {"role": "assistant", "content": r["content"]}, {"role": "user", "content": [{"type": "tool_result", "tool_use_id": tu["id"], "content": [{"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": PNG1x1}}]}]}]}, beta="computer-use-2025-11-24") call("h3_computer_20250124_haiku", {"model": HAIKU, "max_tokens": 100, "tools": [{"type": "computer_20250124", "name": "computer", "display_width_px": 1024, "display_height_px": 768}], "messages": [{"role": "user", "content": "Take a screenshot."}]}, beta="computer-use-2025-01-24") call("h4_computer_toolset_20260801_sonnet5", {"model": SONNET5, "max_tokens": 100, "tools": [{"type": "computer_toolset_20260801", "configs": {"zoom": {"enabled": False}}}], "messages": [{"role": "user", "content": "Take a screenshot."}]}) call("h5_computer_toolset_20260801_haiku_unsupported", {"model": HAIKU, "max_tokens": 20, "tools": [{"type": "computer_toolset_20260801"}], "messages": [{"role": "user", "content": "Take a screenshot."}]}) call("h6_text_editor_20250728", {"model": HAIKU, "max_tokens": 120, "tools": [{"type": "text_editor_20250728", "name": "str_replace_based_edit_tool"}], "messages": [{"role": "user", "content": "View the directory /tmp."}]}) call("h7_bash_20250124", {"model": HAIKU, "max_tokens": 120, "tools": [{"type": "bash_20250124", "name": "bash"}], "messages": [{"role": "user", "content": "Run `echo OK` in the shell."}]}) call("h8_memory_20250818", {"model": HAIKU, "max_tokens": 150, "tools": [{"type": "memory_20250818", "name": "memory"}], "messages": [{"role": "user", "content": "Remember that my favorite color is blue."}]}) call("h9_text_editor_wrong_name_400", {"model": HAIKU, "max_tokens": 20, "tools": [{"type": "text_editor_20250728", "name": "editor"}], "messages": [{"role": "user", "content": "Reply OK"}]}) # ---------------- (i) MCP connector ---------------- MCP = {"model": HAIKU, "max_tokens": 200, "mcp_servers": [{"type": "url", "url": "https://mcp.deepwiki.com/mcp", "name": "deepwiki"}], "tools": [{"type": "mcp_toolset", "mcp_server_name": "deepwiki"}], "messages": [{"role": "user", "content": "Call the read_wiki_structure tool for the GitHub repo anthropics/anthropic-sdk-python and reply with only the first 3 topic names."}]} call("i1_mcp_connector_deepwiki", MCP, beta="mcp-client-2025-11-20") call("i2_mcp_connector_no_beta_header", dict(MCP, max_tokens=20), beta=None) # ---------------- (j) count_tokens ---------------- call("j1_count_no_tools", {"model": HAIKU, "messages": [{"role": "user", "content": "Weather in Paris?"}]}, path="/v1/messages/count_tokens") call("j2_count_one_tool", {"model": HAIKU, "tools": [WEATHER], "messages": [{"role": "user", "content": "Weather in Paris?"}]}, path="/v1/messages/count_tokens") call("j3_count_tool_choice_any", {"model": HAIKU, "tools": [WEATHER], "tool_choice": {"type": "any"}, "messages": [{"role": "user", "content": "Weather in Paris?"}]}, path="/v1/messages/count_tokens") call("j4_count_bash", {"model": HAIKU, "tools": [{"type": "bash_20250124", "name": "bash"}], "messages": [{"role": "user", "content": "Weather in Paris?"}]}, path="/v1/messages/count_tokens") call("j5_count_text_editor", {"model": HAIKU, "tools": [{"type": "text_editor_20250728", "name": "str_replace_based_edit_tool"}], "messages": [{"role": "user", "content": "Weather in Paris?"}]}, path="/v1/messages/count_tokens") call("j6_count_memory", {"model": HAIKU, "tools": [{"type": "memory_20250818", "name": "memory"}], "messages": [{"role": "user", "content": "Weather in Paris?"}]}, path="/v1/messages/count_tokens") call("j7_count_web_search_400", {"model": HAIKU, "tools": [{"type": "web_search_20250305", "name": "web_search"}], "messages": [{"role": "user", "content": "Weather in Paris?"}]}, path="/v1/messages/count_tokens") call("j8_count_computer_20250124", {"model": HAIKU, "tools": [{"type": "computer_20250124", "name": "computer", "display_width_px": 1024, "display_height_px": 768}], "messages": [{"role": "user", "content": "Weather in Paris?"}]}, beta="computer-use-2025-01-24", path="/v1/messages/count_tokens") if SUMMARY: live.save_sanitized({"total_est_cost_usd": round(TOTAL, 4), "steps": SUMMARY}, OUT / "_summary.json") print(f"TOTAL est cost: ${TOTAL:.4f}")