#!/usr/bin/env python3 """Live probe of the Anthropic Messages API core surface (run once on 2026-09-18). STATUS: LIVE_VERIFIED — every call below ran against api.anthropic.com with the key in .env. Sanitized raw responses are written to tmp-live/anthropic-core/.json and every call is logged to reports/live-requests.jsonl by scripts/live.py. Budget: ~40 tiny calls on claude-haiku-4-5-20251001 (one on claude-sonnet-5), well under $0.30 total. Run: .venv/bin/python examples/anthropic/messages/probe_core_live.py """ from __future__ import annotations import base64 import json import sys import zlib from pathlib import Path ROOT = Path(__file__).resolve().parents[3] sys.path.insert(0, str(ROOT)) from scripts.live import anthropic_request, interesting_headers, save_sanitized # noqa: E402 OUT = ROOT / "tmp-live" / "anthropic-core" MODEL = "claude-haiku-4-5-20251001" PRICE = {"claude-haiku-4-5-20251001": (1.0, 5.0), "claude-sonnet-5": (3.0, 15.0)} # USD per 1M in/out (Haiku documented; Sonnet 5 assumed) SUMMARY: list[dict] = [] TOTAL_COST = 0.0 def cost(model: str, body) -> float: if not isinstance(body, dict) or "usage" not in body: return 0.0 u = body["usage"] pin, pout = PRICE.get(model, (3.0, 15.0)) return (u.get("input_tokens", 0) + (u.get("cache_creation_input_tokens") or 0)) * pin / 1e6 + u.get("output_tokens", 0) * pout / 1e6 def call(name: str, method: str, path: str, body=None, *, beta=None, headers=None, note="", model=MODEL): global TOTAL_COST st, out, hdrs = anthropic_request(method, path, body, beta=beta, extra_headers=headers, note=f"core-probe {name}: {note}") c = cost(model, out) TOTAL_COST += c rec = {"name": name, "method": method, "path": path, "status": st, "request": body, "beta": beta, "extra_headers": {k: ("***" if k.lower() == "x-api-key" else v) for k, v in (headers or {}).items()}, "response_headers": interesting_headers(hdrs), "response": out if not isinstance(out, bytes) else out.decode("utf-8", "replace"), "est_cost_usd": round(c, 6)} save_sanitized(rec, OUT / f"{name}.json") SUMMARY.append({"name": name, "status": st, "est_cost_usd": round(c, 6), "stop_reason": out.get("stop_reason") if isinstance(out, dict) else None, "error": out.get("error") if isinstance(out, dict) and out.get("type") == "error" else None}) print(f"[{st}] {name}") return st, out, hdrs def msg(text="Reply with OK.", **kw): b = {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": text}]} b.update(kw) return b def stream_probe(name: str, body: dict, *, beta=None): """Capture the exact ordered SSE event list.""" global TOTAL_COST st, gen, hdrs = anthropic_request("POST", "/v1/messages", body, beta=beta, stream=True, note=f"core-probe {name}: stream") events, raw_lines = [], [] if st == 200: cur = {} for line in gen: raw_lines.append(line) if line.startswith("event:"): cur = {"event": line[6:].strip()} elif line.startswith("data:"): try: cur["data"] = json.loads(line[5:].strip()) except Exception: # noqa: BLE001 cur["data"] = line[5:].strip() events.append(cur) cur = {} final_usage = next((e["data"].get("usage") for e in events if e["event"] == "message_delta"), None) start_msg = next((e["data"]["message"] for e in events if e["event"] == "message_start"), {}) c = 0.0 if final_usage: pin, pout = PRICE[MODEL] c = start_msg.get("usage", {}).get("input_tokens", 0) * pin / 1e6 + final_usage.get("output_tokens", 0) * pout / 1e6 TOTAL_COST += c else: c = 0.0 events = [{"event": "http_error", "data": gen if not isinstance(gen, bytes) else gen.decode()}] rec = {"name": name, "status": st, "request": body, "response_headers": interesting_headers(hdrs), "event_sequence": [e["event"] for e in events], "events": events, "raw_lines": raw_lines, "est_cost_usd": round(c, 6)} save_sanitized(rec, OUT / f"{name}.json") SUMMARY.append({"name": name, "status": st, "est_cost_usd": round(c, 6), "event_sequence": [e["event"] for e in events]}) print(f"[{st}] {name}: {[e['event'] for e in events]}") return events def one_px_png() -> str: def chunk(t, d): return len(d).to_bytes(4, "big") + t + d + zlib.crc32(t + d).to_bytes(4, "big") ihdr = (1).to_bytes(4, "big") + (1).to_bytes(4, "big") + bytes([8, 2, 0, 0, 0]) idat = zlib.compress(b"\x00\xff\x00\x00") png = b"\x89PNG\r\n\x1a\n" + chunk(b"IHDR", ihdr) + chunk(b"IDAT", idat) + chunk(b"IEND", b"") return base64.b64encode(png).decode() TOOL = {"name": "get_weather", "description": "Get current weather for a city.", "input_schema": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"]}} if __name__ == "__main__": OUT.mkdir(parents=True, exist_ok=True) # (a) minimal call("a_minimal", "POST", "/v1/messages", msg(), note="minimal; record undocumented fields") # (b) stream stream_probe("b_stream", msg(stream=True)) # (c) system + sampling params + stop sequence st, out, _ = call("c_system_sampling_stop", "POST", "/v1/messages", msg("Count: 1 2 3 4 5", system="You are terse. Output only what is asked.", temperature=0.2, top_p=0.9, top_k=40, stop_sequences=["3"], max_tokens=32), note="temperature+top_p+top_k together + stop_sequences") if st == 400: call("c2_temperature_only_stop", "POST", "/v1/messages", msg("Count: 1 2 3 4 5", system="You are terse. Output only what is asked.", temperature=0.2, top_k=40, stop_sequences=["3"], max_tokens=32), note="retry without top_p") # (d) max_tokens 1 call("d_max_tokens_1", "POST", "/v1/messages", msg("Write a paragraph about oceans.", max_tokens=1), note="expect stop_reason max_tokens") # (e) metadata.user_id call("e_metadata_user_id", "POST", "/v1/messages", msg(metadata={"user_id": "atlas-user-0001"}), note="metadata.user_id") # (f) prefill prefill = [{"role": "user", "content": "Say the word OK and nothing else."}, {"role": "assistant", "content": "The word is:"}] call("f_prefill_haiku", "POST", "/v1/messages", {"model": MODEL, "max_tokens": 8, "messages": prefill}, note="assistant prefill on Haiku 4.5") call("f2_prefill_sonnet5", "POST", "/v1/messages", {"model": "claude-sonnet-5", "max_tokens": 8, "messages": prefill}, note="assistant prefill on claude-sonnet-5 (expect 400)", model="claude-sonnet-5") call("f3_multiturn", "POST", "/v1/messages", {"model": MODEL, "max_tokens": 8, "messages": [ {"role": "user", "content": "Remember the code word: PLUM."}, {"role": "assistant", "content": "Noted."}, {"role": "user", "content": "Reply with only the code word."}]}, note="multi-turn") # (g) count_tokens base_ct = {"model": MODEL, "messages": [{"role": "user", "content": "Reply with OK."}]} call("g1_count_simple", "POST", "/v1/messages/count_tokens", base_ct, note="count_tokens simple") call("g2_count_system", "POST", "/v1/messages/count_tokens", {**base_ct, "system": "You are a helpful assistant."}, note="with system") call("g3_count_tool", "POST", "/v1/messages/count_tokens", {**base_ct, "tools": [TOOL]}, note="with tool") call("g4_count_image", "POST", "/v1/messages/count_tokens", {"model": MODEL, "messages": [{"role": "user", "content": [ {"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": one_px_png()}}, {"type": "text", "text": "Reply with OK."}]}]}, note="with 1x1 PNG") call("g5_count_document", "POST", "/v1/messages/count_tokens", {"model": MODEL, "messages": [{"role": "user", "content": [ {"type": "document", "source": {"type": "text", "media_type": "text/plain", "data": "The sky is blue."}, "title": "note", "citations": {"enabled": True}}, {"type": "text", "text": "Reply with OK."}]}]}, note="with text document + citations") call("g6_count_thinking", "POST", "/v1/messages/count_tokens", {**base_ct, "thinking": {"type": "enabled", "budget_tokens": 1024}}, note="with thinking enabled") call("g7_count_output_config", "POST", "/v1/messages/count_tokens", {**base_ct, "output_config": {"format": {"type": "json_schema", "schema": { "type": "object", "properties": {"answer": {"type": "string"}}, "required": ["answer"], "additionalProperties": False}}}}, note="with output_config.format") call("g8_count_server_tool", "POST", "/v1/messages/count_tokens", {**base_ct, "tools": [{"type": "web_search_20250305", "name": "web_search"}]}, note="server tool (docs say rejected)") # (h) service_tier call("h1_service_tier_standard_only", "POST", "/v1/messages", msg(service_tier="standard_only"), note="service_tier standard_only") call("h2_service_tier_auto", "POST", "/v1/messages", msg(service_tier="auto"), note="service_tier auto") # (i) errors call("i1_invalid_model", "POST", "/v1/messages", msg(model="claude-does-not-exist"), note="expect 404 not_found_error") b = msg(); del b["max_tokens"] call("i2_missing_max_tokens", "POST", "/v1/messages", b, note="expect 400") call("i3_bogus_api_key", "POST", "/v1/messages", msg(), headers={"x-api-key": "invalid-key-placeholder"}, note="expect 401") call("i4_max_tokens_too_large", "POST", "/v1/messages", msg(max_tokens=1_000_000), note="expect 400") call("i5_invalid_version", "POST", "/v1/messages", msg(), headers={"anthropic-version": "2020-01-01"}, note="invalid anthropic-version") call("i6_unknown_beta", "POST", "/v1/messages", msg(), beta="does-not-exist-2099-01-01", note="unknown beta header value") call("i7_temperature_5", "POST", "/v1/messages", msg(temperature=5), note="expect 400") call("i8_top_k_string", "POST", "/v1/messages", msg(top_k="abc"), note="wrong type") call("i9_empty_messages", "POST", "/v1/messages", msg(messages=[]), note="empty messages") st, out, hdrs = anthropic_request("POST", "/v1/messages", data=b'{"model": ', note="core-probe i10_bad_json") save_sanitized({"name": "i10_bad_json", "status": st, "response": out if not isinstance(out, bytes) else out.decode(), "response_headers": interesting_headers(hdrs)}, OUT / "i10_bad_json.json") SUMMARY.append({"name": "i10_bad_json", "status": st, "error": out.get("error") if isinstance(out, dict) else None}); print(f"[{st}] i10_bad_json") call("i11_not_found_path", "GET", "/v1/does-not-exist", None, note="unknown path") call("i12_tool_use_without_result", "POST", "/v1/messages", {"model": MODEL, "max_tokens": 8, "tools": [TOOL], "messages": [ {"role": "user", "content": "Weather in Paris?"}, {"role": "assistant", "content": [{"type": "tool_use", "id": "toolu_01ABCDEFGHIJKLMNOPQRSTUV", "name": "get_weather", "input": {"city": "Paris"}}]}, {"role": "user", "content": "thanks"}]}, note="tool_use without tool_result") call("i13_thinking_adaptive_haiku", "POST", "/v1/messages", msg(thinking={"type": "adaptive"}), note="adaptive thinking on 4.5 model (expect 400)") call("i14_output_format_legacy_field", "POST", "/v1/messages/count_tokens", {**base_ct, "output_format": {"type": "json_schema", "schema": {"type": "object"}}}, note="legacy output_format field without beta") # (j) models call("j1_get_model_alias", "GET", "/v1/models/claude-haiku-4-5", None, note="alias resolution") call("j2_list_models", "GET", "/v1/models?limit=3", None, note="list models limit 3") # (k) legacy complete call("k1_complete_legacy", "POST", "/v1/complete", {"model": "claude-2.1", "max_tokens_to_sample": 8, "prompt": "\n\nHuman: Reply with OK.\n\nAssistant:"}, note="legacy text completions, retired model", model="claude-2.1") call("k2_complete_haiku", "POST", "/v1/complete", {"model": MODEL, "max_tokens_to_sample": 8, "prompt": "\n\nHuman: Reply with OK.\n\nAssistant:"}, note="legacy text completions with current model") # (l) old version header call("l_version_2023_01_01", "POST", "/v1/messages", msg(), headers={"anthropic-version": "2023-01-01"}, note="anthropic-version 2023-01-01") # tool_use stop reason (cheap) + stream with tool for input_json_delta call("m1_tool_use_stop", "POST", "/v1/messages", {"model": MODEL, "max_tokens": 64, "tools": [TOOL], "tool_choice": {"type": "tool", "name": "get_weather"}, "messages": [{"role": "user", "content": "Weather in Paris?"}]}, note="forced tool_use → stop_reason tool_use") stream_probe("m2_stream_tool_use", {"model": MODEL, "max_tokens": 64, "stream": True, "tools": [TOOL], "tool_choice": {"type": "tool", "name": "get_weather"}, "messages": [{"role": "user", "content": "Weather in Paris?"}]}) # thinking stream on Haiku 4.5 (enabled, budget 1024) — small max_tokens to bound cost stream_probe("m3_stream_thinking", {"model": MODEL, "max_tokens": 1100, "stream": True, "thinking": {"type": "enabled", "budget_tokens": 1024}, "messages": [{"role": "user", "content": "Reply with OK."}]}) # cache_control (top-level) + max_tokens 0 pre-warm (documented) — tiny call("n1_max_tokens_0_prewarm", "POST", "/v1/messages", msg(max_tokens=0, cache_control={"type": "ephemeral"}), note="max_tokens 0 + top-level cache_control") # inference_geo param existence (may 400 depending on workspace) call("n2_inference_geo_us", "POST", "/v1/messages", msg(inference_geo="us"), note="inference_geo us") # stop_details / container / context_management presence with beta header for context management call("n3_beta_context_management", "POST", "/v1/messages", msg(context_management={"edits": [{"type": "clear_tool_uses_20250919"}]}), beta="context-management-2025-06-27", note="beta context_management shape") call("n4_speed_fast_haiku", "POST", "/v1/messages", msg(speed="fast"), beta="fast-mode-2026-02-01", note="speed fast on Haiku (expect 400)") call("n5_betas_field_in_body", "POST", "/v1/messages", msg(betas=["context-management-2025-06-27"]), note="betas as body field (SDK-only param?)") summary = {"probe": "anthropic-core", "date": "2026-09-18", "model": MODEL, "total_est_cost_usd": round(TOTAL_COST, 6), "calls": SUMMARY} save_sanitized(summary, OUT / "_summary.json") print(json.dumps({"total_est_cost_usd": round(TOTAL_COST, 6), "n_calls": len(SUMMARY)}))