"""Prompt caching with the official Python SDK (anthropic 1.7): explicit breakpoint on a ~5.8k-token system block, then a second identical call, then a third call using top-level *automatic* caching and a 1h TTL variant on a fresh prefix. STATUS: LIVE_VERIFIED 2026-09-18 (claude-haiku-4-5-20251001): cold probe write 5809 → read 5809 → auto: read + create 7; example run (prefix already warm from the .sh example): read 5251 on the first two calls, auto read 5251 + create 7; 1h variant: cache_creation.ephemeral_1h_input_tokens 5809. ≈$0.02. Run: .venv/bin/python examples/anthropic/prompt-caching/cache_system_prompt.py """ from __future__ import annotations import os import sys from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parents[3])) from scripts import live # noqa: E402 loads .env, log_request import anthropic # noqa: E402 MODEL = os.environ.get("ANTHROPIC_MODEL", "claude-haiku-4-5-20251001") client = anthropic.Anthropic(max_retries=1) def filler(seed: str, n: int = 150) -> str: return " ".join(f"Section {seed}-{i}: The quarterly logistics review covers warehouse throughput, carrier performance, " "customs documentation, cold-chain compliance, and regional demand forecasting for the coming period." for i in range(n)) def show(label: str, msg: anthropic.types.Message) -> None: u = msg.usage print(f"{label}: input={u.input_tokens} creation={u.cache_creation_input_tokens} read={u.cache_read_input_tokens} " f"detail={u.cache_creation.model_dump() if u.cache_creation else None}") live.log_request("anthropic", "POST", "/v1/messages", 200, 0.004, f"example prompt-caching/cache_system_prompt.py {label} {MODEL}") system_a = [{"type": "text", "text": filler("A"), "cache_control": {"type": "ephemeral"}}] user = [{"role": "user", "content": "Reply with OK."}] show("explicit write", client.messages.create(model=MODEL, max_tokens=8, system=system_a, messages=user)) show("explicit read ", client.messages.create(model=MODEL, max_tokens=8, system=system_a, messages=user)) # Automatic caching: one top-level cache_control, no block markers. Lookback finds the system entry written above. show("automatic ", client.messages.create(model=MODEL, max_tokens=8, system=[{"type": "text", "text": filler("A")}], messages=user, cache_control={"type": "ephemeral"})) # 1-hour TTL (2x write price) on a fresh prefix so the write is visible under cache_creation.ephemeral_1h_input_tokens. system_b = [{"type": "text", "text": filler("B"), "cache_control": {"type": "ephemeral", "ttl": "1h"}}] show("1h write ", client.messages.create(model=MODEL, max_tokens=8, system=system_b, messages=user))