Python 88.3%
TypeScript 7.6%
Shell 4.1%
1"""Prompt caching with the official Python SDK (anthropic 1.7): explicit breakpoint on a ~5.8k-token system2block, then a second identical call, then a third call using top-level *automatic* caching and a 1h TTL variant3on a fresh prefix.4STATUS: LIVE_VERIFIED 2026-09-18 (claude-haiku-4-5-20251001): cold probe write 5809 → read 5809 → auto: read + create 7;5example run (prefix already warm from the .sh example): read 5251 on the first two calls, auto read 5251 + create 7;61h variant: cache_creation.ephemeral_1h_input_tokens 5809. ≈$0.02.7Run: .venv/bin/python examples/anthropic/prompt-caching/cache_system_prompt.py8"""9from __future__ import annotations1011import os12import sys13from pathlib import Path1415sys.path.insert(0, str(Path(__file__).resolve().parents[3]))16from scripts import live # noqa: E402 loads .env, log_request1718import anthropic # noqa: E4021920MODEL = os.environ.get("ANTHROPIC_MODEL", "claude-haiku-4-5-20251001")21client = anthropic.Anthropic(max_retries=1)222324def filler(seed: str, n: int = 150) -> str:25 return " ".join(f"Section {seed}-{i}: The quarterly logistics review covers warehouse throughput, carrier performance, "26 "customs documentation, cold-chain compliance, and regional demand forecasting for the coming period." for i in range(n))272829def show(label: str, msg: anthropic.types.Message) -> None:30 u = msg.usage31 print(f"{label}: input={u.input_tokens} creation={u.cache_creation_input_tokens} read={u.cache_read_input_tokens} "32 f"detail={u.cache_creation.model_dump() if u.cache_creation else None}")33 live.log_request("anthropic", "POST", "/v1/messages", 200, 0.004, f"example prompt-caching/cache_system_prompt.py {label} {MODEL}")343536system_a = [{"type": "text", "text": filler("A"), "cache_control": {"type": "ephemeral"}}]37user = [{"role": "user", "content": "Reply with OK."}]38show("explicit write", client.messages.create(model=MODEL, max_tokens=8, system=system_a, messages=user))39show("explicit read ", client.messages.create(model=MODEL, max_tokens=8, system=system_a, messages=user))40# Automatic caching: one top-level cache_control, no block markers. Lookback finds the system entry written above.41show("automatic ", client.messages.create(model=MODEL, max_tokens=8, system=[{"type": "text", "text": filler("A")}],42 messages=user, cache_control={"type": "ephemeral"}))43# 1-hour TTL (2x write price) on a fresh prefix so the write is visible under cache_creation.ephemeral_1h_input_tokens.44system_b = [{"type": "text", "text": filler("B"), "cache_control": {"type": "ephemeral", "ttl": "1h"}}]45show("1h write ", client.messages.create(model=MODEL, max_tokens=8, system=system_b, messages=user))46