"""Documents with the Python SDK: (1) a generated 1-page PDF as a base64 document block, (2) a text/plain document block, (3) count_tokens for an image to see the visual-token cost, (4) the `transformations.oversized_image` field on an image block. STATUS: LIVE_VERIFIED 2026-09-18 (claude-haiku-4-5-20251001): PDF answer "Green" (~1.6k input tokens), text doc "Green", count_tokens(1x1 PNG + 'Describe.') = 15, transformations accepted. ≈$0.002. Run: .venv/bin/python examples/anthropic/vision/pdf_and_text_document.py """ from __future__ import annotations import base64 import os import struct import sys import zlib from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parents[3])) from scripts import live # noqa: E402 import anthropic # noqa: E402 MODEL = os.environ.get("ANTHROPIC_MODEL", "claude-haiku-4-5-20251001") client = anthropic.Anthropic(max_retries=1) Q = {"type": "text", "text": "What color is grass? One word."} def tiny_pdf(text: str) -> bytes: stream = f"BT /F1 18 Tf 40 750 Td ({text}) Tj ET".encode() objs = [b"<< /Type /Catalog /Pages 2 0 R >>", b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>", b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>", b"<< /Length %d >>\nstream\n" % len(stream) + stream + b"\nendstream", b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>"] out, offs = b"%PDF-1.4\n", [] for i, o in enumerate(objs, 1): offs.append(len(out)); out += f"{i} 0 obj\n".encode() + o + b"\nendobj\n" xref = len(out) out += f"xref\n0 {len(objs)+1}\n0000000000 65535 f \n".encode() + b"".join(f"{o:010d} 00000 n \n".encode() for o in offs) return out + f"trailer\n<< /Size {len(objs)+1} /Root 1 0 R >>\nstartxref\n{xref}\n%%EOF\n".encode() def tiny_png() -> str: def chunk(t, d): return struct.pack(">I", len(d)) + t + d + struct.pack(">I", zlib.crc32(t + d) & 0xFFFFFFFF) png = b"\x89PNG\r\n\x1a\n" + chunk(b"IHDR", struct.pack(">IIBBBBB", 1, 1, 8, 2, 0, 0, 0)) + chunk(b"IDAT", zlib.compress(b"\x00\xff\x00\x00")) + chunk(b"IEND", b"") return base64.b64encode(png).decode() def ask(label: str, block: dict, cost: float) -> None: m = client.messages.create(model=MODEL, max_tokens=10, messages=[{"role": "user", "content": [block, Q]}]) print(f"{label}: {m.content[0].text!r} input_tokens={m.usage.input_tokens}") live.log_request("anthropic", "POST", "/v1/messages", 200, cost, f"example vision/pdf_and_text_document.py {label} {MODEL}") ask("pdf base64", {"type": "document", "source": {"type": "base64", "media_type": "application/pdf", "data": base64.b64encode(tiny_pdf("The sky is blue. Grass is green.")).decode()}, "title": "Colors"}, 0.0016) ask("text/plain", {"type": "document", "source": {"type": "text", "media_type": "text/plain", "data": "Grass is green."}}, 0.0001) img = {"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": tiny_png()}} cnt = client.messages.count_tokens(model=MODEL, messages=[{"role": "user", "content": [img, {"type": "text", "text": "Describe."}]}]) print("count_tokens(1x1 PNG + 'Describe.') =", cnt.input_tokens) live.log_request("anthropic", "POST", "/v1/messages/count_tokens", 200, 0, "example vision/pdf_and_text_document.py count_tokens image") ask("image + transformations", {**img, "transformations": {"oversized_image": "error"}}, 0.00004)