#!/usr/bin/env python3 # ============================================================================= # Project : modelmap # File : experiments/micro/expH_capture_cost_frontier/implementation/benchmark_cold.py # Purpose : Run #2 — cold-cache storage-format throughput (purge per repeat) # Author : Simon-Pierre Boucher # Contact : contact@spboucher.ai # Website : https://modelmap.io # Created : 2026-08-12 # Modified : 2026-08-12 # Platform : macOS / Apple Silicon (arm64) # License : All rights reserved (research code) # ============================================================================= """expH run #2 (see hypothesis.md, registered before this run). Storage-only cold-cache variant: `sudo purge` empties the unified buffer cache before EVERY timed read repetition. Designed to run standalone on a MacLustr node (numpy + zarr + safetensors only, no torch/mlx). Results JSON is self-contained (embeds a local hardware manifest) and is collected back into results/ by the driver on the laptop. Usage: python3 benchmark_cold.py --workdir /path --out results.json \ [--purge-cmd "sudo -n purge"] """ from __future__ import annotations import argparse import json import platform import shutil import subprocess import time from pathlib import Path import numpy as np REPEATS = 3 ROWS, DIM = 200_000, 4096 # ~1.6 GB fp16 per format WRITE_CHUNK = 4_096 BATCH = 4_096 N_BATCHES = 24 SEED = 0 def sysctl(key: str) -> str: try: return subprocess.run(["sysctl", "-n", key], capture_output=True, text=True, check=True).stdout.strip() except subprocess.CalledProcessError: return "" def local_manifest() -> dict: return { "author": "Simon-Pierre Boucher", "chip": sysctl("machdep.cpu.brand_string"), "cores": int(sysctl("hw.ncpu") or 0), "unified_gb": round(int(sysctl("hw.memsize") or 0) / 2**30, 1), "os": platform.mac_ver()[0], "python": platform.python_version(), "numpy": np.__version__, "host": platform.node(), } def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("--workdir", required=True) ap.add_argument("--out", required=True) ap.add_argument("--purge-cmd", default="") args = ap.parse_args() workdir = Path(args.workdir) workdir.mkdir(parents=True, exist_ok=True) def purge(): if args.purge_cmd: subprocess.run(args.purge_cmd, shell=True, check=True, capture_output=True) def timed_cold(fn): out = [] for _ in range(REPEATS): purge() t0 = time.perf_counter() fn() out.append(time.perf_counter() - t0) return out rng = np.random.default_rng(SEED) data = rng.standard_normal((WRITE_CHUNK, DIM)).astype(np.float16) batches = [rng.integers(0, ROWS, BATCH) for _ in range(N_BATCHES)] total_bytes = ROWS * DIM * 2 batch_bytes = BATCH * DIM * 2 * N_BATCHES results = [] def record(fmt, op, times, nbytes): results.append({"format": fmt, "op": op, "bytes": nbytes, "seconds": times, "gb_per_s_mean": nbytes / 2**30 / np.mean(times), "cache": "cold" if op != "write" else "warm"}) print(f" {fmt:18s} {op:12s} {nbytes/2**30/np.mean(times):8.2f} GB/s", flush=True) # ---- raw np.memmap p = workdir / "acts.raw" def write_raw(): m = np.memmap(p, dtype=np.float16, mode="w+", shape=(ROWS, DIM)) for i in range(0, ROWS, WRITE_CHUNK): end = min(i + WRITE_CHUNK, ROWS) m[i:end] = data[: end - i] m.flush(); del m t0 = time.perf_counter(); write_raw() record("raw-mmap", "write", [time.perf_counter() - t0], total_bytes) def seq_raw(): m = np.memmap(p, dtype=np.float16, mode="r", shape=(ROWS, DIM)) float(np.asarray(m).sum(dtype=np.float32)); del m def rnd_raw(): m = np.memmap(p, dtype=np.float16, mode="r", shape=(ROWS, DIM)) for b in batches: m[b].sum(dtype=np.float32) del m record("raw-mmap", "seq-scan", timed_cold(seq_raw), total_bytes) record("raw-mmap", "random-batch", timed_cold(rnd_raw), batch_bytes) # ---- safetensors from safetensors import safe_open from safetensors.numpy import save_file ps = workdir / "acts.safetensors" full = np.memmap(p, dtype=np.float16, mode="r", shape=(ROWS, DIM)) t0 = time.perf_counter(); save_file({"acts": np.asarray(full)}, str(ps)) record("safetensors", "write", [time.perf_counter() - t0], total_bytes) del full def seq_st(): f = safe_open(str(ps), framework="np") float(f.get_tensor("acts").sum(dtype=np.float32)) def rnd_st(): f = safe_open(str(ps), framework="np") t = f.get_tensor("acts") for b in batches: t[b].sum(dtype=np.float32) record("safetensors", "seq-scan", timed_cold(seq_st), total_bytes) record("safetensors", "random-batch", timed_cold(rnd_st), batch_bytes) # ---- zarr variants import zarr for codec, name in ((None, "zarr-uncompressed"), ("default", "zarr-zstd")): pz = workdir / f"acts_{name}.zarr" kwargs = {} if codec == "default" else {"compressors": None} if pz.exists(): shutil.rmtree(pz) t0 = time.perf_counter() z = zarr.create_array(store=str(pz), shape=(ROWS, DIM), chunks=(WRITE_CHUNK, DIM), dtype=np.float16, **kwargs) for i in range(0, ROWS, WRITE_CHUNK): end = min(i + WRITE_CHUNK, ROWS) z[i:end] = data[: end - i] record(name, "write", [time.perf_counter() - t0], total_bytes) def seq_z(pz=pz): zz = zarr.open_array(store=str(pz), mode="r") float(zz[:].sum(dtype=np.float32)) def rnd_z(pz=pz): zz = zarr.open_array(store=str(pz), mode="r") for b in batches: zz[np.sort(b)].sum(dtype=np.float32) record(name, "seq-scan", timed_cold(seq_z), total_bytes) record(name, "random-batch", timed_cold(rnd_z), batch_bytes) doc = { "experiment": "expH_capture_cost_frontier", "run": 2, "scope": "storage formats, COLD cache (purge per repeat), second hardware", "config": {"rows": ROWS, "dim": DIM, "write_chunk": WRITE_CHUNK, "batch": BATCH, "n_batches": N_BATCHES, "repeats": REPEATS, "purge_cmd": args.purge_cmd or "(none — warm!)", "seed": SEED}, "manifest": local_manifest(), "storage": results, } Path(args.out).write_text(json.dumps(doc, indent=2) + "\n") print(f"results -> {args.out}") return 0 if __name__ == "__main__": raise SystemExit(main())