Phase 1: literature sweep complete — 5 theme notes, ~300-source bibliography
research/notes/{quantization,sparsity_pruning,out_of_core_memory_systems,
decomposition_progressive,speculation_error_stability}.md + merged
bibliography.md + LOG entry with converged findings.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Showing 10 changed files with +3,981 and −2
added
experiments/micro/expH_ssd_feasibility/full_run.log
+83 −0
@@ -0,0 +1,83 @@ | ||
| 1 | +creating 8.0 GiB test file (once)… | |
| 2 | + random 4 KiB nocache=1 t=1: 66.9 MB/s | |
| 3 | + random 4 KiB nocache=1 t=4: 254.2 MB/s | |
| 4 | + random 4 KiB nocache=1 t=8: 479.9 MB/s | |
| 5 | + random 16 KiB nocache=1 t=1: 271.5 MB/s | |
| 6 | + random 16 KiB nocache=1 t=4: 1020.1 MB/s | |
| 7 | + random 16 KiB nocache=1 t=8: 1914.0 MB/s | |
| 8 | + random 64 KiB nocache=1 t=1: 769.5 MB/s | |
| 9 | + random 64 KiB nocache=1 t=4: 2917.4 MB/s | |
| 10 | + random 64 KiB nocache=1 t=8: 5137.6 MB/s | |
| 11 | + random 256 KiB nocache=1 t=1: 2302.5 MB/s | |
| 12 | + random 256 KiB nocache=1 t=4: 8134.0 MB/s | |
| 13 | + random 256 KiB nocache=1 t=8: 11590.5 MB/s | |
| 14 | + random 1024 KiB nocache=1 t=1: 4555.8 MB/s | |
| 15 | + random 1024 KiB nocache=1 t=4: 13627.6 MB/s | |
| 16 | + random 1024 KiB nocache=1 t=8: 13788.8 MB/s | |
| 17 | + random 4096 KiB nocache=1 t=1: 12402.6 MB/s | |
| 18 | + random 4096 KiB nocache=1 t=4: 48302.3 MB/s | |
| 19 | + random 4096 KiB nocache=1 t=8: 54669.7 MB/s | |
| 20 | + sequential 4 KiB nocache=1 t=1: 533.2 MB/s | |
| 21 | + sequential 4 KiB nocache=1 t=4: 808.9 MB/s | |
| 22 | + sequential 4 KiB nocache=1 t=8: 1278.1 MB/s | |
| 23 | + sequential 16 KiB nocache=1 t=1: 1683.5 MB/s | |
| 24 | + sequential 16 KiB nocache=1 t=4: 3235.6 MB/s | |
| 25 | + sequential 16 KiB nocache=1 t=8: 4610.0 MB/s | |
| 26 | + sequential 64 KiB nocache=1 t=1: 2297.1 MB/s | |
| 27 | + sequential 64 KiB nocache=1 t=4: 8909.2 MB/s | |
| 28 | + sequential 64 KiB nocache=1 t=8: 14453.3 MB/s | |
| 29 | + sequential 256 KiB nocache=1 t=1: 6645.8 MB/s | |
| 30 | + sequential 256 KiB nocache=1 t=4: 24105.9 MB/s | |
| 31 | + sequential 256 KiB nocache=1 t=8: 27404.6 MB/s | |
| 32 | + sequential 1024 KiB nocache=1 t=1: 12139.1 MB/s | |
| 33 | + sequential 1024 KiB nocache=1 t=4: 17844.4 MB/s | |
| 34 | + sequential 1024 KiB nocache=1 t=8: 34075.1 MB/s | |
| 35 | + sequential 4096 KiB nocache=1 t=1: 14030.8 MB/s | |
| 36 | + sequential 4096 KiB nocache=1 t=4: 46677.5 MB/s | |
| 37 | + sequential 4096 KiB nocache=1 t=8: 52861.6 MB/s | |
| 38 | + random 4 KiB nocache=0 t=1: 4833.8 MB/s | |
| 39 | + random 4 KiB nocache=0 t=4: 1297.7 MB/s | |
| 40 | + random 4 KiB nocache=0 t=8: 838.6 MB/s | |
| 41 | + random 16 KiB nocache=0 t=1: 13794.4 MB/s | |
| 42 | + random 16 KiB nocache=0 t=4: 6611.9 MB/s | |
| 43 | + random 16 KiB nocache=0 t=8: 3239.0 MB/s | |
| 44 | + random 64 KiB nocache=0 t=1: 25056.9 MB/s | |
| 45 | + random 64 KiB nocache=0 t=4: 18890.1 MB/s | |
| 46 | + random 64 KiB nocache=0 t=8: 12634.3 MB/s | |
| 47 | + random 256 KiB nocache=0 t=1: 37768.7 MB/s | |
| 48 | + random 256 KiB nocache=0 t=4: 76369.7 MB/s | |
| 49 | + random 256 KiB nocache=0 t=8: 46092.8 MB/s | |
| 50 | + random 1024 KiB nocache=0 t=1: 42521.2 MB/s | |
| 51 | + random 1024 KiB nocache=0 t=4: 117590.7 MB/s | |
| 52 | + random 1024 KiB nocache=0 t=8: 136162.5 MB/s | |
| 53 | + random 4096 KiB nocache=0 t=1: 42051.2 MB/s | |
| 54 | + random 4096 KiB nocache=0 t=4: 90258.2 MB/s | |
| 55 | + random 4096 KiB nocache=0 t=8: 104391.2 MB/s | |
| 56 | + sequential 4 KiB nocache=0 t=1: 9169.6 MB/s | |
| 57 | + sequential 4 KiB nocache=0 t=4: 1458.5 MB/s | |
| 58 | + sequential 4 KiB nocache=0 t=8: 869.9 MB/s | |
| 59 | + sequential 16 KiB nocache=0 t=1: 21617.5 MB/s | |
| 60 | + sequential 16 KiB nocache=0 t=4: 6491.6 MB/s | |
| 61 | + sequential 16 KiB nocache=0 t=8: 3326.3 MB/s | |
| 62 | + sequential 64 KiB nocache=0 t=1: 29957.4 MB/s | |
| 63 | + sequential 64 KiB nocache=0 t=4: 17569.6 MB/s | |
| 64 | + sequential 64 KiB nocache=0 t=8: 12489.8 MB/s | |
| 65 | + sequential 256 KiB nocache=0 t=1: 39164.7 MB/s | |
| 66 | + sequential 256 KiB nocache=0 t=4: 76053.6 MB/s | |
| 67 | + sequential 256 KiB nocache=0 t=8: 45630.6 MB/s | |
| 68 | + sequential 1024 KiB nocache=0 t=1: 42748.7 MB/s | |
| 69 | + sequential 1024 KiB nocache=0 t=4: 115452.8 MB/s | |
| 70 | + sequential 1024 KiB nocache=0 t=8: 135141.7 MB/s | |
| 71 | + sequential 4096 KiB nocache=0 t=1: 41845.6 MB/s | |
| 72 | + sequential 4096 KiB nocache=0 t=4: 89032.2 MB/s | |
| 73 | + sequential 4096 KiB nocache=0 t=8: 102950.7 MB/s | |
| 74 | +recreating test file to evict cached pages before GPU-load phase… | |
| 75 | +re-running key cells under concurrent Metal (MLX) matmul load… | |
| 76 | +GPU+ random 4 KiB nocache=1 t=8: 478.4 MB/s | |
| 77 | +GPU+ random 16 KiB nocache=1 t=8: 1888.0 MB/s | |
| 78 | +GPU+ random 64 KiB nocache=1 t=8: 5351.3 MB/s | |
| 79 | +GPU+ random 256 KiB nocache=1 t=8: 11609.0 MB/s | |
| 80 | +GPU+ random 1024 KiB nocache=1 t=8: 13393.4 MB/s | |
| 81 | +GPU+ random 4096 KiB nocache=1 t=8: 30955.8 MB/s | |
| 82 | + | |
| 83 | +wrote /Users/simon-pierreboucher/Desktop/localvm-research/results/expH_ssd_feasibility/20260812T034359Z/results.json | |
modified
research/LOG.md
+36 −0
@@ -27,3 +27,39 @@ Format per entry: date/time (local, with timezone) · question · experiment · | ||
| 27 | 27 | - **Decision:** Begin Phase 1 (ultra-deep literature research, charter §4) immediately, |
| 28 | 28 | fanning out across the ten mandated areas (§4.1–§4.10). Deliverables: per-topic notes in |
| 29 | 29 | `research/notes/`, every source logged in `research/bibliography.md` with URL and access date. |
| 30 | + | |
| 31 | +--- | |
| 32 | + | |
| 33 | +## 2026-08-12 00:15 EDT — Phase 1 complete: literature sweep across §4.1–§4.10 | |
| 34 | + | |
| 35 | +- **Question:** What is already known about post-training transforms + out-of-core execution | |
| 36 | + that decouple checkpoint size from resident memory / bytes-per-token, and what is missing? | |
| 37 | +- **Experiment:** Five parallel deep literature sweeps (web, arXiv, GitHub, proceedings), | |
| 38 | + one per theme cluster. Deliverables in `research/notes/` (5 files, ~300 sources, all with | |
| 39 | + access dates, merged into `research/bibliography.md`). | |
| 40 | +- **Result (key facts):** | |
| 41 | + - Nested/progressive weight encodings exist (Any-Precision LLM, MatQuant, BitStack, RRQ, | |
| 42 | + CALDERA Q+LR) but ALL choose the operating point statically, keep everything resident, | |
| 43 | + never page residuals from storage, and are CUDA-only. | |
| 44 | + - 2–3-bit SOTA (QuIP#, AQLM, QTIP, GPTVQ) has zero Metal implementations; LUT-heavy decode | |
| 45 | + is compute-bound on Apple GPUs — Apple formats must keep decode shift/mask-cheap. | |
| 46 | + - A representational cliff sits between 2 and 3 bits (ParetoQ/EfficientQAT): a 2-bit base | |
| 47 | + is the floor for staying in the pretrained basin. | |
| 48 | + - Closest prior art on our hardware: Apple "LLM in a flash" (0.2 GB/token OPT-6.7B on | |
| 49 | + M1 Max via predictors + windowing; ReLU-only, FFN-only, fp16, no code) and PowerInfer-2 | |
| 50 | + (47B on a 24 GB phone; requires dReLU retraining). Nobody has built predictor/threshold- | |
| 51 | + driven sparse weight paging on macOS/Metal/unified memory. | |
| 52 | + - Measured decision stability: 4-bit quants agree with fp16 on ~90–91% of greedy tokens; | |
| 53 | + the joint (cheap-pass margin × agreement) distribution is UNPUBLISHED — cheap for us to | |
| 54 | + measure (expG) and decisive for any escalation design. | |
| 55 | + - Precision-level self-speculation exists across tokens (QSpec, Apple QuantSpec ~2.5×, | |
| 56 | + >90% acceptance) but nobody gates *weight loading* on per-token decision uncertainty. | |
| 57 | + - OS/DB ideas unapplied to LLM weights: MRU/DBMIN for cyclic dense scans (LRU provably | |
| 58 | + worst-case for our pattern), ARC ghost lists for expert caches, anti-caching's | |
| 59 | + never-block-on-miss, purgeable MTLHeaps as an OS-cooperative cache tier. | |
| 60 | +- **Interpretation:** Independent sweeps converged on the same gap: a *progressive, | |
| 61 | + residency-tiered weight representation* (low-bit resident base + SSD-resident residuals) | |
| 62 | + with *decision-uncertainty-driven refinement* is unbuilt, and Apple unified memory + | |
| 63 | + fast NVMe is the substrate where it is most plausible. | |
| 64 | +- **Decision:** Proceed to Phase 2 (state-of-the-art map synthesized from notes), then | |
| 65 | + Phase 3 gap generation (≥20 ideas). expH (SSD envelope) running concurrently. | |
modified
research/bibliography.md
+329 −2
@@ -10,6 +10,333 @@ status: draft | ||
| 10 | 10 | # Bibliography |
| 11 | 11 | |
| 12 | 12 | Every consulted source, with URL and access date. Grouped by theme (mirrors `research/notes/`). |
| 13 | −Entries are appended as Phase 1 progresses; nothing is deleted. | |
| 13 | +Entries are appended as research progresses; nothing is deleted. | |
| 14 | 14 | |
| 15 | −*(Populated during Phase 1 — see per-theme sections below as they are added.)* | |
| 15 | +## Quantization (§4.1) | |
| 16 | + | |
| 17 | +From `research/notes/quantization.md` (69 sources). | |
| 18 | + | |
| 19 | +- GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers (Frantar et al., ICLR 2023) — https://arxiv.org/abs/2210.17323 (accessed 2026-08-11) | |
| 20 | +- AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration (Lin et al., MLSys 2024) — https://arxiv.org/abs/2306.00978 (accessed 2026-08-11) | |
| 21 | +- SmoothQuant: Accurate and Efficient Post-Training Quantization for LLMs (Xiao et al., ICML 2023) — https://arxiv.org/abs/2211.10438 (accessed 2026-08-11) | |
| 22 | +- A Practical Guide to INT4 Quantization for SLMs: GPTQ vs AWQ (Microsoft Data Science, Medium) — https://medium.com/data-science-at-microsoft/a-practical-guide-to-int4-quantization-for-slms-gptq-vs-awq-olive-and-real-world-results-2f63d6963d1d (accessed 2026-08-11) | |
| 23 | +- Combining multiple post-training techniques to achieve most efficient quantized LLMs (MX formats + GPTQ/SmoothQuant) — https://arxiv.org/html/2405.07135v1 (accessed 2026-08-11) | |
| 24 | +- QuIP: 2-Bit Quantization of Large Language Models With Guarantees (Chee et al., NeurIPS 2023) — https://neurips.cc/virtual/2023/poster/69982 (accessed 2026-08-11) | |
| 25 | +- QuIP#: Even Better LLM Quantization with Hadamard Incoherence and Lattice Codebooks (Tseng et al., ICML 2024) — slides — https://icml.cc/media/icml-2024/Slides/34816.pdf (accessed 2026-08-11) | |
| 26 | +- QuIP# codebase (Cornell RelaxML) — https://github.com/Cornell-RelaxML/quip-sharp (accessed 2026-08-11) | |
| 27 | +- QuIP# full text (PMC mirror, RVQ details) — https://pmc.ncbi.nlm.nih.gov/articles/PMC12395268 (accessed 2026-08-11) | |
| 28 | +- QTIP: Quantization with Trellises and Incoherence Processing (Tseng et al., NeurIPS 2024) — https://arxiv.org/html/2406.11235v1 (accessed 2026-08-11) | |
| 29 | +- Even Better, Even Faster Quantized LLMs with QTIP (Together AI blog) — https://www.together.ai/blog/even-better-even-faster-quantized-llms-with-qtip (accessed 2026-08-11) | |
| 30 | +- AQLM: Extreme Compression of Large Language Models via Additive Quantization (Egiazarian et al., ICML 2024) — https://arxiv.org/html/2401.06118v2 (accessed 2026-08-11) | |
| 31 | +- AQLM + PV-Tuning official repository — https://github.com/vahe1994/AQLM (accessed 2026-08-11) | |
| 32 | +- PV-Tuning: Beyond Straight-Through Estimation for Extreme LLM Compression (NeurIPS 2024) — https://proceedings.neurips.cc/paper_files/paper/2024/file/091166620a04a289c555f411d8899049-Paper-Conference.pdf (accessed 2026-08-11) | |
| 33 | +- The Evolution of Extreme LLM Compression: From QuIP to AQLM with PV-Tuning (Yandex, Medium) — https://medium.com/yandex/the-evolution-of-extreme-llm-compression-from-quip-to-aqlm-with-pv-tuning-19c44b91af96 (accessed 2026-08-11) | |
| 34 | +- GPTVQ: The Blessing of Dimensionality for LLM Quantization (van Baalen et al., Qualcomm) — https://arxiv.org/abs/2402.15319 (accessed 2026-08-11) | |
| 35 | +- GPTVQ repository — https://github.com/Qualcomm-AI-research/gptvq (accessed 2026-08-11) | |
| 36 | +- NestQuant: Nested Lattice Quantization for Matrix Products and LLMs (Savkin et al., ICML 2025) — https://arxiv.org/abs/2502.09720 (accessed 2026-08-11) | |
| 37 | +- Learning Grouped Lattice Vector Quantizers for Low-Bit LLM Compression (NeurIPS 2025) — https://neurips.cc/virtual/2025/poster/117396 (accessed 2026-08-11) | |
| 38 | +- SqueezeLLM: Dense-and-Sparse Quantization (Kim et al., ICML 2024) — https://arxiv.org/html/2306.07629v4 (accessed 2026-08-11) | |
| 39 | +- Half-Quadratic Quantization of Large Machine Learning Models (Mobius Labs, via Dropbox Tech) — https://dropbox.tech/machine-learning/halfquadratic-quantization-of-large-machine-learning-models (accessed 2026-08-11) | |
| 40 | +- QuaRot: Outlier-Free 4-Bit Inference in Rotated LLMs (Ashkboos et al., NeurIPS 2024) — https://neurips.cc/virtual/2024/poster/94328 (accessed 2026-08-11) | |
| 41 | +- SpinQuant: LLM Quantization with Learned Rotations (Liu et al., ICLR 2025) — https://proceedings.iclr.cc/paper_files/paper/2025/file/e5b1c0d4866f72393c522c8a00eed4eb-Paper-Conference.pdf (accessed 2026-08-11) | |
| 42 | +- Rotation-based quantization with QuaRot (AMD Quark docs, R1–R4 rotations) — https://quark.docs.amd.com/release-0.9/pytorch/tutorial_quarot.html (accessed 2026-08-11) | |
| 43 | +- The Era of 1-bit LLMs: All Large Language Models are in 1.58 Bits (Ma et al., Microsoft) — https://arxiv.org/abs/2402.17764 (accessed 2026-08-11) | |
| 44 | +- 1-bit AI Infra Part 1.1: Fast and Lossless BitNet b1.58 Inference on CPUs (bitnet.cpp) — https://arxiv.org/html/2410.16144v1 (accessed 2026-08-11) | |
| 45 | +- Bitnet.cpp: Efficient Edge Inference for Ternary LLMs (ACL 2025; TL/I2_S kernels, M2 Ultra 100B result) — https://aclanthology.org/2025.acl-long.457.pdf (accessed 2026-08-11) | |
| 46 | +- microsoft/BitNet official inference framework — https://github.com/microsoft/BitNet (accessed 2026-08-11) | |
| 47 | +- BiLLM: Pushing the Limit of Post-Training Quantization for LLMs (Huang et al., ICML 2024) — https://github.com/Aaronhuang-778/BiLLM (accessed 2026-08-11) | |
| 48 | +- ParetoQ: Scaling Laws in Extremely Low-bit LLM Quantization (Liu et al., Meta, NeurIPS 2025) — https://arxiv.org/html/2502.02631v2 (accessed 2026-08-11) | |
| 49 | +- ParetoQ (PyTorch blog) — https://pytorch.org/blog/paretoq-scaling-laws-in-extremely-low-bit-llm-quantization (accessed 2026-08-11) | |
| 50 | +- EfficientQAT: Efficient Quantization-Aware Training for LLMs (Chen et al., ACL 2025) — https://arxiv.org/abs/2407.11062 (accessed 2026-08-11) | |
| 51 | +- EfficientQAT repository (w2g64 PPL tables) — https://github.com/OpenGVLab/EfficientQAT (accessed 2026-08-11) | |
| 52 | +- Any-Precision LLM: Low-Cost Deployment of Multiple, Different-Sized LLMs (Park et al., ICML 2024 oral) — https://arxiv.org/html/2402.10517v4 (accessed 2026-08-11) | |
| 53 | +- Any-Precision LLM repository (bitplane engine) — https://github.com/SNU-ARC/any-precision-llm (accessed 2026-08-11) | |
| 54 | +- Matryoshka Quantization (Nair et al., Google DeepMind, ICLR 2025 oral) — https://iclr.cc/virtual/2025/10000114 (accessed 2026-08-11) | |
| 55 | +- Matryoshka Quantization topic overview (Emergent Mind) — https://www.emergentmind.com/topics/matryoshka-quantization-matquant (accessed 2026-08-11) | |
| 56 | +- Multi-Bitwidth Quantization for LLMs Using Additive Codebooks ("Drop-by-Drop", successive refinement) — https://arxiv.org/html/2606.12876v1 (accessed 2026-08-11) | |
| 57 | +- Progressive Mixed-Precision Decoding for Efficient LLM Inference (Chen et al., ICLR 2025) — https://arxiv.org/abs/2410.13461 (accessed 2026-08-11) | |
| 58 | +- Mixed-Precision Quantization for Language Models (survey, Oct 2025; PMDP/MPMLC taxonomy) — https://arxiv.org/html/2510.16805v1 (accessed 2026-08-11) | |
| 59 | +- SeedLM: Compressing LLM Weights into Seeds of Pseudo-Random Generators (Apple ML Research) — https://machinelearning.apple.com/research/seedlm-compressing (accessed 2026-08-11) | |
| 60 | +- SeedLM (arXiv full text) — https://arxiv.org/html/2410.10714v1 (accessed 2026-08-11) | |
| 61 | +- DFloat11: 70% Size, 100% Accuracy — Lossless LLM Compression via Dynamic-Length Float — https://huggingface.co/papers/2504.11651 (accessed 2026-08-11) | |
| 62 | +- DFloat11 repository (LeanModels, NeurIPS 2025) — https://github.com/LeanModels/DFloat11 (accessed 2026-08-11) | |
| 63 | +- DFloat11 throughput caveats (Hacker News discussion incl. appendix numbers) — https://news.ycombinator.com/item?id=43796935 (accessed 2026-08-11) | |
| 64 | +- KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache (Liu et al., ICML 2024) — https://arxiv.org/abs/2402.02750 (accessed 2026-08-11) | |
| 65 | +- KIVI repository — https://github.com/jy-yuan/KIVI (accessed 2026-08-11) | |
| 66 | +- KVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization (Hooper et al., NeurIPS 2024) — https://arxiv.org/abs/2401.18079 (accessed 2026-08-11) | |
| 67 | +- KV Cache is 1 Bit Per Channel: Coupled Quantization (NeurIPS 2024) — https://proceedings.neurips.cc/paper_files/paper/2024/file/05d6b5b6901fb57d2c287e1d3ce6d63c-Paper-Conference.pdf (accessed 2026-08-11) | |
| 68 | +- mlx.core.quantize documentation (affine/mxfp4/mxfp8/nvfp4 modes) — https://ml-explore.github.io/mlx/build/html/python/_autosummary/mlx.core.quantize.html (accessed 2026-08-11) | |
| 69 | +- mlx.nn.quantize documentation (quantize_input, class predicates) — https://ml-explore.github.io/mlx/build/html/python/_autosummary/mlx.nn.quantize.html (accessed 2026-08-11) | |
| 70 | +- mlx-lm LEARNED_QUANTS.md (DWQ, dynamic_quant, AWQ/GPTQ recipes) — https://github.com/ml-explore/mlx-lm/blob/main/mlx_lm/LEARNED_QUANTS.md (accessed 2026-08-11) | |
| 71 | +- MLX Quantization on Apple Silicon: dynamic_quant vs AWQ vs GPTQ vs DWQ (Hannecke, Medium) — https://medium.com/@michael.hannecke/mlx-quantization-on-apple-silicon-dynamic-quant-vs-awq-vs-gptq-vs-dwq-8b2a5af2b53f (accessed 2026-08-11) | |
| 72 | +- Better inference quality and performance for MLX on Apple Silicon (Feldman; K-quant vs MLX affine KL measurements) — https://www.linkedin.com/pulse/better-inference-quality-performance-mlx-apple-silicon-asher-feldman-ztm0e (accessed 2026-08-11) | |
| 73 | +- Very slow IQ quant performance on Apple Silicon (llama.cpp discussion #5617, ikawrakow measurements) — https://github.com/ggml-org/llama.cpp/discussions/5617 (accessed 2026-08-11) | |
| 74 | +- Overview of GGUF quantization methods (r/LocalLLaMA; i-quant LUT bottleneck notes) — https://www.reddit.com/r/LocalLLaMA/comments/1ba55rj/overview_of_gguf_quantization_methods (accessed 2026-08-11) | |
| 75 | +- LLM Quantization Formats Compared: GGUF vs MLX vs EXL3 vs GPTQ vs AWQ vs FP8 (D-Central; format inventories) — https://d-central.tech/llm-quantization-formats (accessed 2026-08-11) | |
| 76 | +- GGUF vs MLX Quantization Formats on Apple Silicon (Contra Collective, 2026) — https://contracollective.com/blog/gguf-vs-mlx-quantization-formats-apple-silicon-2026 (accessed 2026-08-11) | |
| 77 | +- llama.cpp Metal Backend vs MLX: Compute Path Comparison (Contra Collective, 2026) — https://contracollective.com/blog/llama-cpp-metal-vs-mlx-backend-apple-silicon-2026 (accessed 2026-08-11) | |
| 78 | +- llama.cpp supports gpt-oss in native MXFP4 (discussion #15095) — https://github.com/ggml-org/llama.cpp/discussions/15095 (accessed 2026-08-11) | |
| 79 | +- exllamav3 / EXL3 trellis format (turboderp) — https://github.com/turboderp-org/exllamav3 (accessed 2026-08-11) | |
| 80 | +- KV Cache and Context Length on Apple Silicon (Contra Collective, 2026; llama.cpp/mlx-lm KV flags) — https://contracollective.com/blog/kv-cache-context-length-apple-silicon-local-inference-2026 (accessed 2026-08-11) | |
| 81 | +- Running LLMs locally on a Mac (MacKinlay; KV cache quant flags across runtimes) — https://danmackinlay.name/notebook/local_llm_mac.html (accessed 2026-08-11) | |
| 82 | +- KVSplit: differentiated K/V precision on Apple Silicon (Show HN) — https://news.ycombinator.com/item?id=44009321 (accessed 2026-08-11) | |
| 83 | +- TurboQuant — Extreme KV Cache Quantization with Metal kernels (llama.cpp discussion #20969) — https://github.com/ggml-org/llama.cpp/discussions/20969 (accessed 2026-08-11) | |
| 84 | +- M2Cache: Harnessing Your DRAM and SSD for Sustainable and Accessible LLM Inference with Mixed-Precision and Multi-level Caching — https://arxiv.org/abs/2410.14740 (accessed 2026-08-11) | |
| 85 | +- SSD Offloading for LLM MoE Weights Considered Harmful in Energy Efficiency — https://www.alphaxiv.org/overview/2508.06978 (accessed 2026-08-11) | |
| 86 | +- Agent Memory Below the Prompt: Persistent Q4 KV Cache for Multi-Agent LLM Inference on Edge Devices (MLX Q4-KV state of play) — https://arxiv.org/html/2603.04428v1 (accessed 2026-08-11) | |
| 87 | +- MLX vs llama.cpp on Apple Silicon: Benchmarks, M5 Neural Accelerators, Ollama switch — https://yage.ai/share/mlx-apple-silicon-en-20260331.html (accessed 2026-08-11) | |
| 88 | + | |
| 89 | +## Activation & weight sparsity, pruning (§4.2–4.3) | |
| 90 | + | |
| 91 | +From `research/notes/sparsity_pruning.md` (48 sources). | |
| 92 | + | |
| 93 | +- Deja Vu: Contextual Sparsity for Efficient LLMs at Inference Time — https://arxiv.org/abs/2310.17157 (accessed 2026-08-11) | |
| 94 | +- Deja Vu (OpenReview, ICML 2023) — https://openreview.net/forum?id=wIPIhHd00i (accessed 2026-08-11) | |
| 95 | +- PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU — https://arxiv.org/abs/2312.12456 (accessed 2026-08-11) | |
| 96 | +- PowerInfer (SOSP 2024 paper PDF, IPADS/SJTU) — https://ipads.se.sjtu.edu.cn/_media/publications/song-sosp24.pdf (accessed 2026-08-11) | |
| 97 | +- PowerInfer GitHub (macOS/Metal support status, supported ReLU models) — https://github.com/SJTU-IPADS/PowerInfer (accessed 2026-08-11) | |
| 98 | +- PowerInfer-2: Fast Large Language Model Inference on a Smartphone — https://arxiv.org/abs/2406.06282 (accessed 2026-08-11) | |
| 99 | +- PowerInfer-2 project page — https://powerinfer.ai/v2/ (accessed 2026-08-11) | |
| 100 | +- LLM in a flash: Efficient Large Language Model Inference with Limited Memory (Apple) — https://arxiv.org/abs/2312.11514 (accessed 2026-08-11) | |
| 101 | +- ReLU Strikes Back: Exploiting Activation Sparsity in Large Language Models (Apple, ICLR 2024) — https://arxiv.org/abs/2310.04564 (accessed 2026-08-11) | |
| 102 | +- ReLU Strikes Back — Apple Machine Learning Research page — https://machinelearning.apple.com/research/relu (accessed 2026-08-11) | |
| 103 | +- The Lazy Neuron Phenomenon: On Emergence of Activation Sparsity in Transformers — https://arxiv.org/abs/2210.06313 (accessed 2026-08-11) | |
| 104 | +- TEAL: Training-Free Activation Sparsity in Large Language Models — https://arxiv.org/abs/2408.14690 (accessed 2026-08-11) | |
| 105 | +- TEAL — Together AI blog — https://www.together.ai/blog/teal-training-free-activation-sparsity-in-large-language-models (accessed 2026-08-11) | |
| 106 | +- CATS: Contextually-Aware Thresholding for Sparsity in Large Language Models (COLM 2024) — https://arxiv.org/abs/2404.08763 (accessed 2026-08-11) | |
| 107 | +- CATS GitHub — https://github.com/ScalingIntelligence/CATS (accessed 2026-08-11) | |
| 108 | +- GRIFFIN: Prompt-prompted Adaptive Structured Pruning for Efficient LLM Generation (ICML 2024) — https://arxiv.org/abs/2404.01365 (accessed 2026-08-11) | |
| 109 | +- GRIFFIN GitHub — https://github.com/hdong920/GRIFFIN (accessed 2026-08-11) | |
| 110 | +- ShadowLLM: Predictor-based Contextual Sparsity for Large Language Models (EMNLP 2024) — https://arxiv.org/abs/2406.16635 (accessed 2026-08-11) | |
| 111 | +- ShadowLLM — ACL Anthology — https://aclanthology.org/2024.emnlp-main.1068/ (accessed 2026-08-11) | |
| 112 | +- ProSparse: Introducing and Enhancing Intrinsic Activation Sparsity within Large Language Models — https://arxiv.org/abs/2402.13516 (accessed 2026-08-11) | |
| 113 | +- Turbo Sparse: Achieving LLM SOTA Performance with Minimal Activated Parameters — https://arxiv.org/abs/2406.05955 (accessed 2026-08-11) | |
| 114 | +- Q-Sparse: All Large Language Models can be Fully Sparsely-Activated (NeurIPS 2024) — https://arxiv.org/abs/2407.10969 (accessed 2026-08-11) | |
| 115 | +- Sirius: Contextual Sparsity with Correction for Efficient LLMs (NeurIPS 2024) — https://arxiv.org/abs/2409.03856 (accessed 2026-08-11) | |
| 116 | +- SparQ Attention: Bandwidth-Efficient LLM Inference (ICML 2024) — https://arxiv.org/abs/2312.04985 (accessed 2026-08-11) | |
| 117 | +- SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot (ICML 2023) — https://arxiv.org/abs/2301.00774 (accessed 2026-08-11) | |
| 118 | +- Wanda: A Simple and Effective Pruning Approach for Large Language Models (ICLR 2024) — https://arxiv.org/abs/2306.11695 (accessed 2026-08-11) | |
| 119 | +- ShortGPT: Layers in Large Language Models are More Redundant Than You Expect — https://arxiv.org/abs/2403.03853 (accessed 2026-08-11) | |
| 120 | +- Sheared LLaMA: Accelerating Language Model Pre-training via Structured Pruning (ICLR 2024) — https://arxiv.org/abs/2310.06694 (accessed 2026-08-11) | |
| 121 | +- Compact Language Models via Pruning and Knowledge Distillation (Minitron, NVIDIA) — https://arxiv.org/abs/2407.14679 (accessed 2026-08-11) | |
| 122 | +- LLM Pruning and Distillation in Practice: The Minitron Approach — https://arxiv.org/pdf/2408.11796 (accessed 2026-08-11) | |
| 123 | +- SliceGPT: Compress Large Language Models by Deleting Rows and Columns (ICLR 2024) — https://arxiv.org/abs/2401.15024 (accessed 2026-08-11) | |
| 124 | +- LLM-Pruner: On the Structural Pruning of Large Language Models (NeurIPS 2023) — https://arxiv.org/abs/2305.11627 (accessed 2026-08-11) | |
| 125 | +- M2Cache: Harnessing Your DRAM and SSD for Sustainable and Accessible LLM Inference with Mixed-Precision and Multi-level Caching — https://arxiv.org/abs/2410.14740 (accessed 2026-08-11) | |
| 126 | +- Ripple/Neuralink: Accelerating LLM Inference on Smartphones with Correlation-Aware Neuron Management / Neuron Co-Activation Linking — https://arxiv.org/abs/2410.19274 (accessed 2026-08-11) | |
| 127 | +- DIP: Efficient LLM Inference using Dynamic Input Pruning and Cache-Aware Masking (Qualcomm AI Research) — https://arxiv.org/abs/2412.01380 (accessed 2026-08-11) | |
| 128 | +- Endor: Hardware-Friendly Sparse Format for Offloaded LLM Inference — https://arxiv.org/pdf/2406.11674 (accessed 2026-08-11) | |
| 129 | +- SparseInfer: Training-free Prediction of Activation Sparsity for Fast LLM Inference — https://arxiv.org/pdf/2411.12692 (accessed 2026-08-11) | |
| 130 | +- Post-Training Statistical Calibration for Higher Activation Sparsity — https://arxiv.org/pdf/2412.07174 (accessed 2026-08-11) | |
| 131 | +- R-Sparse: Rank-Aware Activation Sparsity for Efficient LLM Inference — https://arxiv.org/abs/2504.19449 (accessed 2026-08-11) | |
| 132 | +- Spark Transformer: Reactivating Sparsity in FFN and Attention (NeurIPS 2025) — https://arxiv.org/html/2506.06644v2 (accessed 2026-08-11) | |
| 133 | +- Universal Properties of Activation Sparsity in Modern Large Language Models — https://arxiv.org/abs/2509.00454 (accessed 2026-08-11) | |
| 134 | +- RAP: Runtime Adaptive Pruning for LLM Inference — https://arxiv.org/pdf/2505.17138 (accessed 2026-08-11) | |
| 135 | +- DuoGPT: Training-free Dual Sparsity through Activation-aware Pruning in LLMs — https://arxiv.org/html/2506.20194 (accessed 2026-08-11) | |
| 136 | +- Motivating Next-Gen Accelerators with Flexible (N:M) Activation Sparsity — https://arxiv.org/pdf/2509.22166 (accessed 2026-08-11) | |
| 137 | +- VLM in a flash: I/O-Efficient Sparsification of Vision-Language Model via Neuron Chunking — https://arxiv.org/html/2511.18692 (accessed 2026-08-11) | |
| 138 | +- On-Demand Multi-Task Sparsity for Efficient Large-Model Deployment on Edge Devices — https://arxiv.org/pdf/2511.19986 (accessed 2026-08-11) | |
| 139 | +- Fast Forward: Accelerating LLM Prefill with Predictive FFN Sparsity — https://arxiv.org/pdf/2602.00397 (accessed 2026-08-11) | |
| 140 | +- Dynamic sparsity in tree-structured feed-forward layers at scale — https://arxiv.org/pdf/2604.08565 (accessed 2026-08-11) | |
| 141 | + | |
| 142 | +## Out-of-core inference & memory systems (§4.4, §4.8) | |
| 143 | + | |
| 144 | +From `research/notes/out_of_core_memory_systems.md` (59 sources). | |
| 145 | + | |
| 146 | +- FlexGen: High-Throughput Generative Inference of Large Language Models with a Single GPU — https://arxiv.org/abs/2303.06865 (accessed 2026-08-11) | |
| 147 | +- FlexLLMGen (FlexGen) README, FMInference — https://github.com/FMInference/FlexLLMGen/blob/main/README.md (accessed 2026-08-11) | |
| 148 | +- ZeRO-Inference: Democratizing massive model inference — https://www.deepspeed.ai/2022/09/09/zero-inference.html (accessed 2026-08-11) | |
| 149 | +- DeepSpeed Inference: Enabling Efficient Inference of Transformer Models at Unprecedented Scale — https://arxiv.org/pdf/2207.00032 (accessed 2026-08-11) | |
| 150 | +- DeepNVMe: Affordable I/O scaling for Deep Learning Applications (PyTorch blog) — https://pytorch.org/blog/deepnvme-affordable-i-o-scaling-for-deep-learning-applications/ (accessed 2026-08-11) | |
| 151 | +- LLM in a flash: Efficient Large Language Model Inference with Limited Memory — https://arxiv.org/abs/2312.11514 (accessed 2026-08-11) | |
| 152 | +- LLM in a flash (HTML full text, hardware/throughput details) — https://arxiv.org/html/2312.11514v3 (accessed 2026-08-11) | |
| 153 | +- PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU — https://arxiv.org/abs/2312.12456 (accessed 2026-08-11) | |
| 154 | +- PowerInfer (SOSP '24 proceedings) — https://dl.acm.org/doi/10.1145/3694715.3695964 (accessed 2026-08-11) | |
| 155 | +- PowerInfer-2: Fast Large Language Model Inference on a Smartphone — https://arxiv.org/abs/2406.06282 (accessed 2026-08-11) | |
| 156 | +- PowerInfer-2 project page — https://powerinfer.ai/v2/ (accessed 2026-08-11) | |
| 157 | +- M2Cache: Harnessing Your DRAM and SSD for Sustainable and Accessible LLM Inference with Mixed-Precision and Multi-level Caching — https://arxiv.org/abs/2410.14740 (accessed 2026-08-11) | |
| 158 | +- SolidAttention: Low-Latency SSD-based Serving on Memory-Constrained PCs (FAST '26) — https://www.usenix.org/system/files/fast26-zheng.pdf (accessed 2026-08-11) | |
| 159 | +- InstInfer: In-Storage Attention Offloading for Cost-Effective Long-Context LLM Inference — https://arxiv.org/pdf/2409.04992 (accessed 2026-08-11) | |
| 160 | +- Swarm: Co-Activation Aware KVCache Offloading Across Multiple SSDs — https://arxiv.org/html/2603.17803v1 (accessed 2026-08-11) | |
| 161 | +- FlexInfer: Breaking Memory Constraint via Flexible and Efficient Offloading for On-Device LLM Inference — https://arxiv.org/abs/2503.03777 (accessed 2026-08-11) | |
| 162 | +- Glinthawk: A Two-Tiered Architecture for Offline LLM Inference — https://arxiv.org/pdf/2501.11779 (accessed 2026-08-11) | |
| 163 | +- Fast Inference of Mixture-of-Experts Language Models with Offloading (Eliseev & Mazur) — https://arxiv.org/pdf/2312.17238 (accessed 2026-08-11) | |
| 164 | +- In-Depth Analysis on Caching and Pre-Fetching in Mixture of Experts Offloading — https://arxiv.org/pdf/2511.05814 (accessed 2026-08-11) | |
| 165 | +- Mixture of Cache-Conditional Experts for Efficient Mobile Device Inference — https://arxiv.org/pdf/2412.00099 (accessed 2026-08-11) | |
| 166 | +- MoBiLE: Efficient Mixture-of-Experts Inference on Consumer GPU with Mixture of Big Little Experts — https://arxiv.org/pdf/2510.12357 (accessed 2026-08-11) | |
| 167 | +- Petals: Run LLMs at home, BitTorrent-style — https://github.com/bigscience-workshop/petals (accessed 2026-08-11) | |
| 168 | +- Petals project page — https://petals.dev/ (accessed 2026-08-11) | |
| 169 | +- AirLLM and "70B on a 4GB GPU" — What's Actually Going On? — https://rohit-shirke.medium.com/airllm-and-70b-on-a-4gb-gpu-whats-actually-going-on-3bf0e102252e (accessed 2026-08-11) | |
| 170 | +- llama.cpp: Should use mmap for model loading (issue #91) — https://github.com/ggml-org/llama.cpp/issues/91 (accessed 2026-08-11) | |
| 171 | +- llama.cpp: Memory-mapping weights while loading the model (discussion #9999) — https://github.com/ggml-org/llama.cpp/discussions/9999 (accessed 2026-08-11) | |
| 172 | +- llama.cpp: Mmap faster than direct I/O for MoE models (discussion #18758, incl. M5 Pro/AP1024Z expert-layout measurements) — https://github.com/ggml-org/llama.cpp/discussions/18758 (accessed 2026-08-11) | |
| 173 | +- llama.cpp: Share readonly GPU model weights across processes — Metal reads mmap buffers via MTLResourceStorageModeShared (discussion #21223) — https://github.com/ggml-org/llama.cpp/discussions/21223 (accessed 2026-08-11) | |
| 174 | +- llama.cpp: Two-tier GPU+RAM expert cache for MoE offload, pluggable eviction (issue #20757) — https://github.com/ggml-org/llama.cpp/issues/20757 (accessed 2026-08-11) | |
| 175 | +- llama.cpp: Avoid memcpy for mmap-ed weights on Unified Memory architectures (issue #21827) — https://github.com/ggml-org/llama.cpp/issues/21827 (accessed 2026-08-11) | |
| 176 | +- Performant local mixture-of-experts CPU inference with GPU acceleration in llama.cpp (HF blog) — https://huggingface.co/blog/Doctor-Shotgun/llamacpp-moe-offload-guide (accessed 2026-08-11) | |
| 177 | +- MLX Unified Memory documentation — https://ml-explore.github.io/mlx/build/html/usage/unified_memory.html (accessed 2026-08-11) | |
| 178 | +- MLX: Loading models with mmap (discussion #615, incl. 70GB-on-64GB 0.025 tok/s prototype result) — https://github.com/ml-explore/mlx/discussions/615 (accessed 2026-08-11) | |
| 179 | +- mlx-swift wired-memory documentation (residency/wired limit) — https://github.com/ml-explore/mlx-swift/blob/main/Source/MLX/Documentation.docc/Articles/wired-memory.md (accessed 2026-08-11) | |
| 180 | +- mlx-lm: mlx_lm.server causes macOS kernel panic (IOGPUMemory) via unbounded wired growth (issue #883) — https://github.com/ml-explore/mlx-lm/issues/883 (accessed 2026-08-11) | |
| 181 | +- fcntl F_NOCACHE option behavior (Apple Developer Forums thread 25464) — https://developer.apple.com/forums/thread/25464 (accessed 2026-08-11) | |
| 182 | +- OSX fcntl(fd, F_NOCACHE, 1) not equivalent to O_DIRECT on Linux (fio issue #48) — https://github.com/axboe/fio/issues/48 (accessed 2026-08-11) | |
| 183 | +- ronomon/direct-io: Direct IO helpers for FreeBSD, Linux, macOS, Windows (F_NOCACHE alignment notes) — https://github.com/ronomon/direct-io (accessed 2026-08-11) | |
| 184 | +- makeBuffer(bytesNoCopy:length:options:deallocator:) — Apple Developer Documentation — https://developer.apple.com/documentation/metal/mtldevice/makebuffer(bytesnocopy:length:options:deallocator:) (accessed 2026-08-11) | |
| 185 | +- MTLStorageMode.shared — Apple Developer Documentation — https://developer.apple.com/documentation/metal/mtlstoragemode/shared (accessed 2026-08-11) | |
| 186 | +- MTLHeap (incl. setPurgeableState) — Apple Developer Documentation — https://developer.apple.com/documentation/metal/mtlheap (accessed 2026-08-11) | |
| 187 | +- newBufferWithBytesNoCopy pointer alignment requirement (Apple Developer Forums thread 8011) — https://developer.apple.com/forums/thread/8011 (accessed 2026-08-11) | |
| 188 | +- iOS/macOS writeback behavior for mmap(MAP_SHARED) dirty pages (Apple Developer Forums thread 763058) — https://developer.apple.com/forums/thread/763058 (accessed 2026-08-11) | |
| 189 | +- How to Increase VRAM Allocation on Apple Silicon Mac (iogpu.wired_limit_mb) — https://osxdaily.com/2025/05/07/how-to-increase-vram-allocation-on-apple-silicon-mac/ (accessed 2026-08-11) | |
| 190 | +- Adjust wired limits to allocate more memory to the GPU with Apple Silicon (gist) — https://gist.github.com/havenwood/f2f5c49c2c90c6787ae2295e9805adbe (accessed 2026-08-11) | |
| 191 | +- Disk speed testing on Apple Silicon: AmorphousDiskMark, Blackmagic, etc. (MacRumors, 4K QD1 results) — https://forums.macrumors.com/threads/disk-speed-testing-on-apple-silicon-amorphousdiskmark-blackmagic-etc-merged.2378298/ (accessed 2026-08-11) | |
| 192 | +- M1 Pro SSD speeds (MacRumors, 4K QD1 ~32 MB/s report) — https://forums.macrumors.com/threads/m1-pro-ssd-speeds.2319853/ (accessed 2026-08-11) | |
| 193 | +- MacBook Pro (16-inch, M5 Pro or M5 Max) — Tech Specs (memory bandwidth) — https://support.apple.com/en-us/126319 (accessed 2026-08-11) | |
| 194 | +- Unified Buffer Cache (UBC) — Mac OS X Internals: A Systems Approach (excerpt) — https://flylib.com/books/en/3.126.1.93/1/ (accessed 2026-08-11) | |
| 195 | +- Apple XNU WKdm fast memory page compressor (source mirror) — https://github.com/berkus/wkdm (accessed 2026-08-11) | |
| 196 | +- Virtual memory compression (WKdm background) — https://en.wikipedia.org/wiki/Virtual_memory_compression (accessed 2026-08-11) | |
| 197 | +- The working set model for program behavior (Denning, 1968; publications index) — http://denninginstitute.com/pjd/PUBS/Workingsets.html (accessed 2026-08-11) | |
| 198 | +- Working Set Analytics (Denning, ACM Computing Surveys) — https://dl.acm.org/doi/10.1145/3399709 (accessed 2026-08-11) | |
| 199 | +- ARC: A Self-Tuning, Low Overhead Replacement Cache (Megiddo & Modha, FAST '03) — https://www.usenix.org/legacy/events/fast03/tech/full_papers/megiddo/megiddo.pdf (accessed 2026-08-11) | |
| 200 | +- An Evaluation of Buffer Management Strategies for Relational Database Systems (Chou & DeWitt, VLDB '85 — DBMIN/QLSM) — https://www.cs.cmu.edu/~natassa/courses/15-721/papers/P127.PDF (accessed 2026-08-11) | |
| 201 | +- Anti-Caching: A New Approach to Database Management System Architecture (DeBrabant et al., VLDB 2013) — https://www.vldb.org/pvldb/vol6/p1942-debrabant.pdf (accessed 2026-08-11) | |
| 202 | +- TPP: Transparent Page Placement for CXL-Enabled Tiered-Memory (ASPLOS '23) — https://arxiv.org/abs/2206.02878 (accessed 2026-08-11) | |
| 203 | +- Pythia: A Customizable Hardware Prefetching Framework Using Online Reinforcement Learning (MICRO 2021) — https://arxiv.org/pdf/2109.12021 (accessed 2026-08-11) | |
| 204 | +- Evolution of Buffer Management in Database Systems: From Classical Algorithms to Machine Learning and Disaggregated Memory (survey) — https://arxiv.org/pdf/2512.22995 (accessed 2026-08-11) | |
| 205 | + | |
| 206 | +## Decomposition & progressive computation (§4.5–4.6) | |
| 207 | + | |
| 208 | +From `research/notes/decomposition_progressive.md` (62 sources). | |
| 209 | + | |
| 210 | +- SVD-LLM: Truncation-aware Singular Value Decomposition for Large Language Model Compression (ICLR 2025) — https://arxiv.org/html/2403.07378v3 (accessed 2026-08-11) | |
| 211 | +- SVD-LLM (ICLR 2025 proceedings abstract) — https://proceedings.iclr.cc/paper_files/paper/2025/hash/3104e1ab39875cf54fe1eb4473e7c5a1-Abstract-Conference.html (accessed 2026-08-11) | |
| 212 | +- SVD-LLM GitHub (AIoT-MLSys-Lab) — https://github.com/AIoT-MLSys-Lab/SVD-LLM (accessed 2026-08-11) | |
| 213 | +- ASVD: Activation-aware Singular Value Decomposition for Compressing LLMs — https://arxiv.org/abs/2312.05821 (accessed 2026-08-11) | |
| 214 | +- Language model compression with weighted low-rank factorization (FWSVD, ICLR 2022) — https://arxiv.org/abs/2207.00112 (accessed 2026-08-11) | |
| 215 | +- The Truth is in There: Improving Reasoning in Language Models with Layer-Selective Rank Reduction (LASER, ICLR 2024) — https://arxiv.org/abs/2312.13558 (accessed 2026-08-11) | |
| 216 | +- LASER project page — https://pratyushasharma.github.io/laser (accessed 2026-08-11) | |
| 217 | +- SliceGPT: Compress Large Language Models by Deleting Rows and Columns (ICLR 2024) — https://arxiv.org/abs/2401.15024 (accessed 2026-08-11) | |
| 218 | +- Compressing Large Language Models using Low Rank and Low Precision Decomposition (CALDERA, NeurIPS 2024) — https://arxiv.org/abs/2405.18886 (accessed 2026-08-11) | |
| 219 | +- CALDERA GitHub (pilancilab) — https://github.com/pilancilab/caldera (accessed 2026-08-11) | |
| 220 | +- Matrix Compression via Randomized Low Rank and Low Precision Factorization (NeurIPS 2023) — https://neurips.cc/virtual/2023/poster/70291 (accessed 2026-08-11) | |
| 221 | +- Extreme Compression of Large Language Models via Additive Quantization (AQLM) — https://arxiv.org/html/2401.06118v2 (accessed 2026-08-11) | |
| 222 | +- AQLM GitHub (incl. ~1-bit Llama-2-7B result) — https://github.com/vahe1994/AQLM (accessed 2026-08-11) | |
| 223 | +- QuIP#: Even Better LLM Quantization with Hadamard Incoherence and Lattice Codebooks (ICML 2024) — https://proceedings.mlr.press/v235/tseng24a.html (accessed 2026-08-11) | |
| 224 | +- GPTVQ: The Blessing of Dimensionality for LLM Quantization — https://arxiv.org/abs/2402.15319 (accessed 2026-08-11) | |
| 225 | +- VPTQ: Extreme Low-bit Vector Post-Training Quantization for LLMs (Microsoft Research) — https://www.microsoft.com/en-us/research/publication/vptq-extreme-low-bit-vector-post-training-quantization-for-large-language-models (accessed 2026-08-11) | |
| 226 | +- Relaxed Recursive Transformers: Effective Parameter Sharing with Layer-wise LoRA (ICLR 2025) — https://arxiv.org/html/2410.20672v1 (accessed 2026-08-11) | |
| 227 | +- Subformer: Exploring Weight Sharing for Parameter Efficiency (Findings of EMNLP 2021) — https://aclanthology.org/2021.findings-emnlp.344.pdf (accessed 2026-08-11) | |
| 228 | +- Basis Sharing: Cross-Layer Parameter Sharing for LLM Compression (ICLR 2025) — https://arxiv.org/abs/2410.03765 (accessed 2026-08-11) | |
| 229 | +- Basis Sharing (ICLR 2025 proceedings PDF) — https://proceedings.iclr.cc/paper_files/paper/2025/file/238c98450b1d9e8055f94d22f303bb57-Paper-Conference.pdf (accessed 2026-08-11) | |
| 230 | +- DeltaLLM: Compress LLMs with Low-Rank Deltas between Shared Weights — https://arxiv.org/abs/2501.18596 (accessed 2026-08-11) | |
| 231 | +- ResidualTransformer: Residual Low-Rank Learning with Weight-Sharing for Transformer Layers (ICASSP 2024) — https://arxiv.org/abs/2310.02489 (accessed 2026-08-11) | |
| 232 | +- BitDelta: Your Fine-Tune May Only Be Worth One Bit (NeurIPS 2024) — https://arxiv.org/html/2402.10193v3 (accessed 2026-08-11) | |
| 233 | +- BitDelta NeurIPS poster page — https://neurips.cc/virtual/2024/poster/94736 (accessed 2026-08-11) | |
| 234 | +- DeltaZip: Compression for Foundation Models (EuroSys 2025; repo lists delta-compression literature) — https://github.com/eth-easl/deltazip (accessed 2026-08-11) | |
| 235 | +- Delta-CoMe: Training-Free Delta-Compression with Mixed-Precision for LLMs (NeurIPS 2024) — https://arxiv.org/abs/2406.08903 (accessed 2026-08-11) | |
| 236 | +- Kronecker Decomposition for GPT Compression (KnGPT2, ACL 2022) — https://aclanthology.org/2022.acl-short.24.pdf (accessed 2026-08-11) | |
| 237 | +- TensorGPT: Efficient Compression of LLMs based on Tensor-Train Decomposition — https://arxiv.org/html/2307.00526v2 (accessed 2026-08-11) | |
| 238 | +- BitStack: Any-Size Compression of Large Language Models in Variable Memory Environments (ICLR 2025) — https://arxiv.org/abs/2410.23918 (accessed 2026-08-11) | |
| 239 | +- Any-Precision LLM: Low-Cost Deployment of Multiple, Different-Sized LLMs (ICML 2024 oral) — https://arxiv.org/pdf/2402.10517 (accessed 2026-08-11) | |
| 240 | +- Any-Precision LLM GitHub (SNU-ARC) — https://github.com/SNU-ARC/any-precision-llm (accessed 2026-08-11) | |
| 241 | +- Matryoshka Quantization (MatQuant, ICLR 2025 oral) — https://openreview.net/forum?id=phVWcUSGYP (accessed 2026-08-11) | |
| 242 | +- Recurrent Residual Quantization: A Progressive Multi-Precision Representation for LLMs — https://arxiv.org/abs/2608.04048 (accessed 2026-08-11) | |
| 243 | +- Multi-Bitwidth Quantization for LLMs Using Additive Codebooks (Drop-by-Drop) — https://arxiv.org/html/2606.12876v1 (accessed 2026-08-11) | |
| 244 | +- Intrinsic Dimensionality Explains the Effectiveness of Language Model Fine-Tuning — https://arxiv.org/abs/2012.13255 (accessed 2026-08-11) | |
| 245 | +- ShortGPT: Layers in Large Language Models are More Redundant Than You Expect — https://arxiv.org/html/2403.03853v1 (accessed 2026-08-11) | |
| 246 | +- The Unreasonable Ineffectiveness of the Deeper Layers (ICLR 2025) — https://arxiv.org/abs/2403.17887 (accessed 2026-08-11) | |
| 247 | +- Your Transformer is Secretly Linear (ACL 2024) — https://arxiv.org/abs/2405.12250 (accessed 2026-08-11) | |
| 248 | +- Confident Adaptive Language Modeling (CALM, NeurIPS 2022) — https://proceedings.neurips.cc/paper_files/paper/2022/hash/6fac9e316a4ae75ea244ddcef1982c71-Abstract-Conference.html (accessed 2026-08-11) | |
| 249 | +- Google Research blog: Accelerating text generation with CALM — https://research.google/blog/accelerating-text-generation-with-confident-adaptive-language-modeling-calm (accessed 2026-08-11) | |
| 250 | +- LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding — https://arxiv.org/html/2404.16710v1 (accessed 2026-08-11) | |
| 251 | +- Depth-Adaptive Transformer (ICLR 2020) — https://arxiv.org/abs/1910.10073 (accessed 2026-08-11) | |
| 252 | +- Adaptive Computation Time for Recurrent Neural Networks (Graves 2016) — https://arxiv.org/abs/1603.08983 (accessed 2026-08-11) | |
| 253 | +- PonderNet: Learning to Ponder — https://arxiv.org/abs/2107.05407 (accessed 2026-08-11) | |
| 254 | +- Multi-Scale Dense Networks for Resource Efficient Image Classification (MSDNet) — https://arxiv.org/abs/1703.09844 (accessed 2026-08-11) | |
| 255 | +- Mixture-of-Depths: Dynamically allocating compute in transformer-based language models — https://arxiv.org/abs/2404.02258 (accessed 2026-08-11) | |
| 256 | +- Multiplying Matrices Without Multiplying (MADDNESS, ICML 2021) — https://proceedings.mlr.press/v139/blalock21a/blalock21a.pdf (accessed 2026-08-11) | |
| 257 | +- Fast Monte Carlo Algorithms for Matrices I: Approximating Matrix Multiplication (Drineas, Kannan, Mahoney, SIAM J. Comput. 2006) — https://epubs.siam.org/doi/10.1137/S0097539704442684 (accessed 2026-08-11) | |
| 258 | +- Accelerating the Solution of Linear Systems by Iterative Refinement in Three Precisions (Carson & Higham, SIAM SISC 2018) — https://epubs.siam.org/doi/10.1137/17M1140819 (accessed 2026-08-11) | |
| 259 | +- Five-precision GMRES-based Iterative Refinement (Amestoy et al.) — https://eprints.maths.manchester.ac.uk/2852/1/paper.pdf (accessed 2026-08-11) | |
| 260 | +- What Is Iterative Refinement? (Nick Higham) — https://nhigham.com/2023/03/13/what-is-iterative-refinement (accessed 2026-08-11) | |
| 261 | +- zfp Compression Ratio and Quality (LLNL) — https://computing.llnl.gov/projects/zfp/zfp-compression-ratio-and-quality (accessed 2026-08-11) | |
| 262 | +- Error Analysis of ZFP Compression for Floating-Point Data (SIAM) — https://epubs.siam.org/doi/10.1137/18M1168832 (accessed 2026-08-11) | |
| 263 | +- Fast Error-bounded Lossy HPC Data Compression with SZ (Di & Cappello, IPDPS 2016) — https://www.mcs.anl.gov/papers/P5437-1115.pdf (accessed 2026-08-11) | |
| 264 | +- Embedded zerotrees of wavelet transforms (EZW) — https://en.wikipedia.org/wiki/Embedded_zerotrees_of_wavelet_transforms (accessed 2026-08-11) | |
| 265 | +- Wavelet and image compression: EZW / SPIHT / JPEG2000-EBCOT lecture notes (Cagnazzo, Télécom Paris) — https://perso.telecom-paristech.fr/tupin/ATHENS/COURSES/wavelet_athens_2012.pdf (accessed 2026-08-11) | |
| 266 | +- Progressive Meshes (Hoppe, SIGGRAPH 1996) — https://www.cs.jhu.edu/~misha/ReadingSeminar/Papers/Hoppe96.pdf (accessed 2026-08-11) | |
| 267 | +- Nanite Virtualized Geometry (Unreal Engine documentation) — https://dev.epicgames.com/documentation/unreal-engine/nanite-virtualized-geometry-in-unreal-engine?lang=en-US (accessed 2026-08-11) | |
| 268 | +- BlinkDB: Queries with Bounded Errors and Bounded Response Times on Very Large Data (EuroSys 2013) — https://dl.acm.org/doi/10.1145/2465351.2465355 (accessed 2026-08-11) | |
| 269 | +- Readings in Database Systems (Red Book) ch. 8: Interactive Analytics — online aggregation & AQP context — http://www.redbook.io/ch8-interactive.html (accessed 2026-08-11) | |
| 270 | +- Unweight: how we compressed an LLM 22% without sacrificing quality (Cloudflare engineering, bandwidth-bound inference evidence) — https://blog.cloudflare.com/unweight-tensor-compression (accessed 2026-08-11) | |
| 271 | +- mlx.core.quantize documentation (supported modes, group sizes, bit widths) — https://ml-explore.github.io/mlx/build/html/python/_autosummary/mlx.core.quantize.html (accessed 2026-08-11) | |
| 272 | + | |
| 273 | +## Speculation, error analysis, decision stability (§4.7, §4.9–4.10) | |
| 274 | + | |
| 275 | +From `research/notes/speculation_error_stability.md` (66 sources). | |
| 276 | + | |
| 277 | +- Fast Inference from Transformers via Speculative Decoding (Leviathan, Kalman, Matias; ICML 2023) — https://arxiv.org/abs/2211.17192 (accessed 2026-08-11) | |
| 278 | +- Accelerating Large Language Model Decoding with Speculative Sampling (Chen et al., DeepMind) — https://arxiv.org/abs/2302.01318 (accessed 2026-08-11) | |
| 279 | +- Looking back at speculative decoding (Google Research blog) — https://research.google/blog/looking-back-at-speculative-decoding (accessed 2026-08-11) | |
| 280 | +- Speculative decoding — Wikipedia — https://en.wikipedia.org/wiki/Speculative_decoding (accessed 2026-08-11) | |
| 281 | +- Speculative Decoding: Exploiting Speculative Execution for Accelerating Seq2seq Generation (Xia et al., EMNLP 2023 Findings) — https://aclanthology.org/2023.findings-emnlp.257.pdf (accessed 2026-08-11) | |
| 282 | +- Beyond the Speculative Game: A Survey of Speculative Execution in Large Language Models — https://arxiv.org/html/2404.14897v1 (accessed 2026-08-11) | |
| 283 | +- Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads (Cai et al.) — https://arxiv.org/abs/2401.10774 (accessed 2026-08-11) | |
| 284 | +- EAGLE-3: Scaling up Inference Acceleration of Large Language Models via Training-Time Test — https://arxiv.org/html/2503.01840v1 (accessed 2026-08-11) | |
| 285 | +- EAGLE-3 (NeurIPS 2025 poster) — https://neurips.cc/virtual/2025/poster/119930 (accessed 2026-08-11) | |
| 286 | +- Get 3× Faster LLM Inference with Speculative Decoding (BentoML; real-world EAGLE-3 acceptance rates) — https://www.bentoml.com/blog/3x-faster-llm-inference-with-speculative-decoding (accessed 2026-08-11) | |
| 287 | +- Break the Sequential Dependency of LLM Inference Using Lookahead Decoding (Fu et al., ICML 2024) — https://arxiv.org/html/2402.02057v1 (accessed 2026-08-11) | |
| 288 | +- Lookahead decoding blog (LMSYS) — https://www.lmsys.org/blog/2023-11-21-lookahead-decoding (accessed 2026-08-11) | |
| 289 | +- Draft & Verify: Lossless Large Language Model Acceleration via Self-Speculative Decoding (Zhang et al.) — https://arxiv.org/abs/2309.08168 (accessed 2026-08-11) | |
| 290 | +- LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding (Elhoushi et al., ACL 2024) — https://arxiv.org/html/2404.16710v1 (accessed 2026-08-11) | |
| 291 | +- Faster Text Generation with Self-Speculative Decoding (Hugging Face LayerSkip blog) — https://huggingface.co/blog/layerskip (accessed 2026-08-11) | |
| 292 | +- Kangaroo: Lossless Self-Speculative Decoding via Double Early Exiting (NeurIPS 2024) — https://neurips.cc/virtual/2024/poster/93829 (accessed 2026-08-11) | |
| 293 | +- SWIFT: On-the-Fly Self-Speculative Decoding for LLM Inference Acceleration (ICLR 2025) — https://arxiv.org/pdf/2410.06916 (accessed 2026-08-11) | |
| 294 | +- CLaSp: In-Context Layer Skip for Self-Speculative Decoding — https://arxiv.org/html/2505.24196v1 (accessed 2026-08-11) | |
| 295 | +- QSpec: Speculative Decoding with Complementary Quantization Schemes (EMNLP 2025) — https://aclanthology.org/2025.emnlp-main.240.pdf and https://arxiv.org/abs/2410.11305 (accessed 2026-08-11) | |
| 296 | +- QuantSpec: Self-Speculative Decoding with Hierarchical Quantized KV Cache (Apple ML Research, ICML 2025) — https://machinelearning.apple.com/research/quantspec (accessed 2026-08-11) | |
| 297 | +- ML-SpecQD: Multi-Level Speculative Decoding with Quantized Drafts — https://arxiv.org/html/2503.13565v1 (accessed 2026-08-11) | |
| 298 | +- Speculative Decoding with Big Little Decoder (Kim et al., NeurIPS 2023) — https://arxiv.org/abs/2302.07863 (accessed 2026-08-11) | |
| 299 | +- BigLittleDecoder repository — https://github.com/kssteven418/biglittledecoder (accessed 2026-08-11) | |
| 300 | +- SpecInfer: Accelerating LLM Serving with Tree-based Speculative Inference and Verification (ASPLOS 2024) — https://arxiv.org/abs/2305.09781 (accessed 2026-08-11) | |
| 301 | +- SpecExec: Massively Parallel Speculative Decoding for Interactive LLM Inference on Consumer Devices (NeurIPS 2024) — https://arxiv.org/html/2406.02532v1 (accessed 2026-08-11) | |
| 302 | +- SpecExec results (Together AI blog) — https://www.together.ai/blog/specexec (accessed 2026-08-11) | |
| 303 | +- Recurrent Drafter for Fast Speculative Decoding in Large Language Models (Apple; MLX/Metal benchmarks) — https://arxiv.org/html/2403.09919v5 and https://machinelearning.apple.com/research/recurrent-drafter (accessed 2026-08-11) | |
| 304 | +- Speculative Streaming: Fast LLM Inference Without Auxiliary Models (Apple ML Research) — https://machinelearning.apple.com/research/llm-inference (accessed 2026-08-11) | |
| 305 | +- SPEED: Speculative Pipelined Execution for Efficient Decoding (Hooper et al., NeurIPS-W 2023) — https://arxiv.org/abs/2310.12072 (accessed 2026-08-11) | |
| 306 | +- LLM-42: Enabling Determinism in LLM Inference with Verified Speculation — https://arxiv.org/html/2601.17768v1 (accessed 2026-08-11) | |
| 307 | +- FrugalGPT / cascade & routing results summary — https://neuraltrust.ai/blog/llm-model-routing (accessed 2026-08-11) | |
| 308 | +- Regret Bounds for Model Cascades (survey of FrugalGPT/RouteLLM/Hybrid-LLM numbers) — https://www.tmls.nyc/research/cascade-regret-optimal-stopping (accessed 2026-08-11) | |
| 309 | +- Confident Adaptive Language Modeling (Schuster et al., NeurIPS 2022) — https://arxiv.org/abs/2207.07061 (PDF: https://www.proceedings.com/content/068/068431-1269open.pdf) (accessed 2026-08-11) | |
| 310 | +- Accelerating text generation with CALM (Google Research blog) — https://research.google/blog/accelerating-text-generation-with-confident-adaptive-language-modeling-calm (accessed 2026-08-11) | |
| 311 | +- Consistent Accelerated Inference via Confident Adaptive Transformers (Schuster et al., 2021) — https://neurips2021-nlp.github.io/papers/7/CameraReady/Confident_Early_Exit__Transformer___workshop.pdf (accessed 2026-08-11) | |
| 312 | +- Deja Vu: Contextual Sparsity for Efficient LLMs at Inference Time (Liu et al., ICML 2023) — https://proceedings.mlr.press/v202/liu23am/liu23am.pdf (accessed 2026-08-11) | |
| 313 | +- LLM in a flash: Efficient Large Language Model Inference with Limited Memory (Apple, ACL 2024) — https://arxiv.org/html/2312.11514v2 (accessed 2026-08-11) | |
| 314 | +- MoE-SpeQ: Speculative Quantized Decoding with Proactive Expert Prefetching and Offloading — https://ui.adsabs.harvard.edu/abs/2025arXiv251114102W/abstract (arXiv:2511.14102) (accessed 2026-08-11) | |
| 315 | +- Fate: Fast Edge Inference of Mixture-of-Experts Models via Cross-Layer Gate — https://arxiv.org/html/2502.12224v2 (accessed 2026-08-11) | |
| 316 | +- Speculating Experts Accelerates Inference for Mixture-of-Experts — https://arxiv.org/html/2603.19289v1 (accessed 2026-08-11) | |
| 317 | +- SpecMD: A Comprehensive Study on Speculative Expert Prefetching (Apple ML Research) — https://machinelearning.apple.com/research/specmd-expert-prefetching (accessed 2026-08-11) | |
| 318 | +- The Lipschitz Constant of Self-Attention (Kim, Papamakarios, Mnih; ICML 2021) — https://proceedings.mlr.press/v139/kim21i/kim21i.pdf (accessed 2026-08-11) | |
| 319 | +- How Smooth Is Attention? (Castin et al.; Apple ML Research) — https://arxiv.org/html/2312.14820v2 and https://machinelearning.apple.com/research/how-smooth-is-attention (accessed 2026-08-11) | |
| 320 | +- DeepT: Fast and Precise Certification of Transformers (PLDI 2021) — https://files.sri.inf.ethz.ch/website/papers/pldi21-transformers.pdf (accessed 2026-08-11) | |
| 321 | +- auto_LiRPA: Automatic Linear Relaxation based Perturbation Analysis (NeurIPS 2020; library) — https://github.com/Verified-Intelligence/auto_LiRPA (accessed 2026-08-11) | |
| 322 | +- Towards Tighter LiRPA-based Robustness Certification (COLING 2025; CROWN O(m²n³) complexity discussion) — https://aclanthology.org/2025.coling-main.415.pdf (accessed 2026-08-11) | |
| 323 | +- Mixed-precision iterative refinement using tensor cores (Haidar, Dongarra et al.; surveys Carson–Higham GMRES-IR guarantees) — https://www.netlib.org/utk/people/JackDongarra/PAPERS/mixed-rs-2020.pdf (accessed 2026-08-11) | |
| 324 | +- Three-Precision GMRES-Based Iterative Refinement for Least Squares Problems (Carson, Higham, Pranesh) — https://eprints.maths.manchester.ac.uk/2770/1/paper.pdf (accessed 2026-08-11) | |
| 325 | +- A New Approach to Probabilistic Rounding Error Analysis (Higham & Mary, SIAM SISC 2019) — https://epubs.siam.org/doi/10.1137/18M1226312 (accessed 2026-08-11) | |
| 326 | +- Stochastic Rounding and Its Probabilistic Backward Error Analysis (Connolly, Higham, Mary, SIAM SISC 2021) — https://epubs.siam.org/doi/10.1137/20M1334796 (accessed 2026-08-11) | |
| 327 | +- Quantization Error Propagation: Revisiting Layer-Wise Post-Training Quantization (NeurIPS 2025) — https://arxiv.org/html/2504.09629v3 (accessed 2026-08-11) | |
| 328 | +- Why Do Some Inputs Break Low-Bit LLM Quantization? (EMNLP 2025) — https://aclanthology.org/2025.emnlp-main.168.pdf (accessed 2026-08-11) | |
| 329 | +- Which Quantization Should I Use? A Unified Evaluation of llama.cpp Quantizations — https://arxiv.org/html/2601.14277v1 (accessed 2026-08-11) | |
| 330 | +- Accuracy is Not All You Need (Microsoft; flips + KL under compression) — https://arxiv.org/html/2407.09141v1 (accessed 2026-08-11) | |
| 331 | +- Why accuracy is a misleading metric when evaluating compressed LLMs (flips summary) — https://bdtechtalks.com/2024/08/06/why-accuracy-is-a-misleading-metric-when-evaluating-compressed-llms (accessed 2026-08-11) | |
| 332 | +- llama.cpp quantizer discussion #23853 (KLD percentiles, "Same top p" ≈ 90.9–91.2%) — https://github.com/ggml-org/llama.cpp/discussions/23853 (accessed 2026-08-11) | |
| 333 | +- Blind testing different quants (llama.cpp discussion #5962) — https://github.com/ggml-org/llama.cpp/discussions/5962 (accessed 2026-08-11) | |
| 334 | +- Measuring Model Quantisation Quality with KL Divergence (MLX quant KLD measurements) — https://smcleod.net/2026/04/measuring-model-quantisation-quality-with-kl-divergence (accessed 2026-08-11) | |
| 335 | +- Eliciting Latent Predictions from Transformers with the Tuned Lens (Belrose et al.; "prediction depth") — https://arxiv.org/html/2303.08112v6 (accessed 2026-08-11) | |
| 336 | +- Defeating Nondeterminism in LLM Inference (Thinking Machines) — https://thinkingmachines.ai/blog/defeating-nondeterminism-in-llm-inference (accessed 2026-08-11) | |
| 337 | +- Logit-Gap Steering (Palo Alto Networks Unit 42; measured refusal logit gaps) — https://unit42.paloaltonetworks.com/logit-gap-steering-impact (accessed 2026-08-11) | |
| 338 | +- QuickSilver / Adaptive Matryoshka Quantization (per-token entropy-gated bit-width) — https://arxiv.org/pdf/2506.22396 (accessed 2026-08-11) | |
| 339 | +- FlexQuant: A Flexible and Efficient Dynamic Precision Switching Framework for LLM Quantization — https://arxiv.org/html/2506.12024v3 (accessed 2026-08-11) | |
| 340 | +- DP-LLM: Runtime Model Adaptation with Dynamic Layer-wise Precision Assignment (NeurIPS 2025) — https://neurips.cc/virtual/2025/poster/115920 (accessed 2026-08-11) | |
| 341 | +- MoBiQuant: Mixture-of-Bits Quantization for Token-Adaptive LLM Inference — https://ui.adsabs.harvard.edu/abs/2026arXiv260220191W/abstract (accessed 2026-08-11) | |
| 342 | +- Speculative Decoding Papers (curated list, hemingkx) — https://github.com/hemingkx/SpeculativeDecodingPapers (accessed 2026-08-11) | |
added
research/notes/decomposition_progressive.md
+572 −0
@@ -0,0 +1,572 @@ | ||
| 1 | +--- | |
| 2 | +project: localvm-research | |
| 3 | +document: research/notes/decomposition_progressive | |
| 4 | +author: Simon-Pierre Boucher | |
| 5 | +contact: contact@spboucher.ai | |
| 6 | +created: 2026-08-11 | |
| 7 | +status: draft | |
| 8 | +--- | |
| 9 | + | |
| 10 | +# Model Decomposition (§4.5) & Progressive / Approximate Computation (§4.6) | |
| 11 | + | |
| 12 | +Reading notes for charter sections 4.5 and 4.6. Scope: can pretrained weights be | |
| 13 | +re-represented as `base + residual`, `shared component + per-layer correction`, or | |
| 14 | +`low-rank + sparse/quantized residual` — and can computation itself be made | |
| 15 | +progressive, so that quality scales with bytes loaded rather than being fixed at | |
| 16 | +compile time. All quality numbers below are taken from the cited papers, not | |
| 17 | +reproduced locally. Access date for all sources: 2026-08-11. | |
| 18 | + | |
| 19 | +--- | |
| 20 | + | |
| 21 | +## 1. Landscape | |
| 22 | + | |
| 23 | +The literature splits into four families that almost never talk to each other: | |
| 24 | + | |
| 25 | +1. **Static decomposition** (SVD family, structural slicing, Kronecker/tensor-train, | |
| 26 | + codebook VQ): transform the checkpoint once, into a *fixed-size* smaller | |
| 27 | + representation. State of the art is good at 3–4 bits/param equivalent, usable at | |
| 28 | + 2 bits, and degrades sharply below that. None of these change the *runtime* | |
| 29 | + contract: every byte of the compressed model is still read for every token. | |
| 30 | + | |
| 31 | +2. **Sharing/delta decomposition** (cross-layer sharing, DeltaLLM, delta compression | |
| 32 | + of finetunes): exploit redundancy *between* matrices — between layers of one | |
| 33 | + model, or between a base model and its finetunes. The measured compressibility | |
| 34 | + here (1-bit deltas that are near-lossless) is the strongest published evidence | |
| 35 | + that large fractions of transformer weight information are redundant relative to | |
| 36 | + a reference. | |
| 37 | + | |
| 38 | +3. **Progressive / nested representations** (Any-Precision LLM, Matryoshka | |
| 39 | + Quantization, BitStack, recurrent residual quantization, Matryoshka-style | |
| 40 | + additive codebooks): one stored artifact yields *many* operating points; lower | |
| 41 | + precision is a strict prefix/subset of higher precision. This is the family | |
| 42 | + closest to the localvm-research thesis, and it is young (2024–2026). | |
| 43 | + | |
| 44 | +4. **Adaptive computation** (early exit, dynamic depth, adaptive halting, | |
| 45 | + approximate matmul, mixed-precision iterative refinement, error-bounded lossy | |
| 46 | + compression): vary the *amount of computation or precision per input*, sometimes | |
| 47 | + with provable error control. Almost all of it varies **depth** (skip layers) or | |
| 48 | + **tokens** (route around blocks); essentially none of it varies **how much of | |
| 49 | + each weight matrix is materialized** per token. | |
| 50 | + | |
| 51 | +The gap at the intersection: a representation where the runtime decides, per layer | |
| 52 | +and per token (or per short window), *how many bytes of a nested weight encoding to | |
| 53 | +read*, with the rest resident only on SSD. Family 3 provides the encoding; family 4 | |
| 54 | +provides the control policy; families 1–2 provide the redundancy evidence. No | |
| 55 | +published system combines them (Phase 11 must re-verify this before any claim). | |
| 56 | + | |
| 57 | +--- | |
| 58 | + | |
| 59 | +## 2. Techniques | |
| 60 | + | |
| 61 | +### 2.1 SVD-family low-rank factorization (SVD, FWSVD, ASVD, SVD-LLM) | |
| 62 | + | |
| 63 | +- **Mechanism.** Replace `W (m×n)` with `U_k Σ_k V_kᵀ` (rank k), stored as two thin | |
| 64 | + matrices. FWSVD weights the factorization by Fisher information of rows; | |
| 65 | + ASVD scales by activation statistics; SVD-LLM adds truncation-aware data | |
| 66 | + whitening (direct map from singular values to compression loss) plus a | |
| 67 | + sequential parameter update after truncation. | |
| 68 | +- **Compression / bandwidth.** Parameter ratio = k(m+n)/(mn). Bandwidth savings are | |
| 69 | + proportional and *dense* — low-rank GEMMs are two ordinary GEMMs, no exotic | |
| 70 | + kernels needed. | |
| 71 | +- **Quality (from papers).** Vanilla SVD collapses quickly: at 40–60% compression, | |
| 72 | + LLaMA-7B perplexities for SVD/FWSVD/ASVD reach the hundreds to tens of thousands. | |
| 73 | + SVD-LLM keeps perplexity finite (e.g., ~13.1 at 40%, and 7.73 vs ASVD's 11.14 at | |
| 74 | + 20% compression on LLaMA-7B), a >99% perplexity reduction vs prior SVD baselines | |
| 75 | + — but average downstream accuracy still drops substantially at ≥40% | |
| 76 | + (0.33–0.41 averages vs uncompressed ~0.6+). Consistent picture: **weight matrices | |
| 77 | + are not globally low-rank; only ~20–30% rank reduction is cheap.** | |
| 78 | +- **Calibration.** All modern variants need a small calibration set (whitening / | |
| 79 | + activation stats); no retraining for SVD-LLM's base version. | |
| 80 | +- **Apple Silicon.** Excellent: factors are plain dense matmuls, trivially | |
| 81 | + expressible in MLX; can be combined with MLX affine quantization. | |
| 82 | +- **Limitation.** Fixed rank chosen at compile time; quality cliff past moderate | |
| 83 | + ratios; ignores that the residual `W − U_kΣ_kV_kᵀ` still contains most spectral | |
| 84 | + energy in LLMs (singular values decay slowly). | |
| 85 | +- **Extension opportunity.** Use the low-rank part as a *resident hot path* and keep | |
| 86 | + the residual on SSD, loaded on demand — none of these papers store the residual | |
| 87 | + at all. | |
| 88 | + | |
| 89 | +### 2.2 Structural decomposition and rank-reduction as intervention (SliceGPT, LASER) | |
| 90 | + | |
| 91 | +- **SliceGPT** (ICLR 2024): applies orthogonal rotations (PCA of activations, | |
| 92 | + exploiting computational invariance of pre-norm transformers) then deletes rows | |
| 93 | + and columns, shrinking the embedding dimension. Removes up to 25% of parameters | |
| 94 | + of Llama-2 70B / OPT-66B while keeping 99% zero-shot performance (90% for Phi-2); | |
| 95 | + dense smaller matrices → real speedups (up to 1.55× throughput), no sparse | |
| 96 | + kernels needed. Calibration only. Maps cleanly to MLX. | |
| 97 | +- **LASER** (ICLR 2024): replacing *selected* weight matrices (mostly later-layer | |
| 98 | + MLPs) with low-rank approximations **improves** task accuracy, at times by up to | |
| 99 | + 30 percentage points, with no training. This is an intervention result, not a | |
| 100 | + compression system, but it is first-order evidence that higher-order weight | |
| 101 | + components of specific matrices are noise-like for some tasks. | |
| 102 | +- **Limitation / opportunity.** Both are static. LASER's finding suggests per-matrix | |
| 103 | + rank sensitivity is highly non-uniform — exactly the profile a compile-time | |
| 104 | + analyzer (charter Phase 10 "compilation stage") should measure per layer. | |
| 105 | + | |
| 106 | +### 2.3 Low-rank + quantized residual (CALDERA; randomized LPLR) | |
| 107 | + | |
| 108 | +- **Mechanism.** CALDERA (NeurIPS 2024): `W ≈ Q + LR` where `Q` is a full-rank but | |
| 109 | + aggressively quantized backbone and `L,R` low-rank factors (also quantized, | |
| 110 | + possibly at higher precision), obtained by calibration-aware alternating | |
| 111 | + minimization with approximation-error bounds. Targets the 2–2.5 bit/param regime. | |
| 112 | + Precursor: randomized low-rank + low-precision factorization (NeurIPS 2023) with | |
| 113 | + explicit error bounds vs rank and bit-budget. | |
| 114 | +- **Relevance.** This is literally the `base + residual` shape asked for in §4.5, | |
| 115 | + with the roles inverted (quantized full-rank base + low-rank correction). The | |
| 116 | + decomposition is *additive*, so the terms can in principle be loaded and applied | |
| 117 | + independently — CALDERA never exploits that at runtime. | |
| 118 | +- **Apple Silicon.** `LR` is dense GEMM (fine); `Q` uses QuIP#-style lattice | |
| 119 | + codebooks in the reference implementation (CUDA kernels; Metal port nontrivial). | |
| 120 | +- **Limitation.** Single fixed operating point; joint optimization couples the | |
| 121 | + terms, so dropping one term is not quality-graceful by construction. | |
| 122 | +- **Extension.** Re-derive the decomposition with a *progressivity constraint* | |
| 123 | + (base alone must be usable) — cf. §2.8. | |
| 124 | + | |
| 125 | +### 2.4 Codebook / vector quantization of weights (AQLM, QuIP#, GPTVQ, VPTQ) | |
| 126 | + | |
| 127 | +- **Mechanism.** Represent weight groups as sums of codebook entries: AQLM = | |
| 128 | + additive multi-codebook quantization (from the retrieval literature, i.e. | |
| 129 | + product/additive quantization à la Babenko) optimized against calibration | |
| 130 | + activations; QuIP# = randomized Hadamard incoherence + E8 lattice codebooks + | |
| 131 | + finetuning; GPTVQ, VPTQ = vector PTQ variants ("blessing of dimensionality"). | |
| 132 | +- **Quality.** Current SOTA at 2–3 bits/param. AQLM at ~4 bits: Llama-2 ppl 3.57 vs | |
| 133 | + FP 3.46 (Wiki2, 70B-scale table in paper); AQLM repo reports a ~1-bit Llama-2-7B | |
| 134 | + (1×8-bit codebook) at WikiText-2 ppl 7.85 (2025 update). QuIP# made 2-bit | |
| 135 | + "viable" for the first time. Additive codebooks are *inherently residual*: each | |
| 136 | + additional codebook refines the previous sum. | |
| 137 | +- **Calibration.** Heavy (hours of optimization; QuIP#/AQLM benefit from finetuning). | |
| 138 | +- **Apple Silicon.** Weakest point: efficient decode kernels are CUDA-only today; | |
| 139 | + lattice decodes and LUT-heavy kernels need custom Metal work. MLX ships only | |
| 140 | + affine (2/3/4/5/6/8-bit, group 32/64/128) and mxfp4/mxfp8/nvfp4 modes. | |
| 141 | +- **Limitation.** Fixed bit-width at compile time; decoding cost nontrivial. | |
| 142 | +- **Extension.** Drop-by-Drop / Matryoshka-supervised additive codebooks (§2.8) | |
| 143 | + show the additive structure can be made *ordered* so codebooks can be dropped at | |
| 144 | + runtime — the natural bridge to paging. | |
| 145 | + | |
| 146 | +### 2.5 Cross-layer parameter sharing (ALBERT, Subformer, Relaxed Recursive Transformers, Basis Sharing, DeltaLLM, ResidualTransformer) | |
| 147 | + | |
| 148 | +- **Mechanism.** Reuse one block of weights across layers. ALBERT (all layers | |
| 149 | + share), Subformer (sandwich sharing), Universal Transformer (recurrence) — | |
| 150 | + train-from-scratch results. The post-training versions matter more here: | |
| 151 | + - **Relaxed Recursive Transformers** (DeepMind, ICLR 2025): convert an existing | |
| 152 | + LLM into a model that loops a small block of layers, "relaxed" by per-layer | |
| 153 | + LoRA modules initialized via truncated SVD of the layer-vs-shared-weight | |
| 154 | + difference. Recursive models converted from 2×-larger models can outperform | |
| 155 | + same-size pretrained models; with distillation they approach the original. | |
| 156 | + - **Basis Sharing** (ICLR 2025): express weights of *different layers* as | |
| 157 | + combinations of a shared set of SVD-derived basis vectors + per-layer | |
| 158 | + coefficients; outperforms SVD-LLM at 20–50% compression, calibration-only. | |
| 159 | + - **DeltaLLM** (2025): share weights between adjacent transformer blocks and add | |
| 160 | + low-rank per-layer deltas; ~30–40M tokens of light training; 12% parameter | |
| 161 | + reduction retaining ~90% performance on Llama/Phi; DeltaPhi 2.9B (24% | |
| 162 | + reduction) matches a finetuned SlicedPhi 3.3B. ResidualTransformer (ICASSP | |
| 163 | + 2024) is the same idea for speech models. | |
| 164 | +- **Compression / bandwidth.** This is the interesting part for us: a shared block | |
| 165 | + resident in RAM amortizes across layers; per-layer deltas are small. A "one base | |
| 166 | + layer + N cheap diffs" model turns per-layer weight traffic into per-layer | |
| 167 | + *delta* traffic. | |
| 168 | +- **Limitation.** Post-training conversion still needs some uptraining (RRT, | |
| 169 | + DeltaLLM); pure zero-shot layer tying degrades badly. Compression ratios so far | |
| 170 | + are modest (12–25%), far from the 10× regime. | |
| 171 | +- **Extension.** Nobody treats the shared block as a *cache-resident core* and the | |
| 172 | + deltas as *SSD-resident pages*. Also untested: sharing + progressive delta | |
| 173 | + precision (delta rank/bits as a knob per layer). | |
| 174 | + | |
| 175 | +### 2.6 Delta compression of finetunes (BitDelta, DeltaZip, Delta-CoMe) | |
| 176 | + | |
| 177 | +- **Mechanism.** Decompose a finetuned model as `base + Δ` and compress Δ: | |
| 178 | + BitDelta quantizes Δ to **1 bit** (sign + per-matrix scale, scales distilled in | |
| 179 | + minutes); DeltaZip (EuroSys 2025) uses GPTQ-style compression of Δ (~10×) inside | |
| 180 | + a multi-tenant serving system; Delta-CoMe (NeurIPS 2024) allocates mixed | |
| 181 | + precision to Δ's singular vectors by singular-value magnitude (near-lossless at | |
| 182 | + ~1-bit average, and unlike BitDelta it also holds up on math/code finetunes). | |
| 183 | +- **Quality.** BitDelta: minimal degradation across Llama-2/Mistral/MPT up to 70B; | |
| 184 | + >10× memory reduction for multi-model serving. BitDelta's own ablation is | |
| 185 | + notable for §4.6: applying BitDelta *successively* (compress, treat result as new | |
| 186 | + base, compress the new delta…) yields an increasingly granular stack of 1-bit | |
| 187 | + masks whose quality *approaches the original monotonically* — an accidental | |
| 188 | + progressive code. | |
| 189 | +- **Relevance as evidence.** Finetuning information ≈ 1 bit/param. This does not | |
| 190 | + directly compress a base model, but it proves that "model = reference + | |
| 191 | + extremely compressible correction" is a real structure in modern LLM weight | |
| 192 | + space, and it motivates trying the same decomposition *within* one model | |
| 193 | + (layer_i = layer_j + cheap delta; model = quantized self + cheap residual). | |
| 194 | +- **Apple Silicon.** Sign matrices + scale are trivially Metal-friendly (1-bit | |
| 195 | + masks decode to ±scale; MLX has no built-in kernel but the op is simple). | |
| 196 | + | |
| 197 | +### 2.7 Kronecker and tensor-network decompositions (KnGPT2, TensorGPT, tensor trains) | |
| 198 | + | |
| 199 | +- **Mechanism.** `W ≈ A ⊗ B` (Kronecker; nearest-Kronecker via rank-1 SVD of | |
| 200 | + reshaped W) or tensor-train factorization of reshaped weights/embeddings. | |
| 201 | +- **Quality.** Results are only convincing at GPT-2/BERT scale with mandatory | |
| 202 | + retraining (KnGPT2, ACL 2022). TensorGPT compresses *embedding layers* 2×–65× | |
| 203 | + training-free on GPT-2-class models, but embeddings are a small fraction of a | |
| 204 | + modern LLM. No competitive 7B+ results without heavy retraining found. | |
| 205 | +- **Verdict for us.** Low priority: high implementation cost, weak post-training | |
| 206 | + evidence at scale, and Kronecker-structured matmul kernels for Metal would be | |
| 207 | + bespoke. Worth keeping only as a candidate basis for *shared dictionaries*. | |
| 208 | + | |
| 209 | +### 2.8 Progressive / nested weight representations (Any-Precision LLM, MatQuant, BitStack, RRQ, Drop-by-Drop) | |
| 210 | + | |
| 211 | +The family that matters most for localvm-research. | |
| 212 | + | |
| 213 | +- **Any-Precision LLM** (ICML 2024 oral): stores an n-bit (8-bit) "parent" model | |
| 214 | + such that every k-bit child (3≤k<8) is obtained by taking the **most significant | |
| 215 | + bits** — bit-plane overlay. Built post-training by "incremental upscaling" from a | |
| 216 | + 3-bit seed (< 1 minute for 7B after seed quantization); ships a specialized | |
| 217 | + (GPU) engine with bit-plane-aware memory layout. Memory: supporting {3..8}-bit | |
| 218 | + Llama-2-7B costs 8.4 GB vs 29.9 GB for separate models (3.56×). Each bit-width | |
| 219 | + matches SOTA quality for that width. | |
| 220 | +- **Matryoshka Quantization** (MatQuant, DeepMind, ICLR 2025 oral): co-trains one | |
| 221 | + int8 quantized model whose int4/int2 slices (MSBs) are all optimized jointly; | |
| 222 | + int2 slices become up to ~10% more accurate than dedicated int2 QAT/OmniQuant — | |
| 223 | + an int2-FFN Gemma-2 9B beats an int8-FFN Gemma-2 2B. Also allows layer-wise | |
| 224 | + mix'n'match of precisions at inference. Requires QAT-style training. | |
| 225 | +- **BitStack** (ICLR 2025): training-free. Iterative significance-weighted | |
| 226 | + decomposition produces ~1-bit-per-parameter **residual blocks**; blocks are | |
| 227 | + sorted (universally, across the whole model, by importance) and stacked in | |
| 228 | + storage as transmission units; the runtime loads as many blocks as current | |
| 229 | + memory allows → **megabyte-level tradeoff between resident size and quality**, | |
| 230 | + matching or beating GPTQ/AWQ at extreme ratios. This is the closest existing | |
| 231 | + system to "quality scales with bytes loaded." Known weakness (noted in follow-up | |
| 232 | + work, e.g. the AMQ paper): on-the-fly weight *reconstruction from residual | |
| 233 | + blocks slows inference notably*. | |
| 234 | +- **Recurrent Residual Quantization** (RRQ, arXiv 2608.04048, 2026): calibration- | |
| 235 | + free, additive stage-wise scheme — 2-bit RTN base + successive 2-bit RTN | |
| 236 | + residual corrections gives 4/6/8-bit operating points from one package; the | |
| 237 | + whole multi-precision package for Qwen3-8B builds in ~1,293 s (3.3× faster than | |
| 238 | + MatGPTQ-style joint optimization). | |
| 239 | +- **Drop-by-Drop additive codebooks** (arXiv 2606.12876, 2026): AQLM-style multi- | |
| 240 | + codebook quantization with Matryoshka supervision so codebooks are ordered | |
| 241 | + coarse→fine and can be dropped at inference for progressive compression. | |
| 242 | +- **Apple Silicon feasibility.** Bit-plane and residual-stage layouts are exactly | |
| 243 | + the kind of thing unified memory + mmap should be good at: each precision level | |
| 244 | + is a separate contiguous region; upgrading precision = reading another region, | |
| 245 | + not rewriting the resident one. No published Metal/MLX implementation of any of | |
| 246 | + these exists (all engines are CUDA); MLX's affine quant kernels (2–8 bit) could | |
| 247 | + serve stages if each stage is expressed as an affine-quantized tensor. | |
| 248 | +- **Common limitation.** All of them select the operating point **statically** | |
| 249 | + (per deployment, per memory budget). None selects precision per token; none ties | |
| 250 | + the residual stages to storage paging; none reports bytes-read-per-token. | |
| 251 | + | |
| 252 | +### 2.9 Early exit and dynamic depth (ACT, PonderNet, MSDNet, Depth-Adaptive Transformer, CALM, LayerSkip, Mixture-of-Depths) | |
| 253 | + | |
| 254 | +- **Mechanism lineage.** ACT (Graves 2016): learned halting for RNN steps. | |
| 255 | + PonderNet (2021): stabilized probabilistic halting. MSDNet (ICLR 2018): anytime | |
| 256 | + prediction with multi-scale features — the canonical "anytime NN". Depth- | |
| 257 | + Adaptive Transformer (ICLR 2020): per-token decoder depth. CALM (NeurIPS 2022): | |
| 258 | + confidence-gated early exit for LM generation with *sequence-level calibrated | |
| 259 | + guarantees* (up to ~3× compute reduction, provably maintaining quality; | |
| 260 | + addresses missing-KV problem of exited tokens). LayerSkip (Meta, 2024): layer | |
| 261 | + dropout + shared early-exit head during training, then **self-speculative | |
| 262 | + decoding** — early layers draft, remaining layers verify — 1.34–2.16× speedup | |
| 263 | + with exact final quality. Mixture-of-Depths (2024): learned top-k token routing | |
| 264 | + per block under a static compute budget. | |
| 265 | +- **Bandwidth reality check.** Early exit saves *depth* — and therefore also the | |
| 266 | + weight bytes of skipped layers for that token — but batch dynamics and KV | |
| 267 | + bookkeeping erode the savings; and for us the key limit is that exit decisions | |
| 268 | + gate *whole layers*, the coarsest possible granularity. | |
| 269 | +- **Retraining.** CALM/LayerSkip/MoD all need training or finetuning with exit | |
| 270 | + losses; nothing here is drop-in post-training on a frozen checkpoint (LayerSkip | |
| 271 | + ships finetuned checkpoints; naive early exit on frozen models is poor). | |
| 272 | +- **Apple Silicon.** Conceptually trivial to port (it is control flow, not | |
| 273 | + kernels); single-request local decoding on a Mac is actually the *friendly* case | |
| 274 | + (no batch synchronization problem). | |
| 275 | +- **Extension opportunity.** LayerSkip's draft-then-verify structure is | |
| 276 | + depth-based self-speculation. The unexplored dual: **precision-based | |
| 277 | + self-speculation** — draft with a resident low-bit base (prefix of a nested | |
| 278 | + representation, §2.8), verify/refine with residual planes only when the draft's | |
| 279 | + top-1 margin is small. Verification reads extra bytes *only on demand*. (Charter | |
| 280 | + Experiments D and G test exactly the preconditions.) | |
| 281 | + | |
| 282 | +### 2.10 Approximate matrix multiplication (Drineas–Kannan–Mahoney, Bolt, MADDNESS) | |
| 283 | + | |
| 284 | +- **Mechanism.** (a) Randomized sampling: sample columns/rows with length-squared | |
| 285 | + probabilities → unbiased estimate of `AB` with Frobenius error `O(‖A‖‖B‖/√c)` | |
| 286 | + (DKM, SIAM J. Comput. 2006; foundation of RandNLA). (b) Learned LUT methods: | |
| 287 | + Bolt, MADDNESS (ICML 2021) — replace one operand's inner products with learned | |
| 288 | + hash-bucket lookups; up to 10× better speed-quality than prior AMM on small | |
| 289 | + matrices, ~100× vs exact in the best cases. | |
| 290 | +- **Reality for LLMs.** MADDNESS-class methods shine when one matrix is fixed and | |
| 291 | + *tall-thin* regimes apply (classifier layers, kernels); accuracy at transformer | |
| 292 | + scale is unproven, and LUT-gather-heavy inner loops are a poor match for GPU | |
| 293 | + matmul pipelines (they beat CPUs, not tensor cores). Sampling-based AMM gives | |
| 294 | + clean error bounds but errors are relative to matrix norms — too loose to | |
| 295 | + certify token decisions directly. | |
| 296 | +- **Value to us.** Not as a drop-in kernel, but as the theory toolbox for | |
| 297 | + **partial GEMM with error bars** (Experiment E): length-squared/leverage | |
| 298 | + sampling tells us *which blocks matter most* and what error skipping the rest | |
| 299 | + costs — i.e., a principled block-ordering for progressive evaluation. | |
| 300 | + | |
| 301 | +### 2.11 Adaptive-precision numerical computing (Wilkinson iterative refinement → GMRES-IR, five-precision IR) | |
| 302 | + | |
| 303 | +- **Mechanism.** Solve `Ax=b` with an LU factorization computed in *low* precision | |
| 304 | + (cheap, fast), then iteratively refine: compute residual in high precision, solve | |
| 305 | + a correction system (possibly by GMRES preconditioned with the low-precision | |
| 306 | + factors), update. Carson & Higham (SIAM SISC 2018) formalized three-precision | |
| 307 | + IR; Amestoy et al. extended to five precisions; NVIDIA/Dongarra demonstrated | |
| 308 | + FP16-tensor-core factorizations refined to FP64 accuracy at ~4× speed. | |
| 309 | +- **Why it matters here.** This is the *canonical proof* in numerical computing | |
| 310 | + that "cheap approximate operator + residual-driven correction loop" recovers | |
| 311 | + full accuracy while doing most work at low precision. The transformer analogue — | |
| 312 | + run layers with a low-bit base, monitor a residual/confidence signal, apply | |
| 313 | + stored higher-precision corrections only when needed — is structurally identical | |
| 314 | + and appears untried for *weights* (speculative decoding is the analogue for | |
| 315 | + *tokens*). | |
| 316 | +- **Caveat.** IR has a convergence theory because `A` is the exact operator and | |
| 317 | + the residual is exactly computable; in an LLM the "exact" layer output is not | |
| 318 | + available without loading the full weights. The honest transferable idea is | |
| 319 | + *correction-on-demand plus a cheap instability detector* (logit margins, §4.10 | |
| 320 | + of the charter), not certified refinement. | |
| 321 | + | |
| 322 | +### 2.12 Error-bounded lossy compression from HPC (ZFP, SZ) | |
| 323 | + | |
| 324 | +- **Mechanism.** ZFP (Lindstrom, TVCG 2014): fixed-rate or fixed-accuracy block | |
| 325 | + transform coding of floating-point arrays, with *published round-off error | |
| 326 | + analysis* (SIAM 2019) and random-access decode of 4^d blocks. SZ (Di & Cappello, | |
| 327 | + IPDPS 2016): prediction + error-controlled quantization with strict pointwise | |
| 328 | + error bounds; typically higher ratios than ZFP at equal bounds on many datasets. | |
| 329 | +- **Relevance.** These are mature, *error-budgeted*, block-random-access codecs | |
| 330 | + for float arrays — exactly the engineering shape a weight-paging store needs | |
| 331 | + (bounded per-block reconstruction error → feeds §4.9 perturbation analysis; | |
| 332 | + block random access → mmap-friendly pages). ZFP's fixed-rate mode gives | |
| 333 | + predictable page sizes. Neither has been evaluated as an LLM weight format | |
| 334 | + (weights are not smooth fields, so their predictors may underperform; needs | |
| 335 | + Experiment-H-style measurement). | |
| 336 | +- **Apple Silicon.** Both are C/C++ libraries that build on arm64; decode | |
| 337 | + throughput vs Apple NVMe read speed is the number to measure. | |
| 338 | + | |
| 339 | +--- | |
| 340 | + | |
| 341 | +## 3. Evidence of exploitable redundancy in pretrained transformers | |
| 342 | + | |
| 343 | +The strongest *measured* facts found, ordered by how directly they support a | |
| 344 | +base+residual execution model: | |
| 345 | + | |
| 346 | +1. **Finetune deltas carry ≈1 bit/param of information.** BitDelta quantizes the | |
| 347 | + full delta of 7B–70B finetunes to 1 bit with minimal degradation; Delta-CoMe is | |
| 348 | + near-lossless at ~1-bit average even for math/code finetunes. GPT-Zip/DeltaZip | |
| 349 | + independently report ~10× delta compressibility. → Weight space has directions | |
| 350 | + that are dramatically cheaper to encode *relative to a reference*. | |
| 351 | + | |
| 352 | +2. **Adjacent layers are highly similar / near-linear.** "Your Transformer is | |
| 353 | + Secretly Linear" (ACL 2024) measures Procrustes linearity ≈0.99 between | |
| 354 | + consecutive decoder layer embeddings across GPT/LLaMA/OPT/BLOOM, and shows some | |
| 355 | + of the most-linear blocks can be removed or replaced by linear approximations | |
| 356 | + with little loss. ShortGPT's Block Influence metric (cosine similarity between | |
| 357 | + layer input and output) finds many layers barely transform the hidden state; | |
| 358 | + removing them ("more redundant than you expect") costs little on benchmarks. | |
| 359 | + "The Unreasonable Ineffectiveness of the Deeper Layers" (ICLR 2025) prunes | |
| 360 | + large contiguous blocks of *deep* layers (selected by representational | |
| 361 | + similarity) with minimal QA degradation after light QLoRA healing. | |
| 362 | + | |
| 363 | +3. **Layer weights are compressible against each other.** DeltaLLM: adjacent-block | |
| 364 | + sharing + low-rank deltas retains ~90% performance at 12% reduction with only | |
| 365 | + 30–40M tokens of training. Basis Sharing: one shared SVD basis serves multiple | |
| 366 | + layers' weights with per-layer coefficients and beats per-layer SVD-LLM at | |
| 367 | + 20–50% ratios. Relaxed Recursive Transformers: a looped shared block + | |
| 368 | + SVD-initialized per-layer LoRA recovers most of the original model — i.e., much | |
| 369 | + of a layer's identity is "shared trunk + small correction." | |
| 370 | + | |
| 371 | +4. **Selective rank reduction can even help.** LASER: replacing selected later-MLP | |
| 372 | + matrices by low-rank approximations improves accuracy (up to +30 points on some | |
| 373 | + tasks) — high-order components of specific matrices are noise-like. | |
| 374 | + Complementary: SliceGPT removes 25% of parameters via activation-PCA rotation | |
| 375 | + with 99% zero-shot retention on Llama-2 70B / OPT-66B. | |
| 376 | + | |
| 377 | +5. **Task adaptation is intrinsically low-dimensional.** Aghajanyan et al. (2020): | |
| 378 | + RoBERTa-scale models can be finetuned to ~90% of full performance inside a | |
| 379 | + random subspace of only ~hundreds of dimensions; pretraining *reduces* intrinsic | |
| 380 | + dimension. → capability deltas, not just finetune deltas, are low-dimensional. | |
| 381 | + | |
| 382 | +6. **But global low-rankness of weights is a myth.** The SVD-family results (§2.1) | |
| 383 | + consistently show steep quality loss past ~25–40% rank compression ("features | |
| 384 | + are low-rank, weights are not"). Redundancy is *structured* (cross-layer, | |
| 385 | + relative-to-reference, task-conditional) rather than uniform spectral decay. | |
| 386 | + | |
| 387 | +7. **Inference is bandwidth-bound, so redundancy = latency.** Every token reads | |
| 388 | + every weight byte; on H100-class hardware compute outruns memory delivery by | |
| 389 | + ~600× (Cloudflare "Unweight" engineering measurement). On Apple Silicon the | |
| 390 | + ratio is smaller but the regime is the same — any byte not read per token is | |
| 391 | + ~proportional latency, which is why decoupling "stored bytes" from "read bytes" | |
| 392 | + (charter §2) is the right objective. | |
| 393 | + | |
| 394 | +--- | |
| 395 | + | |
| 396 | +## 4. Progressive encodings from other fields | |
| 397 | + | |
| 398 | +Transferable design patterns, from oldest to newest: | |
| 399 | + | |
| 400 | +- **Embedded wavelet coding (EZW 1993, SPIHT 1996, JPEG2000/EBCOT).** Coefficients | |
| 401 | + are transmitted in *significance order*, bit-plane by bit-plane; the bitstream | |
| 402 | + can be truncated at any byte and decodes to the best possible image for that | |
| 403 | + byte count ("each new bit conveys the maximum information"). JPEG2000's EBCOT | |
| 404 | + adds independently coded blocks with optimized truncation points — i.e., | |
| 405 | + rate-distortion-optimal *per-block* truncation. **Transfer:** encode weight | |
| 406 | + blocks as significance-ordered bit-planes/residual stages; "bytes loaded per | |
| 407 | + matrix" becomes a continuous quality knob, and per-block truncation points can | |
| 408 | + be optimized against layer sensitivity (Experiment F) instead of PSNR. BitStack | |
| 409 | + (§2.8) is an unwitting rediscovery of this with residual SVD blocks; nobody has | |
| 410 | + connected it to the mature R-D-optimal truncation machinery. | |
| 411 | + | |
| 412 | +- **Progressive meshes (Hoppe, SIGGRAPH 1996) and Nanite (UE5).** A mesh is stored | |
| 413 | + as a coarse base + an ordered stream of refinements (vertex splits), giving | |
| 414 | + lossless, continuous LoD, streaming, and *selective refinement* (refine only | |
| 415 | + where the camera looks). Nanite industrializes this: fixed-size clusters in a | |
| 416 | + hierarchical DAG, **streamed on demand so only visible detail resides in | |
| 417 | + memory**, with LoD chosen per-cluster per-frame at ~pixel-error tolerance. | |
| 418 | + **Transfer:** this is the exact architecture shape for weight paging — fixed-size | |
| 419 | + weight "clusters" at multiple precisions, a residency set updated per token/ | |
| 420 | + window by a cheap importance signal (attention/activation statistics as the | |
| 421 | + "camera"), error tolerance expressed in logit margin instead of pixels. | |
| 422 | + | |
| 423 | +- **Approximate query processing (Online Aggregation, SIGMOD 1997; BlinkDB, | |
| 424 | + EuroSys 2013).** Answer first, refine continuously, with statistical error bars; | |
| 425 | + BlinkDB answers queries over 17 TB in <2 s within 2–10% error by choosing among | |
| 426 | + precomputed stratified samples given a per-query time or error budget. | |
| 427 | + **Transfer:** the *interface* idea — inference under an explicit | |
| 428 | + (latency | error) budget, where the runtime chooses how much of the model to | |
| 429 | + consult and can report confidence; and the *offline* idea — precompute multiple | |
| 430 | + "samples" (precision profiles) of the model optimized for expected workloads. | |
| 431 | + | |
| 432 | +- **Mixed-precision iterative refinement (Wilkinson 1963 → Carson–Higham 2018).** | |
| 433 | + See §2.11: do the O(n³) work in cheap precision once, recover accuracy with | |
| 434 | + cheap corrective iterations. The pattern "expensive operator approximated + | |
| 435 | + residual-driven correction + convergence monitor" is the numerical-analysis | |
| 436 | + ancestor of any progressive-weight-refinement runtime. | |
| 437 | + | |
| 438 | +- **Error-bounded scientific compression (ZFP/SZ).** See §2.12: block random | |
| 439 | + access + guaranteed per-element error bounds is the storage-format discipline a | |
| 440 | + weight pager should adopt (bounded weight perturbation → bounded logit | |
| 441 | + perturbation via layer Lipschitz estimates, rather than hoping). | |
| 442 | + | |
| 443 | +--- | |
| 444 | + | |
| 445 | +## 5. Relevance to localvm-research | |
| 446 | + | |
| 447 | +**Could a progressive base+residual representation let quality scale with bytes | |
| 448 | +loaded?** The evidence says the ingredients all exist and individually work: | |
| 449 | + | |
| 450 | +- Nested/progressive weight codes exist and are near-SOTA at each operating point | |
| 451 | + (Any-Precision LLM, MatQuant, BitStack, RRQ). BitStack already demonstrates | |
| 452 | + monotone quality-vs-resident-megabytes on 7B–70B models, training-free. | |
| 453 | +- Redundancy is real and structured (§3): a 2-bit-class base plausibly carries | |
| 454 | + most behavior, and corrections are cheap *relative to* the base (BitDelta's | |
| 455 | + iterated 1-bit masks converge to the original). | |
| 456 | +- Control policies with quality guarantees exist for the depth dimension (CALM's | |
| 457 | + calibrated exits; LayerSkip's exact self-speculative verification). | |
| 458 | +- The systems patterns for demand-paged, error-budgeted, progressively refined | |
| 459 | + data are mature in other fields (EBCOT truncation, Nanite residency, BlinkDB | |
| 460 | + budgets, GMRES-IR refinement). | |
| 461 | + | |
| 462 | +**What has NOT been tried (candidate gaps for `research_gaps.md`):** | |
| 463 | + | |
| 464 | +1. **Token-/layer-conditional residual loading.** Every progressive system picks | |
| 465 | + its operating point statically per deployment. No published system decides *per | |
| 466 | + token* (or per small window) and *per layer* how many residual stages to apply, | |
| 467 | + despite Experiment-B/D-style predictability being the obvious enabler. The | |
| 468 | + marriage BitStack × CALM does not exist. | |
| 469 | +2. **Precision-based self-speculation on a nested code.** Draft with the resident | |
| 470 | + low-bit prefix; verify/refine with SSD-resident residual planes only when the | |
| 471 | + top-1 margin is small (LayerSkip's mechanism, transposed from depth to | |
| 472 | + precision, with draft and verifier *sharing the same bytes*). Needs a Phase 11 | |
| 473 | + novelty sweep (search terms: quantized self-speculation, precision cascade | |
| 474 | + decoding, progressive dequantization inference). | |
| 475 | +3. **Bytes-read-per-token as the optimized objective.** None of the §2.8 papers | |
| 476 | + measures SSD/DRAM traffic per generated token; they measure resident size. On a | |
| 477 | + 48 GB M5 Max with a fast NVMe, the interesting regime is: base resident | |
| 478 | + (~2 bits/param), residual planes mmap'd, and a policy that keeps *average* | |
| 479 | + bytes/token far below checkpoint size. This is measurable with our | |
| 480 | + instrumentation plan (fs_usage, vm_stat) and no one has published it. | |
| 481 | +4. **R-D-optimal truncation for weights.** Port EBCOT-style per-block optimized | |
| 482 | + truncation to weight blocks, with distortion measured as calibration-set logit | |
| 483 | + KL (not MSE), producing a *layer-sensitivity-aware* progressive layout at | |
| 484 | + compile time (fits the charter's "compilation stage" exactly). | |
| 485 | +5. **Shared-basis trunk as the resident core.** Basis Sharing / RRT / DeltaLLM | |
| 486 | + suggest "shared trunk resident + per-layer deltas paged". Unexplored as a | |
| 487 | + memory-hierarchy assignment rather than a compression ratio. | |
| 488 | + | |
| 489 | +**Apple Silicon specifics.** Low-rank factors, sign-mask deltas, and affine- | |
| 490 | +quantized stages (2–8 bit, group 32/64/128) map directly onto today's MLX kernels; | |
| 491 | +bit-plane overlays and additive codebooks would need custom Metal kernels (all | |
| 492 | +published engines are CUDA). Unified memory removes the CPU↔GPU copy that makes | |
| 493 | +progressive loading painful on discrete GPUs: a residual plane read from NVMe into | |
| 494 | +a mapped buffer is immediately GPU-visible. The BitStack-reported reconstruction | |
| 495 | +slowdown is the main engineering risk — reconstruction must be fused into the | |
| 496 | +matmul (dequant-in-kernel, as MLX already does for affine quant) rather than | |
| 497 | +materialized. These claims about MLX/Metal feasibility are assessments to be | |
| 498 | +validated in Experiments D/E/H, not established facts. | |
| 499 | + | |
| 500 | +**Failure modes to respect** (charter §17): if per-token stage selection turns out | |
| 501 | +to need near-all stages for acceptable quality (working set ≈ whole model), or if | |
| 502 | +random 4–64 KB residual reads on Apple NVMe are too slow/thermally throttled | |
| 503 | +(Experiment H), the progressive-paging premise dies; the fallback value of this | |
| 504 | +literature is then "best static compressed format for MLX," which is already | |
| 505 | +well-served by existing work. | |
| 506 | + | |
| 507 | +--- | |
| 508 | + | |
| 509 | +## Sources | |
| 510 | + | |
| 511 | +- SVD-LLM: Truncation-aware Singular Value Decomposition for Large Language Model Compression (ICLR 2025) — https://arxiv.org/html/2403.07378v3 (accessed 2026-08-11) | |
| 512 | +- SVD-LLM (ICLR 2025 proceedings abstract) — https://proceedings.iclr.cc/paper_files/paper/2025/hash/3104e1ab39875cf54fe1eb4473e7c5a1-Abstract-Conference.html (accessed 2026-08-11) | |
| 513 | +- SVD-LLM GitHub (AIoT-MLSys-Lab) — https://github.com/AIoT-MLSys-Lab/SVD-LLM (accessed 2026-08-11) | |
| 514 | +- ASVD: Activation-aware Singular Value Decomposition for Compressing LLMs — https://arxiv.org/abs/2312.05821 (accessed 2026-08-11) | |
| 515 | +- Language model compression with weighted low-rank factorization (FWSVD, ICLR 2022) — https://arxiv.org/abs/2207.00112 (accessed 2026-08-11) | |
| 516 | +- The Truth is in There: Improving Reasoning in Language Models with Layer-Selective Rank Reduction (LASER, ICLR 2024) — https://arxiv.org/abs/2312.13558 (accessed 2026-08-11) | |
| 517 | +- LASER project page — https://pratyushasharma.github.io/laser (accessed 2026-08-11) | |
| 518 | +- SliceGPT: Compress Large Language Models by Deleting Rows and Columns (ICLR 2024) — https://arxiv.org/abs/2401.15024 (accessed 2026-08-11) | |
| 519 | +- Compressing Large Language Models using Low Rank and Low Precision Decomposition (CALDERA, NeurIPS 2024) — https://arxiv.org/abs/2405.18886 (accessed 2026-08-11) | |
| 520 | +- CALDERA GitHub (pilancilab) — https://github.com/pilancilab/caldera (accessed 2026-08-11) | |
| 521 | +- Matrix Compression via Randomized Low Rank and Low Precision Factorization (NeurIPS 2023) — https://neurips.cc/virtual/2023/poster/70291 (accessed 2026-08-11) | |
| 522 | +- Extreme Compression of Large Language Models via Additive Quantization (AQLM) — https://arxiv.org/html/2401.06118v2 (accessed 2026-08-11) | |
| 523 | +- AQLM GitHub (incl. ~1-bit Llama-2-7B result) — https://github.com/vahe1994/AQLM (accessed 2026-08-11) | |
| 524 | +- QuIP#: Even Better LLM Quantization with Hadamard Incoherence and Lattice Codebooks (ICML 2024) — https://proceedings.mlr.press/v235/tseng24a.html (accessed 2026-08-11) | |
| 525 | +- GPTVQ: The Blessing of Dimensionality for LLM Quantization — https://arxiv.org/abs/2402.15319 (accessed 2026-08-11) | |
| 526 | +- VPTQ: Extreme Low-bit Vector Post-Training Quantization for LLMs (Microsoft Research) — https://www.microsoft.com/en-us/research/publication/vptq-extreme-low-bit-vector-post-training-quantization-for-large-language-models (accessed 2026-08-11) | |
| 527 | +- Relaxed Recursive Transformers: Effective Parameter Sharing with Layer-wise LoRA (ICLR 2025) — https://arxiv.org/html/2410.20672v1 (accessed 2026-08-11) | |
| 528 | +- Subformer: Exploring Weight Sharing for Parameter Efficiency (Findings of EMNLP 2021) — https://aclanthology.org/2021.findings-emnlp.344.pdf (accessed 2026-08-11) | |
| 529 | +- Basis Sharing: Cross-Layer Parameter Sharing for LLM Compression (ICLR 2025) — https://arxiv.org/abs/2410.03765 (accessed 2026-08-11) | |
| 530 | +- Basis Sharing (ICLR 2025 proceedings PDF) — https://proceedings.iclr.cc/paper_files/paper/2025/file/238c98450b1d9e8055f94d22f303bb57-Paper-Conference.pdf (accessed 2026-08-11) | |
| 531 | +- DeltaLLM: Compress LLMs with Low-Rank Deltas between Shared Weights — https://arxiv.org/abs/2501.18596 (accessed 2026-08-11) | |
| 532 | +- ResidualTransformer: Residual Low-Rank Learning with Weight-Sharing for Transformer Layers (ICASSP 2024) — https://arxiv.org/abs/2310.02489 (accessed 2026-08-11) | |
| 533 | +- BitDelta: Your Fine-Tune May Only Be Worth One Bit (NeurIPS 2024) — https://arxiv.org/html/2402.10193v3 (accessed 2026-08-11) | |
| 534 | +- BitDelta NeurIPS poster page — https://neurips.cc/virtual/2024/poster/94736 (accessed 2026-08-11) | |
| 535 | +- DeltaZip: Compression for Foundation Models (EuroSys 2025; repo lists delta-compression literature) — https://github.com/eth-easl/deltazip (accessed 2026-08-11) | |
| 536 | +- Delta-CoMe: Training-Free Delta-Compression with Mixed-Precision for LLMs (NeurIPS 2024) — https://arxiv.org/abs/2406.08903 (accessed 2026-08-11) | |
| 537 | +- Kronecker Decomposition for GPT Compression (KnGPT2, ACL 2022) — https://aclanthology.org/2022.acl-short.24.pdf (accessed 2026-08-11) | |
| 538 | +- TensorGPT: Efficient Compression of LLMs based on Tensor-Train Decomposition — https://arxiv.org/html/2307.00526v2 (accessed 2026-08-11) | |
| 539 | +- BitStack: Any-Size Compression of Large Language Models in Variable Memory Environments (ICLR 2025) — https://arxiv.org/abs/2410.23918 (accessed 2026-08-11) | |
| 540 | +- Any-Precision LLM: Low-Cost Deployment of Multiple, Different-Sized LLMs (ICML 2024 oral) — https://arxiv.org/pdf/2402.10517 (accessed 2026-08-11) | |
| 541 | +- Any-Precision LLM GitHub (SNU-ARC) — https://github.com/SNU-ARC/any-precision-llm (accessed 2026-08-11) | |
| 542 | +- Matryoshka Quantization (MatQuant, ICLR 2025 oral) — https://openreview.net/forum?id=phVWcUSGYP (accessed 2026-08-11) | |
| 543 | +- Recurrent Residual Quantization: A Progressive Multi-Precision Representation for LLMs — https://arxiv.org/abs/2608.04048 (accessed 2026-08-11) | |
| 544 | +- Multi-Bitwidth Quantization for LLMs Using Additive Codebooks (Drop-by-Drop) — https://arxiv.org/html/2606.12876v1 (accessed 2026-08-11) | |
| 545 | +- Intrinsic Dimensionality Explains the Effectiveness of Language Model Fine-Tuning — https://arxiv.org/abs/2012.13255 (accessed 2026-08-11) | |
| 546 | +- ShortGPT: Layers in Large Language Models are More Redundant Than You Expect — https://arxiv.org/html/2403.03853v1 (accessed 2026-08-11) | |
| 547 | +- The Unreasonable Ineffectiveness of the Deeper Layers (ICLR 2025) — https://arxiv.org/abs/2403.17887 (accessed 2026-08-11) | |
| 548 | +- Your Transformer is Secretly Linear (ACL 2024) — https://arxiv.org/abs/2405.12250 (accessed 2026-08-11) | |
| 549 | +- Confident Adaptive Language Modeling (CALM, NeurIPS 2022) — https://proceedings.neurips.cc/paper_files/paper/2022/hash/6fac9e316a4ae75ea244ddcef1982c71-Abstract-Conference.html (accessed 2026-08-11) | |
| 550 | +- Google Research blog: Accelerating text generation with CALM — https://research.google/blog/accelerating-text-generation-with-confident-adaptive-language-modeling-calm (accessed 2026-08-11) | |
| 551 | +- LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding — https://arxiv.org/html/2404.16710v1 (accessed 2026-08-11) | |
| 552 | +- Depth-Adaptive Transformer (ICLR 2020) — https://arxiv.org/abs/1910.10073 (accessed 2026-08-11) | |
| 553 | +- Adaptive Computation Time for Recurrent Neural Networks (Graves 2016) — https://arxiv.org/abs/1603.08983 (accessed 2026-08-11) | |
| 554 | +- PonderNet: Learning to Ponder — https://arxiv.org/abs/2107.05407 (accessed 2026-08-11) | |
| 555 | +- Multi-Scale Dense Networks for Resource Efficient Image Classification (MSDNet) — https://arxiv.org/abs/1703.09844 (accessed 2026-08-11) | |
| 556 | +- Mixture-of-Depths: Dynamically allocating compute in transformer-based language models — https://arxiv.org/abs/2404.02258 (accessed 2026-08-11) | |
| 557 | +- Multiplying Matrices Without Multiplying (MADDNESS, ICML 2021) — https://proceedings.mlr.press/v139/blalock21a/blalock21a.pdf (accessed 2026-08-11) | |
| 558 | +- Fast Monte Carlo Algorithms for Matrices I: Approximating Matrix Multiplication (Drineas, Kannan, Mahoney, SIAM J. Comput. 2006) — https://epubs.siam.org/doi/10.1137/S0097539704442684 (accessed 2026-08-11) | |
| 559 | +- Accelerating the Solution of Linear Systems by Iterative Refinement in Three Precisions (Carson & Higham, SIAM SISC 2018) — https://epubs.siam.org/doi/10.1137/17M1140819 (accessed 2026-08-11) | |
| 560 | +- Five-precision GMRES-based Iterative Refinement (Amestoy et al.) — https://eprints.maths.manchester.ac.uk/2852/1/paper.pdf (accessed 2026-08-11) | |
| 561 | +- What Is Iterative Refinement? (Nick Higham) — https://nhigham.com/2023/03/13/what-is-iterative-refinement (accessed 2026-08-11) | |
| 562 | +- zfp Compression Ratio and Quality (LLNL) — https://computing.llnl.gov/projects/zfp/zfp-compression-ratio-and-quality (accessed 2026-08-11) | |
| 563 | +- Error Analysis of ZFP Compression for Floating-Point Data (SIAM) — https://epubs.siam.org/doi/10.1137/18M1168832 (accessed 2026-08-11) | |
| 564 | +- Fast Error-bounded Lossy HPC Data Compression with SZ (Di & Cappello, IPDPS 2016) — https://www.mcs.anl.gov/papers/P5437-1115.pdf (accessed 2026-08-11) | |
| 565 | +- Embedded zerotrees of wavelet transforms (EZW) — https://en.wikipedia.org/wiki/Embedded_zerotrees_of_wavelet_transforms (accessed 2026-08-11) | |
| 566 | +- Wavelet and image compression: EZW / SPIHT / JPEG2000-EBCOT lecture notes (Cagnazzo, Télécom Paris) — https://perso.telecom-paristech.fr/tupin/ATHENS/COURSES/wavelet_athens_2012.pdf (accessed 2026-08-11) | |
| 567 | +- Progressive Meshes (Hoppe, SIGGRAPH 1996) — https://www.cs.jhu.edu/~misha/ReadingSeminar/Papers/Hoppe96.pdf (accessed 2026-08-11) | |
| 568 | +- Nanite Virtualized Geometry (Unreal Engine documentation) — https://dev.epicgames.com/documentation/unreal-engine/nanite-virtualized-geometry-in-unreal-engine?lang=en-US (accessed 2026-08-11) | |
| 569 | +- BlinkDB: Queries with Bounded Errors and Bounded Response Times on Very Large Data (EuroSys 2013) — https://dl.acm.org/doi/10.1145/2465351.2465355 (accessed 2026-08-11) | |
| 570 | +- Readings in Database Systems (Red Book) ch. 8: Interactive Analytics — online aggregation & AQP context — http://www.redbook.io/ch8-interactive.html (accessed 2026-08-11) | |
| 571 | +- Unweight: how we compressed an LLM 22% without sacrificing quality (Cloudflare engineering, bandwidth-bound inference evidence) — https://blog.cloudflare.com/unweight-tensor-compression (accessed 2026-08-11) | |
| 572 | +- mlx.core.quantize documentation (supported modes, group sizes, bit widths) — https://ml-explore.github.io/mlx/build/html/python/_autosummary/mlx.core.quantize.html (accessed 2026-08-11) | |
added
research/notes/out_of_core_memory_systems.md
+578 −0
@@ -0,0 +1,578 @@ | ||
| 1 | +--- | |
| 2 | +project: localvm-research | |
| 3 | +document: research/notes/out_of_core_memory_systems | |
| 4 | +author: Simon-Pierre Boucher | |
| 5 | +contact: contact@spboucher.ai | |
| 6 | +created: 2026-08-11 | |
| 7 | +status: draft | |
| 8 | +--- | |
| 9 | + | |
| 10 | +# Out-of-core LLM inference and memory systems (charter §4.4 + §4.8) | |
| 11 | + | |
| 12 | +Research notes on out-of-core inference systems, macOS/Apple Silicon substrate facts, and | |
| 13 | +memory-systems ideas from OS/architecture/databases that have not yet been translated to | |
| 14 | +neural-weight execution. Target hardware framing throughout: **Apple M5 Max, 48 GB unified | |
| 15 | +memory, macOS 27, internal Apple NVMe AP2048Z (2 TB), Metal/MLX**. | |
| 16 | + | |
| 17 | +All sources listed in §6 were located and, where marked, fetched on 2026-08-11. | |
| 18 | + | |
| 19 | +--- | |
| 20 | + | |
| 21 | +## 1. Landscape | |
| 22 | + | |
| 23 | +Out-of-core inference means executing a model whose weights (and/or KV cache) exceed fast | |
| 24 | +memory, by staging data across a hierarchy: GPU HBM → CPU DRAM → NVMe → (network). Every | |
| 25 | +published system is a point in a small design space defined by four questions: | |
| 26 | + | |
| 27 | +1. **What is the unit of paging?** Whole layers (AirLLM, ZeRO-Inference), tensors/blocks | |
| 28 | + (FlexGen, llama.cpp pages), neurons/neuron clusters (PowerInfer-1/2, LLM-in-a-flash, | |
| 29 | + M2Cache), experts (MoE offloading), or KV blocks (SolidAttention, InstInfer, Swarm). | |
| 30 | +2. **What decides residency?** Static placement (FlexGen's linear program, PowerInfer's | |
| 31 | + offline hot/cold profiling), OS demand paging (llama.cpp mmap), explicit LRU/segmented | |
| 32 | + caches (Eliseev–Mazur, PowerInfer-2, M2Cache), or predictors (LLM-in-a-flash sparsity | |
| 33 | + predictors, speculative expert prefetch). | |
| 34 | +3. **What hides transfer latency?** Batch-level pipelining (FlexGen, ZeRO-Inference), | |
| 35 | + compute/I-O overlap at neuron-cluster granularity (PowerInfer-2), speculative prefetch | |
| 36 | + (Eliseev–Mazur, FlexInfer), or nothing (AirLLM). | |
| 37 | +4. **What is the true bottleneck?** Almost never capacity. For dense models streamed per | |
| 38 | + token it is **storage bandwidth**; for sparse/selective access it is **small-random-read | |
| 39 | + latency and read amplification**; for discrete-GPU systems it is **PCIe bandwidth and | |
| 40 | + CPU↔GPU synchronization**; for batch-throughput systems the bottleneck is deliberately | |
| 41 | + traded against latency. | |
| 42 | + | |
| 43 | +A key structural observation for this project: **most of the literature assumes | |
| 44 | +Linux + discrete GPU**, where there are two memory tiers above storage (VRAM and DRAM) | |
| 45 | +joined by PCIe (~16–64 GB/s). On Apple Silicon the VRAM/DRAM distinction vanishes | |
| 46 | +(unified memory), so the hierarchy collapses to exactly two tiers: **unified RAM | |
| 47 | +(~460–614 GB/s on M5 Max) and internal NVMe (~5–7 GB/s sequential, far less for small | |
| 48 | +random reads)**. That is a ~100:1 bandwidth cliff with no intermediate tier — but also no | |
| 49 | +PCIe copy, no pinned-memory staging, and a legitimate zero-copy path from mmap'd file | |
| 50 | +pages into GPU-visible Metal buffers. Several published designs (FlexGen's DRAM tier, | |
| 51 | +PowerInfer's GPU/CPU split) are meaningless on this substrate; others (LLM-in-a-flash, | |
| 52 | +PowerInfer-2) map almost directly onto it. | |
| 53 | + | |
| 54 | +The single most important accounting identity (charter §2/§8.3): out-of-core systems | |
| 55 | +decouple *capacity* (checkpoint can exceed RAM) but only the sparsity/selectivity systems | |
| 56 | +decouple **bytes-read-per-token** from checkpoint size. AirLLM proves capacity decoupling | |
| 57 | +alone is nearly worthless (full checkpoint read per token → minutes/token); LLM-in-a-flash | |
| 58 | +and MoE-offloading prove bytes/token decoupling is where all the leverage is. | |
| 59 | + | |
| 60 | +--- | |
| 61 | + | |
| 62 | +## 2. Systems | |
| 63 | + | |
| 64 | +### 2.1 FlexGen (ICML 2023) | |
| 65 | + | |
| 66 | +- **Mechanism:** Offloading for *throughput-oriented* (latency-insensitive) generation on | |
| 67 | + one commodity GPU. Formulates tensor placement/compute scheduling across GPU/CPU/disk as | |
| 68 | + a linear program over a "zig-zag" block schedule; compresses weights and KV cache to 4 | |
| 69 | + bit. OPT-175B on a single 16 GB T4-class GPU at ~1 token/s *effective* throughput with | |
| 70 | + very large batches. | |
| 71 | +- **Resident vs streamed:** weights partitioned across GPU/CPU/NVMe by the LP; entire | |
| 72 | + active layer streamed per use; KV cache also tiered. | |
| 73 | +- **Bottleneck:** PCIe bandwidth + disk bandwidth, amortized over huge batches; per-request | |
| 74 | + latency is minutes. Bytes/token for the *batch* ≈ full model per layer-pass, divided by | |
| 75 | + batch size per request. | |
| 76 | +- **Hardware assumptions:** Linux, discrete GPU, separate CPU DRAM tier, pinned-memory DMA. | |
| 77 | + None of this maps to Apple Silicon; the LP-placement idea does (placement between RAM | |
| 78 | + and SSD), the batching trick does not help interactive local use. | |
| 79 | +- **macOS arm64:** not supported (CUDA-only). | |
| 80 | +- **Extension opportunity:** the *formalization* — solve residency as an optimization | |
| 81 | + problem given measured tier bandwidths — is reusable for a RAM/SSD split on Mac. | |
| 82 | + | |
| 83 | +### 2.2 DeepSpeed-Inference / ZeRO-Inference / DeepNVMe | |
| 84 | + | |
| 85 | +- **Mechanism:** ZeRO-Inference streams model weights layer-by-layer from CPU DRAM or NVMe | |
| 86 | + into GPU, overlapping fetch of layer *k+1* with compute of layer *k*; recent versions add | |
| 87 | + weight quantization and KV-cache offload ("20× throughput" claims). DeepNVMe adds | |
| 88 | + io_uring/NVIDIA-GDS-based direct NVMe↔GPU transfer paths. Reported: OPT-175B full-offload | |
| 89 | + ~43 tokens/s (CPU-offload) and ~30 tokens/s (NVMe) *aggregate* with large batches; | |
| 90 | + prefetching improves NVMe offload 1.13–1.21×. | |
| 91 | +- **Resident vs streamed:** essentially nothing resident; **entire model read per forward | |
| 92 | + pass** — bytes/token = checkpoint size unless batched. | |
| 93 | +- **Bottleneck:** PCIe/NVMe *bandwidth* (sequential; layer-granular reads are large, so | |
| 94 | + latency and random I/O are non-issues). Throughput scales with batch size only. | |
| 95 | +- **Hardware assumptions:** Linux (io_uring, O_DIRECT, GDS), discrete GPU. io_uring and | |
| 96 | + O_DIRECT do not exist on macOS. | |
| 97 | +- **Extension opportunity:** demonstrates the hard ceiling of dense layer-streaming: | |
| 98 | + tokens/s ≤ storage_bandwidth / bytes_not_resident. On our SSD (~5–7 GB/s), a 70 GB | |
| 99 | + non-resident model can never exceed ~0.1 tok/s dense, no matter the engineering. | |
| 100 | + | |
| 101 | +### 2.3 "LLM in a flash" (Apple, arXiv 2312.11514) — the closest prior art to this project | |
| 102 | + | |
| 103 | +- **Mechanism:** keep dense attention weights + embeddings in DRAM (~50 % of model), keep | |
| 104 | + FFN weights on flash; a low-rank predictor (r=128 early layers → r=1024 late layers, | |
| 105 | + <2.4 % overhead) predicts which ReLU-sparse FFN neurons will activate; only those rows | |
| 106 | + are read. **Windowing** (reuse neurons active in the last k=4 tokens; only load the | |
| 107 | + delta) and **row–column bundling** (store up-proj column + down-proj row of the same | |
| 108 | + neuron contiguously, doubling read chunk size) reduce and enlarge I/O respectively. | |
| 109 | +- **What's resident vs streamed:** ~52 % of model in DRAM (OPT-6.7B: 52.1 %, Falcon-7B: | |
| 110 | + 52.9 %); per token only **2.4 % (OPT) / 3.1 % (Falcon) of FFN neurons** are loaded. | |
| 111 | +- **Measured (on macOS!):** Apple **M1 Max, 1 TB SSD** and M2 Ultra 2 TB: >6 GiB/s for a | |
| 112 | + 1 GiB linear read, but random-read throughput *increases with chunk size and thread | |
| 113 | + count*; effective ~2.25 GB/s after bundling. I/O latency per token dropped from 2196 ms | |
| 114 | + (naive half-model reload) to **105 ms** on M1 Max. 4–5× CPU and 20–25× GPU speedup vs | |
| 115 | + naive loading; runs models ~2× DRAM size. | |
| 116 | +- **Bottleneck:** flash random-read latency / read amplification — attacked by making | |
| 117 | + reads bigger (bundling) and fewer (windowing, predictor). | |
| 118 | +- **Limitations:** depends on **ReLU-level activation sparsity** (OPT, ReLU-fied Falcon); | |
| 119 | + modern SwiGLU models (Llama-3, Qwen) have far weaker natural sparsity. No public code. | |
| 120 | + Only ~2× DRAM demonstrated. | |
| 121 | +- **Extension opportunity:** it validates the whole premise of localvm-research *on our | |
| 122 | + exact platform*, and its flash-throughput-vs-chunk-size measurements are directly | |
| 123 | + reusable priors for expH. Open gap: doing this for SwiGLU/MoE models, and integrating | |
| 124 | + with Metal GPU compute rather than CPU. | |
| 125 | + | |
| 126 | +### 2.4 PowerInfer (SOSP 2024) | |
| 127 | + | |
| 128 | +- **Mechanism:** neuron activations follow a power law: a small set of "hot" neurons fire | |
| 129 | + for most inputs. Hot neurons are preloaded into limited GPU VRAM; cold neurons are | |
| 130 | + computed on CPU from DRAM (avoiding PCIe transfer); adaptive per-layer activation | |
| 131 | + predictors + neuron-aware sparse kernels. Up to 11.69× over llama.cpp on a 4090; runs | |
| 132 | + OPT-175B on one consumer GPU. | |
| 133 | +- **Resident vs streamed:** hot neurons resident in VRAM, cold in DRAM; nothing streamed | |
| 134 | + from disk in the base design (model ≤ DRAM assumed). | |
| 135 | +- **Bottleneck addressed:** PCIe transfer + VRAM capacity; **assumes model fits in DRAM** | |
| 136 | + — it is a VRAM/DRAM tiering system, not a true out-of-core system. | |
| 137 | +- **macOS mapping:** the GPU/CPU split is meaningless under unified memory; the analog is | |
| 138 | + **hot neurons wired in RAM / cold neurons on SSD**, i.e., exactly the LLM-in-a-flash | |
| 139 | + regime. The transferable asset is the *power-law hotness statistics* and the offline | |
| 140 | + profiling methodology (feeds our expA/expC). | |
| 141 | + | |
| 142 | +### 2.5 PowerInfer-2 (arXiv 2406.06282) | |
| 143 | + | |
| 144 | +- **Mechanism:** smartphone (24 GB Snapdragon, UFS 4.0 flash) inference of models beyond | |
| 145 | + DRAM. Decomposes matmuls into **neuron clusters**; NPU handles dense prefill with large | |
| 146 | + clusters, CPU handles sparse decode with small clusters; **segmented neuron cache** with | |
| 147 | + per-segment policies; fine-grained I/O/compute pipelining at cluster granularity; I/O | |
| 148 | + reads sized/aligned to flash characteristics. | |
| 149 | +- **Reported:** first to run a 47B model on a phone; up to 11.68 tokens/s for TurboSparse- | |
| 150 | + Mixtral-47B; ~22× faster than llama.cpp-class baselines when the model doesn't fit. | |
| 151 | +- **Bottleneck:** UFS random-read latency + bandwidth, hidden with cluster-level pipelining; | |
| 152 | + synchronization overhead of fine-grained pipelining explicitly engineered around. | |
| 153 | +- **Hardware assumptions:** Android/Linux, heterogeneous NPU/CPU, UFS (much slower than | |
| 154 | + Apple NVMe). **This is the closest architectural template for a Mac**: single shared | |
| 155 | + memory pool + flash, no discrete GPU. Its flash is ~4× slower than ours; unified-memory | |
| 156 | + Metal compute is far stronger than a phone CPU — the design should transfer favorably. | |
| 157 | +- **Limitation:** requires activation-sparse ("TurboSparse") fine-tuned model variants — | |
| 158 | + violates our "original model is the source model" constraint unless sparsity is achieved | |
| 159 | + post-hoc. | |
| 160 | + | |
| 161 | +### 2.6 M2Cache (arXiv 2410.14740) | |
| 162 | + | |
| 163 | +- **Mechanism:** neuron-level **mixed-precision** + **three-tier cache**: neuron-level | |
| 164 | + mixed-precision LRU cache in GPU HBM → layer-aware DRAM cache → full model on SSD. A | |
| 165 | + predictor scores neuron activity; important neurons run FP16, less active ones are | |
| 166 | + quantized more aggressively or demoted. | |
| 167 | +- **Relevance:** first system I found that combines *precision* and *tier* as one axis — | |
| 168 | + i.e., the residency decision and the precision decision are unified. Directly relevant | |
| 169 | + to charter expD (progressive reconstruction): the "compressed tier" idea (§4.8 zswap | |
| 170 | + analogy) already has one instantiation. | |
| 171 | +- **Bottleneck:** SSD bandwidth; hidden with precision reduction (fewer bytes) rather than | |
| 172 | + only prediction. | |
| 173 | +- **macOS:** CUDA; concepts portable. | |
| 174 | + | |
| 175 | +### 2.7 SolidAttention (FAST 2026) | |
| 176 | + | |
| 177 | +- **Mechanism:** KV-cache (not weights) on SSD for memory-constrained PCs; identifies the | |
| 178 | + conflict between *dynamic sparse attention* (wants small random reads) and *SSD | |
| 179 | + characteristics* (want large sequential reads); consolidates KV pairs into blocks as the | |
| 180 | + transfer unit, transforming irregular access into coarse-grained sequential access. | |
| 181 | + Up to 3.1× faster inference, 98 % KV memory reduction at 128k context, ≤11 % throughput | |
| 182 | + degradation vs in-memory. | |
| 183 | +- **Lesson:** identical shape to LLM-in-a-flash's bundling lesson, proven for KV instead of | |
| 184 | + weights: **on flash, the paging unit must be chosen by the storage medium, not by the | |
| 185 | + model's natural granularity**. Any localvm design must co-design block layout with the | |
| 186 | + ~16–256 KB sweet spot of Apple NVMe. | |
| 187 | + | |
| 188 | +### 2.8 InstInfer / computational storage (arXiv 2409.04992) | |
| 189 | + | |
| 190 | +- **Mechanism:** offloads attention computation *into* CSD (computational storage drives), | |
| 191 | + so KV never crosses the host bus. Near-storage inference. | |
| 192 | +- **macOS relevance:** none directly (Apple SSD controllers are closed), but conceptually: | |
| 193 | + Apple's SSD controller already does inline encryption/compression; "compute where data | |
| 194 | + lives" on a Mac translates to *decompression/dequantization on GPU at load time*, which | |
| 195 | + MLX quantized kernels already do. | |
| 196 | + | |
| 197 | +### 2.9 llama.cpp (mmap + Metal) — most important existing macOS evidence | |
| 198 | + | |
| 199 | +- **Mechanism:** since PR #613/issue #91 (2023), model files are `mmap`'d; loading is | |
| 200 | + lazy via page faults and the OS page cache (100× faster warm loads, half the memory — | |
| 201 | + weights live once in the unified buffer cache, shared across processes). On Apple | |
| 202 | + Silicon, the **Metal backend accesses the mmap'd weights zero-copy via | |
| 203 | + `MTLResourceStorageModeShared`** (ggerganov: Metal "looks directly at the memory mapped | |
| 204 | + buffers"; CUDA by contrast copies into device buffers). GGUF tensor data is alignment- | |
| 205 | + padded, which is what makes wrapping file pages in Metal buffers possible. | |
| 206 | +- **Out-of-core behavior:** when a model (esp. MoE) exceeds RAM, execution *works* by | |
| 207 | + OS demand paging: hot expert pages stay in the UBC, cold ones fault in from SSD. | |
| 208 | + Discussion #18758 (Dec 2025–2026) measured this concretely, including **on an M5 Pro | |
| 209 | + (Apple SSD AP1024Z) with Qwen3-Next-80B-A3B**: expert weights are 95.4 % of file bytes; | |
| 210 | + replacing fault-based streaming with explicit layout-aware slice reads gave +13–14 % | |
| 211 | + end-to-end and reduced cold-decode reads from 1418 to 370 per token (**2.23× faster | |
| 212 | + cold-decode I/O**); mmap beat direct I/O because the page cache retains the hot expert | |
| 213 | + working set across tokens/runs. | |
| 214 | +- **Bottleneck:** SSD random-read amplification (16 KB page faults scattered across expert | |
| 215 | + tensors) + page-cache eviction unpredictability. Layout (contiguous per-expert | |
| 216 | + placement) is as important as caching policy. | |
| 217 | +- **Limitations:** replacement policy is the kernel's (approximate LRU, opaque, | |
| 218 | + scan-vulnerable); no model-aware prefetch (a router decision is known *before* the | |
| 219 | + expert FFN runs, but nothing uses it); feature request #20757 (two-tier GPU+RAM expert | |
| 220 | + cache with pluggable eviction) is open — i.e., **the gap is acknowledged and unfilled**. | |
| 221 | +- **Runs on macOS arm64:** yes, first-class. | |
| 222 | + | |
| 223 | +### 2.10 MLX (Apple) | |
| 224 | + | |
| 225 | +- **Mechanism:** arrays live in unified memory; device chosen *per operation* | |
| 226 | + (`stream=mx.cpu/mx.gpu`) with automatic cross-stream dependencies; **lazy evaluation** | |
| 227 | + builds a graph and materializes arrays only when needed; `mx.load` on safetensors/GGUF | |
| 228 | + is lazy — weights are read from file when first evaluated, not all at load. | |
| 229 | +- **Out-of-core status:** none. Maintainer (awni, discussion #615) states mmap wouldn't | |
| 230 | + solve the problem since weights must still be materialized in (wired) memory for GPU | |
| 231 | + use; lazy loading is the offered substitute. A community mmap prototype (antbob) found | |
| 232 | + the practical blockers: safetensors tensor offsets are not page-aligned (Metal | |
| 233 | + `bytesNoCopy` needs page alignment), and once the model exceeds RAM, uncontrolled page- | |
| 234 | + cache eviction collapsed a 70 GB model on 64 GB hardware to **0.025 tokens/s** (vs 6 | |
| 235 | + tok/s quantized-fits-in-RAM). This is the negative result that defines our problem: naive | |
| 236 | + mmap + kernel LRU is catastrophically bad for cyclic dense weight access. | |
| 237 | +- **Memory control:** `mx.set_wired_limit` / Metal **residency sets** pin working memory; | |
| 238 | + macOS `iogpu.wired_limit_mb` caps total GPU-wired memory (default ~66–75 % of RAM). | |
| 239 | + mlx-lm issue #883 documents the failure mode when wiring is unbounded: IOGPUMemory | |
| 240 | + kernel panic — wired memory is invisible to compressor/jetsam. | |
| 241 | +- **Extension opportunity:** MLX is the natural host for our runtime (custom Metal | |
| 242 | + kernels, lazy graph, quantized matmuls), but every paging/caching mechanism must be | |
| 243 | + built *by us* — MLX offers none. | |
| 244 | + | |
| 245 | +### 2.11 MoE offloading line: Eliseev & Mazur; caching/prefetch analyses; MoBiLE; cache-conditional experts | |
| 246 | + | |
| 247 | +- **Eliseev & Mazur (arXiv 2312.17238):** Mixtral-8x7B on 11–16 GB consumer GPUs. | |
| 248 | + Exploits (a) temporal locality of expert choice between adjacent tokens → **LRU expert | |
| 249 | + cache**; (b) hidden state of layer *k* already predicts layer *k+1*'s router choice → | |
| 250 | + **speculative expert prefetch** (apply next layer's gate to current hidden state). | |
| 251 | + 2–3 tokens/s on T4/RTX 3060-class GPUs with mixed quantization. This is "branch | |
| 252 | + prediction for weights" in embryonic form. | |
| 253 | +- **In-depth caching/prefetching analysis (arXiv 2511.05814):** measures expert reuse and | |
| 254 | + prefetch accuracy across MoE models — confirms LRU-friendly temporal locality and | |
| 255 | + cross-layer predictability are general, not Mixtral quirks. | |
| 256 | +- **Mixture of cache-conditional experts (arXiv 2412.00099):** *inverts* the problem — | |
| 257 | + biases the router toward experts already in cache (cache-aware routing), trading a | |
| 258 | + little quality for dramatically fewer misses. Notable: the model adapts to the memory | |
| 259 | + system rather than vice versa. | |
| 260 | +- **MoBiLE (arXiv 2510.12357):** "big/little" experts on consumer GPUs — fallback to | |
| 261 | + smaller substitutes for missing experts rather than blocking on I/O. A | |
| 262 | + quality-for-latency miss handler — architecturally interesting for us: a miss need not | |
| 263 | + stall if a low-precision resident approximation exists (connects to expD). | |
| 264 | + | |
| 265 | +### 2.12 Petals (BitTorrent-style distributed inference) | |
| 266 | + | |
| 267 | +- **Mechanism:** transformer blocks sharded across volunteer consumer GPUs; activations | |
| 268 | + forwarded peer-to-peer (DHT discovery); Llama-3.1-405B / BLOOM-176B at ~1 step/s — | |
| 269 | + claimed up to 10× faster than local disk offloading. | |
| 270 | +- **Relevance:** replaces the SSD tier with a network tier (both ~GB/s, both high-latency) | |
| 271 | + — confirms that *activations are the cheap thing to move; weights are the expensive | |
| 272 | + thing*. For localvm the analogous observation: moving hidden states between compute | |
| 273 | + contexts is ~MB/token; moving weights is ~GB/token. Any decomposition should ship | |
| 274 | + activations, not weights. (Also relevant to the MacLustr cluster as a side path, though | |
| 275 | + out of scope for the single-Mac charter.) | |
| 276 | + | |
| 277 | +### 2.13 AirLLM (layer-by-layer streaming) | |
| 278 | + | |
| 279 | +- **Mechanism:** load layer → compute → free → next layer; peak memory ~4 GB for a 70B | |
| 280 | + model. Bytes/token = **entire checkpoint** (every layer re-read per token unless cached); | |
| 281 | + a single response takes 15–30 minutes. | |
| 282 | +- **Value:** the perfect *straw-man baseline* for our harness — pure capacity decoupling | |
| 283 | + with zero bytes/token decoupling. Its existence proves the charter's core inequality | |
| 284 | + (total size ≠ resident size ≠ bytes/token) is the entire game. | |
| 285 | + | |
| 286 | +### 2.14 FlexInfer (arXiv 2503.03777) and Glinthawk (arXiv 2501.11779) | |
| 287 | + | |
| 288 | +- **FlexInfer:** on-device offloading with asynchronous prefetching, **balanced memory | |
| 289 | + locking** (explicitly budgeting pinned vs pageable memory), and flexible tensor | |
| 290 | + preservation; up to 12.5× over existing offloading under tight memory. The | |
| 291 | + "balanced memory locking" idea maps directly onto the macOS wired-limit / residency-set | |
| 292 | + tension documented in §3. | |
| 293 | +- **Glinthawk:** two-tier architecture for *offline* batch inference (throughput regime, | |
| 294 | + like FlexGen) — noted for completeness; wrong latency regime for us. | |
| 295 | + | |
| 296 | +--- | |
| 297 | + | |
| 298 | +## 3. macOS / Apple Silicon substrate (API-level facts) | |
| 299 | + | |
| 300 | +### 3.1 Files, page cache, and the absence of direct I/O | |
| 301 | + | |
| 302 | +- **No `O_DIRECT`, no `posix_fadvise`.** The macOS substitute is | |
| 303 | + `fcntl(fd, F_NOCACHE, 1)`: it *hints* that pages should not be cached going forward, but | |
| 304 | + (a) it does **not purge already-cached pages** — subsequent reads still hit them; (b) | |
| 305 | + other processes can keep re-populating the cache; (c) it is per-fd advisory, not a DMA | |
| 306 | + path (Apple dev forums #25464; fio issue #48). Community direct-I/O libraries document | |
| 307 | + that with F_NOCACHE, **unaligned reads are still buffered**; to actually bypass the | |
| 308 | + cache, offset/length should be 4096-byte (better: 16 KB page) aligned (ronomon/direct-io). | |
| 309 | +- **Unified Buffer Cache (UBC):** since Mac OS X, the buffer cache and VM page cache are | |
| 310 | + one; `mmap`'d file pages *are* the file cache pages. Consequences: warm model loads are | |
| 311 | + ~free (llama.cpp's 100× warm-load speedup); memory shows as "cached files," is | |
| 312 | + reclaimable, and is shared across processes mapping the same model file. | |
| 313 | +- **Eviction is opaque:** the kernel's replacement is approximate LRU over the whole | |
| 314 | + system; there is no `fadvise(DONTNEED/WILLNEED)`; `madvise` exists but with weaker | |
| 315 | + semantics. `purge(8)` clears the disk cache for cold-start experiments; `vm_stat` | |
| 316 | + exposes pageins/pageouts/compressor counters for instrumentation (expH must use both). | |
| 317 | +- **Writeback of `mmap(MAP_SHARED)` dirty pages** is at the OS's discretion until | |
| 318 | + `msync` (Apple dev forums #763058) — matters if we ever write compiled-model caches | |
| 319 | + through mmap. | |
| 320 | +- **APFS:** copy-on-write, 4 KB blocks; clones and sparse files are free — useful for | |
| 321 | + storing multiple weight layouts of the same checkpoint without duplicating cold data. | |
| 322 | + (Performance note: APFS metadata ops are slow relative to data reads; large flat blob | |
| 323 | + files with internal indexing beat many-small-files layouts.) | |
| 324 | + | |
| 325 | +### 3.2 Memory: compression, wiring, jetsam | |
| 326 | + | |
| 327 | +- **Compressor:** since OS X 10.9, LRU-cold anonymous pages are compressed (WKdm-family | |
| 328 | + algorithm, tiny 16-entry dictionary, ~2:1 on pointer/integer-rich data) before any swap. | |
| 329 | + **Quantized/fp16 weights are near-incompressible entropy**, so the compressor gives ~0 | |
| 330 | + benefit on weight pages while burning CPU — weight overflow should go to *file-backed* | |
| 331 | + (evict-don't-compress) memory, never anonymous memory. This asymmetry (file-backed pages | |
| 332 | + get dropped, anonymous pages get compressed/swapped) is a design lever. | |
| 333 | +- **Wired memory:** GPU-active buffers must be wired (non-pageable, non-compressible). | |
| 334 | + Cap is `iogpu.wired_limit_mb` — default ≈ 66–75 % of RAM (≈ 32–36 GB on our 48 GB | |
| 335 | + M5 Max), adjustable via `sudo sysctl iogpu.wired_limit_mb=N` (resets on reboot; | |
| 336 | + unsupported by Apple; leave 8–16 GB headroom). MLX exposes `set_wired_limit` / | |
| 337 | + residency sets. Over-wiring does not trigger graceful jetsam — mlx-lm #883 shows it can | |
| 338 | + end in an **IOGPUMemory kernel panic**; macOS prefers compression over jetsam-style | |
| 339 | + killing, but wired memory is exempt from both, hence the panic path. | |
| 340 | +- **Page size is 16 KB** on Apple Silicon — the natural minimum paging unit for any | |
| 341 | + weight-block store (also the fault granularity that produced the 1418 reads/token in | |
| 342 | + llama.cpp #18758). | |
| 343 | + | |
| 344 | +### 3.3 Metal: zero-copy, heaps, purgeability | |
| 345 | + | |
| 346 | +- **`newBufferWithBytesNoCopy` / `makeBuffer(bytesNoCopy:)`** wraps existing memory as a | |
| 347 | + `MTLBuffer` with **no copy**, but the pointer must be page-aligned and the length a | |
| 348 | + multiple of page size, and the memory must come from `mmap`/`vm_allocate` (not | |
| 349 | + `malloc`) (Apple docs; dev forums #8011). This is exactly how llama.cpp gets the GPU to | |
| 350 | + read weights straight out of the page cache. **Design consequence: our compiled weight | |
| 351 | + format must place every independently-pageable block on a 16 KB boundary** (GGUF does | |
| 352 | + alignment padding; safetensors does not — the root cause of MLX's mmap dead-end). | |
| 353 | +- **`MTLStorageModeShared`:** CPU and GPU share the allocation in system memory — the | |
| 354 | + default and correct mode on Apple Silicon (no managed/private copies needed). | |
| 355 | +- **`MTLHeap`:** suballocate many buffers from one allocation; resources can alias; | |
| 356 | + **`setPurgeableState`** on a heap makes its whole backing memory *volatile* — the OS may | |
| 357 | + reclaim it under pressure and tells you on reacquire whether contents survived. **A | |
| 358 | + purgeable MTLHeap is a kernel-cooperative weight cache**: warm blocks live there, the OS | |
| 359 | + reclaims them instead of paging/killing, and we re-fault from SSD on loss. No LLM | |
| 360 | + runtime uses this today (see §4). | |
| 361 | +- **Residency sets / wired limit:** the modern mechanism to guarantee the *hot* tier stays | |
| 362 | + resident during command-buffer execution (MLX's wired-memory doc). | |
| 363 | + | |
| 364 | +### 3.4 Apple NVMe (AP-class) measured behavior | |
| 365 | + | |
| 366 | +- **Sequential:** recent MacBook Pro internal SSDs (AP2048/AP4096-class) measure ~5.4–7.3 | |
| 367 | + GB/s reads (Blackmagic/AmorphousDiskMark reports for M4 Max 2–4 TB; M5-generation press | |
| 368 | + claims ~2× M4 SSD speed). "LLM in a flash" measured **>6 GiB/s for a 1 GiB linear read | |
| 369 | + on an M1 Max 1 TB**. | |
| 370 | +- **Small random reads are the cliff:** community AmorphousDiskMark results consistently | |
| 371 | + show a **4K QD1 dip on Apple Silicon** — on the order of tens of MB/s (one documented | |
| 372 | + M1 Pro result: ~32 MB/s = ~8 K IOPS = ~120 µs effective latency), i.e. **~200× below | |
| 373 | + sequential**. Throughput recovers with (a) larger blocks and (b) concurrency: Apple's | |
| 374 | + paper reports random-read throughput "increases with the size of sequential chunks and | |
| 375 | + the number of threads," reaching ~2.25 GB/s effective with 32 KB-bundled multi-threaded | |
| 376 | + reads. Rule of thumb for design: **≥256 KB blocks at QD≥8, or don't bother**; expH must | |
| 377 | + measure our exact AP2048Z across 16 KB–4 MB, QD1–32, cold (`purge`) vs warm, with and | |
| 378 | + without F_NOCACHE, and *while Metal compute runs* (shared memory-controller contention | |
| 379 | + is unmeasured in the literature). | |
| 380 | +- **Bandwidth ratio on target:** M5 Max unified memory = 460–614 GB/s (per Apple specs; | |
| 381 | + config-dependent) vs ~6 GB/s SSD sequential → **~80–100:1**; vs realistic mixed random | |
| 382 | + ~2 GB/s → **~250:1**. Every design decision follows from this ratio. | |
| 383 | + | |
| 384 | +--- | |
| 385 | + | |
| 386 | +## 4. OS/architecture/database ideas not yet translated to neural-weight execution | |
| 387 | + | |
| 388 | +1. **Working-set theory (Denning 1968).** No LLM runtime measures a τ-window weight | |
| 389 | + working set W(τ), yet the concept transfers exactly: the set of weight blocks touched | |
| 390 | + in the last τ tokens. Denning's thrashing criterion (if RAM < W(τ), throughput | |
| 391 | + collapses) gives a *principled admission test*: measure W(τ) per model/workload offline | |
| 392 | + and predict — before running — whether a config will thrash (this is what MLX's 0.025 | |
| 393 | + tok/s mmap failure was: W(τ) = whole model for dense access). Working-set *analytics* | |
| 394 | + (Denning's later surveys) also give the math for choosing cache sizes from reuse- | |
| 395 | + distance histograms — directly applicable to expB. | |
| 396 | +2. **Scan-resistant / adaptive replacement (ARC, 2Q, LIRS).** Kernel page cache and all | |
| 397 | + published expert caches use (approximate) LRU. But transformer weight access is a | |
| 398 | + *mixed* workload: dense layers are pure cyclic scans (LRU's pathological worst case — | |
| 399 | + it evicts every block just before reuse), while MoE experts have recency+frequency | |
| 400 | + structure. ARC's ghost lists would auto-partition between the two. **Nobody has | |
| 401 | + published an ARC/LIRS weight cache.** Even better than ARC: | |
| 402 | +3. **MRU for cyclic scans (DBMIN's insight).** Databases learned in 1985 that for a | |
| 403 | + looping sequential scan over N pages with a buffer of B<N, **MRU is optimal and LRU is | |
| 404 | + worst-case**. Dense-decode weight access *is* a looping sequential scan (same order | |
| 405 | + every token). A dense model overflowing RAM by X GB under MRU keeps a stable | |
| 406 | + (model−X) resident set and re-reads exactly X GB/token — the theoretical floor — | |
| 407 | + whereas LRU re-reads everything. Trivial to implement, apparently never applied. | |
| 408 | +4. **Query-informed buffer management (DBMIN / QLSM).** DBMIN allocates a *separate | |
| 409 | + buffer pool per file instance with a policy chosen from the known access pattern of the | |
| 410 | + query plan*. A transformer's "query plan" is fully known: layer order is static; MoE | |
| 411 | + router output is known *milliseconds before* the expert is needed; attention head usage | |
| 412 | + is measurable. Per-tensor-class pools (embeddings: pin; dense scan: MRU ring; experts: | |
| 413 | + ARC; KV: sliding window) with plan-derived sizes is a direct, unexplored translation. | |
| 414 | +5. **Anti-caching (H-Store/VLDB 2013).** Inverts caching: memory is primary, cold data is | |
| 415 | + *evicted* to disk with tombstones; a transaction touching evicted data **aborts, the | |
| 416 | + data is fetched asynchronously, and the transaction restarts** — no thread ever blocks | |
| 417 | + on disk. Translation: a token step that needs a non-resident expert could proceed | |
| 418 | + speculatively with a resident low-precision substitute (MoBiLE-style) or abort-and- | |
| 419 | + replay that layer after async fetch, keeping the GPU busy. "Never block compute on a | |
| 420 | + miss" as an architectural invariant has no LLM instantiation. | |
| 421 | +6. **Prefetching as branch prediction (Pythia, MICRO 2021).** Hardware prefetchers use | |
| 422 | + program context + online RL to issue accurate, *bandwidth-aware* prefetches. The LLM | |
| 423 | + analog has richer context than any CPU: hidden states. Eliseev–Mazur's one-layer-ahead | |
| 424 | + gate trick is a static 1-bit predictor by comparison; a small online-learned prefetcher | |
| 425 | + consuming hidden-state features and issuing SSD reads N layers ahead (with reward = | |
| 426 | + hit-rate minus wasted bandwidth, Pythia-style) is unexplored. | |
| 427 | +7. **Tiered-memory page placement (TPP/ASPLOS'23, Pond).** Hot/cold page *promotion and | |
| 428 | + demotion* between DRAM and CXL based on lightweight access sampling. Translation: | |
| 429 | + background promotion/demotion of weight blocks between wired-RAM / purgeable-RAM / SSD | |
| 430 | + tiers driven by per-block access counters — no runtime does tier migration of weights | |
| 431 | + *during* inference; placements are static after profiling (PowerInfer) or purely | |
| 432 | + reactive (page faults). | |
| 433 | +8. **Compressed memory tier (zswap / macOS compressor).** The OS compressor is useless on | |
| 434 | + weight entropy (§3.2), but the *architecture* — a middle tier holding a cheaper | |
| 435 | + representation, faulting to the full representation — translates as: low-bit weights | |
| 436 | + resident in RAM as the "compressed tier," full-precision residuals on SSD, fetched only | |
| 437 | + when needed (error-driven). This is M2Cache's direction and our expD; the OS analogy | |
| 438 | + suggests the policy structure (compress on demotion, decompress on promotion, track | |
| 439 | + compression benefit per page). | |
| 440 | +9. **Purgeable/volatile memory (macOS-specific, §3.3).** Kernel-cooperative caches | |
| 441 | + (volatile heaps the OS may reclaim, with reclaim notification) have existed since iOS's | |
| 442 | + NSPurgeableData era. No inference runtime marks its warm weight cache purgeable; doing | |
| 443 | + so converts "jetsam/panic risk" into "graceful quality/latency degradation under | |
| 444 | + pressure." | |
| 445 | +10. **Economic residency rules (five-minute-rule style).** Databases decide RAM residency | |
| 446 | + by break-even between storage cost and access frequency. Per-block: wire it if | |
| 447 | + (expected accesses/s × fetch cost) exceeds its RAM rent. Gives a closed-form split of | |
| 448 | + a 48 GB budget across embeddings/dense/experts/KV rather than ad-hoc tuning. (See the | |
| 449 | + buffer-management evolution survey for the modern framing incl. learned policies.) | |
| 450 | +11. **TLB/superpage thinking.** 16 KB pages mean a 100 GB mapping has ~6.5 M PTEs; fault | |
| 451 | + storms are Mach-message-expensive. Batching faults via explicit large-block reads into | |
| 452 | + pre-mapped wired arenas (as llama.cpp #18758's slice-read experiment did: 1418→370 | |
| 453 | + reads/token) is the superpage lesson in disguise; nobody states it as policy. | |
| 454 | + | |
| 455 | +--- | |
| 456 | + | |
| 457 | +## 5. Relevance to localvm-research (M5 Max, 48 GB, AP2048Z) | |
| 458 | + | |
| 459 | +**Where the bottleneck actually is on our target.** With 460–614 GB/s RAM and ~6 GB/s | |
| 460 | +(sequential) / ~2 GB/s (practical random) SSD: | |
| 461 | + | |
| 462 | +- *If the active working set fits in RAM*, decode is RAM-bandwidth-bound: | |
| 463 | + ceiling ≈ 460 GB/s ÷ resident-active-bytes. A fully-resident 40 GB (Q4 70B) model | |
| 464 | + → ~11 tok/s ceiling. This is the regime llama.cpp/MLX already serve. | |
| 465 | +- *If a dense model overflows RAM by X GB*, the SSD must supply ≥X GB/token (MRU floor): | |
| 466 | + X = 10 GB → ≥1.7–5 s/token. **Dense overflow is irrecoverable by systems engineering | |
| 467 | + alone** — confirmed by ZeRO-Inference math, AirLLM, and the MLX mmap prototype (0.025 | |
| 468 | + tok/s). Therefore bytes/token must be decoupled from checkpoint size *before* paging can | |
| 469 | + help: activation sparsity (LLM-in-a-flash: 2–3 % of FFN/token), MoE routing (only active | |
| 470 | + experts), or progressive precision (low-bit resident core + on-demand residuals). | |
| 471 | +- *Once access is selective*, the bottleneck flips from bandwidth to **random-read latency | |
| 472 | + and read amplification** — the fight moves to layout (16 KB-aligned, co-located bundles; | |
| 473 | + SolidAttention/row-column-bundling lesson), replacement policy (ARC/MRU vs kernel LRU), | |
| 474 | + and prefetch (router/hidden-state-driven, Pythia-style), with QD≥8 large-block reads | |
| 475 | + overlapped with Metal compute. | |
| 476 | + | |
| 477 | +**What is uniquely favorable on macOS.** (1) Zero-copy SSD→page-cache→GPU: file pages can | |
| 478 | +be wrapped in Metal buffers with no copy (llama.cpp proves it in production) — discrete-GPU | |
| 479 | +systems can't do this, so most published overheads (PCIe staging, pinned pools) simply | |
| 480 | +vanish. (2) Purgeable heaps + residency sets give a three-tier RAM hierarchy | |
| 481 | +(wired-hot / purgeable-warm / evictable page cache) that no other OS exposes as cleanly. | |
| 482 | +(3) The page cache is shared and persistent across processes/runs — warm-start economics | |
| 483 | +are excellent. What is uniquely *unfavorable*: no O_DIRECT/io_uring (F_NOCACHE is a weak | |
| 484 | +hint; async I/O = thread pools or POSIX AIO), opaque eviction, 16 KB fault granularity, | |
| 485 | +and the wired-limit/panic cliff. | |
| 486 | + | |
| 487 | +**Three most promising openings** (ranked): | |
| 488 | + | |
| 489 | +1. **Model-aware weight pager for MoE/sparse models on Metal** — a compiled, 16 KB-aligned | |
| 490 | + block store (APFS-friendly single blob) + ARC/ghost-list expert cache in a purgeable | |
| 491 | + MTLHeap + router-lookahead prefetch (gate of layer k+1 applied at layer k, à la | |
| 492 | + Eliseev–Mazur, generalized to N-layer Pythia-style learned prefetch) + explicit QD≥8 | |
| 493 | + large-slice reads instead of fault streaming. Every ingredient has isolated evidence | |
| 494 | + (llama.cpp #18758: 2.23× cold I/O from layout alone; #20757 open feature request; MoE | |
| 495 | + temporal locality confirmed); the composition doesn't exist anywhere, least of all on | |
| 496 | + macOS. | |
| 497 | +2. **DBMIN-style per-tensor-class buffer management with working-set admission control** — | |
| 498 | + per-class policies (pin embeddings; MRU ring for dense cyclic scans; ARC for experts; | |
| 499 | + economic wiring of the hot tier under the iogpu wired limit), driven by offline W(τ) | |
| 500 | + profiles (feeds directly on expA/expB/expC outputs). Cheap to build, high novelty (the | |
| 501 | + MRU-for-cyclic-weight-scans observation appears to be unpublished), and it defines the | |
| 502 | + measurement framework we need anyway. | |
| 503 | +3. **Precision-tiered residency ("compressed tier for weights")** — low-bit core resident | |
| 504 | + and wired; residuals on SSD fetched on demand (error- or importance-driven), misses | |
| 505 | + served by the resident approximation so compute never blocks (anti-caching invariant, | |
| 506 | + MoBiLE fallback). This is the systems-side realization of expD and the only opening | |
| 507 | + that helps *dense* models, where pure paging provably cannot. | |
| 508 | + | |
| 509 | +Immediate experimental consequences: expH (SSD feasibility) should replicate Apple's | |
| 510 | +chunk-size×threads throughput surface on the AP2048Z including concurrent-Metal-compute | |
| 511 | +contention; expB (token stability) should be scored as reuse-distance histograms so | |
| 512 | +working-set/ARC math applies directly; and the baseline harness must include llama.cpp | |
| 513 | +mmap-overflow and AirLLM-style layer streaming as the two "pure capacity decoupling" | |
| 514 | +controls. | |
| 515 | + | |
| 516 | +--- | |
| 517 | + | |
| 518 | +## Sources | |
| 519 | + | |
| 520 | +- FlexGen: High-Throughput Generative Inference of Large Language Models with a Single GPU — https://arxiv.org/abs/2303.06865 (accessed 2026-08-11) | |
| 521 | +- FlexLLMGen (FlexGen) README, FMInference — https://github.com/FMInference/FlexLLMGen/blob/main/README.md (accessed 2026-08-11) | |
| 522 | +- ZeRO-Inference: Democratizing massive model inference — https://www.deepspeed.ai/2022/09/09/zero-inference.html (accessed 2026-08-11) | |
| 523 | +- DeepSpeed Inference: Enabling Efficient Inference of Transformer Models at Unprecedented Scale — https://arxiv.org/pdf/2207.00032 (accessed 2026-08-11) | |
| 524 | +- DeepNVMe: Affordable I/O scaling for Deep Learning Applications (PyTorch blog) — https://pytorch.org/blog/deepnvme-affordable-i-o-scaling-for-deep-learning-applications/ (accessed 2026-08-11) | |
| 525 | +- LLM in a flash: Efficient Large Language Model Inference with Limited Memory — https://arxiv.org/abs/2312.11514 (accessed 2026-08-11) | |
| 526 | +- LLM in a flash (HTML full text, hardware/throughput details) — https://arxiv.org/html/2312.11514v3 (accessed 2026-08-11) | |
| 527 | +- PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU — https://arxiv.org/abs/2312.12456 (accessed 2026-08-11) | |
| 528 | +- PowerInfer (SOSP '24 proceedings) — https://dl.acm.org/doi/10.1145/3694715.3695964 (accessed 2026-08-11) | |
| 529 | +- PowerInfer-2: Fast Large Language Model Inference on a Smartphone — https://arxiv.org/abs/2406.06282 (accessed 2026-08-11) | |
| 530 | +- PowerInfer-2 project page — https://powerinfer.ai/v2/ (accessed 2026-08-11) | |
| 531 | +- M2Cache: Harnessing Your DRAM and SSD for Sustainable and Accessible LLM Inference with Mixed-Precision and Multi-level Caching — https://arxiv.org/abs/2410.14740 (accessed 2026-08-11) | |
| 532 | +- SolidAttention: Low-Latency SSD-based Serving on Memory-Constrained PCs (FAST '26) — https://www.usenix.org/system/files/fast26-zheng.pdf (accessed 2026-08-11) | |
| 533 | +- InstInfer: In-Storage Attention Offloading for Cost-Effective Long-Context LLM Inference — https://arxiv.org/pdf/2409.04992 (accessed 2026-08-11) | |
| 534 | +- Swarm: Co-Activation Aware KVCache Offloading Across Multiple SSDs — https://arxiv.org/html/2603.17803v1 (accessed 2026-08-11) | |
| 535 | +- FlexInfer: Breaking Memory Constraint via Flexible and Efficient Offloading for On-Device LLM Inference — https://arxiv.org/abs/2503.03777 (accessed 2026-08-11) | |
| 536 | +- Glinthawk: A Two-Tiered Architecture for Offline LLM Inference — https://arxiv.org/pdf/2501.11779 (accessed 2026-08-11) | |
| 537 | +- Fast Inference of Mixture-of-Experts Language Models with Offloading (Eliseev & Mazur) — https://arxiv.org/pdf/2312.17238 (accessed 2026-08-11) | |
| 538 | +- In-Depth Analysis on Caching and Pre-Fetching in Mixture of Experts Offloading — https://arxiv.org/pdf/2511.05814 (accessed 2026-08-11) | |
| 539 | +- Mixture of Cache-Conditional Experts for Efficient Mobile Device Inference — https://arxiv.org/pdf/2412.00099 (accessed 2026-08-11) | |
| 540 | +- MoBiLE: Efficient Mixture-of-Experts Inference on Consumer GPU with Mixture of Big Little Experts — https://arxiv.org/pdf/2510.12357 (accessed 2026-08-11) | |
| 541 | +- Petals: Run LLMs at home, BitTorrent-style — https://github.com/bigscience-workshop/petals (accessed 2026-08-11) | |
| 542 | +- Petals project page — https://petals.dev/ (accessed 2026-08-11) | |
| 543 | +- AirLLM and "70B on a 4GB GPU" — What's Actually Going On? — https://rohit-shirke.medium.com/airllm-and-70b-on-a-4gb-gpu-whats-actually-going-on-3bf0e102252e (accessed 2026-08-11) | |
| 544 | +- llama.cpp: Should use mmap for model loading (issue #91) — https://github.com/ggml-org/llama.cpp/issues/91 (accessed 2026-08-11) | |
| 545 | +- llama.cpp: Memory-mapping weights while loading the model (discussion #9999) — https://github.com/ggml-org/llama.cpp/discussions/9999 (accessed 2026-08-11) | |
| 546 | +- llama.cpp: Mmap faster than direct I/O for MoE models (discussion #18758, incl. M5 Pro/AP1024Z expert-layout measurements) — https://github.com/ggml-org/llama.cpp/discussions/18758 (accessed 2026-08-11) | |
| 547 | +- llama.cpp: Share readonly GPU model weights across processes — Metal reads mmap buffers via MTLResourceStorageModeShared (discussion #21223) — https://github.com/ggml-org/llama.cpp/discussions/21223 (accessed 2026-08-11) | |
| 548 | +- llama.cpp: Two-tier GPU+RAM expert cache for MoE offload, pluggable eviction (issue #20757) — https://github.com/ggml-org/llama.cpp/issues/20757 (accessed 2026-08-11) | |
| 549 | +- llama.cpp: Avoid memcpy for mmap-ed weights on Unified Memory architectures (issue #21827) — https://github.com/ggml-org/llama.cpp/issues/21827 (accessed 2026-08-11) | |
| 550 | +- Performant local mixture-of-experts CPU inference with GPU acceleration in llama.cpp (HF blog) — https://huggingface.co/blog/Doctor-Shotgun/llamacpp-moe-offload-guide (accessed 2026-08-11) | |
| 551 | +- MLX Unified Memory documentation — https://ml-explore.github.io/mlx/build/html/usage/unified_memory.html (accessed 2026-08-11) | |
| 552 | +- MLX: Loading models with mmap (discussion #615, incl. 70GB-on-64GB 0.025 tok/s prototype result) — https://github.com/ml-explore/mlx/discussions/615 (accessed 2026-08-11) | |
| 553 | +- mlx-swift wired-memory documentation (residency/wired limit) — https://github.com/ml-explore/mlx-swift/blob/main/Source/MLX/Documentation.docc/Articles/wired-memory.md (accessed 2026-08-11) | |
| 554 | +- mlx-lm: mlx_lm.server causes macOS kernel panic (IOGPUMemory) via unbounded wired growth (issue #883) — https://github.com/ml-explore/mlx-lm/issues/883 (accessed 2026-08-11) | |
| 555 | +- fcntl F_NOCACHE option behavior (Apple Developer Forums thread 25464) — https://developer.apple.com/forums/thread/25464 (accessed 2026-08-11) | |
| 556 | +- OSX fcntl(fd, F_NOCACHE, 1) not equivalent to O_DIRECT on Linux (fio issue #48) — https://github.com/axboe/fio/issues/48 (accessed 2026-08-11) | |
| 557 | +- ronomon/direct-io: Direct IO helpers for FreeBSD, Linux, macOS, Windows (F_NOCACHE alignment notes) — https://github.com/ronomon/direct-io (accessed 2026-08-11) | |
| 558 | +- makeBuffer(bytesNoCopy:length:options:deallocator:) — Apple Developer Documentation — https://developer.apple.com/documentation/metal/mtldevice/makebuffer(bytesnocopy:length:options:deallocator:) (accessed 2026-08-11) | |
| 559 | +- MTLStorageMode.shared — Apple Developer Documentation — https://developer.apple.com/documentation/metal/mtlstoragemode/shared (accessed 2026-08-11) | |
| 560 | +- MTLHeap (incl. setPurgeableState) — Apple Developer Documentation — https://developer.apple.com/documentation/metal/mtlheap (accessed 2026-08-11) | |
| 561 | +- newBufferWithBytesNoCopy pointer alignment requirement (Apple Developer Forums thread 8011) — https://developer.apple.com/forums/thread/8011 (accessed 2026-08-11) | |
| 562 | +- iOS/macOS writeback behavior for mmap(MAP_SHARED) dirty pages (Apple Developer Forums thread 763058) — https://developer.apple.com/forums/thread/763058 (accessed 2026-08-11) | |
| 563 | +- How to Increase VRAM Allocation on Apple Silicon Mac (iogpu.wired_limit_mb) — https://osxdaily.com/2025/05/07/how-to-increase-vram-allocation-on-apple-silicon-mac/ (accessed 2026-08-11) | |
| 564 | +- Adjust wired limits to allocate more memory to the GPU with Apple Silicon (gist) — https://gist.github.com/havenwood/f2f5c49c2c90c6787ae2295e9805adbe (accessed 2026-08-11) | |
| 565 | +- Disk speed testing on Apple Silicon: AmorphousDiskMark, Blackmagic, etc. (MacRumors, 4K QD1 results) — https://forums.macrumors.com/threads/disk-speed-testing-on-apple-silicon-amorphousdiskmark-blackmagic-etc-merged.2378298/ (accessed 2026-08-11) | |
| 566 | +- M1 Pro SSD speeds (MacRumors, 4K QD1 ~32 MB/s report) — https://forums.macrumors.com/threads/m1-pro-ssd-speeds.2319853/ (accessed 2026-08-11) | |
| 567 | +- MacBook Pro (16-inch, M5 Pro or M5 Max) — Tech Specs (memory bandwidth) — https://support.apple.com/en-us/126319 (accessed 2026-08-11) | |
| 568 | +- Unified Buffer Cache (UBC) — Mac OS X Internals: A Systems Approach (excerpt) — https://flylib.com/books/en/3.126.1.93/1/ (accessed 2026-08-11) | |
| 569 | +- Apple XNU WKdm fast memory page compressor (source mirror) — https://github.com/berkus/wkdm (accessed 2026-08-11) | |
| 570 | +- Virtual memory compression (WKdm background) — https://en.wikipedia.org/wiki/Virtual_memory_compression (accessed 2026-08-11) | |
| 571 | +- The working set model for program behavior (Denning, 1968; publications index) — http://denninginstitute.com/pjd/PUBS/Workingsets.html (accessed 2026-08-11) | |
| 572 | +- Working Set Analytics (Denning, ACM Computing Surveys) — https://dl.acm.org/doi/10.1145/3399709 (accessed 2026-08-11) | |
| 573 | +- ARC: A Self-Tuning, Low Overhead Replacement Cache (Megiddo & Modha, FAST '03) — https://www.usenix.org/legacy/events/fast03/tech/full_papers/megiddo/megiddo.pdf (accessed 2026-08-11) | |
| 574 | +- An Evaluation of Buffer Management Strategies for Relational Database Systems (Chou & DeWitt, VLDB '85 — DBMIN/QLSM) — https://www.cs.cmu.edu/~natassa/courses/15-721/papers/P127.PDF (accessed 2026-08-11) | |
| 575 | +- Anti-Caching: A New Approach to Database Management System Architecture (DeBrabant et al., VLDB 2013) — https://www.vldb.org/pvldb/vol6/p1942-debrabant.pdf (accessed 2026-08-11) | |
| 576 | +- TPP: Transparent Page Placement for CXL-Enabled Tiered-Memory (ASPLOS '23) — https://arxiv.org/abs/2206.02878 (accessed 2026-08-11) | |
| 577 | +- Pythia: A Customizable Hardware Prefetching Framework Using Online Reinforcement Learning (MICRO 2021) — https://arxiv.org/pdf/2109.12021 (accessed 2026-08-11) | |
| 578 | +- Evolution of Buffer Management in Database Systems: From Classical Algorithms to Machine Learning and Disaggregated Memory (survey) — https://arxiv.org/pdf/2512.22995 (accessed 2026-08-11) | |
added
research/notes/quantization.md
+256 −0
@@ -0,0 +1,256 @@ | ||
| 1 | +--- | |
| 2 | +project: localvm-research | |
| 3 | +document: research/notes/quantization | |
| 4 | +author: Simon-Pierre Boucher | |
| 5 | +contact: contact@spboucher.ai | |
| 6 | +created: 2026-08-11 | |
| 7 | +status: draft | |
| 8 | +--- | |
| 9 | + | |
| 10 | +# Quantization — deep literature note (charter §4.1) | |
| 11 | + | |
| 12 | +Scope: post-training quantization (PTQ), quantization-aware training (QAT), 8→1-bit and ternary regimes, mixed/per-layer/per-channel/dynamic/progressive precision, residual/additive/vector/lattice/trellis (codebook) quantization, weight-only vs weight+activation quantization, KV-cache quantization, and extreme low-bit inference — with explicit attention to (a) **memory-capacity reduction vs memory-bandwidth reduction as distinct effects**, and (b) **which formats have efficient Metal/MLX kernels vs CUDA-only implementations**. All sources listed in §Sources were accessed 2026-08-11. | |
| 13 | + | |
| 14 | +--- | |
| 15 | + | |
| 16 | +## 1. Landscape overview | |
| 17 | + | |
| 18 | +The field has converged on a fairly stable picture as of mid-2026: | |
| 19 | + | |
| 20 | +1. **4-bit weight-only is a solved commodity.** GPTQ (ICLR 2023), AWQ (MLSys 2024 best paper), and calibration-free HQQ all produce 4-bit models within a few percent (often <0.1–0.3 PPL) of FP16. Every serious Mac runtime (llama.cpp Metal, MLX) executes 4-bit weights natively at close to memory-bandwidth limits. | |
| 21 | +2. **2–3-bit is where the frontier is.** State of the art moved from scalar quantization (GPTQ) → incoherence processing + scalar (QuIP) → vector/lattice codebooks (QuIP#, AQLM, GPTVQ) → trellis-coded quantization (QTIP, EXL3), each step buying quality at the same bitrate. QuIP# was the first PTQ method where 3-bit scaled better than 4-bit. The catch for us: **all SOTA sub-3-bit codebook/trellis kernels are CUDA-first**; on Apple GPUs the analogous formats (llama.cpp i-quants) decode measurably slower than simple affine formats. | |
| 22 | +3. **Weight+activation quantization (W8A8, W4A4) is a compute/throughput play, not primarily a capacity play** — relevant for prefill speed, mostly irrelevant to the decode-time bandwidth wall on a Mac at batch 1. Rotation methods (QuaRot, SpinQuant) made W4A4 viable by killing activation outliers. | |
| 23 | +4. **Sub-2-bit exists but changes character.** BiLLM (PTQ, ~1.08 bit) is "viable" but degraded; BitNet-style ternary needs (re)training; ParetoQ shows a sharp representational transition between 2 and 3 bits — below it, weights leave the pretrained basin. For a system that must preserve the original model's behavior (charter §14 hard constraint), 2-bit is the practical floor for a *base* representation, not an endpoint. | |
| 24 | +5. **The newest and, for us, most important thread is multi-precision / nested / progressive representations**: Any-Precision LLM (bitplanes, ICML 2024 oral), Matryoshka Quantization (MSB-nesting, ICLR 2025 oral), Drop-by-Drop additive codebooks with successive-refinement theory (2026), and Progressive Mixed-Precision Decoding (ICLR 2025). These decouple *stored* bits from *read* bits — exactly the `total size ≠ bytes read per token` decoupling in our core research question. None of them targets Apple Silicon, and none streams residual precision from SSD on demand. | |
| 25 | + | |
| 26 | +Key systems distinction used throughout: at batch-1 decode, token latency ≈ (bytes of weights+KV read per token) / (achievable memory bandwidth). A format reduces **memory** if the resident representation is smaller; it reduces **bandwidth per token** only if the bytes actually *traversed* per token shrink and the dequant compute doesn't become the new bottleneck. These usually coincide for dense inference (every weight is read every token) but *diverge* for: codebook lookups (extra random reads), lossless coding (decode compute), bitplane/nested formats (read fewer planes than stored), and any offload scheme. | |
| 27 | + | |
| 28 | +--- | |
| 29 | + | |
| 30 | +## 2. Technique families | |
| 31 | + | |
| 32 | +### 2.1 Round-to-nearest and second-order weight-only PTQ (GPTQ family) | |
| 33 | + | |
| 34 | +**What it does.** GPTQ ("Accurate Post-Training Quantization for Generative Pre-trained Transformers", Frantar et al., ICLR 2023) quantizes weights column-by-column, using a per-layer Hessian proxy (from ~128 calibration samples) to update remaining unquantized weights and cancel accumulated error. Quantized OPT-175B to 3–4 bits in ~4 GPU-hours. | |
| 35 | + | |
| 36 | +- **Memory reduction:** ~4× at 4-bit, ~5.3× at 3-bit vs FP16. | |
| 37 | +- **Bandwidth reduction:** proportional to memory for dense decode (all weights read each token); reported 3.25× (A100) / 4.5× (A6000) generation speedups vs FP16. | |
| 38 | +- **Quality:** near-lossless at 4-bit (e.g., OPT-1.3B: FP16 PPL 14.63; naive RTN 4-bit 48.24; GPTQ 4-bit 15.47). 3-bit usable; 2-bit/ternary "viable" only with tiny groups. AWQ's authors document GPTQ overfitting to the calibration distribution (2.3–4.9 PPL worse under calibration/eval distribution shift, vs 0.5–0.6 for AWQ). | |
| 39 | +- **Calibration/retraining:** calibration data required; no retraining. | |
| 40 | +- **Apple Silicon status:** the *algorithm* is platform-neutral; GPTQ-produced integer weights convert to GGUF/MLX layouts fine. The fast GPTQ inference kernels (Marlin, ExLlama) are CUDA-only. mlx-lm ships a GPTQ-style learned-quant recipe natively. | |
| 41 | +- **Main limitation:** scalar, uniform grids; hits a wall below 3 bits. | |
| 42 | +- **Extension opportunity for us:** GPTQ's Hessian machinery is reusable for *allocating* precision (which blocks deserve residuals) rather than just rounding. | |
| 43 | + | |
| 44 | +**AWQ** (Lin et al., MLSys 2024 Best Paper, arXiv 2306.00978): observation that ~1% of weight channels are salient *as measured by activation magnitude*; protecting them via per-channel scaling (no mixed precision needed) preserves quality. No backprop; more robust to calibration shift than GPTQ. TinyChat gives >3× over HF FP16 on desktop/mobile GPUs. mlx-lm includes an AWQ-style learned quant. Same memory/bandwidth profile as GPTQ. | |
| 45 | + | |
| 46 | +**HQQ** (Mobius Labs, blog Nov 2023): calibration-**free** — solves a per-group half-quadratic optimization on weights only. Llama-2-70B quantized in <5 min (>50× faster than GPTQ); their 2-bit 70B beats FP16 Llama-2-13B in PPL at comparable memory. Useful to us as the cheapest way to generate base representations at many bit-widths for experiments. Python/PyTorch; runs anywhere (incl. MPS backend) but fast fused kernels are CUDA. | |
| 47 | + | |
| 48 | +**SqueezeLLM** (Kim et al., ICML 2024, arXiv 2306.07629): sensitivity-based **non-uniform** (k-means) codebooks per output channel + **dense-and-sparse decomposition** (0.05–0.45% of outlier/sensitive weights kept in FP16 sparse format). 3-bit LLaMA-7B: sensitivity-weighted clustering takes PPL from 18.08 (unweighted k-means) to 7.75. Explicitly frames single-batch LLM inference as **memory-bandwidth bound**, and shows LUT dequant can still hit ~2.3× GPU speedup. The dense+sparse split is a primitive "base + residual" — the sparse part is a tiny, importance-selected correction stream. CUDA kernels only. | |
| 49 | + | |
| 50 | +### 2.2 Weight + activation quantization (SmoothQuant → rotations) | |
| 51 | + | |
| 52 | +**SmoothQuant** (Xiao et al., ICML 2023, arXiv 2211.10438): migrates activation-outlier difficulty into weights via per-channel equivalent scaling → W8A8 training-free, negligible loss on OPT/BLOOM/Llama; up to 1.56× speedup and 2× memory reduction; enabled serving 530B in one node. | |
| 53 | + | |
| 54 | +**QuaRot** (Ashkboos et al., NeurIPS 2024): computational-invariance rotations (randomized Hadamard) applied to hidden state, FFN activations, attention, and KV cache remove outliers *without changing model output*, enabling **end-to-end 4-bit** (weights, activations, KV). Llama-2-70B W4A4: ≤0.47 WikiText-2 PPL loss, 99% zero-shot retention; lossless W6/W8 with plain RTN and *no calibration data*. | |
| 55 | + | |
| 56 | +**SpinQuant** (Liu et al., ICLR 2025): same invariance idea but with *learned* rotations (Cayley-optimized R1/R2 absorbed offline, plus online Hadamard R3/R4), closing more of the W4A4KV4 gap than random rotations. | |
| 57 | + | |
| 58 | +- **Memory/bandwidth:** activation quantization does not shrink the checkpoint; its win is compute density (INT8/INT4 matmul) and KV size. On a Mac at batch 1 decode this matters mainly through the KV cache and prefill speed. | |
| 59 | +- **Apple Silicon status:** no shipped Metal W4A4 path. MLX recently added `quantize_input` and mxfp8/nvfp4 modes (see §2.4), which is the beginning of an activation-quantization story; fast Hadamard transform kernels for Metal would need to be written. Note M5-class Neural Accelerators change the compute/bandwidth ratio in favor of such schemes. | |
| 60 | +- **Relevance:** incoherence/rotation preprocessing is *upstream-compatible* with any base representation we choose — it makes weights more Gaussian, which is exactly what codebook, lattice, and progressive-residual encodings want. QuIP#/QTIP already rely on it. | |
| 61 | + | |
| 62 | +### 2.3 GGUF block formats: k-quants, i-quants, ternary types (llama.cpp) | |
| 63 | + | |
| 64 | +**What they are.** llama.cpp's zoo: legacy Q4_0/Q4_1/Q5_x/Q8_0 (per-block scale ± min); **k-quants** (Q2_K…Q6_K) with two-level scale hierarchies and per-tensor mixes (the `_S/_M/_L` file types give sensitive tensors more bits); **i-quants** (IQ1_S…IQ4_XS) which are codebook/grid-based sub-4-bit formats requiring an **importance matrix** (imatrix, calibration-derived) for quality; ternary **TQ1_0/TQ2_0**; and **MXFP4** (added Aug 2025 for gpt-oss, whose FFN weights ship natively in MXFP4). | |
| 65 | + | |
| 66 | +- **Memory/bandwidth:** proportional (dense decode). Q4_K_M ≈ 4.9 GB for an 8B model. Practical measurements put Q4_K_M ~+0.08 PPL over FP16 on Llama-3-8B class models. | |
| 67 | +- **Quality per bit:** k-quants beat MLX affine at matched bpw — one practitioner measurement (Feldman, 2026): Q4_K_M at 4.88 bpw gives 0.0208 nats KL-to-FP16 vs MLX affine 4-bit (4.69 bpw) 0.0577 nats, ~2.8× better; similar at 5-bit. The reason is the two-level scale hierarchy and mixed per-tensor precision. | |
| 68 | +- **Apple Silicon / Metal status — the critical bandwidth-vs-compute datapoint:** all GGUF types have hand-written Metal kernels, but **i-quants pay a large decode penalty on Apple GPUs**. ikawrakow (llama.cpp discussion #5617): 7B IQ-quant ~53.9 t/s vs Q4_0 63.1 t/s on M2 Max 30-core GPU, despite reading ~half the bytes; vs an RTX-4080 the gap for IQ2_XS is 3.5× (only 2× for Q4_0). Community measurements agree the codebook lookups make dequant compute-bound on Apple Silicon (and older CPUs). **Lesson: on Metal, "fewer bits" only converts to "faster tokens" if the decode path stays trivially cheap (shift/mask + multiply-add), which is exactly what k-quants and MLX affine do and what LUT-heavy formats do not.** | |
| 69 | +- **Limitation:** static, uniform-per-tensor decisions; no notion of loading *part* of a weight's information. | |
| 70 | +- **Extension:** the imatrix (importance) infrastructure is a ready-made per-block sensitivity signal usable for residual allocation. | |
| 71 | + | |
| 72 | +### 2.4 MLX native quantization (our primary substrate) | |
| 73 | + | |
| 74 | +**Formats** (mlx.core.quantize / mlx.nn.quantize docs, MLX 0.32): default **affine** mode — bits ∈ {2,3,4,5,6,8}, group size ∈ {32,64,128} (default 4-bit/g64), per-group scale+bias, ŵ = round((w−β)/s); plus **mxfp4** (E2M1, g32, E8M0 scale), **mxfp8** (g32), **nvfp4** (g16, E4M3 scale). `quantize_input=True` enables input/activation quantization for linear layers. All modes have first-class Metal kernels (fused dequant-matmul); quantized-KV attention supported via `mlx_lm.generate --kv-bits {2,4,8} --kv-group-size`. | |
| 75 | + | |
| 76 | +**Learned quants in mlx-lm** (LEARNED_QUANTS.md): **DWQ** (distilled weight quantization — distills a 16/8-bit teacher into the *quantization parameters* (scales/biases) of a low-bit student; works best at 2–4 bit; community measurements ≈ +0.6 effective bits of quality); **AWQ** port; **GPTQ**-style; **dynamic_quant** (per-layer sensitivity-based bit allocation: layers that hurt most get more bits). This is the closest thing to a maintained, Apple-first learned-PTQ toolchain, and it's pure Python on top of MLX — directly hackable for our experiments. | |
| 77 | + | |
| 78 | +- **Memory/bandwidth:** proportional; MLX 4-bit ≈ 4.5 GB for 8B. MLX vs llama.cpp decode speed on identical Macs is within ±10–20% either way depending on version/model; MLX tends to win on 4-bit 7–30B decode, llama.cpp on prefill (mature simdgroup matmuls). Ollama switched its Apple Silicon backend to MLX (2026), and M5 "Neural Accelerators" give MLX further headroom. | |
| 79 | +- **Quality:** plain affine 4-bit/g64 trails Q4_K_M slightly (see §2.3); group-size 32 and/or DWQ closes the gap. | |
| 80 | +- **Main limitation:** uniform affine only — no non-uniform codebooks, no two-level scales, no sub-2-bit, no nested/progressive layout. Quantization is chosen once at convert time; the runtime has no concept of refining a weight after load. | |
| 81 | +- **Extension:** MLX's `mode` plug-point plus custom Metal kernels is where a progressive/residual format would be implemented. The DWQ distillation loop is also the obvious way to *calibrate a low-bit base to be maximally correctable by its residuals*. | |
| 82 | + | |
| 83 | +### 2.5 Vector / additive / lattice / trellis codebook quantization (the 2-bit frontier) | |
| 84 | + | |
| 85 | +**QuIP** (Chee et al., NeurIPS 2023): incoherence processing (random orthogonal pre/post rotations) + adaptive rounding; first theoretical analysis at LLM scale; first "viable" 2-bit. | |
| 86 | + | |
| 87 | +**QuIP#** (Tseng et al., ICML 2024): randomized Hadamard incoherence + **E8-lattice 8-dimensional codebook (E8P)** + inter-layer fine-tuning. Higher bitrates built via **residual vector quantization** (RVQ — quantize, then quantize the residual with another codebook; e.g., 4-bit = 2+2). Llama-2-70B Wiki2 PPL (no FT): FP16 3.12 → E8P 2-bit 4.16 (vs 5.90 for QuIP scalar 2-bit). First PTQ where 3-bit scaled better than 4-bit. >3× faster than FP16 inference (CUDA); codebook decodable in <4 instructions/weight due to E8 symmetry. | |
| 88 | + | |
| 89 | +**AQLM** (Egiazarian et al., ICML 2024, arXiv 2401.06118): **additive quantization** — each weight group is a *sum of several codewords* from learned 8-dimensional, 2^16-entry codebooks, trained by beam search + block-wise then end-to-end fine-tuning. First scheme Pareto-optimal below 3 bits; Pareto-optimal bitwidth ≈ 2.5 bpw. **PV-Tuning** (NeurIPS 2024) adds representation-agnostic fine-tuning of discrete+continuous params → first Pareto-optimal 2-bit Llama-2. Cost: ~720 GPU-hours for a 70B; 1 MiB codebooks blow L1 cache → AQLM decode is *slow* (20.6 tok/s vs QuIP# 106.3 on 2-7B, per QTIP measurements). | |
| 90 | + | |
| 91 | +**GPTVQ** (van Baalen et al., Qualcomm, ICML 2024): GPTQ-style Hessian-interleaved updates extended to non-uniform VQ (1–4D); codebooks themselves compressed (int + SVD). 70B processed in 3–11 h. Notably demonstrated **simultaneous DRAM-footprint and latency reduction on a mobile-class Arm CPU** — one of the few codebook methods validated on unified-memory consumer silicon rather than discrete GPUs. | |
| 92 | + | |
| 93 | +**QTIP** (Tseng et al., NeurIPS 2024): **trellis-coded quantization** — stateful codes over long sequences instead of fixed-dim VQ; with compute-based "bitshift trellis" codebooks there is no large LUT, decode is ~2 instructions/weight, and matvecs run at >80% of peak GPU memory bandwidth. Beats QuIP#/AQLM at all bitrates; QTIP-without-finetune ≈ QuIP#/AQLM-with-finetune. **EXL3** (exllamav3) is a streamlined QTIP variant productized for consumer GPUs — Llama-3.1-70B coherent at 1.6 bpw, 70B in <16 GB VRAM. **CUDA-only.** | |
| 94 | + | |
| 95 | +**NestQuant** (Savkin et al., ICML 2025, arXiv 2502.09720): self-similar **nested lattices** (Gosset/E8), information-theoretically near-optimal for low-precision matmul, covering weights *and* activations. Also NeurIPS 2025 work on learned grouped lattice VQ (learnable generator matrices, Babai rounding). Lattice methods are converging with VQ methods. | |
| 96 | + | |
| 97 | +- **Memory reduction:** the whole point — 2.0–2.5 bpw with usable quality (~8× vs FP16). | |
| 98 | +- **Bandwidth reduction:** *conditional*. Weight bytes drop 8×, but decode cost decides whether that becomes tokens/sec: QTIP/QuIP# reach near-bandwidth-limit on NVIDIA; AQLM does not; and the Apple-GPU evidence from i-quants (§2.3) says LUT-based decode may leave Metal compute-bound. No published Metal implementation of QuIP#/AQLM/QTIP exists as of this writing. | |
| 99 | +- **Calibration:** all need calibration; AQLM/QuIP#/QTIP quality depends significantly on (expensive) fine-tuning. | |
| 100 | +- **Main limitation for us:** CUDA-first ecosystems; heavy encode cost; fixed bitrate at encode time. | |
| 101 | +- **Extension opportunity:** **RVQ/additive structure is inherently progressive** — codeword sums can be truncated (see Drop-by-Drop, §2.7). A Metal "bitshift-trellis" decoder (no LUT) is plausibly the right way to get QTIP-class quality on Apple GPUs; nobody has published one. | |
| 102 | + | |
| 103 | +### 2.6 Extreme low-bit: ternary, 1-bit, and low-bit QAT | |
| 104 | + | |
| 105 | +**BitNet b1.58** (Ma et al., Microsoft, arXiv 2402.17764): ternary {−1,0,+1} weights, *trained from scratch*; claims parity with FP16 at same params/tokens, with large latency/memory/energy wins. Not a post-training transform → excluded as a solution by charter §14, but relevant as an existence proof of ~1.58-bit information sufficiency and for its inference stack. **bitnet.cpp** (arXiv 2410.16144; ACL 2025): CPU LUT kernels — I2_S (lossless MAD-based), TL1 (ARM), TL2 (x86); 1.37–5.07× speedups on ARM; runs a demo on Apple M2, and TL2_0 reaches 7.45 tok/s for a **100B ternary model on an M2 Ultra** — an interesting datapoint that sub-2-bit makes 100B-class models CPU-feasible on Macs. Note: ternary kernels are **CPU-only** (NEON/AVX LUTs), no Metal GPU path; llama.cpp's TQ1_0/TQ2_0 exist but were measured badly inaccurate for BitNet models by the bitnet.cpp authors. | |
| 106 | + | |
| 107 | +**BiLLM** (Huang et al., ICML 2024): first 1-bit PTQ — Hessian-selected salient weights get **binary residual approximation** (a second binary pass over the residual: again base+residual!), non-salient bell-shaped weights get optimal-split binarization. ~1.08–1.11 bpw; LLaMA2-70B PPL 8.41; 7B binarized in 0.5 h on one GPU. Quality is far from FP16 (usable-ish, clearly degraded); ARB-LLM (2024) refines it. | |
| 108 | + | |
| 109 | +**ParetoQ** (Liu et al., Meta, NeurIPS 2025, arXiv 2502.02631): unified QAT-finetune framework across 1/1.58/2/3/4-bit. Key findings: ~10% of training budget on QAT finetuning suffices; **sharp representational transition between 2 and 3 bits** — at ≥3 bits finetuned models stay close to the pretrained distribution ("compensation"); at ≤2 bits representations change drastically ("reconstruction"). Ternary/2-bit/3-bit beat 4-bit on size-accuracy Pareto in their runs. | |
| 110 | + | |
| 111 | +**EfficientQAT** (Chen et al., ACL 2025, arXiv 2407.11062): block-wise QAT (Block-AP) + end-to-end training of quant params only (E2E-QP). 2-bit Llama-2-70B on a single A100 in 41 h, −3% accuracy vs FP16 (69.48 vs 72.41); w2g64 Llama-2-7B PPL 6.86 vs 5.47 FP16. This is the realistic quality ceiling for "2-bit that still behaves like the original model" with modest compute. | |
| 112 | + | |
| 113 | +- **Implication for localvm-research:** ParetoQ's 2↔3-bit transition + EfficientQAT numbers suggest a **2-bit base is recoverable with light training but is near the edge**; a 2.5–3-bit-effective base (or 2-bit + streamed residual) is the safer floor if the base must preserve routing/decision structure without full QAT. | |
| 114 | + | |
| 115 | +### 2.7 Multi-precision, nested, progressive, and residual representations ⭐ (most relevant family) | |
| 116 | + | |
| 117 | +**Any-Precision LLM** (Park et al., ICML 2024 **oral**, arXiv 2402.10517; code SNU-ARC/any-precision-llm): store ONE n-bit "parent" model (non-uniform, incremental-upscaling from a 3-bit seed) in **bitplane layout**; any child bit-width 3…n is obtained by reading only the top-k bitplanes. Memory: supporting {3,4,5,6,7,8} bits costs 8.4 GB instead of 29.9 GB for separate models (3.56× saving). Crucially, **"any runtime request of reduced bit-width directly translates into proportional speedup, as we can simply load the specified number of bits"** — i.e., bytes-read-per-token scales with *chosen* precision, not stored precision. Engine has custom **CUDA** kernels (bitplane layout, bit-transpose, merged table lookups). No Metal port exists. | |
| 118 | + | |
| 119 | +**Matryoshka Quantization / MatQuant** (Nair et al., Google DeepMind, ICLR 2025 oral, arXiv 2502.06786): exploit MSB-nesting of integers — co-optimize (via QAT or OmniQuant-style PTQ) a single int8 weight tensor so that slicing its top 4 or 2 bits yields good int4/int2 models; int2 extracted this way is up to 10% more accurate than dedicated int2 QAT; an int2-FFN Gemma-2 9B beats an int8-FFN Gemma-2 2B. Interpolative bit-widths (int3/int6) come for free. | |
| 120 | + | |
| 121 | +**Drop-by-Drop / multi-bitwidth additive codebooks** (arXiv 2606.12876, Feb 2026): grounds multi-precision PTQ in **information-theoretic successive refinement**; trains additive (AQLM-style) codebooks with Matryoshka supervision so that *ordered subsets of codebooks* give accurate partial reconstructions — progressive compression by literally dropping codebooks at inference. Explicitly motivates "run-time hardware-aware dynamic loading". This is the closest published object to a "progressive residual base representation," and it is brand new — no systems implementation, no SSD tier, no Apple port. | |
| 122 | + | |
| 123 | +**Progressive Mixed-Precision Decoding (PMPD)** (Chen et al., Samsung AI, ICLR 2025, arXiv 2410.13461): *phase-aware* precision — higher-precision weights for prefill, lower for decode, and **progressively lower precision as generation deepens** (later tokens tolerate more error), with task-/prompt-adaptive schedulers. 3.8–8.0× decode throughput on an NPU. Validates, at small scale, the premise that required weight precision is *token-position-dependent* and can be scheduled at runtime. | |
| 124 | + | |
| 125 | +Related: **AnyBCQ** (2025) — binary-coded multi-precision with hardware-efficient bit-plane access; **Squeeze10-LLM** (2025) staged sub-2-bit PTQ; the **mixed-precision survey** arXiv 2510.16805 (Oct 2025) taxonomizes this whole space (§3.1 covers Any-Precision, PMPD, M2Cache). | |
| 126 | + | |
| 127 | +- **Memory:** one artifact serves all precisions (≈ cost of the highest). | |
| 128 | +- **Bandwidth:** the *defining* feature — bytes/token = f(precision requested now), decoupled from storage. This is breakthrough criterion B in embryo. | |
| 129 | +- **Quality:** MatQuant int2-slice ≈ or better than dedicated int2; Any-Precision children match dedicated SqueezeLLM-class models at each width. | |
| 130 | +- **Apple status:** none. All engines are CUDA (or NPU simulators). Bitplane decode is bit-manipulation-heavy — Apple GPU/AMX behavior unknown; a Metal bitplane-matvec kernel is an obvious micro-experiment (ties to charter Exp. D/E). | |
| 131 | +- **Limitation:** all of these still *load the full chosen precision for every weight every token* — precision is scheduled globally (per phase/token), not per weight-block; and the lowest usable slice is ~2–3 bits. | |
| 132 | + | |
| 133 | +### 2.8 Random-basis and lossless recompression (orthogonal tricks) | |
| 134 | + | |
| 135 | +**SeedLM** (Shafipour et al., **Apple** + Meta, ICLR 2025, arXiv 2410.10714): compress each weight block into a **seed + coefficients of an LFSR-generated pseudo-random basis** — at inference, regenerate the basis from the seed and reconstruct the block. Data-free, 3–4-bit effective, ~zero-shot parity with calibration methods at 4-bit; trades **memory bandwidth for free compute** (PRNG regeneration), which is exactly the right trade on bandwidth-bound hardware — FPGA demo ~4× speedup at 70B. Apple authored this: conceptually adjacent to "weights need not be stored, only recoverable." | |
| 136 | + | |
| 137 | +**DFloat11** (Zhang et al., NeurIPS 2025, arXiv 2504.11651): **lossless** Huffman coding of BF16 exponents → ~30% size cut (≈11 bpw), bit-identical outputs, GPU online-decompression kernels. Cautionary datapoint: even with careful kernels, decode costs ~2–3× tokens/s vs uncompressed-in-VRAM on 8–32B models (it only wins when the alternative is offload). **Lossless coding shrinks capacity, not effective bandwidth** — entropy decode sits on the critical path. CUDA-only. | |
| 138 | + | |
| 139 | +### 2.9 KV-cache quantization | |
| 140 | + | |
| 141 | +- **KIVI** (Liu et al., ICML 2024, arXiv 2402.02750): tuning-free 2-bit KV — **keys per-channel, values per-token** (asymmetric, matches outlier structure), small FP16 sliding window; 2.6× peak-memory cut, 2.35–3.47× throughput (batch effect). Inspired HF Transformers' KV quantization. | |
| 142 | +- **KVQuant** (Hooper et al., NeurIPS 2024, arXiv 2401.18079): per-channel pre-RoPE key quant + non-uniform sensitivity-weighted datatypes + per-vector dense-and-sparse outliers → **3-bit KV with <0.1 PPL degradation**; enables 1M-token context for LLaMA-7B on one A100. | |
| 143 | +- **Coupled Quantization** (NeurIPS 2024): exploits inter-channel dependence to reach ~1 bit/channel-equivalent KV. | |
| 144 | +- **Apple Silicon status — good:** llama.cpp `--cache-type-k/v {q8_0,q5_0,q4_0,iq4_nl}` with flash attention on Metal (q8_0 ≈ half KV memory, negligible loss; q4_0 noticeable on long reasoning); mlx-lm `--kv-bits {2,4,8}` (group 64 default). Community work (KVSplit) confirms asymmetric K/V precision pays on Metal. Google's 2026 sub-3-bit KV result (picked up in llama.cpp discussion #20969, "TurboQuant", with working Metal kernels at 3.25/4.25 bits) is being adopted. | |
| 145 | +- **Relevance:** KV bytes/token grow with context and can rival weight bytes at long context on 48 GB; any working-set argument must model both. KV quantization is the *already-solved* part of the working-set problem on Macs. | |
| 146 | + | |
| 147 | +### 2.10 Precision granularity: per-layer / per-channel / per-token / dynamic | |
| 148 | + | |
| 149 | +Cross-cutting rather than a single method: per-channel scales are standard (AWQ/GPTQ groups); per-layer bit allocation is in mlx-lm `dynamic_quant` and GGUF `_M` mixes; per-token/phase precision is PMPD (§2.7); dynamic runtime precision selection appears in Any-Precision serving and M2Cache (below). The mixed-precision survey (arXiv 2510.16805) is the map of this space. **MoE note:** for MoE models, per-expert precision (hot experts high-bit in RAM, cold experts low-bit near SSD) is an obvious composite nobody ships on Mac; but the energy study arXiv 2508.06978 warns that *naive* SSD expert-offload raises per-token energy up to ~12× vs HBM — prefetch hides latency, not energy or bandwidth. | |
| 150 | + | |
| 151 | +### 2.11 Quantization + offload hybrids (closest existing systems to our target) | |
| 152 | + | |
| 153 | +**M2Cache** (arXiv 2410.14740): neuron-level modularization + importance ranking + **dynamic sparse mixed-precision quantization** + three-level cache (GPU HBM ← DRAM ← SSD). Highest-importance neurons stay high-precision and cached; low-importance ones live at low precision on lower tiers. Up to 14× throughput vs baseline offload on RTX 3090-class hardware, 70B on 24 GB VRAM. This is the closest published architecture to charter §13 — but: Linux/CUDA, discrete-GPU tiering assumptions (PCIe copy costs that don't exist on unified memory), static importance ranking, and no progressive-precision refinement (a neuron is fetched at one precision, not refined). | |
| 154 | + | |
| 155 | +--- | |
| 156 | + | |
| 157 | +## 3. Relevance to localvm-research | |
| 158 | + | |
| 159 | +### 3.1 What quantization gives us as a base representation | |
| 160 | + | |
| 161 | +The working-set-decoupling question needs a representation where `bytes read per token` is a *runtime variable*. The literature supplies four composable building blocks: | |
| 162 | + | |
| 163 | +1. **A 2–4-bit affine base with native Metal speed** (MLX affine / k-quants) — the only formats today where fewer bits reliably equal proportionally fewer nanoseconds on Apple GPUs. A 2-bit/g32 (+DWQ-calibrated) MLX base is implementable *now*; expected quality per EfficientQAT/ParetoQ: degraded but structurally faithful (PPL +~1.4 on 7B-class at w2g64 with training; worse pure-PTQ). | |
| 164 | +2. **Nested/bitplane layouts** (Any-Precision, MatQuant) — store 6–8 bits, *read* 2–4. MSB-slicing means the "residual" is literally the next bitplane: refinement = read more planes of the same tensor, perfectly sequential, prefetchable, and idempotent. This is the natural on-SSD layout for progressive weight materialization (charter Exp. D). | |
| 165 | +3. **Additive/RVQ codebooks with successive-refinement training** (QuIP#'s RVQ, Drop-by-Drop) — higher quality per bit than bitplanes at the same budget, refinement = add codewords; but decode-cost risk on Metal (i-quant lesson) unless a QTIP-style computed (LUT-free) codebook is used. | |
| 166 | +4. **Error-side instruments**: SqueezeLLM/BiLLM's dense-and-sparse and KVQuant's outlier streams show that a *tiny, importance-ranked sparse residual* captures outsized quality — a cheap, cache-friendly correction channel that can be paged independently of the dense base. | |
| 167 | + | |
| 168 | +### 3.2 The specific opening (what is NOT in the literature) | |
| 169 | + | |
| 170 | +Verified against everything above, the following combination does not exist: | |
| 171 | + | |
| 172 | +- **Progressive precision as a memory hierarchy.** Any-Precision/MatQuant/Drop-by-Drop keep all bitplanes/codebooks *resident* and slice for latency; PMPD schedules precision *globally per token phase*. **Nobody stores the low-bit base resident in unified memory and treats higher-order residual planes as a demand-paged tier on NVMe**, fetched per-block, conditioned on (layer sensitivity × current-token need × cache state). That system would make RAM bound the *base* size (e.g., 2 bits/param ≈ 17.5 GB for a 70B) while total quality lives on disk — precisely `resident size ≠ total size ≠ bytes/token`. | |
| 173 | +- **Decision-stability-driven refinement.** No published method decides *how many residual planes to fetch* based on whether the token decision (top-1 margin / top-k set) is already stable — PMPD's schedulers are position-based, not confidence-based; Drop-by-Drop's profiles are static. This connects quantization directly to charter §4.10 / Exp. G: fetch residuals only when the low-bit forward pass is uncertain. | |
| 174 | +- **Apple Silicon kernels for any of it.** No Metal implementations exist for: bitplane matvec (Any-Precision), MSB-sliced decode (MatQuant), trellis decode (QTIP/EXL3), E8/additive codebooks (QuIP#/AQLM). Unified memory actually *helps* here versus CUDA: a refined block is written once and visible to the GPU without PCIe traffic, and the SSD→RAM→GPU path has no copy step. The i-quant Metal evidence defines the design constraint: decode must be shift/mask-cheap (bitplanes and bitshift-trellises qualify; big LUTs do not). | |
| 175 | +- **Residual-aware calibration.** DWQ/EfficientQAT optimize a single bit-width; MatQuant co-optimizes slices; but nobody calibrates a base *jointly with a sparse importance-ranked residual stream under a bytes-per-token budget* ("make the 2-bit base maximally correctable by its cheapest residuals"). mlx-lm's DWQ loop is the natural place to prototype this on-device. | |
| 176 | + | |
| 177 | +### 3.3 Falsifiable premises to test first (feeds Exp. D/E/H) | |
| 178 | + | |
| 179 | +1. **Metal bitplane decode cost:** does a 2-of-8-bitplane matvec on M-series GPU run at ≥70% of the speed of a dense 2-bit affine matvec? (If not, MSB-sliced *packed* MatQuant layout instead of bitplanes.) | |
| 180 | +2. **Quality-vs-planes curve:** on a 7–8B model, measure PPL/KL/greedy-token-agreement at 2, 2+1, 2+2 … planes (charter Exp. D). MatQuant/Any-Precision predict smooth improvement; the open question is how *few blocks* need refinement to recover most of it. | |
| 181 | +3. **Residual locality:** are the blocks whose refinement matters stable across tokens/domains (Exp. B/C)? If yes, an SSD residual tier with an LRU of refined blocks beats static mixed precision; if no, bandwidth math likely kills it (cf. arXiv 2508.06978's energy warning). | |
| 182 | +4. **Bandwidth accounting:** a 48 GB M5 Max has O(500+) GB/s unified memory vs O(6–8) GB/s NVMe — residual fetches must therefore be ≤~1–2% of weight bytes per token, or fully overlapped/amortized across tokens. This ratio, not quality, is the most likely failure mode; measure before building (Exp. H). | |
| 183 | + | |
| 184 | +--- | |
| 185 | + | |
| 186 | +## Sources | |
| 187 | + | |
| 188 | +- GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers (Frantar et al., ICLR 2023) — https://arxiv.org/abs/2210.17323 (accessed 2026-08-11) | |
| 189 | +- AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration (Lin et al., MLSys 2024) — https://arxiv.org/abs/2306.00978 (accessed 2026-08-11) | |
| 190 | +- SmoothQuant: Accurate and Efficient Post-Training Quantization for LLMs (Xiao et al., ICML 2023) — https://arxiv.org/abs/2211.10438 (accessed 2026-08-11) | |
| 191 | +- A Practical Guide to INT4 Quantization for SLMs: GPTQ vs AWQ (Microsoft Data Science, Medium) — https://medium.com/data-science-at-microsoft/a-practical-guide-to-int4-quantization-for-slms-gptq-vs-awq-olive-and-real-world-results-2f63d6963d1d (accessed 2026-08-11) | |
| 192 | +- Combining multiple post-training techniques to achieve most efficient quantized LLMs (MX formats + GPTQ/SmoothQuant) — https://arxiv.org/html/2405.07135v1 (accessed 2026-08-11) | |
| 193 | +- QuIP: 2-Bit Quantization of Large Language Models With Guarantees (Chee et al., NeurIPS 2023) — https://neurips.cc/virtual/2023/poster/69982 (accessed 2026-08-11) | |
| 194 | +- QuIP#: Even Better LLM Quantization with Hadamard Incoherence and Lattice Codebooks (Tseng et al., ICML 2024) — slides — https://icml.cc/media/icml-2024/Slides/34816.pdf (accessed 2026-08-11) | |
| 195 | +- QuIP# codebase (Cornell RelaxML) — https://github.com/Cornell-RelaxML/quip-sharp (accessed 2026-08-11) | |
| 196 | +- QuIP# full text (PMC mirror, RVQ details) — https://pmc.ncbi.nlm.nih.gov/articles/PMC12395268 (accessed 2026-08-11) | |
| 197 | +- QTIP: Quantization with Trellises and Incoherence Processing (Tseng et al., NeurIPS 2024) — https://arxiv.org/html/2406.11235v1 (accessed 2026-08-11) | |
| 198 | +- Even Better, Even Faster Quantized LLMs with QTIP (Together AI blog) — https://www.together.ai/blog/even-better-even-faster-quantized-llms-with-qtip (accessed 2026-08-11) | |
| 199 | +- AQLM: Extreme Compression of Large Language Models via Additive Quantization (Egiazarian et al., ICML 2024) — https://arxiv.org/html/2401.06118v2 (accessed 2026-08-11) | |
| 200 | +- AQLM + PV-Tuning official repository — https://github.com/vahe1994/AQLM (accessed 2026-08-11) | |
| 201 | +- PV-Tuning: Beyond Straight-Through Estimation for Extreme LLM Compression (NeurIPS 2024) — https://proceedings.neurips.cc/paper_files/paper/2024/file/091166620a04a289c555f411d8899049-Paper-Conference.pdf (accessed 2026-08-11) | |
| 202 | +- The Evolution of Extreme LLM Compression: From QuIP to AQLM with PV-Tuning (Yandex, Medium) — https://medium.com/yandex/the-evolution-of-extreme-llm-compression-from-quip-to-aqlm-with-pv-tuning-19c44b91af96 (accessed 2026-08-11) | |
| 203 | +- GPTVQ: The Blessing of Dimensionality for LLM Quantization (van Baalen et al., Qualcomm) — https://arxiv.org/abs/2402.15319 (accessed 2026-08-11) | |
| 204 | +- GPTVQ repository — https://github.com/Qualcomm-AI-research/gptvq (accessed 2026-08-11) | |
| 205 | +- NestQuant: Nested Lattice Quantization for Matrix Products and LLMs (Savkin et al., ICML 2025) — https://arxiv.org/abs/2502.09720 (accessed 2026-08-11) | |
| 206 | +- Learning Grouped Lattice Vector Quantizers for Low-Bit LLM Compression (NeurIPS 2025) — https://neurips.cc/virtual/2025/poster/117396 (accessed 2026-08-11) | |
| 207 | +- SqueezeLLM: Dense-and-Sparse Quantization (Kim et al., ICML 2024) — https://arxiv.org/html/2306.07629v4 (accessed 2026-08-11) | |
| 208 | +- Half-Quadratic Quantization of Large Machine Learning Models (Mobius Labs, via Dropbox Tech) — https://dropbox.tech/machine-learning/halfquadratic-quantization-of-large-machine-learning-models (accessed 2026-08-11) | |
| 209 | +- QuaRot: Outlier-Free 4-Bit Inference in Rotated LLMs (Ashkboos et al., NeurIPS 2024) — https://neurips.cc/virtual/2024/poster/94328 (accessed 2026-08-11) | |
| 210 | +- SpinQuant: LLM Quantization with Learned Rotations (Liu et al., ICLR 2025) — https://proceedings.iclr.cc/paper_files/paper/2025/file/e5b1c0d4866f72393c522c8a00eed4eb-Paper-Conference.pdf (accessed 2026-08-11) | |
| 211 | +- Rotation-based quantization with QuaRot (AMD Quark docs, R1–R4 rotations) — https://quark.docs.amd.com/release-0.9/pytorch/tutorial_quarot.html (accessed 2026-08-11) | |
| 212 | +- The Era of 1-bit LLMs: All Large Language Models are in 1.58 Bits (Ma et al., Microsoft) — https://arxiv.org/abs/2402.17764 (accessed 2026-08-11) | |
| 213 | +- 1-bit AI Infra Part 1.1: Fast and Lossless BitNet b1.58 Inference on CPUs (bitnet.cpp) — https://arxiv.org/html/2410.16144v1 (accessed 2026-08-11) | |
| 214 | +- Bitnet.cpp: Efficient Edge Inference for Ternary LLMs (ACL 2025; TL/I2_S kernels, M2 Ultra 100B result) — https://aclanthology.org/2025.acl-long.457.pdf (accessed 2026-08-11) | |
| 215 | +- microsoft/BitNet official inference framework — https://github.com/microsoft/BitNet (accessed 2026-08-11) | |
| 216 | +- BiLLM: Pushing the Limit of Post-Training Quantization for LLMs (Huang et al., ICML 2024) — https://github.com/Aaronhuang-778/BiLLM (accessed 2026-08-11) | |
| 217 | +- ParetoQ: Scaling Laws in Extremely Low-bit LLM Quantization (Liu et al., Meta, NeurIPS 2025) — https://arxiv.org/html/2502.02631v2 (accessed 2026-08-11) | |
| 218 | +- ParetoQ (PyTorch blog) — https://pytorch.org/blog/paretoq-scaling-laws-in-extremely-low-bit-llm-quantization (accessed 2026-08-11) | |
| 219 | +- EfficientQAT: Efficient Quantization-Aware Training for LLMs (Chen et al., ACL 2025) — https://arxiv.org/abs/2407.11062 (accessed 2026-08-11) | |
| 220 | +- EfficientQAT repository (w2g64 PPL tables) — https://github.com/OpenGVLab/EfficientQAT (accessed 2026-08-11) | |
| 221 | +- Any-Precision LLM: Low-Cost Deployment of Multiple, Different-Sized LLMs (Park et al., ICML 2024 oral) — https://arxiv.org/html/2402.10517v4 (accessed 2026-08-11) | |
| 222 | +- Any-Precision LLM repository (bitplane engine) — https://github.com/SNU-ARC/any-precision-llm (accessed 2026-08-11) | |
| 223 | +- Matryoshka Quantization (Nair et al., Google DeepMind, ICLR 2025 oral) — https://iclr.cc/virtual/2025/10000114 (accessed 2026-08-11) | |
| 224 | +- Matryoshka Quantization topic overview (Emergent Mind) — https://www.emergentmind.com/topics/matryoshka-quantization-matquant (accessed 2026-08-11) | |
| 225 | +- Multi-Bitwidth Quantization for LLMs Using Additive Codebooks ("Drop-by-Drop", successive refinement) — https://arxiv.org/html/2606.12876v1 (accessed 2026-08-11) | |
| 226 | +- Progressive Mixed-Precision Decoding for Efficient LLM Inference (Chen et al., ICLR 2025) — https://arxiv.org/abs/2410.13461 (accessed 2026-08-11) | |
| 227 | +- Mixed-Precision Quantization for Language Models (survey, Oct 2025; PMDP/MPMLC taxonomy) — https://arxiv.org/html/2510.16805v1 (accessed 2026-08-11) | |
| 228 | +- SeedLM: Compressing LLM Weights into Seeds of Pseudo-Random Generators (Apple ML Research) — https://machinelearning.apple.com/research/seedlm-compressing (accessed 2026-08-11) | |
| 229 | +- SeedLM (arXiv full text) — https://arxiv.org/html/2410.10714v1 (accessed 2026-08-11) | |
| 230 | +- DFloat11: 70% Size, 100% Accuracy — Lossless LLM Compression via Dynamic-Length Float — https://huggingface.co/papers/2504.11651 (accessed 2026-08-11) | |
| 231 | +- DFloat11 repository (LeanModels, NeurIPS 2025) — https://github.com/LeanModels/DFloat11 (accessed 2026-08-11) | |
| 232 | +- DFloat11 throughput caveats (Hacker News discussion incl. appendix numbers) — https://news.ycombinator.com/item?id=43796935 (accessed 2026-08-11) | |
| 233 | +- KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache (Liu et al., ICML 2024) — https://arxiv.org/abs/2402.02750 (accessed 2026-08-11) | |
| 234 | +- KIVI repository — https://github.com/jy-yuan/KIVI (accessed 2026-08-11) | |
| 235 | +- KVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization (Hooper et al., NeurIPS 2024) — https://arxiv.org/abs/2401.18079 (accessed 2026-08-11) | |
| 236 | +- KV Cache is 1 Bit Per Channel: Coupled Quantization (NeurIPS 2024) — https://proceedings.neurips.cc/paper_files/paper/2024/file/05d6b5b6901fb57d2c287e1d3ce6d63c-Paper-Conference.pdf (accessed 2026-08-11) | |
| 237 | +- mlx.core.quantize documentation (affine/mxfp4/mxfp8/nvfp4 modes) — https://ml-explore.github.io/mlx/build/html/python/_autosummary/mlx.core.quantize.html (accessed 2026-08-11) | |
| 238 | +- mlx.nn.quantize documentation (quantize_input, class predicates) — https://ml-explore.github.io/mlx/build/html/python/_autosummary/mlx.nn.quantize.html (accessed 2026-08-11) | |
| 239 | +- mlx-lm LEARNED_QUANTS.md (DWQ, dynamic_quant, AWQ/GPTQ recipes) — https://github.com/ml-explore/mlx-lm/blob/main/mlx_lm/LEARNED_QUANTS.md (accessed 2026-08-11) | |
| 240 | +- MLX Quantization on Apple Silicon: dynamic_quant vs AWQ vs GPTQ vs DWQ (Hannecke, Medium) — https://medium.com/@michael.hannecke/mlx-quantization-on-apple-silicon-dynamic-quant-vs-awq-vs-gptq-vs-dwq-8b2a5af2b53f (accessed 2026-08-11) | |
| 241 | +- Better inference quality and performance for MLX on Apple Silicon (Feldman; K-quant vs MLX affine KL measurements) — https://www.linkedin.com/pulse/better-inference-quality-performance-mlx-apple-silicon-asher-feldman-ztm0e (accessed 2026-08-11) | |
| 242 | +- Very slow IQ quant performance on Apple Silicon (llama.cpp discussion #5617, ikawrakow measurements) — https://github.com/ggml-org/llama.cpp/discussions/5617 (accessed 2026-08-11) | |
| 243 | +- Overview of GGUF quantization methods (r/LocalLLaMA; i-quant LUT bottleneck notes) — https://www.reddit.com/r/LocalLLaMA/comments/1ba55rj/overview_of_gguf_quantization_methods (accessed 2026-08-11) | |
| 244 | +- LLM Quantization Formats Compared: GGUF vs MLX vs EXL3 vs GPTQ vs AWQ vs FP8 (D-Central; format inventories) — https://d-central.tech/llm-quantization-formats (accessed 2026-08-11) | |
| 245 | +- GGUF vs MLX Quantization Formats on Apple Silicon (Contra Collective, 2026) — https://contracollective.com/blog/gguf-vs-mlx-quantization-formats-apple-silicon-2026 (accessed 2026-08-11) | |
| 246 | +- llama.cpp Metal Backend vs MLX: Compute Path Comparison (Contra Collective, 2026) — https://contracollective.com/blog/llama-cpp-metal-vs-mlx-backend-apple-silicon-2026 (accessed 2026-08-11) | |
| 247 | +- llama.cpp supports gpt-oss in native MXFP4 (discussion #15095) — https://github.com/ggml-org/llama.cpp/discussions/15095 (accessed 2026-08-11) | |
| 248 | +- exllamav3 / EXL3 trellis format (turboderp) — https://github.com/turboderp-org/exllamav3 (accessed 2026-08-11) | |
| 249 | +- KV Cache and Context Length on Apple Silicon (Contra Collective, 2026; llama.cpp/mlx-lm KV flags) — https://contracollective.com/blog/kv-cache-context-length-apple-silicon-local-inference-2026 (accessed 2026-08-11) | |
| 250 | +- Running LLMs locally on a Mac (MacKinlay; KV cache quant flags across runtimes) — https://danmackinlay.name/notebook/local_llm_mac.html (accessed 2026-08-11) | |
| 251 | +- KVSplit: differentiated K/V precision on Apple Silicon (Show HN) — https://news.ycombinator.com/item?id=44009321 (accessed 2026-08-11) | |
| 252 | +- TurboQuant — Extreme KV Cache Quantization with Metal kernels (llama.cpp discussion #20969) — https://github.com/ggml-org/llama.cpp/discussions/20969 (accessed 2026-08-11) | |
| 253 | +- M2Cache: Harnessing Your DRAM and SSD for Sustainable and Accessible LLM Inference with Mixed-Precision and Multi-level Caching — https://arxiv.org/abs/2410.14740 (accessed 2026-08-11) | |
| 254 | +- SSD Offloading for LLM MoE Weights Considered Harmful in Energy Efficiency — https://www.alphaxiv.org/overview/2508.06978 (accessed 2026-08-11) | |
| 255 | +- Agent Memory Below the Prompt: Persistent Q4 KV Cache for Multi-Agent LLM Inference on Edge Devices (MLX Q4-KV state of play) — https://arxiv.org/html/2603.04428v1 (accessed 2026-08-11) | |
| 256 | +- MLX vs llama.cpp on Apple Silicon: Benchmarks, M5 Neural Accelerators, Ollama switch — https://yage.ai/share/mlx-apple-silicon-en-20260331.html (accessed 2026-08-11) | |
added
research/notes/sparsity_pruning.md
+474 −0
@@ -0,0 +1,474 @@ | ||
| 1 | +--- | |
| 2 | +project: localvm-research | |
| 3 | +document: research/notes/sparsity_pruning | |
| 4 | +author: Simon-Pierre Boucher | |
| 5 | +contact: contact@spboucher.ai | |
| 6 | +created: 2026-08-11 | |
| 7 | +status: draft | |
| 8 | +--- | |
| 9 | + | |
| 10 | +# Activation Sparsity & Weight Pruning — Research Notes (charter §4.2, §4.3) | |
| 11 | + | |
| 12 | +Deep literature scan, 2022–2026. Focus: which sparsity/pruning mechanisms decouple | |
| 13 | +`total model size` from `resident model size` and `bytes read per token`, and which of them | |
| 14 | +actually map to Apple Silicon unified memory + Metal + internal NVMe. | |
| 15 | + | |
| 16 | +--- | |
| 17 | + | |
| 18 | +## 1. Landscape overview | |
| 19 | + | |
| 20 | +Two families, one shared premise: | |
| 21 | + | |
| 22 | +- **Activation sparsity (§4.2):** for a given input, only a small input-dependent subset of | |
| 23 | + FFN neurons / attention heads meaningfully contributes to the output. If that subset can be | |
| 24 | + *predicted cheaply before the matmul*, the untouched weights never need to be read — | |
| 25 | + a **bandwidth** and (with an offload tier) a **capacity** win, with no permanent model change. | |
| 26 | +- **Weight pruning (§4.3):** some weights contribute little for *all* inputs and can be removed | |
| 27 | + permanently (one-shot unstructured: SparseGPT, Wanda; structured: SliceGPT, LLM-Pruner, | |
| 28 | + Sheared-LLaMA, Minitron; depth: ShortGPT). A **capacity** win, but paid for with permanent | |
| 29 | + quality loss and (for unstructured) very poor hardware realizability. | |
| 30 | + | |
| 31 | +Key historical arc: | |
| 32 | + | |
| 33 | +1. **2022–2023 (ReLU era):** the "lazy neuron" observation — in ReLU transformers <10% of FFN | |
| 34 | + neurons fire per token (3.0% for T5-Base). DejaVu turns this into *contextual sparsity* with | |
| 35 | + per-layer predictors (up to 85% sparsity, 2× over FasterTransformer on OPT-175B). | |
| 36 | +2. **2023–2024 (offload era):** PowerInfer (hot/cold neurons split across GPU/CPU), | |
| 37 | + **LLM in a flash** (Apple: neurons paged from flash on demand), PowerInfer-2 / Ripple | |
| 38 | + (smartphones, UFS storage). Sparsity becomes a **cache/paging policy**, not just a FLOP saver. | |
| 39 | + This is the closest prior art to the localvm-research goal. | |
| 40 | +3. **2024–2026 (SwiGLU problem & training-free era):** modern models (Llama-2/3, Mistral, Qwen) | |
| 41 | + use SwiGLU — activations are no longer exactly zero. Responses: *ReLUfication* retraining | |
| 42 | + (ReLU Strikes Back, ProSparse, TurboSparse dReLU, Q-Sparse), or *training-free thresholding* | |
| 43 | + of near-zero activations (CATS, TEAL, GRIFFIN, DIP, R-Sparse, SparseInfer, statistical | |
| 44 | + calibration). Sirius (NeurIPS 2024) shows the quality cost is concentrated in reasoning tasks | |
| 45 | + and is *recoverable by correction*. | |
| 46 | +4. **Unresolved:** essentially all fast implementations are CUDA (Triton kernels, GPU/CPU split) | |
| 47 | + or Android/UFS. **No system implements predictor-driven sparse weight paging on | |
| 48 | + macOS/Metal/Apple NVMe.** PowerInfer's own README lists "Metal backend for sparse inference | |
| 49 | + on macOS" as *planned, not implemented* (macOS today = CPU only, "limited" gains). | |
| 50 | + | |
| 51 | +The most important conceptual distinction for us: pruning research asks *"which weights can we | |
| 52 | +delete?"*; offload-sparsity research asks *"which weights must be resident right now?"* The | |
| 53 | +second question converts pruning from a destructive transform into a **cache-residency policy** | |
| 54 | +— weights are demoted to NVMe, not destroyed, and recovered when the input distribution needs | |
| 55 | +them. Only a handful of systems (LLM in a flash, PowerInfer-2, M2Cache, DIP, Ripple) partially | |
| 56 | +do this, none on a Mac. | |
| 57 | + | |
| 58 | +--- | |
| 59 | + | |
| 60 | +## 2. Techniques and systems | |
| 61 | + | |
| 62 | +### 2.1 DejaVu — contextual sparsity (ICML 2023 oral) | |
| 63 | + | |
| 64 | +- **Mechanism.** Hypothesis: small, input-dependent sets of attention heads and FFN neurons | |
| 65 | + reproduce the dense output for each input. Trains small MLP predictors per layer; exploits | |
| 66 | + slowly-changing hidden states across layers to predict layer ℓ's sparsity from layer ℓ−1's | |
| 67 | + input (asynchronous "lookahead" hides predictor latency). | |
| 68 | +- **Memory / bandwidth.** No resident-memory reduction (full model stays in GPU HBM). Bandwidth: | |
| 69 | + up to 85% contextual sparsity ⇒ proportionally fewer weight bytes streamed from HBM per token. | |
| 70 | + Contextual sparsity has up to 7× better efficiency–accuracy trade-off than static sparsity. | |
| 71 | +- **Quality.** "Without compromising model quality" on OPT-175B; >2× latency reduction vs | |
| 72 | + FasterTransformer, >6× vs HuggingFace. | |
| 73 | +- **Predictor overhead.** Small per-layer MLPs; hidden by async lookahead on A100s. | |
| 74 | +- **Hardware assumptions.** OPT (ReLU FFN), multi-A100 discrete-GPU serving. The | |
| 75 | + GPU-resident-model assumption gives it no capacity benefit; irrelevant as-is for a | |
| 76 | + memory-constrained Mac, but the *predictability result* (85%) is foundational. | |
| 77 | +- **Limitation.** ReLU-dependent; whole model must fit in accelerator memory; predictors trained | |
| 78 | + per model. | |
| 79 | +- **Extension for us.** Reuse the "slowly changing hidden states" property as a *prefetch signal* | |
| 80 | + for NVMe reads rather than a FLOP-skip signal. | |
| 81 | + | |
| 82 | +### 2.2 PowerInfer — hot/cold neuron split on one consumer GPU (SOSP 2024) | |
| 83 | + | |
| 84 | +- **Mechanism.** Neuron activations follow a **power-law**: a small "hot" set is activated across | |
| 85 | + almost all inputs, the "cold" majority is input-specific. Hot neurons preloaded to GPU VRAM; | |
| 86 | + cold neurons computed on CPU; adaptive per-layer predictors + neuron-aware sparse operators. | |
| 87 | +- **Numbers.** OPT-175B-class models on one RTX 4090: 13.2 tok/s average, 29.08 peak — only 18% | |
| 88 | + below an A100. Requires ReLU-family models (ReluLLaMA, ProSparse-LLaMA, Bamboo, TurboSparse); | |
| 89 | + explicitly does **not** support vanilla Llama/Mistral/Qwen. | |
| 90 | +- **Memory / bandwidth.** GPU-resident set ≪ model size (capacity win via CPU DRAM as second | |
| 91 | + tier); bandwidth win from computing only predicted-active neurons. | |
| 92 | +- **Hardware assumptions.** **Discrete GPU + PCIe + separate CPU DRAM.** This split is | |
| 93 | + *meaningless on Apple unified memory* — there is no "GPU VRAM vs CPU RAM" distinction; the Mac | |
| 94 | + analogue is RAM (hot) vs NVMe (cold), i.e., exactly the LLM-in-a-flash setting. | |
| 95 | +- **macOS status (verified on repo).** Runs on Apple M chips CPU-only with "limited" improvement; | |
| 96 | + Metal sparse backend is a *planned feature that never shipped*. Project still active | |
| 97 | + (SmallThinker 2025, Tiiny AI Pocket Lab Jan 2026) but the Mac gap remains. | |
| 98 | +- **Extension.** Port the hot/cold *statistical* insight to a RAM/NVMe hierarchy with Metal | |
| 99 | + gather-matvec kernels; hot set pinned in wired memory, cold set demand-paged. | |
| 100 | + | |
| 101 | +### 2.3 LLM in a flash (Apple, ACL 2024) — deepest dive, closest prior art #1 | |
| 102 | + | |
| 103 | +*Alizadeh et al., arXiv 2312.11514. The only major paper that ran sparse weight paging on actual | |
| 104 | +Apple hardware.* | |
| 105 | + | |
| 106 | +**What exactly they did:** | |
| 107 | + | |
| 108 | +- **Hardware:** Apple **M1 Max** (1TB SSD) and **M2 Ultra** (2TB SSD), CPU (fp32) and Metal GPU | |
| 109 | + (fp16) paths; plus Linux RTX 4090 (bf16). Memory budget: ~**half the model size** in DRAM. | |
| 110 | +- **Models:** OPT-6.7B, sparsified/ReLUfied Falcon-7B, Persimmon-8B, Phi-2, FATReLU Llama-2-7B. | |
| 111 | + All ReLU-family FFNs. | |
| 112 | +- **Predictor:** low-rank predictor per FFN layer (OPT-6.7B: rank 128 for layers 1–28, rank 1024 | |
| 113 | + for last 4). Trained on 10k C4 samples, 2 epochs, ~4 h/predictor on A100. Cost: <2.4% of | |
| 114 | + non-embedding weights/FLOPs; ~5% false negatives, 7% false positives; 2.75% (CPU) / 4.8% (GPU) | |
| 115 | + of compute time. | |
| 116 | +- **Windowing:** keep the union of neurons activated in the last k=4–5 tokens in DRAM; per new | |
| 117 | + token only load the *delta*. OPT-6.7B at k=4: each token touches **2.4% of FFN neurons**; | |
| 118 | + FFN occupies only 15.5% of DRAM; total DRAM ≈ **52.1% of model size**. Falcon-7B: 3.1% of FFN | |
| 119 | + neurons/token, DRAM ≈ 52.9%. | |
| 120 | +- **Row–column bundling:** store up-projection column i together with down-projection row i so | |
| 121 | + one neuron = one contiguous 2·d_model read → doubles chunk size. Measured effective flash | |
| 122 | + throughput: ~1.25 GiB/s (predictor+windowing) → **~2.25 GiB/s with bundling**, vs >6 GiB/s | |
| 123 | + sequential on M1 Max. Sweet spot: ≥32 KiB random reads across **32 threads**. | |
| 124 | +- **Bytes/token (OPT-6.7B):** naive 13.4 GB → predictor only 6.7 GB → **predictor+windowing | |
| 125 | + 0.2 GB per token**. (This is the single most important measured number for our charter §8.3.) | |
| 126 | +- **Speedups:** OPT-6.7B total per-token latency 669 ms CPU (4.75× vs naive), 565 ms Metal/M1 | |
| 127 | + (4.23×), 305 ms Metal/M2 Ultra (7.44×), 84 ms CUDA (26.4×); +speculative → 37×. Runs models | |
| 128 | + up to **2× DRAM size**. | |
| 129 | + | |
| 130 | +**What they did NOT do (our opening):** | |
| 131 | + | |
| 132 | +- Attention weights + embeddings kept **permanently in DRAM** (~19–32% of model) — only FFN | |
| 133 | + weights are paged. No paging of attention, no KV-cache tiering. | |
| 134 | +- Predicated on **ReLU sparsity**; dense SwiGLU models "not addressed" (their words: the method | |
| 135 | + "is constructed on the foundation of sparsified networks"). | |
| 136 | +- Only ~2× DRAM oversubscription demonstrated on ≤8B models — not 4–10×, not 30–70B on a Mac. | |
| 137 | +- Single-batch, greedy decoding only; no power measurement; no quantization co-design (fp16 | |
| 138 | + neurons on flash — 4-bit bundles would quadruple effective neuron throughput). | |
| 139 | +- **No code release.** Nothing in MLX/llama.cpp implements it today. | |
| 140 | +- Negative result they report: bundling by *co-activation* (closest coactivated neighbor) failed | |
| 141 | + — hot neurons got loaded repeatedly. (Ripple later solved this with global placement.) | |
| 142 | + | |
| 143 | +### 2.4 PowerInfer-2 — deepest dive, closest prior art #2 | |
| 144 | + | |
| 145 | +*Xue et al., arXiv 2406.06282. First 47B model on a smartphone.* | |
| 146 | + | |
| 147 | +**What exactly they did:** | |
| 148 | + | |
| 149 | +- **Hardware:** OnePlus 12 (24 GB DRAM, 19 GB usable, UFS 4.0, Snapdragon 8 Gen 3) and OnePlus | |
| 150 | + Ace 2 (16 GB, UFS 3.1). UFS 4.0: ~4 GB/s sequential (512 KB), **~1 GB/s at 4 KB random**, | |
| 151 | + 850 MB/s over larger ranges. | |
| 152 | +- **Models:** TurboSparse-Mixtral-47B (dReLU MoE, ~3B activated params/token), | |
| 153 | + TurboSparse-Mistral-7B, Bamboo-7B, sparse Llama-13B, Qwen2-7B; SiLU Mistral-7B as a stress | |
| 154 | + case. | |
| 155 | +- **Neuron cluster abstraction:** groups of same-layer FFN neurons with similar activation | |
| 156 | + statistics. Hot (frequently active) clusters → large, dense, NPU-computed; cold clusters → | |
| 157 | + small, sparse, CPU-computed. Prefill: NPU dense matmuls while a CPU core streams weights. | |
| 158 | + Decode: NPU handles ~70% (hot dense part), CPU the sparse remainder. | |
| 159 | +- **Segmented neuron cache:** three DRAM regions — (1) pinned attention + KV, (2) hot region | |
| 160 | + (cluster-granularity LRU), (3) cold region (neuron-granularity LRU); ratios adapt to batch | |
| 161 | + size. | |
| 162 | +- **I/O pipeline:** neuron-cluster-level pipelining overlaps compute on cached clusters with UFS | |
| 163 | + reads of missing ones. Cold neurons stored as **Gate-Up-Down bundles** (~80% co-activation | |
| 164 | + across the three matrices). For 4-bit models: two-phase read — 4 KB Gate first, Up/Down 4 KB | |
| 165 | + fetched *only if* Gate output ≠ 0. | |
| 166 | +- **Results:** 11.68 tok/s decoding for TurboSparse-Mixtral-47B on 24 GB phone (up to 27.8–29× | |
| 167 | + vs llama.cpp; 3.84× vs "LLMFlash" i.e. LLM-in-a-flash-style baseline). With ~50% FFN offload | |
| 168 | + on a 7B: ~11.1 tok/s vs ~14.5 in-memory — offload costs only ~23%. | |
| 169 | +- **Overhead accounting (7 GB budget, 47B model):** 1 GB non-FFN weights + **2.6 GB predictors** | |
| 170 | + + 2.7 GB quantization scales + 0.3 GB runtime = 6.6 GB, leaving only 400 MB of neuron cache | |
| 171 | + (1.8% of FFN weights) — still runs. Predictor DRAM cost is *large*, a real design lesson. | |
| 172 | + | |
| 173 | +**What they did NOT do:** | |
| 174 | + | |
| 175 | +- Rooted **Android only**; "iOS portability" claimed but never validated. Nothing on | |
| 176 | + macOS/Metal/ANE. | |
| 177 | +- Depends on **dReLU ReLUfied models** (TurboSparse = SFT-retrained Mistral/Mixtral, ~150B | |
| 178 | + tokens); on stock SiLU models speedup drops to 2.4× (vs 4.6× ReLU). Not post-training-only. | |
| 179 | +- Offline per-device planner required; no cross-device generalization; high tail latency | |
| 180 | + (P99 +40.9% vs mean). | |
| 181 | +- Never tested Apple-class NVMe (6+ GB/s vs their 1 GB/s random UFS): **the Mac's storage is | |
| 182 | + ~4–6× faster than the storage this system was designed around** — its economics should | |
| 183 | + transfer favorably, but nobody has built it. | |
| 184 | + | |
| 185 | +### 2.5 Ripple / Neuralink — co-activation-aware flash layout (2024) | |
| 186 | + | |
| 187 | +- **Mechanism.** Neurons that fire together are placed together in flash ("neuron co-activation | |
| 188 | + linking"), converting many small random reads into fewer large sequential reads — directly | |
| 189 | + attacking the IOPS bound that limits smartphone (and, less severely, Mac) storage. | |
| 190 | +- **Relevance.** Solves exactly the negative result LLM in a flash reported for co-activation | |
| 191 | + bundling, via offline global placement optimization. An offline "layout compiler" of this | |
| 192 | + kind belongs in our compile stage (§14 of charter). Hardware: Android/UFS; no Mac port. | |
| 193 | + | |
| 194 | +### 2.6 ShadowLLM — better importance predictors (EMNLP 2024) | |
| 195 | + | |
| 196 | +- **Mechanism.** Single early-layer predictor "shadows" the whole model instead of per-layer | |
| 197 | + predictors; goes beyond magnitude-based criteria for head/neuron importance. | |
| 198 | +- **Numbers.** >15% end-to-end accuracy improvement over DejaVu-style criteria at equal sparsity; | |
| 199 | + up to 20% speedup over DejaVu; validated on OPT/Llama-2 up to 30B. Code: abdelfattah-lab/shadow_llm. | |
| 200 | +- **Hardware.** CUDA. Predictor design is transferable; a single-point predictor is attractive on | |
| 201 | + Mac because it gives *maximum I/O prefetch lead time* (predict at layer 0 → prefetch layer 30). | |
| 202 | + | |
| 203 | +### 2.7 ReLUfication line: ReLU Strikes Back → ProSparse → TurboSparse → Q-Sparse | |
| 204 | + | |
| 205 | +- **ReLU Strikes Back (Apple, ICLR 2024):** swapping SiLU/GELU→ReLU and fine-tuning has | |
| 206 | + negligible quality impact while enabling up to ~3× less computation/weight transfer at the | |
| 207 | + memory-bound decode step. Establishes that sparsity is *recoverable post-training* with modest | |
| 208 | + fine-tuning. | |
| 209 | +- **ProSparse (2402.13516):** activation substitution + progressive L1 regularization + threshold | |
| 210 | + shifting → **89.3% / 88.8% / 87.9%** activation sparsity on LLaMA2-7B/13B/MiniCPM-1B at parity | |
| 211 | + with the Swish originals. | |
| 212 | +- **TurboSparse (2406.05955):** **dReLU** + data-mix continued training (~150B tokens) → | |
| 213 | + Mistral-7B activates only **2.5B** params/token; Mixtral-47B activates **4.3B**; 2–5× decode | |
| 214 | + speedup; 11 tok/s on phones (feeds PowerInfer-2). | |
| 215 | +- **Q-Sparse (2407.10969, NeurIPS 2024):** top-K activation sparsification with STE during | |
| 216 | + training; full activation sparsity, inference-optimal scaling law for sparse LLMs; works with | |
| 217 | + BitNet 1.58-bit. Training-time method — violates our post-training-only constraint but maps | |
| 218 | + the ceiling. | |
| 219 | +- **Caveat for us:** all of these need GPU-scale fine-tuning (out of scope to *produce*, but the | |
| 220 | + checkpoints — ProSparse-LLaMA, Bamboo-7B, TurboSparse-Mistral/Mixtral — are downloadable and | |
| 221 | + are ideal *test vehicles* on the Mac). | |
| 222 | + | |
| 223 | +### 2.8 Training-free SwiGLU sparsity: CATS, TEAL, GRIFFIN, DIP, and successors | |
| 224 | + | |
| 225 | +- **CATS (COLM 2024):** thresholds the **gate output** of SwiGLU blocks (per-layer thresholds | |
| 226 | + from calibration distributions). ~**99% of base task performance at 50% FFN activation | |
| 227 | + sparsity** on Mistral-7B/Llama2-7B without fine-tuning; but only sparsifies Wup/Wdown ⇒ | |
| 228 | + ~25% model-wide sparsity. Custom GPU kernel gives wall-clock gains. | |
| 229 | +- **TEAL (ICLR 2025):** magnitude-thresholds **hidden states model-wide** (every matrix incl. | |
| 230 | + attention), exploiting zero-mean unimodal activation distributions. **40–50% model-wide | |
| 231 | + sparsity with minimal degradation** on Llama-2/-3/Mistral 7B–70B; 1.53×/1.8× decode speedup at | |
| 232 | + 40/50% via Triton gather kernels; composes with weight quantization. **CUDA-only kernels** — | |
| 233 | + the natural first Metal port target. | |
| 234 | +- **GRIFFIN (ICML 2024):** "flocking" — within a sequence, tokens activate largely the *same* FF | |
| 235 | + neurons. Selects experts once per sequence from the prompt, no training/calibration, works on | |
| 236 | + many non-ReLU activations. **50% of FF parameters with little-to-no degradation** (generation + | |
| 237 | + classification), lower latency. Sequence-level selection = coarse, *prefetch-friendly* | |
| 238 | + granularity (one NVMe read burst per sequence, not per token). | |
| 239 | +- **DIP — Dynamic Input Pruning with Cache-Aware Masking (Qualcomm, 2412.01380):** | |
| 240 | + predictor-free magnitude sparsification of SwiGLU + optional LoRA recovery + **cache-aware | |
| 241 | + masking**: the sparsity mask is chosen considering *what is already in the DRAM cache*, | |
| 242 | + trading tiny accuracy for large cache-hit-rate gains. Phi-3-Medium under mobile DRAM limits: | |
| 243 | + **46% less memory, 40% more throughput, <0.1 ppl loss** vs streaming the dense model. This is | |
| 244 | + the first explicit "sparsity-as-cache-policy" formulation — conceptually the closest paper to | |
| 245 | + our §4.3 key question, but on simulated mobile constraints, CUDA/simulation, no Mac. | |
| 246 | +- **Others (2024–2026):** SparseInfer (training-free sign-bit activation prediction); | |
| 247 | + Post-Training Statistical Calibration (2412.07174); R-Sparse (rank-aware, training-free, | |
| 248 | + attention+FFN); La RoSA (layerwise rotation before sparsification); Spark Transformer | |
| 249 | + (Google, NeurIPS 2025: restores FFN+attention sparsity with low-cost top-k predictor); | |
| 250 | + CETT/“Universal Properties” (2509.00454: sparsity potential *grows with model size*, first | |
| 251 | + diffusion-LLM study); Fast Forward (2602.00397: predictive FFN sparsity for *prefill*); | |
| 252 | + tree-structured FFN dynamic sparsity at scale (2604.08565); flexible N:M *activation* | |
| 253 | + sparsity benchmarking for next-gen accelerators (2509.22166). | |
| 254 | + | |
| 255 | +### 2.9 SparQ Attention — the attention-side analogue (ICML 2024) | |
| 256 | + | |
| 257 | +- Fetches only the KV-cache rows whose keys matter for the current query (rank the query's large | |
| 258 | + components, approximate scores, fetch top keys). **Up to 8× reduction in attention data | |
| 259 | + transfer** with negligible loss on Llama-2/3, Mistral, Gemma, Pythia; no fine-tuning. | |
| 260 | +- Relevance: our working-set question applies to KV as well as weights; SparQ shows | |
| 261 | + demand-driven fetching works for attention state. Complements FFN-side sparsity (weights | |
| 262 | + dominate at short context, KV at long context). | |
| 263 | + | |
| 264 | +### 2.10 One-shot weight pruning: SparseGPT & Wanda (verified numbers) | |
| 265 | + | |
| 266 | +- **SparseGPT (ICML 2023):** layer-wise sparse regression with approximate Hessian-based weight | |
| 267 | + updates; prunes OPT-175B/BLOOM-176B in <4.5 h on one GPU; 50–60% unstructured with small ppl | |
| 268 | + increase at 100B+ scale. | |
| 269 | +- **Wanda (ICLR 2024):** score = |weight| × ‖input activation‖, per-output-row, no weight update; | |
| 270 | + ~300× faster to compute than SparseGPT. | |
| 271 | +- **WikiText-2 perplexity (dense → magnitude / SparseGPT / Wanda at 50% unstructured):** | |
| 272 | + - LLaMA-7B: 5.68 → 17.29 / 7.22 / 7.26 | |
| 273 | + - LLaMA-13B: 5.09 → 20.21 / 6.21 / 6.15 | |
| 274 | + - LLaMA-65B: 3.56 → 5.90 / 4.57 / 4.57 | |
| 275 | + - LLaMA-2-70B: 3.12 → 4.98 / 3.98 / 3.98 | |
| 276 | + - 2:4 structured is much worse at small scale (LLaMA-2-7B: 5.12 → 11.02 Wanda). | |
| 277 | +- **Interpretation:** (i) one-shot 50% is nearly free only at ≥65B scale; at 7B it costs | |
| 278 | + ~25–28% ppl. (ii) Hardware-friendly 2:4 patterns are the most damaging. (iii) End-to-end gain | |
| 279 | + is meager even on NVIDIA (Wanda reports 1.24× e2e with 2:4 on LLaMA-7B) — and **Apple GPUs | |
| 280 | + have no sparse tensor cores at all**, so unstructured weight sparsity yields *zero* bandwidth | |
| 281 | + savings on a Mac unless the representation itself skips loads (which CSR-style indexing | |
| 282 | + overhead largely cancels — see Endor). | |
| 283 | +- **Reframing for us:** Wanda's per-input-statistics score is cheap enough to compute *online*; | |
| 284 | + it can rank weights for **residency** (RAM vs NVMe) instead of deletion. | |
| 285 | + | |
| 286 | +### 2.11 Structured / depth / width pruning: SliceGPT, LLM-Pruner, ShortGPT, Sheared-LLaMA, Minitron | |
| 287 | + | |
| 288 | +- **SliceGPT (ICLR 2024):** orthogonal-rotation + slicing; removes up to 25% of parameters with | |
| 289 | + 99% (Llama-2-70B, OPT-66B) / 90% (Phi-2) zero-shot retention; produces smaller *dense* | |
| 290 | + matrices — the only pruning family that trivially runs fast on Metal (it's just smaller GEMMs). | |
| 291 | +- **LLM-Pruner (NeurIPS 2023):** gradient-based structural pruning + LoRA recovery; ~20% | |
| 292 | + compression at moderate loss. | |
| 293 | +- **ShortGPT (2403.03853):** Block Influence = 1 − cos-sim(layer input, output); middle-to-late | |
| 294 | + layers (e.g. 21–29 of LLaMA-2-7B) are highly redundant; deleting whole layers beats fancier | |
| 295 | + methods on many benchmarks, though generation/reasoning suffer more than classification | |
| 296 | + (follow-ups: Prune&Comp, E³-Pruner, layer-pruning-limits studies 2025–26). | |
| 297 | +- **Sheared-LLaMA (ICLR 2024):** Lagrangian-learned masks over layers/heads/dims + dynamic batch | |
| 298 | + loading; LLaMA2-7B → 1.3B/2.7B using ~50B tokens (3% of from-scratch compute); beats equal-size | |
| 299 | + models trained from scratch. | |
| 300 | +- **Minitron (NVIDIA 2024):** iterative width(+depth) pruning + KD; Nemotron-4 15B → 8B/4B at up | |
| 301 | + to 40× fewer tokens than from-scratch; Llama-3.1-Minitron-4B-Depth distilled on ~380B tokens. | |
| 302 | +- **Assessment:** all of these *permanently* shrink the model — excellent baselines, but they | |
| 303 | + answer the wrong question for us (they reduce total size, not the size↔residency coupling), | |
| 304 | + and the strong ones require retraining budgets we don't have. Their *diagnostics* (Block | |
| 305 | + Influence, activation-norm channel ranking) are directly reusable as residency scores. | |
| 306 | + | |
| 307 | +### 2.12 Sparsity + offload hybrids: M2Cache, Endor, RAP | |
| 308 | + | |
| 309 | +- **M2Cache (2410.14740):** neuron-level mixed precision + three-tier cache **HBM → DRAM → SSD**; | |
| 310 | + important neurons fp16, colder ones more aggressively quantized, coldest on SSD; LRU at neuron | |
| 311 | + granularity. The only published system with an explicit *SSD tier in a neuron cache hierarchy*. | |
| 312 | + Server GPUs, not Mac. | |
| 313 | +- **Endor (2406.11674):** key observation for §4.3 — CSR-style formats for unstructured-pruned | |
| 314 | + LLMs spend the saved bytes on indices, so offloaded pruned models are usually stored *dense*; | |
| 315 | + proposes a hardware-friendly bitmap format so pruned weights actually transfer fewer bytes | |
| 316 | + from SSD/flash. Directly applicable to any Mac NVMe streaming design. | |
| 317 | +- **RAP (2505.17138):** RL-guided *runtime* elastic pruning of weights + KV under a live memory | |
| 318 | + budget — "pruning as a scheduling decision," another step toward pruning-as-policy. | |
| 319 | + | |
| 320 | +### 2.13 Quality reality check: Sirius | |
| 321 | + | |
| 322 | +- **Sirius (NeurIPS 2024):** systematic evaluation shows contextual-sparsity models hold up on | |
| 323 | + prompt-understanding tasks but **significantly degrade on reasoning/deduction/knowledge | |
| 324 | + (GSM8K, coding)**; yet sparse and dense models share problem-solving structure, and correcting | |
| 325 | + only ~**11% of tokens** (dense-model verification, KV/hardware-efficient) restores full | |
| 326 | + accuracy at ~78% of the theoretical efficiency gain. | |
| 327 | +- **Implication:** any localvm design that uses aggressive sparsity should budget a | |
| 328 | + dense-verification / correction path (cheap on unified memory since CPU+GPU share the cache); | |
| 329 | + perplexity alone will overstate quality (CATS/TEAL-style "99% retention" claims are mostly | |
| 330 | + perplexity + short benchmarks, not multi-step reasoning). | |
| 331 | + | |
| 332 | +--- | |
| 333 | + | |
| 334 | +## 3. Relevance to localvm-research | |
| 335 | + | |
| 336 | +### 3.1 How predictable are activation patterns, really? (measured numbers) | |
| 337 | + | |
| 338 | +- **Per-token sparsity (ReLU models):** 85% contextual sparsity (DejaVu, OPT); 2.4–3.1% of FFN | |
| 339 | + neurons touched per token (LLM in a flash, OPT-6.7B/Falcon-7B); ~90% (ProSparse); <10% FFN | |
| 340 | + activation in ReLU transformers generally (Lazy Neuron). | |
| 341 | +- **Temporal stability:** windowing over k=4–5 tokens works — after caching the last-4-token | |
| 342 | + union, the per-token *delta* of new neurons is small enough to cut bytes/token from 6.7 GB | |
| 343 | + (predictor alone) to **0.2 GB** (OPT-6.7B). This is a direct, measured confirmation of our | |
| 344 | + Experiment B hypothesis on real hardware. | |
| 345 | +- **Spatial/structural stability:** power-law hot/cold split (PowerInfer); ~80% co-activation of | |
| 346 | + Gate-Up-Down bundles per neuron (PowerInfer-2); strong cross-neuron co-activation exploitable | |
| 347 | + by layout (Ripple); **flocking** — sequence-level shared expert sets (GRIFFIN, 50% FF params | |
| 348 | + chosen once per prompt). | |
| 349 | +- **Predictor accuracy/cost:** low-rank per-layer predictors reach ~5% FN / 7% FP at <2.4% | |
| 350 | + weight/FLOP overhead (LLM in a flash); but at scale predictors get heavy — 2.6 GB DRAM for a | |
| 351 | + 47B model (PowerInfer-2). Predictor-free thresholding (CATS/TEAL/DIP) eliminates this cost at | |
| 352 | + some accuracy expense; ShadowLLM shows one early predictor can serve all layers (better | |
| 353 | + accuracy and more prefetch lead time). | |
| 354 | +- **Open question for our Experiments A–C:** all the stability numbers above are from ReLU or | |
| 355 | + ReLUfied models; nobody has published Jaccard/transition statistics for *thresholded SwiGLU* | |
| 356 | + working sets (TEAL/CATS-style masks) — we should measure this ourselves on Llama-3/Qwen-class | |
| 357 | + models before building anything. | |
| 358 | + | |
| 359 | +### 3.2 Does sparsity survive in non-ReLU models? | |
| 360 | + | |
| 361 | +Partially, and this is the field's central tension: | |
| 362 | + | |
| 363 | +- Exact zeros: essentially none in SwiGLU (DIP: "little inherent sparsity"). | |
| 364 | +- *Approximate* sparsity: 40–50% model-wide prunable per token with minimal ppl loss (TEAL); | |
| 365 | + 50% FFN with ~99% task retention (CATS); 50% FF via flocking (GRIFFIN); and the recoverable | |
| 366 | + ceiling with fine-tuning is ~85–90% (ProSparse/TurboSparse). Universal-properties (2509.00454) | |
| 367 | + finds effective sparsity *increases with model size* — good news since our targets are big. | |
| 368 | +- But 40–50% ≠ 97%. For SSD paging, a 2× reduction in bytes/token is helpful yet far from the | |
| 369 | + 50× that ReLU windowing achieved. Bridging options: combine threshold sparsity with | |
| 370 | + quantization (TEAL composes), low-rank hot path + sparse cold residual (R-Sparse suggests the | |
| 371 | + decomposition), Sirius-style correction, or accept ReLUfied checkpoints where they exist. | |
| 372 | +- Quality caveat: Sirius's reasoning-degradation result means our quality metrics (§8.2) must | |
| 373 | + include GSM8K/coding-style tasks, not just perplexity. | |
| 374 | + | |
| 375 | +### 3.3 What maps to Apple Silicon, and what doesn't | |
| 376 | + | |
| 377 | +| Assumption in prior work | Apple Silicon reality | | |
| 378 | +|---|---| | |
| 379 | +| GPU VRAM vs CPU DRAM split (DejaVu, PowerInfer) | No split — unified memory. The only meaningful hierarchy is **RAM vs NVMe** | | |
| 380 | +| PCIe transfer cost motivates hot/cold placement | Zero-copy CPU↔GPU; placement = *residency*, not device | | |
| 381 | +| 2:4 sparse tensor cores (Wanda/SparseGPT speedups) | **Absent on Apple GPUs** — N:M weight sparsity gives no free lunch; savings must come from avoided *loads* in custom Metal kernels | | |
| 382 | +| UFS 4.0: ~4 GB/s seq, ~1 GB/s 4K random (PowerInfer-2) | Apple NVMe: >6 GiB/s seq measured (M1 Max); ~2.25 GiB/s effective sparse reads at 32 KiB×32 threads — **the storage is 2–6× better than what PowerInfer-2 was built for** | | |
| 383 | +| Linux O_DIRECT/io_uring | macOS F_NOCACHE + many-threaded pread; APFS page cache behaves differently (our Experiment H) | | |
| 384 | +| Triton/CUDA gather kernels (TEAL, CATS, DejaVu) | **No Metal equivalents exist anywhere** (PowerInfer's Metal sparse backend: never shipped) | | |
| 385 | +| Rooted Android, offline per-device planner (PowerInfer-2) | Full user control of macOS; can pin memory (wired), use mmap, run calibration at "compile" time | | |
| 386 | + | |
| 387 | +### 3.4 The §4.3 key question: pruning as a cache policy — prior-art verdict | |
| 388 | + | |
| 389 | +Searched explicitly for "discarded weights live on SSD and are recovered on demand": | |
| 390 | + | |
| 391 | +- **Partially done:** LLM in a flash (sparsity-driven demand paging of FFN weights from flash — | |
| 392 | + but ReLU-only, ≤2× oversubscription, no code, attention pinned); PowerInfer-2 + Ripple (same | |
| 393 | + idea on Android/UFS with cluster caches and layout optimization); M2Cache (neuron LRU over | |
| 394 | + HBM/DRAM/SSD with mixed precision); DIP (**mask chosen as a function of cache contents** — | |
| 395 | + pruning literally becomes the cache policy, but only simulated mobile constraints); | |
| 396 | + RAP (runtime pruning under memory budget, no SSD recovery); On-Demand Multi-Task Sparsity | |
| 397 | + (2511.19986, edge, task-level sparse deltas from flash); VLM in a flash (2511.18692, neuron | |
| 398 | + chunking for I/O-efficient VLM sparsification — shows the line is alive in 2025–26). | |
| 399 | +- **Not done anywhere (verified gap):** (a) using a **one-shot pruning importance score | |
| 400 | + (Wanda/SparseGPT-style) as the *static tier-assignment* policy** — "pruned" weights demoted to | |
| 401 | + NVMe in an Endor-style dense-readable format and *re-materialized* when a cheap online | |
| 402 | + statistic (input norms, gate outputs, flocking profile) says they matter, restoring the dense | |
| 403 | + model's quality ceiling instead of accepting permanent 7B-scale pruning damage; | |
| 404 | + (b) any of this on **macOS / Metal / unified memory / Apple NVMe**; (c) combination with | |
| 405 | + **layer-level** granularity (ShortGPT-scored cold layers paged in only when a router detects | |
| 406 | + they're needed); (d) 4-bit-quantized neuron bundles on flash (PowerInfer-2 does 4-bit on UFS; | |
| 407 | + LLM in a flash used fp16 — nobody did quantized bundles on Apple NVMe with Metal decode). | |
| 408 | + | |
| 409 | +### 3.5 What's unexplored (candidate experiment seeds) | |
| 410 | + | |
| 411 | +1. **Rebuild the LLM-in-a-flash measurement stack on our M5 Max** (Experiment H + E): reproduce | |
| 412 | + the 32 KiB×32-thread flash-read curve, then measure TEAL-style 50% SwiGLU masks as a *paging* | |
| 413 | + policy (not a FLOP policy) on Llama-3-8B/Qwen-14B class models. Nobody has published | |
| 414 | + bytes/token for thresholded-SwiGLU paging. | |
| 415 | +2. **Working-set statistics for SwiGLU masks** (Experiments A–C): Jaccard(t, t+1), window-union | |
| 416 | + growth curves, cross-prompt domain overlap — the numbers exist only for ReLU models. | |
| 417 | +3. **Wanda-score residency tiers + Sirius-style correction**: keep top-p% weights (by | |
| 418 | + |W|·‖x‖ calibration) resident, stream the rest on demand from NVMe, dense-verify | |
| 419 | + occasionally. This composes three verified results into a system nobody has built, on | |
| 420 | + hardware (fast NVMe + unified memory + Metal) that is *more* favorable than anything prior | |
| 421 | + work targeted. | |
| 422 | + | |
| 423 | +--- | |
| 424 | + | |
| 425 | +## Sources | |
| 426 | + | |
| 427 | +- Deja Vu: Contextual Sparsity for Efficient LLMs at Inference Time — https://arxiv.org/abs/2310.17157 (accessed 2026-08-11) | |
| 428 | +- Deja Vu (OpenReview, ICML 2023) — https://openreview.net/forum?id=wIPIhHd00i (accessed 2026-08-11) | |
| 429 | +- PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU — https://arxiv.org/abs/2312.12456 (accessed 2026-08-11) | |
| 430 | +- PowerInfer (SOSP 2024 paper PDF, IPADS/SJTU) — https://ipads.se.sjtu.edu.cn/_media/publications/song-sosp24.pdf (accessed 2026-08-11) | |
| 431 | +- PowerInfer GitHub (macOS/Metal support status, supported ReLU models) — https://github.com/SJTU-IPADS/PowerInfer (accessed 2026-08-11) | |
| 432 | +- PowerInfer-2: Fast Large Language Model Inference on a Smartphone — https://arxiv.org/abs/2406.06282 (accessed 2026-08-11) | |
| 433 | +- PowerInfer-2 project page — https://powerinfer.ai/v2/ (accessed 2026-08-11) | |
| 434 | +- LLM in a flash: Efficient Large Language Model Inference with Limited Memory (Apple) — https://arxiv.org/abs/2312.11514 (accessed 2026-08-11) | |
| 435 | +- ReLU Strikes Back: Exploiting Activation Sparsity in Large Language Models (Apple, ICLR 2024) — https://arxiv.org/abs/2310.04564 (accessed 2026-08-11) | |
| 436 | +- ReLU Strikes Back — Apple Machine Learning Research page — https://machinelearning.apple.com/research/relu (accessed 2026-08-11) | |
| 437 | +- The Lazy Neuron Phenomenon: On Emergence of Activation Sparsity in Transformers — https://arxiv.org/abs/2210.06313 (accessed 2026-08-11) | |
| 438 | +- TEAL: Training-Free Activation Sparsity in Large Language Models — https://arxiv.org/abs/2408.14690 (accessed 2026-08-11) | |
| 439 | +- TEAL — Together AI blog — https://www.together.ai/blog/teal-training-free-activation-sparsity-in-large-language-models (accessed 2026-08-11) | |
| 440 | +- CATS: Contextually-Aware Thresholding for Sparsity in Large Language Models (COLM 2024) — https://arxiv.org/abs/2404.08763 (accessed 2026-08-11) | |
| 441 | +- CATS GitHub — https://github.com/ScalingIntelligence/CATS (accessed 2026-08-11) | |
| 442 | +- GRIFFIN: Prompt-prompted Adaptive Structured Pruning for Efficient LLM Generation (ICML 2024) — https://arxiv.org/abs/2404.01365 (accessed 2026-08-11) | |
| 443 | +- GRIFFIN GitHub — https://github.com/hdong920/GRIFFIN (accessed 2026-08-11) | |
| 444 | +- ShadowLLM: Predictor-based Contextual Sparsity for Large Language Models (EMNLP 2024) — https://arxiv.org/abs/2406.16635 (accessed 2026-08-11) | |
| 445 | +- ShadowLLM — ACL Anthology — https://aclanthology.org/2024.emnlp-main.1068/ (accessed 2026-08-11) | |
| 446 | +- ProSparse: Introducing and Enhancing Intrinsic Activation Sparsity within Large Language Models — https://arxiv.org/abs/2402.13516 (accessed 2026-08-11) | |
| 447 | +- Turbo Sparse: Achieving LLM SOTA Performance with Minimal Activated Parameters — https://arxiv.org/abs/2406.05955 (accessed 2026-08-11) | |
| 448 | +- Q-Sparse: All Large Language Models can be Fully Sparsely-Activated (NeurIPS 2024) — https://arxiv.org/abs/2407.10969 (accessed 2026-08-11) | |
| 449 | +- Sirius: Contextual Sparsity with Correction for Efficient LLMs (NeurIPS 2024) — https://arxiv.org/abs/2409.03856 (accessed 2026-08-11) | |
| 450 | +- SparQ Attention: Bandwidth-Efficient LLM Inference (ICML 2024) — https://arxiv.org/abs/2312.04985 (accessed 2026-08-11) | |
| 451 | +- SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot (ICML 2023) — https://arxiv.org/abs/2301.00774 (accessed 2026-08-11) | |
| 452 | +- Wanda: A Simple and Effective Pruning Approach for Large Language Models (ICLR 2024) — https://arxiv.org/abs/2306.11695 (accessed 2026-08-11) | |
| 453 | +- ShortGPT: Layers in Large Language Models are More Redundant Than You Expect — https://arxiv.org/abs/2403.03853 (accessed 2026-08-11) | |
| 454 | +- Sheared LLaMA: Accelerating Language Model Pre-training via Structured Pruning (ICLR 2024) — https://arxiv.org/abs/2310.06694 (accessed 2026-08-11) | |
| 455 | +- Compact Language Models via Pruning and Knowledge Distillation (Minitron, NVIDIA) — https://arxiv.org/abs/2407.14679 (accessed 2026-08-11) | |
| 456 | +- LLM Pruning and Distillation in Practice: The Minitron Approach — https://arxiv.org/pdf/2408.11796 (accessed 2026-08-11) | |
| 457 | +- SliceGPT: Compress Large Language Models by Deleting Rows and Columns (ICLR 2024) — https://arxiv.org/abs/2401.15024 (accessed 2026-08-11) | |
| 458 | +- LLM-Pruner: On the Structural Pruning of Large Language Models (NeurIPS 2023) — https://arxiv.org/abs/2305.11627 (accessed 2026-08-11) | |
| 459 | +- M2Cache: Harnessing Your DRAM and SSD for Sustainable and Accessible LLM Inference with Mixed-Precision and Multi-level Caching — https://arxiv.org/abs/2410.14740 (accessed 2026-08-11) | |
| 460 | +- Ripple/Neuralink: Accelerating LLM Inference on Smartphones with Correlation-Aware Neuron Management / Neuron Co-Activation Linking — https://arxiv.org/abs/2410.19274 (accessed 2026-08-11) | |
| 461 | +- DIP: Efficient LLM Inference using Dynamic Input Pruning and Cache-Aware Masking (Qualcomm AI Research) — https://arxiv.org/abs/2412.01380 (accessed 2026-08-11) | |
| 462 | +- Endor: Hardware-Friendly Sparse Format for Offloaded LLM Inference — https://arxiv.org/pdf/2406.11674 (accessed 2026-08-11) | |
| 463 | +- SparseInfer: Training-free Prediction of Activation Sparsity for Fast LLM Inference — https://arxiv.org/pdf/2411.12692 (accessed 2026-08-11) | |
| 464 | +- Post-Training Statistical Calibration for Higher Activation Sparsity — https://arxiv.org/pdf/2412.07174 (accessed 2026-08-11) | |
| 465 | +- R-Sparse: Rank-Aware Activation Sparsity for Efficient LLM Inference — https://arxiv.org/abs/2504.19449 (accessed 2026-08-11) | |
| 466 | +- Spark Transformer: Reactivating Sparsity in FFN and Attention (NeurIPS 2025) — https://arxiv.org/html/2506.06644v2 (accessed 2026-08-11) | |
| 467 | +- Universal Properties of Activation Sparsity in Modern Large Language Models — https://arxiv.org/abs/2509.00454 (accessed 2026-08-11) | |
| 468 | +- RAP: Runtime Adaptive Pruning for LLM Inference — https://arxiv.org/pdf/2505.17138 (accessed 2026-08-11) | |
| 469 | +- DuoGPT: Training-free Dual Sparsity through Activation-aware Pruning in LLMs — https://arxiv.org/html/2506.20194 (accessed 2026-08-11) | |
| 470 | +- Motivating Next-Gen Accelerators with Flexible (N:M) Activation Sparsity — https://arxiv.org/pdf/2509.22166 (accessed 2026-08-11) | |
| 471 | +- VLM in a flash: I/O-Efficient Sparsification of Vision-Language Model via Neuron Chunking — https://arxiv.org/html/2511.18692 (accessed 2026-08-11) | |
| 472 | +- On-Demand Multi-Task Sparsity for Efficient Large-Model Deployment on Edge Devices — https://arxiv.org/pdf/2511.19986 (accessed 2026-08-11) | |
| 473 | +- Fast Forward: Accelerating LLM Prefill with Predictive FFN Sparsity — https://arxiv.org/pdf/2602.00397 (accessed 2026-08-11) | |
| 474 | +- Dynamic sparsity in tree-structured feed-forward layers at scale — https://arxiv.org/pdf/2604.08565 (accessed 2026-08-11) | |
added
research/notes/speculation_error_stability.md
+245 −0
@@ -0,0 +1,245 @@ | ||
| 1 | +--- | |
| 2 | +project: localvm-research | |
| 3 | +document: research/notes/speculation_error_stability | |
| 4 | +author: Simon-Pierre Boucher | |
| 5 | +contact: contact@spboucher.ai | |
| 6 | +created: 2026-08-11 | |
| 7 | +status: draft | |
| 8 | +--- | |
| 9 | + | |
| 10 | +# Speculative execution, numerical error analysis, and output-decision stability (charter §4.7, §4.9, §4.10) | |
| 11 | + | |
| 12 | +Reading notes for three intertwined questions that matter to localvm-research: | |
| 13 | + | |
| 14 | +1. **§4.7** — Can we compute with a *cheap approximation* of the model and pay for the *exact* model only when needed (speculation + verification)? | |
| 15 | +2. **§4.9** — Can we *bound* the error introduced by approximation (quantization, truncated residuals, partial GEMM) tightly enough to know when the cheap answer is already correct? | |
| 16 | +3. **§4.10** — Empirically, how often does an approximate model make the *same decision* (same greedy token, same top-k, close distribution) as the exact model — because that frequency is exactly the fraction of tokens for which we would never need to touch the expensive weights. | |
| 17 | + | |
| 18 | +The through-line: **every speedup number in the speculative-decoding literature is simultaneously a measurement of decision stability.** An acceptance rate of 0.8 for a draft model *is* the statement "the cheap model's greedy/sampled token matches the exact model's 80% of the time." This literature therefore already contains a large body of measured evidence for charter Experiments B, D, E and G. | |
| 19 | + | |
| 20 | +--- | |
| 21 | + | |
| 22 | +## 1. Landscape | |
| 23 | + | |
| 24 | +Three research communities touch our problem and barely cite each other: | |
| 25 | + | |
| 26 | +- **Speculative decoding (ML systems, 2022–now).** Draft-then-verify across *future tokens*. Mathematically mature (exact-distribution rejection sampling), enormous empirical corpus (2–6.5× wall-clock speedups), and, crucially for us, verified on Apple Silicon (Apple ReDrafter in MLX; QuantSpec explicitly targets edge). Verification granularity is always **the whole target model over a block of tokens**. | |
| 27 | +- **Certified robustness / formal verification of NNs (2018–now).** Interval bound propagation, CROWN/LiRPA, zonotopes (DeepT), Lipschitz analysis. Gives *sound* guarantees of the form "no perturbation within ε can flip the argmax," which is exactly the primitive we would want for "certified early termination." Current reality: does not scale past small transformers; self-attention is not even globally Lipschitz. | |
| 28 | +- **Mixed-precision numerical linear algebra (Higham school, 2017–now).** Iterative refinement with low-precision factorizations (GMRES-IR), probabilistic rounding-error analysis with √n instead of n error constants. This community routinely does what we want to do — *compute cheap, refine to full accuracy with a guarantee* — but for linear systems, not for transformer inference. It is the strongest source of transferable ideas that the LLM literature has not absorbed. | |
| 29 | + | |
| 30 | +A fourth, more empirical strand — **quantization-quality measurement** (KL divergence, "flips," same-top-token rates from llama.cpp tooling and the Microsoft "Accuracy is Not All You Need" paper) — provides the measured token-stability numbers §4.10 asks for. | |
| 31 | + | |
| 32 | +--- | |
| 33 | + | |
| 34 | +## 2. Technique-by-technique map | |
| 35 | + | |
| 36 | +### 2.1 Classical speculative decoding (draft model + rejection sampling) | |
| 37 | + | |
| 38 | +- **Mechanism.** A small draft model proposes γ tokens autoregressively; the target model scores all γ+1 positions in *one* forward pass; a rejection-sampling rule accepts a prefix and resamples the first rejected position from a corrected residual distribution. | |
| 39 | +- **Exactness guarantee.** The modified rejection sampling of Leviathan et al. / Chen et al. provably preserves the target model's output distribution *exactly* (token-level acceptance probability min(1, p(x)/q(x)), resample from norm(max(0, p−q))). This is the canonical proof that **an approximate computation plus a cheap correction can equal the exact computation in distribution** — the conceptual anchor for everything in this note. | |
| 40 | +- **Numbers.** Leviathan et al.: 2–3× on T5-XXL (2.6× at temp=1, **3.4× at temp=0** on translation; 2.3×/3.1× summarization). Chen et al.: 2–2.5× on Chinchilla-70B. Note the consistent pattern: **greedy (temp 0) speedups exceed sampling speedups**, i.e., argmax decisions are easier to reproduce than full distributions — direct evidence for §4.10. | |
| 41 | +- **Hardware assumptions.** Works precisely because decode is memory-bandwidth-bound: verifying k tokens costs ≈ the same wall-clock as generating 1, since the weights are read once either way. **Apple Silicon: same regime** (unified memory, bandwidth-bound decode); llama.cpp (`--model-draft`) and mlx_lm both ship draft-model speculation on Metal. | |
| 42 | +- **Limitation.** Needs a well-aligned draft model; acceptance rate α drives everything. Real-world α measured at 0.6–0.8 (BentoML production benchmarks with EAGLE-3-class drafts), not the near-1.0 of idealized analyses. | |
| 43 | +- **Extension opportunity for us.** Reinterpret "draft model" as "cheap *representation* of the same model" (low-bit base) and "target model" as "base + SSD-resident residual." The rejection-sampling correction then makes a *progressive-precision* runtime exact in distribution. | |
| 44 | + | |
| 45 | +### 2.2 Medusa / EAGLE / ReDrafter / Speculative Streaming (auxiliary-head speculation) | |
| 46 | + | |
| 47 | +- **Mechanism.** Instead of a separate draft LLM, small heads (Medusa: parallel FFN heads; EAGLE: a 1-layer autoregressive head over target *features*; ReDrafter: an RNN head; Speculative Streaming: multi-stream attention inside the target) predict a *tree* of candidate continuations, verified with tree attention in one pass. | |
| 48 | +- **Numbers.** Medusa 2.2–3.6× (Vicuna 7B/13B). EAGLE-3: **3.0–6.5×**, average acceptance length up to 7.5 (HumanEval), 20–40% over EAGLE-2, 1.38× throughput even at batch 64 in SGLang. ReDrafter: up to 3.5 tokens/step, 2.8× on H100 — and **on Apple Silicon via MLX: 1.37× on M1 Max, up to 2.3× on M2 Ultra** (the single most directly relevant Apple-hardware datapoint in this literature). Apple's Speculative Streaming: 1.8–3.1× with ~10,000× fewer extra parameters than Medusa. | |
| 49 | +- **Exactness.** All use strict speculative-sampling acceptance ⇒ lossless w.r.t. the target model. | |
| 50 | +- **Limitation.** Heads must be trained (EAGLE-3 needs a training-time test procedure); acceptance is task-dependent (templated code ≫ open chat). | |
| 51 | +- **Opportunity.** EAGLE's key insight — *draft from the target's own hidden features, not from tokens* — transfers to weight-paging: hidden states are a strong predictor of what comes next, hence of *which weights will matter next* (cf. §2.7). | |
| 52 | + | |
| 53 | +### 2.3 Self-speculative decoding: the model drafts with a subset of itself | |
| 54 | + | |
| 55 | +This family is the closest published relative of "speculate with cheap weights, verify with full weights." | |
| 56 | + | |
| 57 | +- **Draft & Verify (Zhang et al.)**: draft by *skipping layers* of the target (selected by Bayesian optimization), verify with the full model. 1.2–1.56× (Llama-2-13B: 1.56× CNN/DM). No extra model, no training. | |
| 58 | +- **LayerSkip (Elhoushi et al., ACL 2024)**: train with layer dropout + early-exit loss so layer-E exits are usable drafts; verification reuses the draft's KV cache (the draft pass is a *prefix* of the verify pass, so drafting work is not thrown away). Speedups 1.34–2.16×. **Measured acceptance: 68.9% (Llama-2-7B, exit at layer 8/32) and 74.5% (13B, exit 15/40)** — i.e., roughly *three quarters of token decisions made with ~25–40% of the depth survive exact verification*. Strong quantitative support for charter hypothesis "decisions stabilize early." | |
| 59 | +- **Kangaroo (NeurIPS 2024)**: fixed shallow sub-network (2–3 layers!) + tiny adapter as the draft, with *confidence-based early stopping of drafting* (stop speculating when the draft's own confidence drops). | |
| 60 | +- **SWIFT (ICLR 2025)**: plug-and-play, optimizes the skipped-layer set *on the fly* per input stream; 1.3–1.6×, no training at all. | |
| 61 | +- **CLaSp**: dynamic layer-skip schedule updated after every verification step using the last verified hidden states; 1.3–1.7× on Llama-3; for Llama-3-70B the optimum skipped **44 of 80 layers**. | |
| 62 | +- **Hardware.** All of these are single-model and memory-light — well suited to unified memory. Nothing architecturally CUDA-specific. | |
| 63 | +- **Limitation.** The draft is a *depth* truncation. Nobody in this family drafts with a *precision* truncation of the weights while verifying with the full-precision weights streamed on demand (see §3). | |
| 64 | + | |
| 65 | +### 2.4 Precision-level self-speculation (QSpec, QuantSpec, ML-SpecQD) | |
| 66 | + | |
| 67 | +The newest and most relevant variant: the draft and target share **the same weights at different precisions**. | |
| 68 | + | |
| 69 | +- **QSpec (EMNLP 2025).** One W4-quantized weight set; draft in **W4A4** (fast low-precision activations), verify in **W4A16**. Token generation between the two modes is "highly similar" (their Sec. 2.2 validation); switching cost near zero because weights and KV are shared. Up to **1.64×** over the high-precision baseline, **no quality loss**, plug-and-play (no training). | |
| 70 | +- **QuantSpec (Apple + Berkeley, ICML 2025).** Self-speculation for long context: draft uses **4-bit hierarchical KV cache + 4-bit weights**, target uses full precision. Acceptance **>90%**, end-to-end up to **~2.5×**, ~1.3× memory reduction. Explicitly motivated by *edge deployment*. | |
| 71 | +- **ML-SpecQD (2025).** Multi-level pipeline: MXFP4 quantized draft (a direct 4-bit cast of the target — "quantized drafts as a turnkey solution, no custom draft pretraining") possibly itself accelerated by a smaller draft; >2× over 16-bit baselines. | |
| 72 | +- **Why this matters to us.** These systems prove the core premise of a progressive-precision runtime: **a 4-bit version of the same model agrees with the 16-bit version on the (large) majority of tokens, and the disagreements can be repaired exactly by a shared-weight verification pass.** What none of them do: exploit the agreement to avoid *storing/loading* the high-precision weights in the first place (their full-precision operands stay resident; the win is arithmetic/bandwidth, not capacity). | |
| 73 | + | |
| 74 | +### 2.5 BiLD, cascades, and routing (approximate-first with deferral — *not* exact) | |
| 75 | + | |
| 76 | +- **BiLD (NeurIPS 2023).** Small model decodes autoregressively; **fallback policy** hands control to the large model when small-model confidence (max prob) is below a threshold; **rollback policy** lets the large model revert the small model's recent tokens when their distance exceeds a threshold. Up to **2.12×** on a T4 with "minimal" quality degradation — but the output is *not* guaranteed to match the large model (thresholded, not rejection-sampled). BiLD is the published prototype of "escalate to expensive weights only when uncertain." | |
| 77 | +- **FrugalGPT (2023) / RouteLLM (ICLR 2025) / cascades generally.** Sequential escalation across models with a learned scorer + threshold: FrugalGPT matches GPT-4 quality at up to **98% lower cost, sending only ~1 query in 6 to the big model** on some tasks; RouteLLM ~2× cost cut at 95% quality with 14% of queries routed strong. Cascade evidence at *request* granularity for what we want at *token/weight* granularity: most work is easy; a calibrated uncertainty signal can concentrate expensive compute on the hard minority. | |
| 78 | +- **Limitation.** Confidence thresholds are heuristic; guarantees are empirical-statistical at best (see CALM for how to make them rigorous). | |
| 79 | + | |
| 80 | +### 2.6 Speculation *within* the forward pass across time: SPEED, lookahead, SpecInfer, SpecExec | |
| 81 | + | |
| 82 | +- **SPEED (Berkeley, NeurIPS-W 2023).** Uses **early-layer hidden states** to predict the next token *before the current token's forward pass finishes*, launching future tokens' passes speculatively in a pipelined fashion, with **invalidation logic** to roll back mispredictions. This is genuine speculative execution *inside* the transformer computation (though for parameter-shared models). Closest published thing to "start computing before you are sure." | |
| 83 | +- **Lookahead decoding (ICML 2024).** Jacobi fixed-point iteration on the token sequence + n-gram cache; no draft model; 1.5–2.3× (best on code). Shows the sequential dependency itself is partially artificial. | |
| 84 | +- **SpecInfer (ASPLOS 2024).** Tree-based speculation + parallel tree verification; **2.6–3.5× for offloading-based inference** on one GPU. First strong signal that speculation is *more* valuable when weights live off-device. | |
| 85 | +- **SpecExec (NeurIPS 2024).** The extreme of that logic, aimed at consumer hardware with RAM offloading: huge draft trees (up to 2048 nodes) from a 7B draft are verified by an offloaded 70B target in one pass; **Llama-2-70B at 4–6 tok/s (4-bit) or 2–3 tok/s (16-bit) on consumer GPUs, 10.6–18.7× over sequential offloaded decoding**; generation rate ~20 accepted tokens per target-model pass. **For localvm this is the key economics result: when reading the big model costs seconds, speculation amortizes one full-model weight sweep over ~20 tokens — bytes-read-per-token drops by the acceptance length.** | |
| 86 | +- **LLM-42 (2026).** Uses *verified speculation* to get deterministic inference cheaply: draft with fast batch-variant kernels, verify with batch-invariant kernels, roll back on mismatch. Notable as prior art for "same weights, cheaper numerics as draft, exact numerics as verifier." | |
| 87 | + | |
| 88 | +### 2.7 Speculative *weight loading* (MoE expert prefetch; LLM-in-a-flash) | |
| 89 | + | |
| 90 | +Speculation applied to *which parameters to fetch*, not which tokens to emit: | |
| 91 | + | |
| 92 | +- **Deja Vu (ICML 2023).** Contextual sparsity: small MLP predictors, fed by layer k's activations, predict which attention heads / FFN neurons layer k(+1) needs (**asynchronous lookahead predictors**). Up to ~85% sparsity with no quality drop measured, >2× latency vs FasterTransformer on OPT-175B. No verification: a mispredicted neuron is simply dropped (lossy, uncorrected). | |
| 93 | +- **LLM in a flash (Apple, ACL 2024).** Keeps attention weights resident; predicts FFN sparsity (low-rank predictor) + sliding-window neuron reuse, loading only ~2% of FFN weights per token from flash; runs models ~2× DRAM size. Again **predict-and-hope: no exactness correction**; relies on ReLU-style sparsity that modern SwiGLU models lack natively. | |
| 94 | +- **MoE expert speculation (2025–26):** MoE-SpeQ (a small on-device draft model predicts the *sequence of experts* future tokens will need and prefetches them, hiding PCIe I/O), Fate (cross-layer gate signals for prefetch, ~90%+ hit rates reported for hash-based SiDA on Switch), pre-gated MoE (retrained gate decouples selection from execution), Apple's SpecMD study of speculative expert prefetching + caching policies. Here misprediction is *not* lossy — the correct expert is fetched late (a stall, not an error). **This is true speculative weight paging with rollback-to-exactness, but it exists only where the architecture already has discrete routable units (experts).** No dense-model equivalent exists. | |
| 95 | + | |
| 96 | +### 2.8 Early exit with statistical guarantees (CALM, CATs) | |
| 97 | + | |
| 98 | +- **CATs (Schuster et al., 2021).** Early-exit BERT with a *meta consistency classifier* + conformal-style calibration: guarantees the early-exit model's prediction equals the full model's with probability ≥ 1−ε on i.i.d. data. Classification only. | |
| 99 | +- **CALM (NeurIPS 2022).** Extends this to autoregressive generation: per-token exit decisions (confidence = e.g. **softmax top-1/top-2 margin**) calibrated via **distribution-free risk control** so that *sequence-level* quality (ROUGE/BLEURT vs. the full model, or exact textual consistency) is provably maintained with user-chosen probability (e.g., 95%). Reported ~3× decode speedups at negligible quality change on summarization/translation/QA. | |
| 100 | +- **Significance for §4.9's key question.** CALM is the strongest published answer to "can computation terminate when additional precision/depth can no longer change the output?" — answered *statistically* (calibrated risk control), not *deterministically* (no worst-case certificate). Nobody has done the same for precision (bit-width) instead of depth. That specific transfer — *"CALM, but the axis is weight precision / residual count rather than layer count"* — appears unclaimed (see §3). | |
| 101 | + | |
| 102 | +### 2.9 Formal error bounds: Lipschitz, IBP/CROWN, zonotopes (§4.9) | |
| 103 | + | |
| 104 | +- **Lipschitz status of attention.** Kim, Papamakarios & Mnih (ICML 2021): standard dot-product self-attention is **not Lipschitz** on unbounded domains (they propose L2 attention that is). Castin et al., "How Smooth Is Attention?" (ICML 2024, Apple co-affiliation): on *compact* input sets the local Lipschitz constant of self-attention grows like **√n in sequence length** (tight for practical n), with mean-field bounds beyond. Verdict for our purposes: per-layer local Lipschitz constants exist but are input-radius-dependent and, **composed over 30–80 layers, give astronomically loose (vacuous) worst-case logit bounds**. Deterministic global certification of "ΔW cannot flip this token" via Lipschitz products is not practical for 7B+ models. | |
| 105 | +- **Bound propagation (IBP/CROWN/auto_LiRPA).** auto_LiRPA (NeurIPS 2020) supports transformers and even **model-weight perturbations** (weights treated as graph inputs — conceptually exactly our ‖ΔW‖ → Δlogits question). But CROWN's complexity is **O(m²n³)** (m layers, n neurons/layer); published successes are BERT-small-scale classifiers and CIFAR/TinyImageNet CNNs. DeepT (PLDI 2021, multi-norm zonotopes) certifies "larger" transformers than prior work — meaning a few layers of BERT against synonym/ℓp attacks, minutes per instance. **Four to five orders of magnitude of scale separate this literature from a 7B decoder.** Usable insight, though: the *margin certificate primitive* — argmax is stable iff top-2 logit margin > 2·(bound on ‖Δlogits‖∞) — is trivial and cheap once you have any bound on Δlogits; the hard part is the bound. | |
| 106 | +- **Practical middle ground.** Layer-wise *empirical* perturbation theory: the PTQ literature effectively works with ‖WX − ŴX‖_F as its per-layer error functional (GPTQ/OBC objective), and **QEP (NeurIPS 2025)** shows these per-layer errors **accumulate near-exponentially with depth** under independent layer-wise quantization (with a theoretical justification under mild conditions), which is why naive low-bit PTQ collapses. "Why Do Some Inputs Break Low-Bit Quantization?" (EMNLP 2025) finds *input-dependent* structure: full-precision **residual-stream magnitudes predict which examples will have large quantization error (ρ = 0.82)**, with RMSNorm inverting magnitude relations and late-layer MLP gates amplifying. Both papers support a per-input, per-layer *empirical* error predictor (cheap features → predicted logit error) as the realistic substitute for formal bounds — i.e., the certificate becomes statistical, as in CALM. | |
| 107 | +- **Numerical-analysis imports.** Carson & Higham's three-precision iterative refinement / **GMRES-IR**: factorize in低 precision (fp16), refine in higher precision, with *proven* convergence to working-precision accuracy for κ(A) up to 10⁸–10¹²; adopted on tensor cores (Haidar et al.). **Higham & Mary's probabilistic rounding-error analysis** and Connolly–Higham–Mary's stochastic-rounding analysis: replacing worst-case nu error constants by **√n·u with high probability** (rounding errors as mean-independent zero-mean RVs). Two transferable ideas: (a) *refinement loops with guarantees* — compute logits with cheap weights, estimate a residual, refine only if needed; (b) *probabilistic rather than worst-case error budgets* — realistic Δlogit estimates scale like √(accumulated variance), not the vacuous worst case. Nobody has written the "GMRES-IR of transformer inference" paper. | |
| 108 | + | |
| 109 | +### 2.10 Dynamic per-token precision (no verification) | |
| 110 | + | |
| 111 | +Recent, mostly 2025–26, and directly on our path: **QuickSilver** (per-token entropy → 8/4/2-bit Matryoshka bit-width mid-network), **FlexQuant** (perplexity-entropy-guided runtime bit-width switching, 1.3× with fine-grained precision management), **DP-LLM** (dynamic layer-wise precision on multi-scale quantized overlays, NeurIPS 2025), **MoBiQuant** (token-sensitivity mixture-of-bits, any-precision weights). All are *feed-forward heuristic* precision assignment — uncertainty gates precision, but **nothing verifies the low-precision tokens afterward**, so all are lossy with empirical-only quality claims. They confirm the mechanism (any-precision/Matryoshka weight layouts where low bits are a prefix of high bits) is implementable; none closes the loop with exactness. | |
| 112 | + | |
| 113 | +--- | |
| 114 | + | |
| 115 | +## 3. Within-forward-pass speculation: prior art or gap? | |
| 116 | + | |
| 117 | +The charter's key §4.7 question: *can speculation happen inside a transformer forward pass — speculate with cheap weights, verify with full weights only when needed — rather than only across future tokens?* | |
| 118 | + | |
| 119 | +**Verdict: the ingredients all exist separately; the specific mechanism does not appear to exist. This is a real gap, but a narrow one, and it must be positioned very carefully against five near-misses.** | |
| 120 | + | |
| 121 | +Documented near-misses, from farthest to closest: | |
| 122 | + | |
| 123 | +1. **Depth-truncated self-speculation** (Draft&Verify, LayerSkip, Kangaroo, SWIFT, CLaSp): draft = same weights, fewer layers; verify = all layers. Speculation is *about future tokens*; the full weights are read on every verification pass regardless. Verification granularity: whole model. | |
| 124 | +2. **Numerics-truncated self-speculation** (QSpec: W4A4 draft vs W4A16 verify; LLM-42: fast kernels vs batch-invariant kernels): draft = same weights, cheaper arithmetic; verify = better arithmetic. **This is literally "speculate cheap, verify with full computation" — but every token is verified, both operand sets stay resident, and the goal is arithmetic speed / determinism, not resident-set or bytes-per-token reduction.** | |
| 125 | +3. **Precision-truncated drafts** (ML-SpecQD, QuantSpec): draft = 4-bit cast of the target (weights and/or KV); verify = 16-bit target. Proves 4-bit-vs-16-bit token agreement is high enough (>90% acceptance) to power speculation — but the 16-bit model is fully resident and fully read. | |
| 126 | +4. **Speculative weight prefetch without exactness** (Deja Vu, LLM in a flash): predict *which* weights the pass will need, load only those. Inside-the-forward-pass speculation about *memory*, but wrong predictions silently change the output (no verify/rollback path). | |
| 127 | +5. **Speculative weight prefetch with implicit exactness** (MoE expert prefetching: MoE-SpeQ, Fate, SpecMD, pre-gated MoE): predicted experts are prefetched; a misprediction causes a stall while the right expert loads — output exactness preserved by *waiting*, not by approximating. Exists only for architectures with discrete routable units; there is no dense-model analog where the "unit" is a precision level or residual correction. | |
| 128 | +6. **Hidden-state speculation across the pipeline** (SPEED): early-layer states used to *start* dependent computation early, with invalidation. Speculates on intermediate values inside the pass, but for pipelining parameter-shared layers, not for avoiding weight loads. | |
| 129 | + | |
| 130 | +**What does not exist (after searching "weight-level speculation," "approximate forward pass verification," "speculative dequantization," "progressive precision verification," "margin-gated precision escalation," cascade/deferral and self-speculation literatures):** a runtime where the *default* forward pass uses a cheap resident representation (low-bit base), a per-token **decision-uncertainty signal** (top-2 logit margin, entropy, or a learned error predictor à la EMNLP-2025 residual-magnitude features) decides whether the cheap decision is trustworthy, and only on low-margin tokens are **full-precision residuals streamed from storage** to re-verify — with either (a) speculative-sampling-style correction giving exactness in distribution, or (b) CALM-style distribution-free risk calibration giving a statistical consistency guarantee, and with **bytes-read-per-token as the optimization target**. The closest conceptual statement in print is BiLD's fallback/rollback (2023) — but with two separate models, both resident, no storage tier, and no guarantee. | |
| 131 | + | |
| 132 | +Honesty requirements for a novelty claim later (Phase 11): the claim cannot be "speculate cheap / verify expensive" (QSpec owns that), nor "load weights on demand by prediction" (Deja Vu / LLM-in-a-flash own that), nor "escalate on low confidence" (BiLD/cascades/CALM own that). The defensible claim is the **composition**: *uncertainty-gated, storage-tiered precision escalation with an exactness or calibrated-risk correction, evaluated in bytes/token* — plus, if we can make refinement *incremental* (reuse the low-bit matmul result and add only a residual term, GMRES-IR-style, instead of recomputing), a genuinely new kernel-level primitive. | |
| 133 | + | |
| 134 | +--- | |
| 135 | + | |
| 136 | +## 4. How stable are token decisions really? (measured numbers) | |
| 137 | + | |
| 138 | +Everything found that quantifies "same useful output despite different computation": | |
| 139 | + | |
| 140 | +**Same-greedy-token / top-1 agreement under quantization** | |
| 141 | +- llama.cpp's KL-divergence tooling reports "Same top p" (fraction of positions where quantized and fp16 models pick the same top token). Measured examples from a 4-bit-class (NVFP4/Q4) quantization run: **Same top p = 90.87 ± 0.08% and 91.23 ± 0.07%** (two variants of the same model), with Mean KLD ≈ 0.054, median KLD 0.018, 99th-percentile KLD 0.53, max KLD ~22–26. Community wisdom (blind test, llama.cpp discussion #5962): Q6_K/Q5_K statistically indistinguishable from fp16 in human preference; IQ2/IQ1 clearly distinguishable. | |
| 142 | +- MLX-ecosystem measurements (smcleod, Qwen3.6-27B dense, top-K sparse KLD): mean KLD **0.014 (8-bit) → 0.029 (6-bit) → 0.059–0.113 (4-bit variants)**; on a 35B-A3B MoE, 4-bit KLD ranges 0.027 (DWQ) to 0.074 (RTN) — and MoE router protection changes rankings, i.e., *which* weights get precision matters more than average bpw. | |
| 143 | +- **Acceptance rates in precision-level self-speculation are direct agreement measurements**: QuantSpec (4-bit weights + 4-bit hierarchical KV draft): **>90% acceptance**; QSpec validates W4A4 vs W4A16 generation as "highly similar"; production EAGLE-3 acceptance α ≈ 0.6–0.8 per token (BentoML). | |
| 144 | +- **Depth truncation**: LayerSkip acceptance **68.9%** (Llama-2-7B drafting from layer 8 of 32) and **74.5%** (13B, layer 15 of 40); CLaSp finds Llama-3-70B tolerates skipping 44/80 layers in drafting at 1.64× peak speedup. Tuned Lens (Belrose et al.) formalizes "**prediction depth**" — the layer after which the top-1 prediction stops changing — and finds many tokens stabilize well before the final layer (easy tokens shallow, hard tokens deep), which is the mechanistic basis for all early-exit acceptance numbers. | |
| 145 | + | |
| 146 | +**Flips: same accuracy ≠ same answers ("Accuracy is Not All You Need", Microsoft 2024)** | |
| 147 | +- Quantized models within **1% aggregate accuracy** of baseline nonetheless flip **up to ~15%** of individual answers (correct↔incorrect symmetrically); layer-dropping/pruning at matched accuracy reaches **25%+ flips**. Flip rate correlates with KL divergence at **Spearman 0.96–0.97**, and with MT-Bench degradation. Consequence for us: *aggregate benchmarks cannot certify a compressed runtime; per-token distance metrics (KL, flips, agreement) are the right quality currency* — matching the charter's §8.2 quality metrics list. | |
| 148 | + | |
| 149 | +**Margins and their fragility** | |
| 150 | +- The whole early-exit line (CALM, CATs, BiLD fallback, Kangaroo's drafting stop) uses the **softmax top-1/top-2 margin as the confidence signal**, and it works — meaning margins are informative — but published *distributions* of logit gaps are surprisingly scarce. Unit 42's "logit-gap steering" measures refusal-vs-affirmation logit gaps in safety contexts and shows small logit shifts close them (margins there are small and exploitable). Thinking Machines' batch-invariance study is the sharpest evidence that **a nontrivial share of tokens sit on a knife's edge**: batch-size-dependent rounding differences alone (perturbations of order 10⁻⁵ relative) caused 1,000 identical greedy (temp-0) prompts to produce dozens of distinct completions on a production stack — any single flipped token then diverges the whole continuation. LLM-42 builds a serving system on precisely this fact. | |
| 151 | +- Synthesis for Experiment G: expect a bimodal picture — most tokens have comfortable margins (hence 90%+ same-top-token at 4-bit, 70%+ at half-depth), but a persistent 5–15% of tokens are genuinely unstable, and *those* tokens are disproportionately the semantically load-bearing ones (flips paper; hard-token analyses in QuickSilver's entropy gating). **Measuring the joint distribution (margin of cheap model, agreement with full model) on our own hardware/models is charter Experiment G and is cheap to run.** No paper we found reports this joint distribution directly — a small but real measurement gap we can fill and publish. | |
| 152 | + | |
| 153 | +**Distribution-level repair** | |
| 154 | +- Speculative sampling's rejection rule is the *only* known mechanism that converts "approximately right most of the time" into "exactly the target distribution always" at bounded extra cost (one target pass per block). Tree variants (SpecInfer) generalize it to multi-candidate verification. Any localvm design wanting exactness should reuse this machinery unchanged. | |
| 155 | + | |
| 156 | +--- | |
| 157 | + | |
| 158 | +## 5. Relevance to localvm-research: uncertainty-gated weight materialization | |
| 159 | + | |
| 160 | +**The proposed regime.** Resident in unified memory: a low-bit base model (2–4 bit, MLX-native layout) + KV cache + a small error/uncertainty apparatus. On SSD: precision residuals (Matryoshka/any-precision layout so higher precision = base bits + extra bit-planes, or additive-quantization residual codebooks). Per token: run the base pass; compute the decision margin (plus, optionally, a learned per-layer error estimate from residual-stream features); if the decision is certifiably/confidently stable, emit; otherwise *defer* — keep drafting with the base and periodically verify the accumulated low-confidence block with residual-augmented weights streamed once per block, LayerSkip-economics style. | |
| 161 | + | |
| 162 | +**What exists to build on (per section above):** | |
| 163 | +- Agreement rates (90%+ at 4-bit; >90% QuantSpec acceptance) say the escalation rate can plausibly be 5–15% of tokens. | |
| 164 | +- SpecExec proves the amortization math on consumer hardware: ~20 accepted tokens per full-weight sweep turns a 4.5 s/token offloaded model into 4–6 tok/s. Our version amortizes *residual streaming* instead of full-model streaming — strictly less data. | |
| 165 | +- Rejection sampling (exactness) or CALM risk calibration (guaranteed consistency at chosen ε) supply the correctness story — nothing new needs to be proven mathematically for either mode. | |
| 166 | +- MLX/Metal feasibility is de-risked by ReDrafter-on-MLX (Apple shipped speculative verification kernels on Metal) and by llama.cpp/mlx_lm speculative modes; batch-of-k verification is bandwidth-neutral on unified memory exactly as on CUDA. | |
| 167 | +- Any-precision/Matryoshka weight layouts (QuickSilver, MoBiQuant, DP-LLM overlays) show prefix-decodable multi-precision storage is practical. | |
| 168 | + | |
| 169 | +**What's missing (the research):** | |
| 170 | +1. **The joint stability measurement** (margin of base vs. agreement with full model, per task domain) — nobody has published it; it decides whether the escalation rate is 5% or 40%. → Experiment G, run first. | |
| 171 | +2. **A cheap, calibrated per-token error signal that accounts for depth-compounding** — QEP says errors compound across layers; EMNLP-2025 says compounding is predictable from residual magnitudes (ρ=0.82). A logit-margin threshold alone may be miscalibrated for the tokens where the base is *confidently wrong* (the dangerous quadrant). This is where formal §4.9 tools are vacuous and a small learned predictor + distribution-free calibration (CALM's recipe) is the realistic instrument. | |
| 172 | +3. **Incremental refinement kernels**: recomputing the whole pass with residuals doubles compute; the GMRES-IR analogy suggests computing `ΔY = (ΔW)X` only (residual weights × cached activations) and adding it — requires caching per-layer activations for the deferred block (memory cost ~ activations × block length, cheap next to weights) and a Metal kernel for sparse/low-rank residual matmul. Nobody has published this primitive; it is also exactly charter Experiment D/E territory. | |
| 173 | +4. **Block-deferred verification policy**: verifying uncertain tokens one-by-one would thrash the SSD; batching them inherits speculative decoding's rollback problem (a flipped early token invalidates later drafted tokens). The right policy (when to flush the uncertain block) is an open scheduling problem — SpecExec's budget analysis and Kangaroo's confidence-stop are the starting points. | |
| 174 | +5. **Failure mode to respect**: if the base is 2-bit rather than 4-bit, agreement may collapse (llama.cpp IQ1/IQ2 blind-test results; max-KLD outliers of 20+ nats even at 4-bit) and escalation could approach 100% on hard domains — the charter's §17 "required working set is nearly the entire model" failure. The measurement in (1) resolves this cheaply before any engineering. | |
| 175 | + | |
| 176 | +--- | |
| 177 | + | |
| 178 | +## Sources | |
| 179 | + | |
| 180 | +- Fast Inference from Transformers via Speculative Decoding (Leviathan, Kalman, Matias; ICML 2023) — https://arxiv.org/abs/2211.17192 (accessed 2026-08-11) | |
| 181 | +- Accelerating Large Language Model Decoding with Speculative Sampling (Chen et al., DeepMind) — https://arxiv.org/abs/2302.01318 (accessed 2026-08-11) | |
| 182 | +- Looking back at speculative decoding (Google Research blog) — https://research.google/blog/looking-back-at-speculative-decoding (accessed 2026-08-11) | |
| 183 | +- Speculative decoding — Wikipedia — https://en.wikipedia.org/wiki/Speculative_decoding (accessed 2026-08-11) | |
| 184 | +- Speculative Decoding: Exploiting Speculative Execution for Accelerating Seq2seq Generation (Xia et al., EMNLP 2023 Findings) — https://aclanthology.org/2023.findings-emnlp.257.pdf (accessed 2026-08-11) | |
| 185 | +- Beyond the Speculative Game: A Survey of Speculative Execution in Large Language Models — https://arxiv.org/html/2404.14897v1 (accessed 2026-08-11) | |
| 186 | +- Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads (Cai et al.) — https://arxiv.org/abs/2401.10774 (accessed 2026-08-11) | |
| 187 | +- EAGLE-3: Scaling up Inference Acceleration of Large Language Models via Training-Time Test — https://arxiv.org/html/2503.01840v1 (accessed 2026-08-11) | |
| 188 | +- EAGLE-3 (NeurIPS 2025 poster) — https://neurips.cc/virtual/2025/poster/119930 (accessed 2026-08-11) | |
| 189 | +- Get 3× Faster LLM Inference with Speculative Decoding (BentoML; real-world EAGLE-3 acceptance rates) — https://www.bentoml.com/blog/3x-faster-llm-inference-with-speculative-decoding (accessed 2026-08-11) | |
| 190 | +- Break the Sequential Dependency of LLM Inference Using Lookahead Decoding (Fu et al., ICML 2024) — https://arxiv.org/html/2402.02057v1 (accessed 2026-08-11) | |
| 191 | +- Lookahead decoding blog (LMSYS) — https://www.lmsys.org/blog/2023-11-21-lookahead-decoding (accessed 2026-08-11) | |
| 192 | +- Draft & Verify: Lossless Large Language Model Acceleration via Self-Speculative Decoding (Zhang et al.) — https://arxiv.org/abs/2309.08168 (accessed 2026-08-11) | |
| 193 | +- LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding (Elhoushi et al., ACL 2024) — https://arxiv.org/html/2404.16710v1 (accessed 2026-08-11) | |
| 194 | +- Faster Text Generation with Self-Speculative Decoding (Hugging Face LayerSkip blog) — https://huggingface.co/blog/layerskip (accessed 2026-08-11) | |
| 195 | +- Kangaroo: Lossless Self-Speculative Decoding via Double Early Exiting (NeurIPS 2024) — https://neurips.cc/virtual/2024/poster/93829 (accessed 2026-08-11) | |
| 196 | +- SWIFT: On-the-Fly Self-Speculative Decoding for LLM Inference Acceleration (ICLR 2025) — https://arxiv.org/pdf/2410.06916 (accessed 2026-08-11) | |
| 197 | +- CLaSp: In-Context Layer Skip for Self-Speculative Decoding — https://arxiv.org/html/2505.24196v1 (accessed 2026-08-11) | |
| 198 | +- QSpec: Speculative Decoding with Complementary Quantization Schemes (EMNLP 2025) — https://aclanthology.org/2025.emnlp-main.240.pdf and https://arxiv.org/abs/2410.11305 (accessed 2026-08-11) | |
| 199 | +- QuantSpec: Self-Speculative Decoding with Hierarchical Quantized KV Cache (Apple ML Research, ICML 2025) — https://machinelearning.apple.com/research/quantspec (accessed 2026-08-11) | |
| 200 | +- ML-SpecQD: Multi-Level Speculative Decoding with Quantized Drafts — https://arxiv.org/html/2503.13565v1 (accessed 2026-08-11) | |
| 201 | +- Speculative Decoding with Big Little Decoder (Kim et al., NeurIPS 2023) — https://arxiv.org/abs/2302.07863 (accessed 2026-08-11) | |
| 202 | +- BigLittleDecoder repository — https://github.com/kssteven418/biglittledecoder (accessed 2026-08-11) | |
| 203 | +- SpecInfer: Accelerating LLM Serving with Tree-based Speculative Inference and Verification (ASPLOS 2024) — https://arxiv.org/abs/2305.09781 (accessed 2026-08-11) | |
| 204 | +- SpecExec: Massively Parallel Speculative Decoding for Interactive LLM Inference on Consumer Devices (NeurIPS 2024) — https://arxiv.org/html/2406.02532v1 (accessed 2026-08-11) | |
| 205 | +- SpecExec results (Together AI blog) — https://www.together.ai/blog/specexec (accessed 2026-08-11) | |
| 206 | +- Recurrent Drafter for Fast Speculative Decoding in Large Language Models (Apple; MLX/Metal benchmarks) — https://arxiv.org/html/2403.09919v5 and https://machinelearning.apple.com/research/recurrent-drafter (accessed 2026-08-11) | |
| 207 | +- Speculative Streaming: Fast LLM Inference Without Auxiliary Models (Apple ML Research) — https://machinelearning.apple.com/research/llm-inference (accessed 2026-08-11) | |
| 208 | +- SPEED: Speculative Pipelined Execution for Efficient Decoding (Hooper et al., NeurIPS-W 2023) — https://arxiv.org/abs/2310.12072 (accessed 2026-08-11) | |
| 209 | +- LLM-42: Enabling Determinism in LLM Inference with Verified Speculation — https://arxiv.org/html/2601.17768v1 (accessed 2026-08-11) | |
| 210 | +- FrugalGPT / cascade & routing results summary — https://neuraltrust.ai/blog/llm-model-routing (accessed 2026-08-11) | |
| 211 | +- Regret Bounds for Model Cascades (survey of FrugalGPT/RouteLLM/Hybrid-LLM numbers) — https://www.tmls.nyc/research/cascade-regret-optimal-stopping (accessed 2026-08-11) | |
| 212 | +- Confident Adaptive Language Modeling (Schuster et al., NeurIPS 2022) — https://arxiv.org/abs/2207.07061 (PDF: https://www.proceedings.com/content/068/068431-1269open.pdf) (accessed 2026-08-11) | |
| 213 | +- Accelerating text generation with CALM (Google Research blog) — https://research.google/blog/accelerating-text-generation-with-confident-adaptive-language-modeling-calm (accessed 2026-08-11) | |
| 214 | +- Consistent Accelerated Inference via Confident Adaptive Transformers (Schuster et al., 2021) — https://neurips2021-nlp.github.io/papers/7/CameraReady/Confident_Early_Exit__Transformer___workshop.pdf (accessed 2026-08-11) | |
| 215 | +- Deja Vu: Contextual Sparsity for Efficient LLMs at Inference Time (Liu et al., ICML 2023) — https://proceedings.mlr.press/v202/liu23am/liu23am.pdf (accessed 2026-08-11) | |
| 216 | +- LLM in a flash: Efficient Large Language Model Inference with Limited Memory (Apple, ACL 2024) — https://arxiv.org/html/2312.11514v2 (accessed 2026-08-11) | |
| 217 | +- MoE-SpeQ: Speculative Quantized Decoding with Proactive Expert Prefetching and Offloading — https://ui.adsabs.harvard.edu/abs/2025arXiv251114102W/abstract (arXiv:2511.14102) (accessed 2026-08-11) | |
| 218 | +- Fate: Fast Edge Inference of Mixture-of-Experts Models via Cross-Layer Gate — https://arxiv.org/html/2502.12224v2 (accessed 2026-08-11) | |
| 219 | +- Speculating Experts Accelerates Inference for Mixture-of-Experts — https://arxiv.org/html/2603.19289v1 (accessed 2026-08-11) | |
| 220 | +- SpecMD: A Comprehensive Study on Speculative Expert Prefetching (Apple ML Research) — https://machinelearning.apple.com/research/specmd-expert-prefetching (accessed 2026-08-11) | |
| 221 | +- The Lipschitz Constant of Self-Attention (Kim, Papamakarios, Mnih; ICML 2021) — https://proceedings.mlr.press/v139/kim21i/kim21i.pdf (accessed 2026-08-11) | |
| 222 | +- How Smooth Is Attention? (Castin et al.; Apple ML Research) — https://arxiv.org/html/2312.14820v2 and https://machinelearning.apple.com/research/how-smooth-is-attention (accessed 2026-08-11) | |
| 223 | +- DeepT: Fast and Precise Certification of Transformers (PLDI 2021) — https://files.sri.inf.ethz.ch/website/papers/pldi21-transformers.pdf (accessed 2026-08-11) | |
| 224 | +- auto_LiRPA: Automatic Linear Relaxation based Perturbation Analysis (NeurIPS 2020; library) — https://github.com/Verified-Intelligence/auto_LiRPA (accessed 2026-08-11) | |
| 225 | +- Towards Tighter LiRPA-based Robustness Certification (COLING 2025; CROWN O(m²n³) complexity discussion) — https://aclanthology.org/2025.coling-main.415.pdf (accessed 2026-08-11) | |
| 226 | +- Mixed-precision iterative refinement using tensor cores (Haidar, Dongarra et al.; surveys Carson–Higham GMRES-IR guarantees) — https://www.netlib.org/utk/people/JackDongarra/PAPERS/mixed-rs-2020.pdf (accessed 2026-08-11) | |
| 227 | +- Three-Precision GMRES-Based Iterative Refinement for Least Squares Problems (Carson, Higham, Pranesh) — https://eprints.maths.manchester.ac.uk/2770/1/paper.pdf (accessed 2026-08-11) | |
| 228 | +- A New Approach to Probabilistic Rounding Error Analysis (Higham & Mary, SIAM SISC 2019) — https://epubs.siam.org/doi/10.1137/18M1226312 (accessed 2026-08-11) | |
| 229 | +- Stochastic Rounding and Its Probabilistic Backward Error Analysis (Connolly, Higham, Mary, SIAM SISC 2021) — https://epubs.siam.org/doi/10.1137/20M1334796 (accessed 2026-08-11) | |
| 230 | +- Quantization Error Propagation: Revisiting Layer-Wise Post-Training Quantization (NeurIPS 2025) — https://arxiv.org/html/2504.09629v3 (accessed 2026-08-11) | |
| 231 | +- Why Do Some Inputs Break Low-Bit LLM Quantization? (EMNLP 2025) — https://aclanthology.org/2025.emnlp-main.168.pdf (accessed 2026-08-11) | |
| 232 | +- Which Quantization Should I Use? A Unified Evaluation of llama.cpp Quantizations — https://arxiv.org/html/2601.14277v1 (accessed 2026-08-11) | |
| 233 | +- Accuracy is Not All You Need (Microsoft; flips + KL under compression) — https://arxiv.org/html/2407.09141v1 (accessed 2026-08-11) | |
| 234 | +- Why accuracy is a misleading metric when evaluating compressed LLMs (flips summary) — https://bdtechtalks.com/2024/08/06/why-accuracy-is-a-misleading-metric-when-evaluating-compressed-llms (accessed 2026-08-11) | |
| 235 | +- llama.cpp quantizer discussion #23853 (KLD percentiles, "Same top p" ≈ 90.9–91.2%) — https://github.com/ggml-org/llama.cpp/discussions/23853 (accessed 2026-08-11) | |
| 236 | +- Blind testing different quants (llama.cpp discussion #5962) — https://github.com/ggml-org/llama.cpp/discussions/5962 (accessed 2026-08-11) | |
| 237 | +- Measuring Model Quantisation Quality with KL Divergence (MLX quant KLD measurements) — https://smcleod.net/2026/04/measuring-model-quantisation-quality-with-kl-divergence (accessed 2026-08-11) | |
| 238 | +- Eliciting Latent Predictions from Transformers with the Tuned Lens (Belrose et al.; "prediction depth") — https://arxiv.org/html/2303.08112v6 (accessed 2026-08-11) | |
| 239 | +- Defeating Nondeterminism in LLM Inference (Thinking Machines) — https://thinkingmachines.ai/blog/defeating-nondeterminism-in-llm-inference (accessed 2026-08-11) | |
| 240 | +- Logit-Gap Steering (Palo Alto Networks Unit 42; measured refusal logit gaps) — https://unit42.paloaltonetworks.com/logit-gap-steering-impact (accessed 2026-08-11) | |
| 241 | +- QuickSilver / Adaptive Matryoshka Quantization (per-token entropy-gated bit-width) — https://arxiv.org/pdf/2506.22396 (accessed 2026-08-11) | |
| 242 | +- FlexQuant: A Flexible and Efficient Dynamic Precision Switching Framework for LLM Quantization — https://arxiv.org/html/2506.12024v3 (accessed 2026-08-11) | |
| 243 | +- DP-LLM: Runtime Model Adaptation with Dynamic Layer-wise Precision Assignment (NeurIPS 2025) — https://neurips.cc/virtual/2025/poster/115920 (accessed 2026-08-11) | |
| 244 | +- MoBiQuant: Mixture-of-Bits Quantization for Token-Adaptive LLM Inference — https://ui.adsabs.harvard.edu/abs/2026arXiv260220191W/abstract (accessed 2026-08-11) | |
| 245 | +- Speculative Decoding Papers (curated list, hemingkx) — https://github.com/hemingkx/SpeculativeDecodingPapers (accessed 2026-08-11) | |
added
results/expH_ssd_feasibility/20260812T034359Z/iostat.log
+336 −0
@@ -0,0 +1,336 @@ | ||
| 1 | + disk0 disk4 | |
| 2 | + KB/t tps MB/s KB/t tps MB/s | |
| 3 | + 51.56 2593 130.57 5.75 1 0.00 | |
| 4 | + 16.12 16037 252.46 0.00 0 0.00 | |
| 5 | + 15.94 16289 253.58 0.00 0 0.00 | |
| 6 | + 16.00 16647 260.11 0.00 0 0.00 | |
| 7 | + 15.97 16595 258.75 0.00 0 0.00 | |
| 8 | + 16.00 16559 258.74 0.00 0 0.00 | |
| 9 | + 15.92 16556 257.43 0.00 0 0.00 | |
| 10 | + 16.00 16292 254.57 0.00 0 0.00 | |
| 11 | + 16.00 16136 252.13 0.00 0 0.00 | |
| 12 | + 16.00 15981 249.71 0.00 0 0.00 | |
| 13 | + 16.00 16190 252.96 0.00 0 0.00 | |
| 14 | + 16.00 15953 249.26 0.00 0 0.00 | |
| 15 | + 16.00 16627 259.78 0.00 0 0.00 | |
| 16 | + 16.00 38944 608.50 0.00 0 0.00 | |
| 17 | + 15.98 62053 968.65 0.00 0 0.00 | |
| 18 | + 15.99 62016 968.38 0.00 0 0.00 | |
| 19 | + 16.00 60441 944.39 0.00 0 0.00 | |
| 20 | + 16.00 62185 971.64 0.00 0 0.00 | |
| 21 | + 16.00 62076 969.94 0.00 0 0.00 | |
| 22 | + 16.00 60783 949.75 0.00 0 0.00 | |
| 23 | + disk0 disk4 | |
| 24 | + KB/t tps MB/s KB/t tps MB/s | |
| 25 | + 16.00 62239 972.48 0.00 0 0.00 | |
| 26 | + 16.00 62163 971.30 0.00 0 0.00 | |
| 27 | + 16.00 68518 1070.59 0.00 0 0.00 | |
| 28 | + 16.00 116964 1827.57 0.00 0 0.00 | |
| 29 | + 16.00 116792 1824.87 0.00 0 0.00 | |
| 30 | + 15.99 111233 1737.12 0.00 0 0.00 | |
| 31 | + 16.00 116397 1818.70 0.00 0 0.00 | |
| 32 | + 16.00 116349 1817.93 0.00 0 0.00 | |
| 33 | + 16.00 112007 1750.12 0.00 0 0.00 | |
| 34 | + 16.00 118523 1851.93 0.00 0 0.00 | |
| 35 | + 16.00 118763 1855.67 0.00 0 0.00 | |
| 36 | + 16.00 95981 1499.70 0.00 0 0.00 | |
| 37 | + 16.00 16315 254.92 0.00 0 0.00 | |
| 38 | + 16.00 16610 259.56 0.00 0 0.00 | |
| 39 | + 16.00 16558 258.73 0.00 0 0.00 | |
| 40 | + 16.00 16669 260.46 0.00 0 0.00 | |
| 41 | + 15.91 16718 259.77 0.00 0 0.00 | |
| 42 | + 16.00 16505 257.90 0.00 0 0.00 | |
| 43 | + 16.00 16518 258.09 0.00 0 0.00 | |
| 44 | + 16.00 16671 260.50 0.00 0 0.00 | |
| 45 | + disk0 disk4 | |
| 46 | + KB/t tps MB/s KB/t tps MB/s | |
| 47 | + 16.00 32941 514.70 0.00 0 0.00 | |
| 48 | + 16.00 61267 957.29 0.00 0 0.00 | |
| 49 | + 16.00 63503 992.20 0.00 0 0.00 | |
| 50 | + 16.00 62117 970.61 0.00 0 0.00 | |
| 51 | + 16.00 61915 967.42 0.00 0 0.00 | |
| 52 | + 15.98 61487 959.55 0.00 0 0.00 | |
| 53 | + 16.00 60847 950.74 0.00 0 0.00 | |
| 54 | + 16.00 62028 969.18 0.00 0 0.00 | |
| 55 | + 16.00 62071 969.86 0.00 0 0.00 | |
| 56 | + 16.00 68604 1071.95 0.00 0 0.00 | |
| 57 | + 16.00 116187 1815.42 0.00 0 0.00 | |
| 58 | + 16.00 115969 1812.01 0.00 0 0.00 | |
| 59 | + 16.00 110965 1733.83 0.00 0 0.00 | |
| 60 | + 16.00 115952 1811.75 0.00 0 0.00 | |
| 61 | + 16.00 115886 1810.73 0.00 0 0.00 | |
| 62 | + 16.00 112896 1764.00 0.00 0 0.00 | |
| 63 | + 16.00 118474 1851.16 0.00 0 0.00 | |
| 64 | + 16.00 118319 1848.72 0.00 0 0.00 | |
| 65 | + 19.49 70678 1345.01 0.00 0 0.00 | |
| 66 | + 63.98 11797 737.16 0.00 0 0.00 | |
| 67 | + disk0 disk4 | |
| 68 | + KB/t tps MB/s KB/t tps MB/s | |
| 69 | + 64.00 11993 749.59 0.00 0 0.00 | |
| 70 | + 63.99 11720 732.37 0.00 0 0.00 | |
| 71 | + 64.00 11755 734.69 0.00 0 0.00 | |
| 72 | + 64.00 11779 736.17 0.00 0 0.00 | |
| 73 | + 63.99 11580 723.66 0.00 0 0.00 | |
| 74 | + 64.00 11730 733.14 0.00 0 0.00 | |
| 75 | + 62.47 11807 720.33 0.00 0 0.00 | |
| 76 | + 63.64 28738 1786.06 0.00 0 0.00 | |
| 77 | + 64.00 44708 2794.25 0.00 0 0.00 | |
| 78 | + 64.00 44584 2786.34 0.00 0 0.00 | |
| 79 | + 64.00 43729 2733.08 0.00 0 0.00 | |
| 80 | + 64.00 44631 2789.39 0.00 0 0.00 | |
| 81 | + 64.00 44604 2787.64 0.00 0 0.00 | |
| 82 | + 64.00 43783 2736.42 0.00 0 0.00 | |
| 83 | + 64.00 44672 2792.01 0.00 0 0.00 | |
| 84 | + 63.85 46502 2899.71 0.00 0 0.00 | |
| 85 | + 64.00 79651 4978.13 0.00 0 0.00 | |
| 86 | + 64.00 71132 4445.76 0.00 0 0.00 | |
| 87 | + 64.00 80040 5002.52 0.00 0 0.00 | |
| 88 | + 64.00 78573 4910.65 0.00 0 0.00 | |
| 89 | + disk0 disk4 | |
| 90 | + KB/t tps MB/s KB/t tps MB/s | |
| 91 | + 65.08 76205 4842.99 0.00 0 0.00 | |
| 92 | + 256.00 8590 2147.52 0.00 0 0.00 | |
| 93 | + 256.00 8543 2135.76 0.00 0 0.00 | |
| 94 | + 256.00 8671 2167.82 0.00 0 0.00 | |
| 95 | + 256.00 9044 2261.02 0.00 0 0.00 | |
| 96 | + 253.70 9102 2254.99 0.00 0 0.00 | |
| 97 | + 255.97 9083 2270.61 0.00 0 0.00 | |
| 98 | + 256.00 8900 2224.91 0.00 0 0.00 | |
| 99 | + 255.92 8644 2160.19 0.00 0 0.00 | |
| 100 | + 255.89 8494 2122.49 0.00 0 0.00 | |
| 101 | + 256.00 27668 6916.90 0.00 0 0.00 | |
| 102 | + 256.00 30970 7742.42 0.00 0 0.00 | |
| 103 | + 255.99 30826 7706.35 0.00 0 0.00 | |
| 104 | + 255.98 40151 10037.11 0.00 0 0.00 | |
| 105 | + 256.00 43750 10937.48 0.00 0 0.00 | |
| 106 | + 317.12 25290 7832.23 0.00 0 0.00 | |
| 107 | + 1023.77 4361 4359.66 0.00 0 0.00 | |
| 108 | + 1024.00 4365 4364.58 0.00 0 0.00 | |
| 109 | + 1024.00 4340 4339.73 0.00 0 0.00 | |
| 110 | + 1017.08 4320 4290.92 0.00 0 0.00 | |
| 111 | + disk0 disk4 | |
| 112 | + KB/t tps MB/s KB/t tps MB/s | |
| 113 | + 1024.00 4340 4339.83 0.00 0 0.00 | |
| 114 | + 1024.00 11736 11735.95 0.00 0 0.00 | |
| 115 | + 1024.00 12987 12986.59 0.00 0 0.00 | |
| 116 | + 1023.77 13115 13111.65 0.00 0 0.00 | |
| 117 | + 952.44 13471 12529.50 0.00 0 0.00 | |
| 118 | + 642.62 10570 6633.21 0.00 0 0.00 | |
| 119 | + 825.39 5566 4486.08 0.00 0 0.00 | |
| 120 | + 989.60 12392 11975.58 0.00 0 0.00 | |
| 121 | + 16.00 38208 597.00 0.00 0 0.00 | |
| 122 | + 15.99 42383 661.92 0.00 0 0.00 | |
| 123 | + 16.00 37086 579.47 0.00 0 0.00 | |
| 124 | + 16.00 35432 553.63 0.00 0 0.00 | |
| 125 | + 16.00 45184 706.00 0.00 0 0.00 | |
| 126 | + 16.00 37732 589.57 0.00 0 0.00 | |
| 127 | + 16.00 34148 533.56 0.00 0 0.00 | |
| 128 | + 16.00 39729 620.77 0.00 0 0.00 | |
| 129 | + 16.00 40573 633.95 0.00 0 0.00 | |
| 130 | + 16.00 38236 597.54 0.00 0 0.00 | |
| 131 | + 16.00 59766 933.83 0.00 0 0.00 | |
| 132 | + 16.00 60772 949.56 0.00 0 0.00 | |
| 133 | + disk0 disk4 | |
| 134 | + KB/t tps MB/s KB/t tps MB/s | |
| 135 | + 16.00 55729 870.74 0.00 0 0.00 | |
| 136 | + 16.00 57991 906.11 0.00 0 0.00 | |
| 137 | + 16.00 59162 924.40 0.00 0 0.00 | |
| 138 | + 15.97 55768 869.88 0.00 0 0.00 | |
| 139 | + 16.00 56541 883.44 0.00 0 0.00 | |
| 140 | + 15.98 59459 927.95 0.00 0 0.00 | |
| 141 | + 16.00 73704 1151.59 0.00 0 0.00 | |
| 142 | + 16.00 93571 1462.00 0.00 0 0.00 | |
| 143 | + 16.00 93667 1463.56 0.00 0 0.00 | |
| 144 | + 16.00 88656 1385.26 0.00 0 0.00 | |
| 145 | + 16.00 93727 1464.56 0.00 0 0.00 | |
| 146 | + 16.00 99183 1549.73 0.00 0 0.00 | |
| 147 | + 16.00 91423 1428.49 0.00 0 0.00 | |
| 148 | + 16.00 90901 1420.32 0.00 0 0.00 | |
| 149 | + 15.98 84671 1321.66 0.00 0 0.00 | |
| 150 | + 16.00 34224 534.73 0.00 0 0.00 | |
| 151 | + 16.00 28573 446.45 0.00 0 0.00 | |
| 152 | + 16.00 29612 462.68 0.00 0 0.00 | |
| 153 | + 16.00 34311 536.12 0.00 0 0.00 | |
| 154 | + 16.00 29231 456.73 0.00 0 0.00 | |
| 155 | + disk0 disk4 | |
| 156 | + KB/t tps MB/s KB/t tps MB/s | |
| 157 | + 16.00 29114 454.91 0.00 0 0.00 | |
| 158 | + 16.00 32543 508.48 0.00 0 0.00 | |
| 159 | + 16.00 32896 513.99 0.00 0 0.00 | |
| 160 | + 15.97 28469 443.91 0.00 0 0.00 | |
| 161 | + 16.00 31565 493.14 0.00 0 0.00 | |
| 162 | + 16.00 58762 918.13 0.00 0 0.00 | |
| 163 | + 16.00 59475 929.31 0.00 0 0.00 | |
| 164 | + 16.00 57386 896.66 0.00 0 0.00 | |
| 165 | + 16.00 58807 918.86 0.00 0 0.00 | |
| 166 | + 16.00 58982 921.58 0.00 0 0.00 | |
| 167 | + 16.00 57015 890.86 0.00 0 0.00 | |
| 168 | + 16.00 59713 933.01 0.00 0 0.00 | |
| 169 | + 16.00 60612 947.07 0.00 0 0.00 | |
| 170 | + 16.01 83808 1309.96 0.00 0 0.00 | |
| 171 | + 16.00 81735 1277.08 0.00 0 0.00 | |
| 172 | + 15.97 85750 1337.43 0.00 0 0.00 | |
| 173 | + 16.00 81619 1275.30 0.00 0 0.00 | |
| 174 | + 16.00 83357 1302.45 0.00 0 0.00 | |
| 175 | + 21.82 45496 969.53 0.00 0 0.00 | |
| 176 | + 63.85 10563 658.60 0.00 0 0.00 | |
| 177 | + disk0 disk4 | |
| 178 | + KB/t tps MB/s KB/t tps MB/s | |
| 179 | + 62.62 10419 637.15 0.00 0 0.00 | |
| 180 | + 64.00 10160 635.01 0.00 0 0.00 | |
| 181 | + 63.36 10722 663.43 0.00 0 0.00 | |
| 182 | + 63.94 11010 687.53 0.00 0 0.00 | |
| 183 | + 63.99 10610 663.06 0.00 0 0.00 | |
| 184 | + 63.98 10736 670.77 0.00 0 0.00 | |
| 185 | + 64.00 10795 674.69 0.00 0 0.00 | |
| 186 | + 64.00 10196 637.24 0.00 0 0.00 | |
| 187 | + 63.99 24947 1558.99 0.00 0 0.00 | |
| 188 | + 64.00 40171 2510.53 0.00 0 0.00 | |
| 189 | + 64.00 40434 2527.13 0.00 0 0.00 | |
| 190 | + 64.00 55060 3441.27 0.00 0 0.00 | |
| 191 | + 64.00 65407 4087.65 0.00 0 0.00 | |
| 192 | + 128.45 18118 2272.77 0.00 0 0.00 | |
| 193 | + 255.87 7460 1864.01 0.00 0 0.00 | |
| 194 | + 255.97 7617 1904.10 0.00 0 0.00 | |
| 195 | + 255.97 7587 1896.47 0.00 0 0.00 | |
| 196 | + 256.00 26463 6615.87 0.00 0 0.00 | |
| 197 | + 255.97 30589 7646.21 0.00 0 0.00 | |
| 198 | + 787.77 4745 3650.17 0.00 0 0.00 | |
| 199 | + disk0 disk4 | |
| 200 | + KB/t tps MB/s KB/t tps MB/s | |
| 201 | + 1022.82 3440 3436.34 0.00 0 0.00 | |
| 202 | + 1023.80 4873 4872.41 0.00 0 0.00 | |
| 203 | + 1023.56 6893 6889.56 0.00 0 0.00 | |
| 204 | + 960.23 6259 5869.09 0.00 0 0.00 | |
| 205 | + 1024.00 3372 3371.96 0.00 0 0.00 | |
| 206 | + 988.85 10505 10144.75 0.00 0 0.00 | |
| 207 | + 709.43 6718 4654.06 0.00 0 0.00 | |
| 208 | + 16.00 1 0.02 0.00 0 0.00 | |
| 209 | + 18.00 4 0.07 0.00 0 0.00 | |
| 210 | + 0.00 0 0.00 0.00 0 0.00 | |
| 211 | + 0.00 0 0.00 0.00 0 0.00 | |
| 212 | + 0.00 0 0.00 0.00 0 0.00 | |
| 213 | + 0.00 0 0.00 0.00 0 0.00 | |
| 214 | + 24.00 1 0.02 0.00 0 0.00 | |
| 215 | + 5.19 100 0.51 0.00 0 0.00 | |
| 216 | + 8.00 3 0.02 0.00 0 0.00 | |
| 217 | + 20.00 2 0.04 0.00 0 0.00 | |
| 218 | + 0.00 0 0.00 0.00 0 0.00 | |
| 219 | + 18.00 4 0.07 0.00 0 0.00 | |
| 220 | + 0.00 0 0.00 0.00 0 0.00 | |
| 221 | + disk0 disk4 | |
| 222 | + KB/t tps MB/s KB/t tps MB/s | |
| 223 | + 0.00 0 0.00 0.00 0 0.00 | |
| 224 | + 0.00 0 0.00 0.00 0 0.00 | |
| 225 | + 0.00 0 0.00 0.00 0 0.00 | |
| 226 | + 0.00 0 0.00 0.00 0 0.00 | |
| 227 | + 7.06 88 0.61 0.00 0 0.00 | |
| 228 | + 8.00 3 0.02 0.00 0 0.00 | |
| 229 | + 0.00 0 0.00 0.00 0 0.00 | |
| 230 | + 0.00 0 0.00 0.00 0 0.00 | |
| 231 | + 18.00 4 0.07 0.00 0 0.00 | |
| 232 | + 0.00 0 0.00 0.00 0 0.00 | |
| 233 | + 4.00 2 0.01 0.00 0 0.00 | |
| 234 | + 24.00 1 0.02 0.00 0 0.00 | |
| 235 | + 24.00 1 0.02 0.00 0 0.00 | |
| 236 | + 0.00 0 0.00 0.00 0 0.00 | |
| 237 | + 6.08 99 0.59 0.00 0 0.00 | |
| 238 | + 11.58 133 1.50 0.00 0 0.00 | |
| 239 | + 0.00 0 0.00 0.00 0 0.00 | |
| 240 | + 0.00 0 0.00 0.00 0 0.00 | |
| 241 | + 15.20 5 0.07 0.00 0 0.00 | |
| 242 | + 4.00 2 0.01 0.00 0 0.00 | |
| 243 | + disk0 disk4 | |
| 244 | + KB/t tps MB/s KB/t tps MB/s | |
| 245 | + 0.00 0 0.00 0.00 0 0.00 | |
| 246 | + 0.00 0 0.00 0.00 0 0.00 | |
| 247 | + 0.00 0 0.00 0.00 0 0.00 | |
| 248 | + 0.00 0 0.00 0.00 0 0.00 | |
| 249 | + 6.12 161 0.96 0.00 0 0.00 | |
| 250 | + 16.00 1 0.02 0.00 0 0.00 | |
| 251 | + 10.67 3 0.03 0.00 0 0.00 | |
| 252 | + 4.00 1 0.00 0.00 0 0.00 | |
| 253 | + 18.00 4 0.07 0.00 0 0.00 | |
| 254 | + 0.00 0 0.00 0.00 0 0.00 | |
| 255 | + 0.00 0 0.00 0.00 0 0.00 | |
| 256 | + 0.00 0 0.00 0.00 0 0.00 | |
| 257 | + 24.00 1 0.02 0.00 0 0.00 | |
| 258 | + 0.00 0 0.00 0.00 0 0.00 | |
| 259 | + 0.00 0 0.00 0.00 0 0.00 | |
| 260 | + 16.00 1 0.02 0.00 0 0.00 | |
| 261 | + 15.43 7 0.10 0.00 0 0.00 | |
| 262 | + 0.00 0 0.00 0.00 0 0.00 | |
| 263 | + 7.93 1345 10.41 0.00 0 0.00 | |
| 264 | + 0.00 0 0.00 0.00 0 0.00 | |
| 265 | + disk0 disk4 | |
| 266 | + KB/t tps MB/s KB/t tps MB/s | |
| 267 | + 6.06 104 0.62 0.00 0 0.00 | |
| 268 | + 22.40 5 0.11 0.00 0 0.00 | |
| 269 | + 0.00 0 0.00 0.00 0 0.00 | |
| 270 | + 5.66 246 1.36 0.00 0 0.00 | |
| 271 | + 10.49 234 2.39 0.00 0 0.00 | |
| 272 | + 16.00 2 0.03 0.00 0 0.00 | |
| 273 | + 0.00 0 0.00 0.00 0 0.00 | |
| 274 | + 0.00 0 0.00 0.00 0 0.00 | |
| 275 | + 18.00 4 0.07 0.00 0 0.00 | |
| 276 | + 0.00 0 0.00 0.00 0 0.00 | |
| 277 | + 8.03 128 1.01 0.00 0 0.00 | |
| 278 | + 4.00 1 0.00 0.00 0 0.00 | |
| 279 | + 24.00 1 0.02 0.00 0 0.00 | |
| 280 | + 5.37 66 0.35 0.00 0 0.00 | |
| 281 | + 0.00 0 0.00 0.00 0 0.00 | |
| 282 | + 16.00 2 0.03 0.00 0 0.00 | |
| 283 | + 4.00 2 0.01 0.00 0 0.00 | |
| 284 | + 24.00 1 0.02 0.00 0 0.00 | |
| 285 | + 18.00 4 0.07 0.00 0 0.00 | |
| 286 | + 0.00 0 0.00 0.00 0 0.00 | |
| 287 | + disk0 disk4 | |
| 288 | + KB/t tps MB/s KB/t tps MB/s | |
| 289 | + 132.00 1 0.13 0.00 0 0.00 | |
| 290 | + 0.00 0 0.00 0.00 0 0.00 | |
| 291 | + 0.00 0 0.00 0.00 0 0.00 | |
| 292 | + 16.00 2 0.03 0.00 0 0.00 | |
| 293 | + 124.80 5 0.61 0.00 0 0.00 | |
| 294 | + 16.00 1 0.02 0.00 0 0.00 | |
| 295 | + 0.00 0 0.00 0.00 0 0.00 | |
| 296 | + 0.00 0 0.00 0.00 0 0.00 | |
| 297 | + 5.64 193 1.07 0.00 0 0.00 | |
| 298 | + 6.52 46 0.29 0.00 0 0.00 | |
| 299 | + 5.91 91 0.53 0.00 0 0.00 | |
| 300 | + 117.60 5 0.57 0.00 0 0.00 | |
| 301 | + 12.80 5 0.06 0.00 0 0.00 | |
| 302 | + 4.00 1 0.00 0.00 0 0.00 | |
| 303 | + 12.21 261 3.11 0.00 0 0.00 | |
| 304 | + 951.83 2543 2363.61 0.00 0 0.00 | |
| 305 | + 905.23 6532 5774.04 0.00 0 0.00 | |
| 306 | + 16.00 110571 1728.03 0.00 0 0.00 | |
| 307 | + 16.00 117808 1840.66 0.00 0 0.00 | |
| 308 | + 16.00 113138 1767.77 0.00 0 0.00 | |
| 309 | + disk0 disk4 | |
| 310 | + KB/t tps MB/s KB/t tps MB/s | |
| 311 | + 16.00 118035 1844.30 0.00 0 0.00 | |
| 312 | + 16.00 116647 1822.61 0.00 0 0.00 | |
| 313 | + 16.00 110491 1726.42 0.00 0 0.00 | |
| 314 | + 16.00 116742 1824.09 0.00 0 0.00 | |
| 315 | + 16.00 116735 1823.98 0.00 0 0.00 | |
| 316 | + 16.00 109081 1704.35 0.00 0 0.00 | |
| 317 | + 16.00 115677 1807.43 0.00 0 0.00 | |
| 318 | + 16.00 115588 1806.06 0.00 0 0.00 | |
| 319 | + 16.00 109279 1707.50 0.00 0 0.00 | |
| 320 | + 16.00 114269 1785.43 0.00 0 0.00 | |
| 321 | + 16.00 116080 1813.76 0.00 0 0.00 | |
| 322 | + 16.00 107318 1676.85 0.00 0 0.00 | |
| 323 | + 16.00 115766 1808.84 0.00 0 0.00 | |
| 324 | + 16.00 116779 1824.67 0.00 0 0.00 | |
| 325 | + 46.58 88651 4032.80 0.00 0 0.00 | |
| 326 | + 64.00 80049 5003.02 0.00 0 0.00 | |
| 327 | + 64.00 81712 5107.00 0.00 0 0.00 | |
| 328 | + 64.00 79099 4943.66 0.00 0 0.00 | |
| 329 | + 64.00 82501 5156.16 0.00 0 0.00 | |
| 330 | + 210.83 48738 10034.73 0.00 0 0.00 | |
| 331 | + disk0 disk4 | |
| 332 | + KB/t tps MB/s KB/t tps MB/s | |
| 333 | + 255.95 43990 10995.49 0.00 0 0.00 | |
| 334 | + 508.97 24373 12114.33 0.00 0 0.00 | |
| 335 | + 1023.68 12748 12743.58 0.00 0 0.00 | |
| 336 | + 728.33 17860 12703.08 0.00 0 0.00 | |
added
results/expH_ssd_feasibility/20260812T034359Z/results.json
+1072 −0
@@ -0,0 +1,1072 @@ | ||
| 1 | +{ | |
| 2 | + "experiment": "expH_ssd_feasibility", | |
| 3 | + "author": "Simon-Pierre Boucher", | |
| 4 | + "contact": "contact@spboucher.ai", | |
| 5 | + "manifest": { | |
| 6 | + "author": "Simon-Pierre Boucher", | |
| 7 | + "contact": "contact@spboucher.ai", | |
| 8 | + "project": "localvm-research", | |
| 9 | + "collected_utc": "2026-08-12T03:43:58.831292+00:00", | |
| 10 | + "chip": { | |
| 11 | + "brand": "Apple M5 Max", | |
| 12 | + "arch": "arm64", | |
| 13 | + "cores_total": 18, | |
| 14 | + "cores_performance": 6, | |
| 15 | + "cores_efficiency": 12, | |
| 16 | + "gpu_cores": 40 | |
| 17 | + }, | |
| 18 | + "memory": { | |
| 19 | + "unified_bytes": 51539607552, | |
| 20 | + "unified_gb": 48.0, | |
| 21 | + "pagesize": 16384 | |
| 22 | + }, | |
| 23 | + "ssd": { | |
| 24 | + "model": "APPLE SSD AP2048Z", | |
| 25 | + "size": "2 TB", | |
| 26 | + "smart_status": "Verified" | |
| 27 | + }, | |
| 28 | + "os": { | |
| 29 | + "product": "macOS", | |
| 30 | + "version": "27.0", | |
| 31 | + "build": "26A5388g", | |
| 32 | + "kernel": "27.0.0" | |
| 33 | + }, | |
| 34 | + "software": { | |
| 35 | + "python": "3.14.4", | |
| 36 | + "mlx": "0.32.0", | |
| 37 | + "mlx_lm": null, | |
| 38 | + "torch": "2.12.0", | |
| 39 | + "numpy": "2.4.4" | |
| 40 | + }, | |
| 41 | + "git": { | |
| 42 | + "commit": "b4a652d2362706cf660f9f1a8082eba53592bc84", | |
| 43 | + "dirty_tree": true | |
| 44 | + }, | |
| 45 | + "thermal_level_at_collect": null | |
| 46 | + }, | |
| 47 | + "config": { | |
| 48 | + "file_gib": 8.0, | |
| 49 | + "repeats": 3, | |
| 50 | + "budget_mib": 256, | |
| 51 | + "quick": false | |
| 52 | + }, | |
| 53 | + "test_file_bytes": 8589934592, | |
| 54 | + "gpu_load_matmul_iterations": 10032, | |
| 55 | + "thermal_level_after": null, | |
| 56 | + "cells": [ | |
| 57 | + { | |
| 58 | + "block_bytes": 4096, | |
| 59 | + "pattern": "random", | |
| 60 | + "nocache": true, | |
| 61 | + "threads": 1, | |
| 62 | + "gpu_load": false, | |
| 63 | + "repeats": 3, | |
| 64 | + "budget_bytes": 268435456, | |
| 65 | + "mb_per_s_mean": 66.9404714855185, | |
| 66 | + "mb_per_s_median": 66.81818957398704, | |
| 67 | + "mb_per_s_std": 0.35694487918688783, | |
| 68 | + "iops_mean": 16342.888546269165 | |
| 69 | + }, | |
| 70 | + { | |
| 71 | + "block_bytes": 4096, | |
| 72 | + "pattern": "random", | |
| 73 | + "nocache": true, | |
| 74 | + "threads": 4, | |
| 75 | + "gpu_load": false, | |
| 76 | + "repeats": 3, | |
| 77 | + "budget_bytes": 779313911, | |
| 78 | + "mb_per_s_mean": 254.1893090701863, | |
| 79 | + "mb_per_s_median": 254.38198941931918, | |
| 80 | + "mb_per_s_std": 0.7176359985579541, | |
| 81 | + "iops_mean": 62057.936784713456 | |
| 82 | + }, | |
| 83 | + { | |
| 84 | + "block_bytes": 4096, | |
| 85 | + "pattern": "random", | |
| 86 | + "nocache": true, | |
| 87 | + "threads": 8, | |
| 88 | + "gpu_load": false, | |
| 89 | + "repeats": 3, | |
| 90 | + "budget_bytes": 1416435497, | |
| 91 | + "mb_per_s_mean": 479.9329254027471, | |
| 92 | + "mb_per_s_median": 477.30629754724333, | |
| 93 | + "mb_per_s_std": 4.914641441447007, | |
| 94 | + "iops_mean": 117171.12436590505 | |
| 95 | + }, | |
| 96 | + { | |
| 97 | + "block_bytes": 16384, | |
| 98 | + "pattern": "random", | |
| 99 | + "nocache": true, | |
| 100 | + "threads": 1, | |
| 101 | + "gpu_load": false, | |
| 102 | + "repeats": 3, | |
| 103 | + "budget_bytes": 792376370, | |
| 104 | + "mb_per_s_mean": 271.5236092855627, | |
| 105 | + "mb_per_s_median": 271.89875725074535, | |
| 106 | + "mb_per_s_std": 0.9362987548937706, | |
| 107 | + "iops_mean": 16572.48591830827 | |
| 108 | + }, | |
| 109 | + { | |
| 110 | + "block_bytes": 16384, | |
| 111 | + "pattern": "random", | |
| 112 | + "nocache": true, | |
| 113 | + "threads": 4, | |
| 114 | + "gpu_load": false, | |
| 115 | + "repeats": 3, | |
| 116 | + "budget_bytes": 3107779103, | |
| 117 | + "mb_per_s_mean": 1020.114525861106, | |
| 118 | + "mb_per_s_median": 1016.582082480907, | |
| 119 | + "mb_per_s_std": 7.121204487376759, | |
| 120 | + "iops_mean": 62262.84947882727 | |
| 121 | + }, | |
| 122 | + { | |
| 123 | + "block_bytes": 16384, | |
| 124 | + "pattern": "random", | |
| 125 | + "nocache": true, | |
| 126 | + "threads": 8, | |
| 127 | + "gpu_load": false, | |
| 128 | + "repeats": 3, | |
| 129 | + "budget_bytes": 5546518370, | |
| 130 | + "mb_per_s_mean": 1914.0060665159742, | |
| 131 | + "mb_per_s_median": 1904.4346643621623, | |
| 132 | + "mb_per_s_std": 20.83233031292896, | |
| 133 | + "iops_mean": 116821.65933325038 | |
| 134 | + }, | |
| 135 | + { | |
| 136 | + "block_bytes": 65536, | |
| 137 | + "pattern": "random", | |
| 138 | + "nocache": true, | |
| 139 | + "threads": 1, | |
| 140 | + "gpu_load": false, | |
| 141 | + "repeats": 3, | |
| 142 | + "budget_bytes": 2282668305, | |
| 143 | + "mb_per_s_mean": 769.5064458059036, | |
| 144 | + "mb_per_s_median": 770.9807098513519, | |
| 145 | + "mb_per_s_std": 7.351182080135739, | |
| 146 | + "iops_mean": 11741.73653878637 | |
| 147 | + }, | |
| 148 | + { | |
| 149 | + "block_bytes": 65536, | |
| 150 | + "pattern": "random", | |
| 151 | + "nocache": true, | |
| 152 | + "threads": 4, | |
| 153 | + "gpu_load": false, | |
| 154 | + "repeats": 3, | |
| 155 | + "budget_bytes": 8202457332, | |
| 156 | + "mb_per_s_mean": 2917.446150981038, | |
| 157 | + "mb_per_s_median": 2921.1746081661267, | |
| 158 | + "mb_per_s_std": 6.740200316834216, | |
| 159 | + "iops_mean": 44516.69541902219 | |
| 160 | + }, | |
| 161 | + { | |
| 162 | + "block_bytes": 65536, | |
| 163 | + "pattern": "random", | |
| 164 | + "nocache": true, | |
| 165 | + "threads": 8, | |
| 166 | + "gpu_load": false, | |
| 167 | + "repeats": 3, | |
| 168 | + "budget_bytes": 8589934592, | |
| 169 | + "mb_per_s_mean": 5137.568745521635, | |
| 170 | + "mb_per_s_median": 5232.037270174379, | |
| 171 | + "mb_per_s_std": 172.5656556041619, | |
| 172 | + "iops_mean": 78393.07778200737 | |
| 173 | + }, | |
| 174 | + { | |
| 175 | + "block_bytes": 262144, | |
| 176 | + "pattern": "random", | |
| 177 | + "nocache": true, | |
| 178 | + "threads": 1, | |
| 179 | + "gpu_load": false, | |
| 180 | + "repeats": 3, | |
| 181 | + "budget_bytes": 7077957286, | |
| 182 | + "mb_per_s_mean": 2302.4898879657267, | |
| 183 | + "mb_per_s_median": 2270.912495936314, | |
| 184 | + "mb_per_s_std": 63.88880522233235, | |
| 185 | + "iops_mean": 8783.30187975207 | |
| 186 | + }, | |
| 187 | + { | |
| 188 | + "block_bytes": 262144, | |
| 189 | + "pattern": "random", | |
| 190 | + "nocache": true, | |
| 191 | + "threads": 4, | |
| 192 | + "gpu_load": false, | |
| 193 | + "repeats": 3, | |
| 194 | + "budget_bytes": 8589934592, | |
| 195 | + "mb_per_s_mean": 8134.0178194677865, | |
| 196 | + "mb_per_s_median": 8149.727408328307, | |
| 197 | + "mb_per_s_std": 29.232837861725862, | |
| 198 | + "iops_mean": 31028.81553446879 | |
| 199 | + }, | |
| 200 | + { | |
| 201 | + "block_bytes": 262144, | |
| 202 | + "pattern": "random", | |
| 203 | + "nocache": true, | |
| 204 | + "threads": 8, | |
| 205 | + "gpu_load": false, | |
| 206 | + "repeats": 3, | |
| 207 | + "budget_bytes": 8589934592, | |
| 208 | + "mb_per_s_mean": 11590.499886788693, | |
| 209 | + "mb_per_s_median": 11578.704613755446, | |
| 210 | + "mb_per_s_std": 46.93897840523587, | |
| 211 | + "iops_mean": 44214.248225359705 | |
| 212 | + }, | |
| 213 | + { | |
| 214 | + "block_bytes": 1048576, | |
| 215 | + "pattern": "random", | |
| 216 | + "nocache": true, | |
| 217 | + "threads": 1, | |
| 218 | + "gpu_load": false, | |
| 219 | + "repeats": 3, | |
| 220 | + "budget_bytes": 8589934592, | |
| 221 | + "mb_per_s_mean": 4555.8072703006865, | |
| 222 | + "mb_per_s_median": 4550.336341890094, | |
| 223 | + "mb_per_s_std": 21.971122405961314, | |
| 224 | + "iops_mean": 4344.756384182631 | |
| 225 | + }, | |
| 226 | + { | |
| 227 | + "block_bytes": 1048576, | |
| 228 | + "pattern": "random", | |
| 229 | + "nocache": true, | |
| 230 | + "threads": 4, | |
| 231 | + "gpu_load": false, | |
| 232 | + "repeats": 3, | |
| 233 | + "budget_bytes": 8589934592, | |
| 234 | + "mb_per_s_mean": 13627.559191790438, | |
| 235 | + "mb_per_s_median": 13623.559341244068, | |
| 236 | + "mb_per_s_std": 10.41341531736513, | |
| 237 | + "iops_mean": 12996.253196516456 | |
| 238 | + }, | |
| 239 | + { | |
| 240 | + "block_bytes": 1048576, | |
| 241 | + "pattern": "random", | |
| 242 | + "nocache": true, | |
| 243 | + "threads": 8, | |
| 244 | + "gpu_load": false, | |
| 245 | + "repeats": 3, | |
| 246 | + "budget_bytes": 8589934592, | |
| 247 | + "mb_per_s_mean": 13788.768917982177, | |
| 248 | + "mb_per_s_median": 13799.81956863669, | |
| 249 | + "mb_per_s_std": 34.65064663137935, | |
| 250 | + "iops_mean": 13149.9947719404 | |
| 251 | + }, | |
| 252 | + { | |
| 253 | + "block_bytes": 4194304, | |
| 254 | + "pattern": "random", | |
| 255 | + "nocache": true, | |
| 256 | + "threads": 1, | |
| 257 | + "gpu_load": false, | |
| 258 | + "repeats": 3, | |
| 259 | + "budget_bytes": 8589934592, | |
| 260 | + "mb_per_s_mean": 12402.590071473736, | |
| 261 | + "mb_per_s_median": 12883.232717044471, | |
| 262 | + "mb_per_s_std": 1709.430853228165, | |
| 263 | + "iops_mean": 2957.0079020199146 | |
| 264 | + }, | |
| 265 | + { | |
| 266 | + "block_bytes": 4194304, | |
| 267 | + "pattern": "random", | |
| 268 | + "nocache": true, | |
| 269 | + "threads": 4, | |
| 270 | + "gpu_load": false, | |
| 271 | + "repeats": 3, | |
| 272 | + "budget_bytes": 8589934592, | |
| 273 | + "mb_per_s_mean": 48302.31789990957, | |
| 274 | + "mb_per_s_median": 48073.32828905314, | |
| 275 | + "mb_per_s_std": 840.6008705042974, | |
| 276 | + "iops_mean": 11516.170001008408 | |
| 277 | + }, | |
| 278 | + { | |
| 279 | + "block_bytes": 4194304, | |
| 280 | + "pattern": "random", | |
| 281 | + "nocache": true, | |
| 282 | + "threads": 8, | |
| 283 | + "gpu_load": false, | |
| 284 | + "repeats": 3, | |
| 285 | + "budget_bytes": 8589934592, | |
| 286 | + "mb_per_s_mean": 54669.70854743223, | |
| 287 | + "mb_per_s_median": 54748.470018810105, | |
| 288 | + "mb_per_s_std": 204.85515698656337, | |
| 289 | + "iops_mean": 13034.27423177534 | |
| 290 | + }, | |
| 291 | + { | |
| 292 | + "block_bytes": 4096, | |
| 293 | + "pattern": "sequential", | |
| 294 | + "nocache": true, | |
| 295 | + "threads": 1, | |
| 296 | + "gpu_load": false, | |
| 297 | + "repeats": 3, | |
| 298 | + "budget_bytes": 1731126863, | |
| 299 | + "mb_per_s_mean": 533.2201141223684, | |
| 300 | + "mb_per_s_median": 537.7438262876947, | |
| 301 | + "mb_per_s_std": 8.055973488534326, | |
| 302 | + "iops_mean": 130180.69192440633 | |
| 303 | + }, | |
| 304 | + { | |
| 305 | + "block_bytes": 4096, | |
| 306 | + "pattern": "sequential", | |
| 307 | + "nocache": true, | |
| 308 | + "threads": 4, | |
| 309 | + "gpu_load": false, | |
| 310 | + "repeats": 3, | |
| 311 | + "budget_bytes": 2325091738, | |
| 312 | + "mb_per_s_mean": 808.948619823331, | |
| 313 | + "mb_per_s_median": 803.7552072780586, | |
| 314 | + "mb_per_s_std": 15.047566187682303, | |
| 315 | + "iops_mean": 197497.22163655542 | |
| 316 | + }, | |
| 317 | + { | |
| 318 | + "block_bytes": 4096, | |
| 319 | + "pattern": "sequential", | |
| 320 | + "nocache": true, | |
| 321 | + "threads": 8, | |
| 322 | + "gpu_load": false, | |
| 323 | + "repeats": 3, | |
| 324 | + "budget_bytes": 3533900213, | |
| 325 | + "mb_per_s_mean": 1278.0558840916142, | |
| 326 | + "mb_per_s_median": 1276.0966617695374, | |
| 327 | + "mb_per_s_std": 16.479351890695185, | |
| 328 | + "iops_mean": 312025.36232705426 | |
| 329 | + }, | |
| 330 | + { | |
| 331 | + "block_bytes": 16384, | |
| 332 | + "pattern": "sequential", | |
| 333 | + "nocache": true, | |
| 334 | + "threads": 1, | |
| 335 | + "gpu_load": false, | |
| 336 | + "repeats": 3, | |
| 337 | + "budget_bytes": 5704339689, | |
| 338 | + "mb_per_s_mean": 1683.4541778440341, | |
| 339 | + "mb_per_s_median": 1682.1212913476447, | |
| 340 | + "mb_per_s_std": 11.042299232794667, | |
| 341 | + "iops_mean": 102749.88878442592 | |
| 342 | + }, | |
| 343 | + { | |
| 344 | + "block_bytes": 16384, | |
| 345 | + "pattern": "sequential", | |
| 346 | + "nocache": true, | |
| 347 | + "threads": 4, | |
| 348 | + "gpu_load": false, | |
| 349 | + "repeats": 3, | |
| 350 | + "budget_bytes": 8589934592, | |
| 351 | + "mb_per_s_mean": 3235.629632580155, | |
| 352 | + "mb_per_s_median": 3235.9738272143163, | |
| 353 | + "mb_per_s_std": 9.75227833806203, | |
| 354 | + "iops_mean": 197487.16019165984 | |
| 355 | + }, | |
| 356 | + { | |
| 357 | + "block_bytes": 16384, | |
| 358 | + "pattern": "sequential", | |
| 359 | + "nocache": true, | |
| 360 | + "threads": 8, | |
| 361 | + "gpu_load": false, | |
| 362 | + "repeats": 3, | |
| 363 | + "budget_bytes": 8589934592, | |
| 364 | + "mb_per_s_mean": 4609.956362698625, | |
| 365 | + "mb_per_s_median": 4604.9110008589405, | |
| 366 | + "mb_per_s_std": 21.40866342375183, | |
| 367 | + "iops_mean": 281369.4069029922 | |
| 368 | + }, | |
| 369 | + { | |
| 370 | + "block_bytes": 65536, | |
| 371 | + "pattern": "sequential", | |
| 372 | + "nocache": true, | |
| 373 | + "threads": 1, | |
| 374 | + "gpu_load": false, | |
| 375 | + "repeats": 3, | |
| 376 | + "budget_bytes": 7712973384, | |
| 377 | + "mb_per_s_mean": 2297.0864023031027, | |
| 378 | + "mb_per_s_median": 2303.205490335651, | |
| 379 | + "mb_per_s_std": 43.60780725006312, | |
| 380 | + "iops_mean": 35050.756871080055 | |
| 381 | + }, | |
| 382 | + { | |
| 383 | + "block_bytes": 65536, | |
| 384 | + "pattern": "sequential", | |
| 385 | + "nocache": true, | |
| 386 | + "threads": 4, | |
| 387 | + "gpu_load": false, | |
| 388 | + "repeats": 3, | |
| 389 | + "budget_bytes": 8589934592, | |
| 390 | + "mb_per_s_mean": 8909.168461018951, | |
| 391 | + "mb_per_s_median": 8913.559953040489, | |
| 392 | + "mb_per_s_std": 13.441699453599812, | |
| 393 | + "iops_mean": 135943.12226896593 | |
| 394 | + }, | |
| 395 | + { | |
| 396 | + "block_bytes": 65536, | |
| 397 | + "pattern": "sequential", | |
| 398 | + "nocache": true, | |
| 399 | + "threads": 8, | |
| 400 | + "gpu_load": false, | |
| 401 | + "repeats": 3, | |
| 402 | + "budget_bytes": 8589934592, | |
| 403 | + "mb_per_s_mean": 14453.31272605767, | |
| 404 | + "mb_per_s_median": 14547.811286041728, | |
| 405 | + "mb_per_s_std": 186.43281335956638, | |
| 406 | + "iops_mean": 220540.0501412608 | |
| 407 | + }, | |
| 408 | + { | |
| 409 | + "block_bytes": 262144, | |
| 410 | + "pattern": "sequential", | |
| 411 | + "nocache": true, | |
| 412 | + "threads": 1, | |
| 413 | + "gpu_load": false, | |
| 414 | + "repeats": 3, | |
| 415 | + "budget_bytes": 8589934592, | |
| 416 | + "mb_per_s_mean": 6645.781042732493, | |
| 417 | + "mb_per_s_median": 6645.407502186923, | |
| 418 | + "mb_per_s_std": 20.755385770241354, | |
| 419 | + "iops_mean": 25351.642771654104 | |
| 420 | + }, | |
| 421 | + { | |
| 422 | + "block_bytes": 262144, | |
| 423 | + "pattern": "sequential", | |
| 424 | + "nocache": true, | |
| 425 | + "threads": 4, | |
| 426 | + "gpu_load": false, | |
| 427 | + "repeats": 3, | |
| 428 | + "budget_bytes": 8589934592, | |
| 429 | + "mb_per_s_mean": 24105.905011831746, | |
| 430 | + "mb_per_s_median": 24137.64635014016, | |
| 431 | + "mb_per_s_std": 112.8076208054412, | |
| 432 | + "iops_mean": 91956.72993405054 | |
| 433 | + }, | |
| 434 | + { | |
| 435 | + "block_bytes": 262144, | |
| 436 | + "pattern": "sequential", | |
| 437 | + "nocache": true, | |
| 438 | + "threads": 8, | |
| 439 | + "gpu_load": false, | |
| 440 | + "repeats": 3, | |
| 441 | + "budget_bytes": 8589934592, | |
| 442 | + "mb_per_s_mean": 27404.562122412142, | |
| 443 | + "mb_per_s_median": 27323.04002007838, | |
| 444 | + "mb_per_s_std": 242.0050048684501, | |
| 445 | + "iops_mean": 104540.10819401605 | |
| 446 | + }, | |
| 447 | + { | |
| 448 | + "block_bytes": 1048576, | |
| 449 | + "pattern": "sequential", | |
| 450 | + "nocache": true, | |
| 451 | + "threads": 1, | |
| 452 | + "gpu_load": false, | |
| 453 | + "repeats": 3, | |
| 454 | + "budget_bytes": 8589934592, | |
| 455 | + "mb_per_s_mean": 12139.10666140986, | |
| 456 | + "mb_per_s_median": 12121.789771248836, | |
| 457 | + "mb_per_s_std": 58.91428212284964, | |
| 458 | + "iops_mean": 11576.754247102604 | |
| 459 | + }, | |
| 460 | + { | |
| 461 | + "block_bytes": 1048576, | |
| 462 | + "pattern": "sequential", | |
| 463 | + "nocache": true, | |
| 464 | + "threads": 4, | |
| 465 | + "gpu_load": false, | |
| 466 | + "repeats": 3, | |
| 467 | + "budget_bytes": 8589934592, | |
| 468 | + "mb_per_s_mean": 17844.382066771657, | |
| 469 | + "mb_per_s_median": 17887.961156996822, | |
| 470 | + "mb_per_s_std": 85.98610137287082, | |
| 471 | + "iops_mean": 17017.72886922041 | |
| 472 | + }, | |
| 473 | + { | |
| 474 | + "block_bytes": 1048576, | |
| 475 | + "pattern": "sequential", | |
| 476 | + "nocache": true, | |
| 477 | + "threads": 8, | |
| 478 | + "gpu_load": false, | |
| 479 | + "repeats": 3, | |
| 480 | + "budget_bytes": 8589934592, | |
| 481 | + "mb_per_s_mean": 34075.14820508285, | |
| 482 | + "mb_per_s_median": 34000.091116193194, | |
| 483 | + "mb_per_s_std": 162.02303904684373, | |
| 484 | + "iops_mean": 32496.59367092404 | |
| 485 | + }, | |
| 486 | + { | |
| 487 | + "block_bytes": 4194304, | |
| 488 | + "pattern": "sequential", | |
| 489 | + "nocache": true, | |
| 490 | + "threads": 1, | |
| 491 | + "gpu_load": false, | |
| 492 | + "repeats": 3, | |
| 493 | + "budget_bytes": 8589934592, | |
| 494 | + "mb_per_s_mean": 14030.78368694708, | |
| 495 | + "mb_per_s_median": 13924.19981054842, | |
| 496 | + "mb_per_s_std": 308.5869591644102, | |
| 497 | + "iops_mean": 3345.199510323305 | |
| 498 | + }, | |
| 499 | + { | |
| 500 | + "block_bytes": 4194304, | |
| 501 | + "pattern": "sequential", | |
| 502 | + "nocache": true, | |
| 503 | + "threads": 4, | |
| 504 | + "gpu_load": false, | |
| 505 | + "repeats": 3, | |
| 506 | + "budget_bytes": 8589934592, | |
| 507 | + "mb_per_s_mean": 46677.461538890755, | |
| 508 | + "mb_per_s_median": 47090.85241001869, | |
| 509 | + "mb_per_s_std": 829.8895672088935, | |
| 510 | + "iops_mean": 11128.774056170167 | |
| 511 | + }, | |
| 512 | + { | |
| 513 | + "block_bytes": 4194304, | |
| 514 | + "pattern": "sequential", | |
| 515 | + "nocache": true, | |
| 516 | + "threads": 8, | |
| 517 | + "gpu_load": false, | |
| 518 | + "repeats": 3, | |
| 519 | + "budget_bytes": 8589934592, | |
| 520 | + "mb_per_s_mean": 52861.64960218013, | |
| 521 | + "mb_per_s_median": 52848.9806852303, | |
| 522 | + "mb_per_s_std": 79.6172666483042, | |
| 523 | + "iops_mean": 12603.199387116461 | |
| 524 | + }, | |
| 525 | + { | |
| 526 | + "block_bytes": 4096, | |
| 527 | + "pattern": "random", | |
| 528 | + "nocache": false, | |
| 529 | + "threads": 1, | |
| 530 | + "gpu_load": false, | |
| 531 | + "repeats": 3, | |
| 532 | + "budget_bytes": 8589934592, | |
| 533 | + "mb_per_s_mean": 4833.8428865859405, | |
| 534 | + "mb_per_s_median": 4861.2682882384925, | |
| 535 | + "mb_per_s_std": 86.58442576019786, | |
| 536 | + "iops_mean": 1180137.4234828956 | |
| 537 | + }, | |
| 538 | + { | |
| 539 | + "block_bytes": 4096, | |
| 540 | + "pattern": "random", | |
| 541 | + "nocache": false, | |
| 542 | + "threads": 4, | |
| 543 | + "gpu_load": false, | |
| 544 | + "repeats": 3, | |
| 545 | + "budget_bytes": 3837545612, | |
| 546 | + "mb_per_s_mean": 1297.7420334022972, | |
| 547 | + "mb_per_s_median": 1300.895252932938, | |
| 548 | + "mb_per_s_std": 13.019775132944018, | |
| 549 | + "iops_mean": 316831.5511236077 | |
| 550 | + }, | |
| 551 | + { | |
| 552 | + "block_bytes": 4096, | |
| 553 | + "pattern": "random", | |
| 554 | + "nocache": false, | |
| 555 | + "threads": 8, | |
| 556 | + "gpu_load": false, | |
| 557 | + "repeats": 3, | |
| 558 | + "budget_bytes": 2496186496, | |
| 559 | + "mb_per_s_mean": 838.6389323935614, | |
| 560 | + "mb_per_s_median": 840.6294270996217, | |
| 561 | + "mb_per_s_std": 3.5395010230721073, | |
| 562 | + "iops_mean": 204745.83310389682 | |
| 563 | + }, | |
| 564 | + { | |
| 565 | + "block_bytes": 16384, | |
| 566 | + "pattern": "random", | |
| 567 | + "nocache": false, | |
| 568 | + "threads": 1, | |
| 569 | + "gpu_load": false, | |
| 570 | + "repeats": 3, | |
| 571 | + "budget_bytes": 8589934592, | |
| 572 | + "mb_per_s_mean": 13794.383075604997, | |
| 573 | + "mb_per_s_median": 13805.74785875728, | |
| 574 | + "mb_per_s_std": 26.797785604933974, | |
| 575 | + "iops_mean": 841942.3263919066 | |
| 576 | + }, | |
| 577 | + { | |
| 578 | + "block_bytes": 16384, | |
| 579 | + "pattern": "random", | |
| 580 | + "nocache": false, | |
| 581 | + "threads": 4, | |
| 582 | + "gpu_load": false, | |
| 583 | + "repeats": 3, | |
| 584 | + "budget_bytes": 8589934592, | |
| 585 | + "mb_per_s_mean": 6611.908832996066, | |
| 586 | + "mb_per_s_median": 6593.998544205867, | |
| 587 | + "mb_per_s_std": 48.69429568828386, | |
| 588 | + "iops_mean": 403558.88873267005 | |
| 589 | + }, | |
| 590 | + { | |
| 591 | + "block_bytes": 16384, | |
| 592 | + "pattern": "random", | |
| 593 | + "nocache": false, | |
| 594 | + "threads": 8, | |
| 595 | + "gpu_load": false, | |
| 596 | + "repeats": 3, | |
| 597 | + "budget_bytes": 8589934592, | |
| 598 | + "mb_per_s_mean": 3239.0038636823183, | |
| 599 | + "mb_per_s_median": 3235.7536007939602, | |
| 600 | + "mb_per_s_std": 15.5407398582331, | |
| 601 | + "iops_mean": 197693.106914204 | |
| 602 | + }, | |
| 603 | + { | |
| 604 | + "block_bytes": 65536, | |
| 605 | + "pattern": "random", | |
| 606 | + "nocache": false, | |
| 607 | + "threads": 1, | |
| 608 | + "gpu_load": false, | |
| 609 | + "repeats": 3, | |
| 610 | + "budget_bytes": 8589934592, | |
| 611 | + "mb_per_s_mean": 25056.861235823093, | |
| 612 | + "mb_per_s_median": 25070.731498877456, | |
| 613 | + "mb_per_s_std": 48.36346752965899, | |
| 614 | + "iops_mean": 382337.36016575765 | |
| 615 | + }, | |
| 616 | + { | |
| 617 | + "block_bytes": 65536, | |
| 618 | + "pattern": "random", | |
| 619 | + "nocache": false, | |
| 620 | + "threads": 4, | |
| 621 | + "gpu_load": false, | |
| 622 | + "repeats": 3, | |
| 623 | + "budget_bytes": 8589934592, | |
| 624 | + "mb_per_s_mean": 18890.134859980062, | |
| 625 | + "mb_per_s_median": 18924.878648963844, | |
| 626 | + "mb_per_s_std": 105.70593456006415, | |
| 627 | + "iops_mean": 288240.5831906138 | |
| 628 | + }, | |
| 629 | + { | |
| 630 | + "block_bytes": 65536, | |
| 631 | + "pattern": "random", | |
| 632 | + "nocache": false, | |
| 633 | + "threads": 8, | |
| 634 | + "gpu_load": false, | |
| 635 | + "repeats": 3, | |
| 636 | + "budget_bytes": 8589934592, | |
| 637 | + "mb_per_s_mean": 12634.34598109216, | |
| 638 | + "mb_per_s_median": 12631.497469912529, | |
| 639 | + "mb_per_s_std": 22.241199376823143, | |
| 640 | + "iops_mean": 192784.8202681299 | |
| 641 | + }, | |
| 642 | + { | |
| 643 | + "block_bytes": 262144, | |
| 644 | + "pattern": "random", | |
| 645 | + "nocache": false, | |
| 646 | + "threads": 1, | |
| 647 | + "gpu_load": false, | |
| 648 | + "repeats": 3, | |
| 649 | + "budget_bytes": 8589934592, | |
| 650 | + "mb_per_s_mean": 37768.7416908379, | |
| 651 | + "mb_per_s_median": 37672.0880532647, | |
| 652 | + "mb_per_s_std": 262.20428371927875, | |
| 653 | + "iops_mean": 144076.31565413627 | |
| 654 | + }, | |
| 655 | + { | |
| 656 | + "block_bytes": 262144, | |
| 657 | + "pattern": "random", | |
| 658 | + "nocache": false, | |
| 659 | + "threads": 4, | |
| 660 | + "gpu_load": false, | |
| 661 | + "repeats": 3, | |
| 662 | + "budget_bytes": 8589934592, | |
| 663 | + "mb_per_s_mean": 76369.7070018444, | |
| 664 | + "mb_per_s_median": 75988.93309625734, | |
| 665 | + "mb_per_s_std": 797.6245473422035, | |
| 666 | + "iops_mean": 291327.3124765183 | |
| 667 | + }, | |
| 668 | + { | |
| 669 | + "block_bytes": 262144, | |
| 670 | + "pattern": "random", | |
| 671 | + "nocache": false, | |
| 672 | + "threads": 8, | |
| 673 | + "gpu_load": false, | |
| 674 | + "repeats": 3, | |
| 675 | + "budget_bytes": 8589934592, | |
| 676 | + "mb_per_s_mean": 46092.76466515209, | |
| 677 | + "mb_per_s_median": 46119.54640399593, | |
| 678 | + "mb_per_s_std": 199.15678583641514, | |
| 679 | + "iops_mean": 175829.9433332523 | |
| 680 | + }, | |
| 681 | + { | |
| 682 | + "block_bytes": 1048576, | |
| 683 | + "pattern": "random", | |
| 684 | + "nocache": false, | |
| 685 | + "threads": 1, | |
| 686 | + "gpu_load": false, | |
| 687 | + "repeats": 3, | |
| 688 | + "budget_bytes": 8589934592, | |
| 689 | + "mb_per_s_mean": 42521.21851423637, | |
| 690 | + "mb_per_s_median": 42672.19699901347, | |
| 691 | + "mb_per_s_std": 336.25170062821485, | |
| 692 | + "iops_mean": 40551.39399932515 | |
| 693 | + }, | |
| 694 | + { | |
| 695 | + "block_bytes": 1048576, | |
| 696 | + "pattern": "random", | |
| 697 | + "nocache": false, | |
| 698 | + "threads": 4, | |
| 699 | + "gpu_load": false, | |
| 700 | + "repeats": 3, | |
| 701 | + "budget_bytes": 8589934592, | |
| 702 | + "mb_per_s_mean": 117590.73300051215, | |
| 703 | + "mb_per_s_median": 117908.10422192531, | |
| 704 | + "mb_per_s_std": 725.5453509369022, | |
| 705 | + "iops_mean": 112143.26190997328 | |
| 706 | + }, | |
| 707 | + { | |
| 708 | + "block_bytes": 1048576, | |
| 709 | + "pattern": "random", | |
| 710 | + "nocache": false, | |
| 711 | + "threads": 8, | |
| 712 | + "gpu_load": false, | |
| 713 | + "repeats": 3, | |
| 714 | + "budget_bytes": 8589934592, | |
| 715 | + "mb_per_s_mean": 136162.510007979, | |
| 716 | + "mb_per_s_median": 136485.28601135127, | |
| 717 | + "mb_per_s_std": 590.72962549765, | |
| 718 | + "iops_mean": 129854.68865201854 | |
| 719 | + }, | |
| 720 | + { | |
| 721 | + "block_bytes": 4194304, | |
| 722 | + "pattern": "random", | |
| 723 | + "nocache": false, | |
| 724 | + "threads": 1, | |
| 725 | + "gpu_load": false, | |
| 726 | + "repeats": 3, | |
| 727 | + "budget_bytes": 8589934592, | |
| 728 | + "mb_per_s_mean": 42051.24976793863, | |
| 729 | + "mb_per_s_median": 41922.47628846252, | |
| 730 | + "mb_per_s_std": 349.48052173829495, | |
| 731 | + "iops_mean": 10025.799219116838 | |
| 732 | + }, | |
| 733 | + { | |
| 734 | + "block_bytes": 4194304, | |
| 735 | + "pattern": "random", | |
| 736 | + "nocache": false, | |
| 737 | + "threads": 4, | |
| 738 | + "gpu_load": false, | |
| 739 | + "repeats": 3, | |
| 740 | + "budget_bytes": 8589934592, | |
| 741 | + "mb_per_s_mean": 90258.19122016594, | |
| 742 | + "mb_per_s_median": 90475.04629189364, | |
| 743 | + "mb_per_s_std": 460.2574373857038, | |
| 744 | + "iops_mean": 21519.229702989087 | |
| 745 | + }, | |
| 746 | + { | |
| 747 | + "block_bytes": 4194304, | |
| 748 | + "pattern": "random", | |
| 749 | + "nocache": false, | |
| 750 | + "threads": 8, | |
| 751 | + "gpu_load": false, | |
| 752 | + "repeats": 3, | |
| 753 | + "budget_bytes": 8589934592, | |
| 754 | + "mb_per_s_mean": 104391.19356704544, | |
| 755 | + "mb_per_s_median": 104843.148232012, | |
| 756 | + "mb_per_s_std": 1480.6322295507905, | |
| 757 | + "iops_mean": 24888.800040971146 | |
| 758 | + }, | |
| 759 | + { | |
| 760 | + "block_bytes": 4096, | |
| 761 | + "pattern": "sequential", | |
| 762 | + "nocache": false, | |
| 763 | + "threads": 1, | |
| 764 | + "gpu_load": false, | |
| 765 | + "repeats": 3, | |
| 766 | + "budget_bytes": 8589934592, | |
| 767 | + "mb_per_s_mean": 9169.630879150798, | |
| 768 | + "mb_per_s_median": 9206.796888636633, | |
| 769 | + "mb_per_s_std": 84.58152137626024, | |
| 770 | + "iops_mean": 2238679.4138551755 | |
| 771 | + }, | |
| 772 | + { | |
| 773 | + "block_bytes": 4096, | |
| 774 | + "pattern": "sequential", | |
| 775 | + "nocache": false, | |
| 776 | + "threads": 4, | |
| 777 | + "gpu_load": false, | |
| 778 | + "repeats": 3, | |
| 779 | + "budget_bytes": 4407421178, | |
| 780 | + "mb_per_s_mean": 1458.5310745058405, | |
| 781 | + "mb_per_s_median": 1459.9012697253913, | |
| 782 | + "mb_per_s_std": 3.743948118625381, | |
| 783 | + "iops_mean": 356086.6881117775 | |
| 784 | + }, | |
| 785 | + { | |
| 786 | + "block_bytes": 4096, | |
| 787 | + "pattern": "sequential", | |
| 788 | + "nocache": false, | |
| 789 | + "threads": 8, | |
| 790 | + "gpu_load": false, | |
| 791 | + "repeats": 3, | |
| 792 | + "budget_bytes": 2608882443, | |
| 793 | + "mb_per_s_mean": 869.924842253859, | |
| 794 | + "mb_per_s_median": 869.6260861814246, | |
| 795 | + "mb_per_s_std": 2.6503208536023317, | |
| 796 | + "iops_mean": 212383.99469088353 | |
| 797 | + }, | |
| 798 | + { | |
| 799 | + "block_bytes": 16384, | |
| 800 | + "pattern": "sequential", | |
| 801 | + "nocache": false, | |
| 802 | + "threads": 1, | |
| 803 | + "gpu_load": false, | |
| 804 | + "repeats": 3, | |
| 805 | + "budget_bytes": 8589934592, | |
| 806 | + "mb_per_s_mean": 21617.538531889717, | |
| 807 | + "mb_per_s_median": 21621.78809793309, | |
| 808 | + "mb_per_s_std": 59.802760544329175, | |
| 809 | + "iops_mean": 1319429.8420342845 | |
| 810 | + }, | |
| 811 | + { | |
| 812 | + "block_bytes": 16384, | |
| 813 | + "pattern": "sequential", | |
| 814 | + "nocache": false, | |
| 815 | + "threads": 4, | |
| 816 | + "gpu_load": false, | |
| 817 | + "repeats": 3, | |
| 818 | + "budget_bytes": 8589934592, | |
| 819 | + "mb_per_s_mean": 6491.621145125613, | |
| 820 | + "mb_per_s_median": 6494.489458206393, | |
| 821 | + "mb_per_s_std": 38.56700686897171, | |
| 822 | + "iops_mean": 396217.11090854567 | |
| 823 | + }, | |
| 824 | + { | |
| 825 | + "block_bytes": 16384, | |
| 826 | + "pattern": "sequential", | |
| 827 | + "nocache": false, | |
| 828 | + "threads": 8, | |
| 829 | + "gpu_load": false, | |
| 830 | + "repeats": 3, | |
| 831 | + "budget_bytes": 8589934592, | |
| 832 | + "mb_per_s_mean": 3326.3441359561525, | |
| 833 | + "mb_per_s_median": 3325.431237638123, | |
| 834 | + "mb_per_s_std": 3.050975584419479, | |
| 835 | + "iops_mean": 203023.934079355 | |
| 836 | + }, | |
| 837 | + { | |
| 838 | + "block_bytes": 65536, | |
| 839 | + "pattern": "sequential", | |
| 840 | + "nocache": false, | |
| 841 | + "threads": 1, | |
| 842 | + "gpu_load": false, | |
| 843 | + "repeats": 3, | |
| 844 | + "budget_bytes": 8589934592, | |
| 845 | + "mb_per_s_mean": 29957.432089618407, | |
| 846 | + "mb_per_s_median": 30189.466110754165, | |
| 847 | + "mb_per_s_std": 470.2460893755195, | |
| 848 | + "iops_mean": 457114.1371096559 | |
| 849 | + }, | |
| 850 | + { | |
| 851 | + "block_bytes": 65536, | |
| 852 | + "pattern": "sequential", | |
| 853 | + "nocache": false, | |
| 854 | + "threads": 4, | |
| 855 | + "gpu_load": false, | |
| 856 | + "repeats": 3, | |
| 857 | + "budget_bytes": 8589934592, | |
| 858 | + "mb_per_s_mean": 17569.562876093267, | |
| 859 | + "mb_per_s_median": 17615.89037433721, | |
| 860 | + "mb_per_s_std": 97.34448506842367, | |
| 861 | + "iops_mean": 268090.253846638 | |
| 862 | + }, | |
| 863 | + { | |
| 864 | + "block_bytes": 65536, | |
| 865 | + "pattern": "sequential", | |
| 866 | + "nocache": false, | |
| 867 | + "threads": 8, | |
| 868 | + "gpu_load": false, | |
| 869 | + "repeats": 3, | |
| 870 | + "budget_bytes": 8589934592, | |
| 871 | + "mb_per_s_mean": 12489.792139874966, | |
| 872 | + "mb_per_s_median": 12524.282391173198, | |
| 873 | + "mb_per_s_std": 69.33482171071729, | |
| 874 | + "iops_mean": 190579.1036968226 | |
| 875 | + }, | |
| 876 | + { | |
| 877 | + "block_bytes": 262144, | |
| 878 | + "pattern": "sequential", | |
| 879 | + "nocache": false, | |
| 880 | + "threads": 1, | |
| 881 | + "gpu_load": false, | |
| 882 | + "repeats": 3, | |
| 883 | + "budget_bytes": 8589934592, | |
| 884 | + "mb_per_s_mean": 39164.65820057082, | |
| 885 | + "mb_per_s_median": 39486.16075602976, | |
| 886 | + "mb_per_s_std": 601.0243079129414, | |
| 887 | + "iops_mean": 149401.31454685522 | |
| 888 | + }, | |
| 889 | + { | |
| 890 | + "block_bytes": 262144, | |
| 891 | + "pattern": "sequential", | |
| 892 | + "nocache": false, | |
| 893 | + "threads": 4, | |
| 894 | + "gpu_load": false, | |
| 895 | + "repeats": 3, | |
| 896 | + "budget_bytes": 8589934592, | |
| 897 | + "mb_per_s_mean": 76053.55966688057, | |
| 898 | + "mb_per_s_median": 76388.16056316164, | |
| 899 | + "mb_per_s_std": 590.3903823692293, | |
| 900 | + "iops_mean": 290121.30610229704 | |
| 901 | + }, | |
| 902 | + { | |
| 903 | + "block_bytes": 262144, | |
| 904 | + "pattern": "sequential", | |
| 905 | + "nocache": false, | |
| 906 | + "threads": 8, | |
| 907 | + "gpu_load": false, | |
| 908 | + "repeats": 3, | |
| 909 | + "budget_bytes": 8589934592, | |
| 910 | + "mb_per_s_mean": 45630.60576793564, | |
| 911 | + "mb_per_s_median": 45669.005339825126, | |
| 912 | + "mb_per_s_std": 166.5428067385005, | |
| 913 | + "iops_mean": 174066.94705175643 | |
| 914 | + }, | |
| 915 | + { | |
| 916 | + "block_bytes": 1048576, | |
| 917 | + "pattern": "sequential", | |
| 918 | + "nocache": false, | |
| 919 | + "threads": 1, | |
| 920 | + "gpu_load": false, | |
| 921 | + "repeats": 3, | |
| 922 | + "budget_bytes": 8589934592, | |
| 923 | + "mb_per_s_mean": 42748.724335353734, | |
| 924 | + "mb_per_s_median": 42961.4873690113, | |
| 925 | + "mb_per_s_std": 398.34237778641517, | |
| 926 | + "iops_mean": 40768.3604577577 | |
| 927 | + }, | |
| 928 | + { | |
| 929 | + "block_bytes": 1048576, | |
| 930 | + "pattern": "sequential", | |
| 931 | + "nocache": false, | |
| 932 | + "threads": 4, | |
| 933 | + "gpu_load": false, | |
| 934 | + "repeats": 3, | |
| 935 | + "budget_bytes": 8589934592, | |
| 936 | + "mb_per_s_mean": 115452.76372983315, | |
| 937 | + "mb_per_s_median": 115383.80137455142, | |
| 938 | + "mb_per_s_std": 360.25037919630205, | |
| 939 | + "iops_mean": 110104.33552726093 | |
| 940 | + }, | |
| 941 | + { | |
| 942 | + "block_bytes": 1048576, | |
| 943 | + "pattern": "sequential", | |
| 944 | + "nocache": false, | |
| 945 | + "threads": 8, | |
| 946 | + "gpu_load": false, | |
| 947 | + "repeats": 3, | |
| 948 | + "budget_bytes": 8589934592, | |
| 949 | + "mb_per_s_mean": 135141.67129088886, | |
| 950 | + "mb_per_s_median": 135055.84872295542, | |
| 951 | + "mb_per_s_std": 234.70894626348195, | |
| 952 | + "iops_mean": 128881.14098633657 | |
| 953 | + }, | |
| 954 | + { | |
| 955 | + "block_bytes": 4194304, | |
| 956 | + "pattern": "sequential", | |
| 957 | + "nocache": false, | |
| 958 | + "threads": 1, | |
| 959 | + "gpu_load": false, | |
| 960 | + "repeats": 3, | |
| 961 | + "budget_bytes": 8589934592, | |
| 962 | + "mb_per_s_mean": 41845.60382065829, | |
| 963 | + "mb_per_s_median": 41902.67348506605, | |
| 964 | + "mb_per_s_std": 222.59592874335817, | |
| 965 | + "iops_mean": 9976.769404568264 | |
| 966 | + }, | |
| 967 | + { | |
| 968 | + "block_bytes": 4194304, | |
| 969 | + "pattern": "sequential", | |
| 970 | + "nocache": false, | |
| 971 | + "threads": 4, | |
| 972 | + "gpu_load": false, | |
| 973 | + "repeats": 3, | |
| 974 | + "budget_bytes": 8589934592, | |
| 975 | + "mb_per_s_mean": 89032.16638537818, | |
| 976 | + "mb_per_s_median": 89044.04777215824, | |
| 977 | + "mb_per_s_std": 306.99513381891666, | |
| 978 | + "iops_mean": 21226.92260393576 | |
| 979 | + }, | |
| 980 | + { | |
| 981 | + "block_bytes": 4194304, | |
| 982 | + "pattern": "sequential", | |
| 983 | + "nocache": false, | |
| 984 | + "threads": 8, | |
| 985 | + "gpu_load": false, | |
| 986 | + "repeats": 3, | |
| 987 | + "budget_bytes": 8589934592, | |
| 988 | + "mb_per_s_mean": 102950.69245037381, | |
| 989 | + "mb_per_s_median": 103379.94695611407, | |
| 990 | + "mb_per_s_std": 1175.062630271669, | |
| 991 | + "iops_mean": 24545.35781154008 | |
| 992 | + }, | |
| 993 | + { | |
| 994 | + "block_bytes": 4096, | |
| 995 | + "pattern": "random", | |
| 996 | + "nocache": true, | |
| 997 | + "threads": 8, | |
| 998 | + "gpu_load": true, | |
| 999 | + "repeats": 3, | |
| 1000 | + "budget_bytes": 1340761293, | |
| 1001 | + "mb_per_s_mean": 478.35036355266675, | |
| 1002 | + "mb_per_s_median": 478.41375277377244, | |
| 1003 | + "mb_per_s_std": 1.0703311360714174, | |
| 1004 | + "iops_mean": 116784.75672672527 | |
| 1005 | + }, | |
| 1006 | + { | |
| 1007 | + "block_bytes": 16384, | |
| 1008 | + "pattern": "random", | |
| 1009 | + "nocache": true, | |
| 1010 | + "threads": 8, | |
| 1011 | + "gpu_load": true, | |
| 1012 | + "repeats": 3, | |
| 1013 | + "budget_bytes": 5442779396, | |
| 1014 | + "mb_per_s_mean": 1888.0411374899622, | |
| 1015 | + "mb_per_s_median": 1887.862781859016, | |
| 1016 | + "mb_per_s_std": 0.7370887499392919, | |
| 1017 | + "iops_mean": 115236.88583312757 | |
| 1018 | + }, | |
| 1019 | + { | |
| 1020 | + "block_bytes": 65536, | |
| 1021 | + "pattern": "random", | |
| 1022 | + "nocache": true, | |
| 1023 | + "threads": 8, | |
| 1024 | + "gpu_load": true, | |
| 1025 | + "repeats": 3, | |
| 1026 | + "budget_bytes": 8589934592, | |
| 1027 | + "mb_per_s_mean": 5351.261002842351, | |
| 1028 | + "mb_per_s_median": 5348.537195537803, | |
| 1029 | + "mb_per_s_std": 6.5762207021706836, | |
| 1030 | + "iops_mean": 81653.76286075366 | |
| 1031 | + }, | |
| 1032 | + { | |
| 1033 | + "block_bytes": 262144, | |
| 1034 | + "pattern": "random", | |
| 1035 | + "nocache": true, | |
| 1036 | + "threads": 8, | |
| 1037 | + "gpu_load": true, | |
| 1038 | + "repeats": 3, | |
| 1039 | + "budget_bytes": 8589934592, | |
| 1040 | + "mb_per_s_mean": 11608.997808656672, | |
| 1041 | + "mb_per_s_median": 11620.467878177564, | |
| 1042 | + "mb_per_s_std": 26.761948567871453, | |
| 1043 | + "iops_mean": 44284.81219732922 | |
| 1044 | + }, | |
| 1045 | + { | |
| 1046 | + "block_bytes": 1048576, | |
| 1047 | + "pattern": "random", | |
| 1048 | + "nocache": true, | |
| 1049 | + "threads": 8, | |
| 1050 | + "gpu_load": true, | |
| 1051 | + "repeats": 3, | |
| 1052 | + "budget_bytes": 8589934592, | |
| 1053 | + "mb_per_s_mean": 13393.447895148627, | |
| 1054 | + "mb_per_s_median": 13397.359252768547, | |
| 1055 | + "mb_per_s_std": 10.659178098289324, | |
| 1056 | + "iops_mean": 12772.987265728594 | |
| 1057 | + }, | |
| 1058 | + { | |
| 1059 | + "block_bytes": 4194304, | |
| 1060 | + "pattern": "random", | |
| 1061 | + "nocache": true, | |
| 1062 | + "threads": 8, | |
| 1063 | + "gpu_load": true, | |
| 1064 | + "repeats": 3, | |
| 1065 | + "budget_bytes": 8589934592, | |
| 1066 | + "mb_per_s_mean": 30955.750339884242, | |
| 1067 | + "mb_per_s_median": 31923.229778403995, | |
| 1068 | + "mb_per_s_std": 12129.557206669355, | |
| 1069 | + "iops_mean": 7380.426011057912 | |
| 1070 | + } | |
| 1071 | + ] | |
| 1072 | +} | |
| \ No newline at end of file | ||
| 1073 | ||