SPB Git

spb/forge Public MIT

Forge — LLM training from scratch in pure C++20 + Metal on Apple Silicon.

C++ 61.2% C 23% Python 7.6% TeX 7.2% CMake 1.1%
1.1 KB · 39 lines json
Raw Blame History
1{2  "_comment": "205.6M params, ONE EPOCH over the full TinyStories corpus on M3U96a (M3 Ultra, 60-core GPU, 96 GB). 8 x 32 x 1024 = 262144 tokens/step x 1553 steps = 407.1M of 407,344,713 tokens (99.94%). Micro-batch 8 halves activation memory vs 16; grad_accum 32 keeps the effective batch (and so the 3e-4 LR) unchanged. precision is parsed but not yet honored - all kernels are f32.",3  "model": {4    "name": "gpt-200m-cluster",5    "n_layers": 16,6    "d_model": 1024,7    "n_heads": 16,8    "n_kv_heads": 8,9    "d_ff": 3072,10    "vocab_size": 4096,11    "context_length": 1024,12    "tied_embeddings": true,13    "use_rope": true,14    "rope_theta": 10000.0,15    "norm": "rmsnorm",16    "norm_eps": 1e-06,17    "activation": "swiglu",18    "dropout": 0.019  },20  "train": {21    "lr": 0.0003,22    "min_lr_ratio": 0.1,23    "warmup_steps": 155,24    "max_steps": 1553,25    "beta1": 0.9,26    "beta2": 0.95,27    "eps": 1e-08,28    "weight_decay": 0.1,29    "grad_clip": 1.0,30    "batch_size": 8,31    "grad_accum_steps": 32,32    "precision": "f32",33    "checkpoint_every": 100,34    "eval_every": 100,35    "eval_batches": 10,36    "seed": 133737  }38}39