spb/forge Public MIT
Forge — LLM training from scratch in pure C++20 + Metal on Apple Silicon.
C++ 61.2%
C 23%
Python 7.6%
TeX 7.2%
CMake 1.1%
1{2 "_comment": "batch_size is the MICRO-batch (8 x 8 x 1024 = 65536 tokens/step); the autograd tape holds all activations. precision is parsed but not yet honored - all kernels are f32 (see README \"Not yet done\").",3 "model": {4 "name": "gpt-25m",5 "n_layers": 8,6 "d_model": 512,7 "n_heads": 8,8 "n_kv_heads": 8,9 "d_ff": 1408,10 "vocab_size": 8192,11 "context_length": 1024,12 "tied_embeddings": true,13 "use_rope": true,14 "rope_theta": 10000.0,15 "norm": "rmsnorm",16 "norm_eps": 1e-06,17 "activation": "swiglu",18 "dropout": 0.019 },20 "train": {21 "lr": 0.0006,22 "min_lr_ratio": 0.1,23 "warmup_steps": 2000,24 "max_steps": 50000,25 "beta1": 0.9,26 "beta2": 0.95,27 "eps": 1e-08,28 "weight_decay": 0.1,29 "grad_clip": 1.0,30 "batch_size": 8,31 "grad_accum_steps": 8,32 "precision": "f32",33 "checkpoint_every": 1000,34 "eval_every": 500,35 "eval_batches": 20,36 "seed": 133737 }38}39