File size: 1,813 Bytes
5629a1a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
{
 "_comment": "Morena-1.5B target: 28L x 2048d, GQA 16/4 (head_dim 128), SwiGLU 6144, tied 64k embeddings, RoPE 5e5, seq 4096 -> 1.48B params (1.35B non-embedding). GLOBAL BATCH = global_batch_seqs 1024 x 4096 = 4.19M tokens/step, independent of node count: train.py sets grad_accum = round(1024 / (micro_batch * n_gpus)). 64 GPUs: accum 4 (exact). 48 GPUs: 5.33 -> 5 (3.9M). 40 GPUs: 6.4 -> 6 (3.9M). 12 GPUs: 21.3 -> 21 (4.1M). Keep the realized global batch constant across resumes (log line [data] prints it). 250B tokens = ~60k steps. fsdp_shard_size 4 = HSDP: parameters sharded inside each 4-GPU node (NVLink all-gathers), gradients all-reduced across nodes -- per-GPU state for 1.48B fp32 master+grad+Muon momentum+Adam (embed only) ~ 19 GB/4 = 5 GB, fits 64 GB with act_ckpt on. Set fsdp_shard_size 0 for full FSDP across all GPUs (lower memory, more inter-node traffic). LR 1e-3 shared (Moonlight scaling), warmup 2k, WSD decay over the last ~10% -- launch the anneal explicitly with --anneal from any stable checkpoint.",
 "model": {
  "vocab_size": 65536,
  "n_layer": 28,
  "d_model": 2048,
  "n_head": 16,
  "n_kv_head": 4,
  "d_ff": 6144,
  "rope_theta": 500000.0,
  "norm_eps": 1e-05,
  "tie_embeddings": true,
  "init_std": 0.02
 },
 "train": {
  "seq_len": 4096,
  "micro_batch": 4,
  "grad_accum": 4,
  "lr": 0.001,
  "min_lr_ratio": 0.1,
  "weight_decay": 0.1,
  "muon_momentum": 0.95,
  "muon_ns_steps": 5,
  "muon_ns_mode": "roundrobin",
  "adam_betas": [
   0.9,
   0.95
  ],
  "adam_eps": 1e-08,
  "grad_clip": 1.0,
  "warmup_steps": 2000,
  "total_steps": 60000,
  "decay_steps": 6000,
  "decay_start": -1,
  "decay_shape": "1-sqrt",
  "attn": "auto",
  "act_ckpt": true,
  "compile": false,
  "seed": 1234,
  "global_batch_seqs": 1024,
  "fsdp_shard_size": 4,
  "optimizer": "muon"
 }
}