ArushBuilds commited on
Commit
030d93a
·
1 Parent(s): 94f7e97

Update config.py

Browse files
Files changed (1) hide show
  1. config.py +33 -95
config.py CHANGED
@@ -4,16 +4,15 @@ xla_on_gpu=False
4
  # ---------------------------------------------------------------------------
5
  # Core shape -- chosen for GPU efficiency, not just "a number that fits"
6
  # ---------------------------------------------------------------------------
7
- CONTEXT = 512
8
- max_gen_tokens = 512
9
- yarn_scale = 1.0 # neutralized YaRN extension ramp for clean RoPE sanity checks
10
  gen_headroom = 128
11
- vocab_size = 8192+4096
12
 
13
- # Missing config fields expected by the shared GPTConfig dataclass and the
14
- # model factory path. Keeping them explicit here removes the fallback-warning
15
- # cascade shown in the traceback and stabilizes object construction.
16
  bias = False
 
 
 
17
 
18
 
19
 
@@ -21,120 +20,59 @@ gradient_checkpointing = False
21
  num_kv_heads = 4
22
  pattern = "dense"
23
  sliding_window_size = 512
24
- # tensor_parallel_size is the knob the config dataclass reads.
25
  tensor_parallel_size = 1
26
 
27
- # YaRN config defaults from the model constructor.
28
 
29
  mod_capacity = 0.25
30
- use_mtp = False # set True to activate
31
- mtp_depth = 1 # 1 = predict 2 tokens ahead; 2 = also predict 3 ahead
32
- mtp_lambda = 0.3
33
  use_nvfp4 = False
34
  use_int8 = False
35
  int8_group_size = 128
36
- mod_gate_entropy_coeff = 0.01
37
- # Experimental hardware-specific paths. These are deliberately opt-in and
38
- # are read live by model.py/prefetch.py so post-import CLI overrides work.
39
 
40
- numberoflayers = 16 # 4 complete pyramid_swa groups (4x4) -- no odd
41
- # trailing layer forced into an incomplete group
42
  numberofheads = 8
43
- D_MODEL = 384 # multiple of 64 -> head_dim = 192/12 = 16, a clean
44
- # Tensor Core tile-friendly size (fp16 wants
45
- # multiples of 16; 16 divides evenly with no
46
- # padding waste on T4/A100 GEMM tiling)
47
- # AdamW-scale LR for a model this size; only
48
- # takes effect if use_static_lr=False below
49
- ALPHA_COMPILE_MODE="reduce-overhead"
50
- # ---------------------------------------------------------------------------
51
- # Attention
52
- # ---------------------------------------------------------------------------
53
- num_kv_heads = 4 # 2x GQA compression on global layers (8->4 kv
54
- # heads); pyramid_swa steps kv heads [1,2,4]
55
- # across each group's 3 SWA layers (MQA->GQA),
56
- # verified against the real build_attention_layers
57
- # logic in model.py, not assumed
58
-
59
- sliding_window_size = 512 # explicit (was previously left to fall back to
60
- # the default with a warning) -- 1/4 of CONTEXT,
61
- # a reasonable local-attention window for this
62
- # sequence length
63
  use_sdpa=False
64
  optimizer_8bit = False
65
  use_wsd = True
66
- wsd_decay_steps = 700 # or None for auto (10% of MAX_STEPS)
67
- # ---------------------------------------------------------------------------
68
- # FFN / MoE
69
- # ---------------------------------------------------------------------------
70
- ffn_mult = 8 // 3 # SwiGLU hidden_dim multiplier. model.py rounds
71
- # d_model*ffn_mult UP to the nearest multiple of
72
- # 64 (256 * 8/3 = 682.67 -> 704), avoiding the
73
- # prime-dimension GEMM padding waste a plain
74
- # round() would produce (683 is prime).
75
- use_moe = False # plain dense SwiGLU_FFN every layer. AryaSparseMoE
76
- # currently has NO auxiliary load-balancing loss
77
- # (expert_bias is trainable but nothing pushes it
78
- # toward balanced routing) -- leave this False
79
- # unless/until that's added, especially at this
80
- # small a scale where routing collapse compounds
81
- # fastest across depth.
82
  n_experts = 8
83
  n_shared = 1
84
  # T4 path
85
 
86
- # (use_int8 ignored on SM100+ -- nvFP4 takes priority)
87
- # ---------------------------------------------------------------------------
88
- # Precision / batching
89
- # ---------------------------------------------------------------------------
90
- PRECISION = 'fp16' # T4 has no bf16 Tensor Cores -- fp16 is correct
91
- # here. On A100+, bf16 is the better choice (same
92
- # speed, wider dynamic range, no GradScaler
93
- # needed) -- change this if/when you move hardware.
94
  GLOBAL_DTYPE = PRECISION
95
  USE_ASYNC_LAYER_PREFETCH = True
96
  intra_op_threads = 2
 
 
97
  DISABLE_MMAP = False
98
- use_flash_outproj_add_rmsnorm = False
99
- use_flash_rope = False
100
- use_flash_swiglu = False
101
- use_fused_add_rmsnorm = False
102
  use_xsa = False
103
- batch_size = 64 # starting point, not a measured optimum -- train.py
104
- # now logs `vram X/Y GB` on every step (added
105
- # recently). Watch that number on your first run
106
- # and raise batch_size toward your actual VRAM
107
- # headroom; this value has NOT been validated
108
- # against a real GPU run for this exact config.
109
- weight_decay = 0.01 # train.py now splits params into decay/no-decay
110
- # groups automatically (biases, RMSNorm gains,
111
- # expert_bias get weight_decay=0.0 regardless of
112
- # this value) -- this number only applies to
113
- # actual weight matrices + the tied embedding.
114
-
115
- # ---------------------------------------------------------------------------
116
- # Schedule
117
- # ---------------------------------------------------------------------------
118
- # Defaults to warmup+cosine (NOT static) for this fresh config -- unlike
119
- # your existing config.py, this file has no prior "it works, don't touch it"
120
- # history behind a flat LR, so there's no reason to ship it with the known
121
- # no-warmup risk baked in. WARMUP_STEPS=500 and MAX_STEPS=20000 are
122
- # hardcoded in train.py's main() (not read from config.py at all) -- change
123
- # them there directly if this run needs a different horizon.
124
  use_static_lr = False
125
- static_lr = 1e-3 # only used if you flip use_static_lr back to True
126
  learning_rate = 1e-3
127
- # ---------------------------------------------------------------------------
128
- # Parallelism / misc
129
- # ---------------------------------------------------------------------------
130
 
131
  max_steps = 4000
132
  data_parallel_only = True
133
- dropout = 0.0 # explicit (was previously left to the
134
- # default with a warning)
135
  use_muon = True
136
- # ---------------------------------------------------------------------------
137
- # Data
138
- # ---------------------------------------------------------------------------
139
  DATA_PATH = './training_data'
140
  data_path = DATA_PATH
 
 
 
4
  # ---------------------------------------------------------------------------
5
  # Core shape -- chosen for GPU efficiency, not just "a number that fits"
6
  # ---------------------------------------------------------------------------
7
+ CONTEXT = 1024
8
+ max_gen_tokens = 1024
 
9
  gen_headroom = 128
10
+ vocab_size = 16384+8192
11
 
 
 
 
12
  bias = False
13
+ d_rope = None
14
+ top_k = 64
15
+ moe_top_k = 2
16
 
17
 
18
 
 
20
  num_kv_heads = 4
21
  pattern = "dense"
22
  sliding_window_size = 512
23
+
24
  tensor_parallel_size = 1
25
 
26
+
27
 
28
  mod_capacity = 0.25
29
+ use_mtp = False
30
+ mtp_depth = 1
31
+ mtp_lambda = 0.1
32
  use_nvfp4 = False
33
  use_int8 = False
34
  int8_group_size = 128
35
+ mod_budget_coeff = 0.05
36
+ mod_gate_entropy_coeff = 0.0
37
+ mod_hard_eval = False
38
 
39
+ numberoflayers = 16
 
40
  numberofheads = 8
41
+ D_MODEL = 384
42
+ ALPHA_COMPILE_MODE="default"
43
+ num_kv_heads = 4
44
+ sliding_window_size = 512
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
45
  use_sdpa=False
46
  optimizer_8bit = False
47
  use_wsd = True
48
+ wsd_decay_steps = 700
49
+ ffn_mult = 8 // 3
50
+ use_moe = False
 
 
 
 
 
 
 
 
 
 
 
 
 
51
  n_experts = 8
52
  n_shared = 1
53
  # T4 path
54
 
55
+ PRECISION = 'fp16'
 
 
 
 
 
 
 
56
  GLOBAL_DTYPE = PRECISION
57
  USE_ASYNC_LAYER_PREFETCH = True
58
  intra_op_threads = 2
59
+ num_workers = 2
60
+ GPU_SIDE_BATCHING = False
61
  DISABLE_MMAP = False
62
+ use_flash_ops = False
 
 
 
63
  use_xsa = False
64
+ batch_size = 32
65
+ weight_decay = 0.01
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
66
  use_static_lr = False
67
+ static_lr = 1e-3
68
  learning_rate = 1e-3
 
 
 
69
 
70
  max_steps = 4000
71
  data_parallel_only = True
72
+ use_fsdp = False
73
+ dropout = 0.0
74
  use_muon = True
 
 
 
75
  DATA_PATH = './training_data'
76
  data_path = DATA_PATH
77
+ TOKENIZER_MODEL_PATH = './tokenizer'
78
+ use_streaming_batcher = False