Spaces:
Running
Running
Commit ·
030d93a
1
Parent(s): 94f7e97
Update config.py
Browse files
config.py
CHANGED
|
@@ -4,16 +4,15 @@ xla_on_gpu=False
|
|
| 4 |
# ---------------------------------------------------------------------------
|
| 5 |
# Core shape -- chosen for GPU efficiency, not just "a number that fits"
|
| 6 |
# ---------------------------------------------------------------------------
|
| 7 |
-
CONTEXT =
|
| 8 |
-
max_gen_tokens =
|
| 9 |
-
yarn_scale = 1.0 # neutralized YaRN extension ramp for clean RoPE sanity checks
|
| 10 |
gen_headroom = 128
|
| 11 |
-
vocab_size =
|
| 12 |
|
| 13 |
-
# Missing config fields expected by the shared GPTConfig dataclass and the
|
| 14 |
-
# model factory path. Keeping them explicit here removes the fallback-warning
|
| 15 |
-
# cascade shown in the traceback and stabilizes object construction.
|
| 16 |
bias = False
|
|
|
|
|
|
|
|
|
|
| 17 |
|
| 18 |
|
| 19 |
|
|
@@ -21,120 +20,59 @@ gradient_checkpointing = False
|
|
| 21 |
num_kv_heads = 4
|
| 22 |
pattern = "dense"
|
| 23 |
sliding_window_size = 512
|
| 24 |
-
|
| 25 |
tensor_parallel_size = 1
|
| 26 |
|
| 27 |
-
|
| 28 |
|
| 29 |
mod_capacity = 0.25
|
| 30 |
-
use_mtp = False
|
| 31 |
-
mtp_depth = 1
|
| 32 |
-
mtp_lambda = 0.
|
| 33 |
use_nvfp4 = False
|
| 34 |
use_int8 = False
|
| 35 |
int8_group_size = 128
|
| 36 |
-
|
| 37 |
-
|
| 38 |
-
|
| 39 |
|
| 40 |
-
numberoflayers = 16
|
| 41 |
-
# trailing layer forced into an incomplete group
|
| 42 |
numberofheads = 8
|
| 43 |
-
D_MODEL = 384
|
| 44 |
-
|
| 45 |
-
|
| 46 |
-
|
| 47 |
-
# AdamW-scale LR for a model this size; only
|
| 48 |
-
# takes effect if use_static_lr=False below
|
| 49 |
-
ALPHA_COMPILE_MODE="reduce-overhead"
|
| 50 |
-
# ---------------------------------------------------------------------------
|
| 51 |
-
# Attention
|
| 52 |
-
# ---------------------------------------------------------------------------
|
| 53 |
-
num_kv_heads = 4 # 2x GQA compression on global layers (8->4 kv
|
| 54 |
-
# heads); pyramid_swa steps kv heads [1,2,4]
|
| 55 |
-
# across each group's 3 SWA layers (MQA->GQA),
|
| 56 |
-
# verified against the real build_attention_layers
|
| 57 |
-
# logic in model.py, not assumed
|
| 58 |
-
|
| 59 |
-
sliding_window_size = 512 # explicit (was previously left to fall back to
|
| 60 |
-
# the default with a warning) -- 1/4 of CONTEXT,
|
| 61 |
-
# a reasonable local-attention window for this
|
| 62 |
-
# sequence length
|
| 63 |
use_sdpa=False
|
| 64 |
optimizer_8bit = False
|
| 65 |
use_wsd = True
|
| 66 |
-
wsd_decay_steps = 700
|
| 67 |
-
|
| 68 |
-
|
| 69 |
-
# ---------------------------------------------------------------------------
|
| 70 |
-
ffn_mult = 8 // 3 # SwiGLU hidden_dim multiplier. model.py rounds
|
| 71 |
-
# d_model*ffn_mult UP to the nearest multiple of
|
| 72 |
-
# 64 (256 * 8/3 = 682.67 -> 704), avoiding the
|
| 73 |
-
# prime-dimension GEMM padding waste a plain
|
| 74 |
-
# round() would produce (683 is prime).
|
| 75 |
-
use_moe = False # plain dense SwiGLU_FFN every layer. AryaSparseMoE
|
| 76 |
-
# currently has NO auxiliary load-balancing loss
|
| 77 |
-
# (expert_bias is trainable but nothing pushes it
|
| 78 |
-
# toward balanced routing) -- leave this False
|
| 79 |
-
# unless/until that's added, especially at this
|
| 80 |
-
# small a scale where routing collapse compounds
|
| 81 |
-
# fastest across depth.
|
| 82 |
n_experts = 8
|
| 83 |
n_shared = 1
|
| 84 |
# T4 path
|
| 85 |
|
| 86 |
-
|
| 87 |
-
# ---------------------------------------------------------------------------
|
| 88 |
-
# Precision / batching
|
| 89 |
-
# ---------------------------------------------------------------------------
|
| 90 |
-
PRECISION = 'fp16' # T4 has no bf16 Tensor Cores -- fp16 is correct
|
| 91 |
-
# here. On A100+, bf16 is the better choice (same
|
| 92 |
-
# speed, wider dynamic range, no GradScaler
|
| 93 |
-
# needed) -- change this if/when you move hardware.
|
| 94 |
GLOBAL_DTYPE = PRECISION
|
| 95 |
USE_ASYNC_LAYER_PREFETCH = True
|
| 96 |
intra_op_threads = 2
|
|
|
|
|
|
|
| 97 |
DISABLE_MMAP = False
|
| 98 |
-
|
| 99 |
-
use_flash_rope = False
|
| 100 |
-
use_flash_swiglu = False
|
| 101 |
-
use_fused_add_rmsnorm = False
|
| 102 |
use_xsa = False
|
| 103 |
-
batch_size =
|
| 104 |
-
|
| 105 |
-
# recently). Watch that number on your first run
|
| 106 |
-
# and raise batch_size toward your actual VRAM
|
| 107 |
-
# headroom; this value has NOT been validated
|
| 108 |
-
# against a real GPU run for this exact config.
|
| 109 |
-
weight_decay = 0.01 # train.py now splits params into decay/no-decay
|
| 110 |
-
# groups automatically (biases, RMSNorm gains,
|
| 111 |
-
# expert_bias get weight_decay=0.0 regardless of
|
| 112 |
-
# this value) -- this number only applies to
|
| 113 |
-
# actual weight matrices + the tied embedding.
|
| 114 |
-
|
| 115 |
-
# ---------------------------------------------------------------------------
|
| 116 |
-
# Schedule
|
| 117 |
-
# ---------------------------------------------------------------------------
|
| 118 |
-
# Defaults to warmup+cosine (NOT static) for this fresh config -- unlike
|
| 119 |
-
# your existing config.py, this file has no prior "it works, don't touch it"
|
| 120 |
-
# history behind a flat LR, so there's no reason to ship it with the known
|
| 121 |
-
# no-warmup risk baked in. WARMUP_STEPS=500 and MAX_STEPS=20000 are
|
| 122 |
-
# hardcoded in train.py's main() (not read from config.py at all) -- change
|
| 123 |
-
# them there directly if this run needs a different horizon.
|
| 124 |
use_static_lr = False
|
| 125 |
-
static_lr = 1e-3
|
| 126 |
learning_rate = 1e-3
|
| 127 |
-
# ---------------------------------------------------------------------------
|
| 128 |
-
# Parallelism / misc
|
| 129 |
-
# ---------------------------------------------------------------------------
|
| 130 |
|
| 131 |
max_steps = 4000
|
| 132 |
data_parallel_only = True
|
| 133 |
-
|
| 134 |
-
|
| 135 |
use_muon = True
|
| 136 |
-
# ---------------------------------------------------------------------------
|
| 137 |
-
# Data
|
| 138 |
-
# ---------------------------------------------------------------------------
|
| 139 |
DATA_PATH = './training_data'
|
| 140 |
data_path = DATA_PATH
|
|
|
|
|
|
|
|
|
| 4 |
# ---------------------------------------------------------------------------
|
| 5 |
# Core shape -- chosen for GPU efficiency, not just "a number that fits"
|
| 6 |
# ---------------------------------------------------------------------------
|
| 7 |
+
CONTEXT = 1024
|
| 8 |
+
max_gen_tokens = 1024
|
|
|
|
| 9 |
gen_headroom = 128
|
| 10 |
+
vocab_size = 16384+8192
|
| 11 |
|
|
|
|
|
|
|
|
|
|
| 12 |
bias = False
|
| 13 |
+
d_rope = None
|
| 14 |
+
top_k = 64
|
| 15 |
+
moe_top_k = 2
|
| 16 |
|
| 17 |
|
| 18 |
|
|
|
|
| 20 |
num_kv_heads = 4
|
| 21 |
pattern = "dense"
|
| 22 |
sliding_window_size = 512
|
| 23 |
+
|
| 24 |
tensor_parallel_size = 1
|
| 25 |
|
| 26 |
+
|
| 27 |
|
| 28 |
mod_capacity = 0.25
|
| 29 |
+
use_mtp = False
|
| 30 |
+
mtp_depth = 1
|
| 31 |
+
mtp_lambda = 0.1
|
| 32 |
use_nvfp4 = False
|
| 33 |
use_int8 = False
|
| 34 |
int8_group_size = 128
|
| 35 |
+
mod_budget_coeff = 0.05
|
| 36 |
+
mod_gate_entropy_coeff = 0.0
|
| 37 |
+
mod_hard_eval = False
|
| 38 |
|
| 39 |
+
numberoflayers = 16
|
|
|
|
| 40 |
numberofheads = 8
|
| 41 |
+
D_MODEL = 384
|
| 42 |
+
ALPHA_COMPILE_MODE="default"
|
| 43 |
+
num_kv_heads = 4
|
| 44 |
+
sliding_window_size = 512
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
use_sdpa=False
|
| 46 |
optimizer_8bit = False
|
| 47 |
use_wsd = True
|
| 48 |
+
wsd_decay_steps = 700
|
| 49 |
+
ffn_mult = 8 // 3
|
| 50 |
+
use_moe = False
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 51 |
n_experts = 8
|
| 52 |
n_shared = 1
|
| 53 |
# T4 path
|
| 54 |
|
| 55 |
+
PRECISION = 'fp16'
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
GLOBAL_DTYPE = PRECISION
|
| 57 |
USE_ASYNC_LAYER_PREFETCH = True
|
| 58 |
intra_op_threads = 2
|
| 59 |
+
num_workers = 2
|
| 60 |
+
GPU_SIDE_BATCHING = False
|
| 61 |
DISABLE_MMAP = False
|
| 62 |
+
use_flash_ops = False
|
|
|
|
|
|
|
|
|
|
| 63 |
use_xsa = False
|
| 64 |
+
batch_size = 32
|
| 65 |
+
weight_decay = 0.01
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 66 |
use_static_lr = False
|
| 67 |
+
static_lr = 1e-3
|
| 68 |
learning_rate = 1e-3
|
|
|
|
|
|
|
|
|
|
| 69 |
|
| 70 |
max_steps = 4000
|
| 71 |
data_parallel_only = True
|
| 72 |
+
use_fsdp = False
|
| 73 |
+
dropout = 0.0
|
| 74 |
use_muon = True
|
|
|
|
|
|
|
|
|
|
| 75 |
DATA_PATH = './training_data'
|
| 76 |
data_path = DATA_PATH
|
| 77 |
+
TOKENIZER_MODEL_PATH = './tokenizer'
|
| 78 |
+
use_streaming_batcher = False
|