Arush kumar commited on
Commit
cf1c5a3
Β·
1 Parent(s): bcf40e7

Update config.py

Browse files
Files changed (1) hide show
  1. config.py +6 -6
config.py CHANGED
@@ -7,12 +7,12 @@
7
 
8
  # ── Architecture ──────────────────────────────────────────────────────────────
9
  CONTEXT = 1024
10
- vocab_size = 6000
11
- D_MODEL = 256+128
12
- numberoflayers = 10
13
  numberofheads = 8
14
  d_Latent = 64
15
- ffn_mult = 3.5
16
  swa_window = 512 # SWA: attend to last 512 tokens (decoder-only, no loss of info at T4 scale)
17
  num_kv_heads = 2 # 4Γ— KV compression via GQA (8 heads / 2 KV heads = 4 query groups)
18
 
@@ -33,10 +33,10 @@ PRECISION = 'bf16' # BF16 for T4 (mixed_bfloat16 policy) β€” no loss scal
33
  GLOBAL_DTYPE = PRECISION # global dtype for model weights, activations, and optimizer
34
 
35
  # ── Data Shuffling ────────────────────────────────────────────────────────────
36
- shuffle_samples = True # shuffle individual samples/lines before tokenization (avoid abrupt domain transitions)
37
  shuffle_seed = 42 # seed for reproducible shuffling (set to different values for different shuffles)
38
  shuffle_buffer_size = 10_000 # number of window indices to hold in shuffle buffer (larger = more randomness, higher memory)
39
  # ── Checkpointing & Inference ────────────────────────────────────────────────
40
  USE_REMAT = False # gradient checkpointing β€” saves ~30% VRAM
41
- MAX_GEN_TOKENS = 256 # RoPE headroom: context + this = table size
42
  LR = 3e-4 # if None, will be auto-scaled based on batch size and warmup
 
7
 
8
  # ── Architecture ──────────────────────────────────────────────────────────────
9
  CONTEXT = 1024
10
+ vocab_size = 24000
11
+ D_MODEL = 256+128
12
+ numberoflayers = 9
13
  numberofheads = 8
14
  d_Latent = 64
15
+ ffn_mult = 2.75
16
  swa_window = 512 # SWA: attend to last 512 tokens (decoder-only, no loss of info at T4 scale)
17
  num_kv_heads = 2 # 4Γ— KV compression via GQA (8 heads / 2 KV heads = 4 query groups)
18
 
 
33
  GLOBAL_DTYPE = PRECISION # global dtype for model weights, activations, and optimizer
34
 
35
  # ── Data Shuffling ────────────────────────────────────────────────────────────
36
+ shuffle_samples = True # shuffle individual samples/lines before tokenization (avoid abrupt domain transitions)
37
  shuffle_seed = 42 # seed for reproducible shuffling (set to different values for different shuffles)
38
  shuffle_buffer_size = 10_000 # number of window indices to hold in shuffle buffer (larger = more randomness, higher memory)
39
  # ── Checkpointing & Inference ────────────────────────────────────────────────
40
  USE_REMAT = False # gradient checkpointing β€” saves ~30% VRAM
41
+ MAX_GEN_TOKENS = 256 # RoPE headroom: context + this = table size
42
  LR = 3e-4 # if None, will be auto-scaled based on batch size and warmup