Spaces:
Sleeping
Sleeping
Download config.py from ArushBuilds/Pragya: direct link, hf CLI and curl.
- Browser
- Download file 3.69 kB
-
https://huggingface.co/spaces/ArushBuilds/Pragya/resolve/main/config.py
- Command line
-
hf download hf://spaces/ArushBuilds/Pragya/config.py
-
curl -L -o config.py https://huggingface.co/spaces/ArushBuilds/Pragya/resolve/main/config.py
3.69 kB
| use_liger=True | |
| xla_on_gpu=False | |
| # --------------------------------------------------------------------------- | |
| # Core shape -- chosen for GPU efficiency, not just "a number that fits" | |
| # --------------------------------------------------------------------------- | |
| CONTEXT = 1024 | |
| max_gen_tokens = 1024 | |
| gen_headroom = 128 | |
| vocab_size = 16384 | |
| bias = False | |
| d_rope = None | |
| top_k = 64 | |
| moe_top_k = 2 | |
| gradient_checkpointing = False | |
| num_kv_heads = 4 | |
| pattern = "dense" | |
| sliding_window_size = 512 | |
| tensor_parallel_size = 1 | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Curriculum learning (token-level mixing across sources) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Master switch. False = old behavior (uniform random over the whole | |
| # flat token array). True = per-source anchor schedule below. | |
| CURRICULUM_ENABLED = False | |
| # Number of training steps the anchor schedule spans. Set equal to | |
| # MAX_STEPS in train.py so the ramp completes at run end. | |
| CURRICULUM_TOTAL_STEPS = 20000 | |
| # Sources, in curriculum order (easiest first). Each entry: | |
| # (name, [glob_patterns], w@0, w@T/3, w@2T/3, w@T) | |
| # Between anchors, weights interpolate linearly. Columns should sum | |
| # to ~1.0 (the code renormalizes if not). | |
| # Files matching none of the globs are EXCLUDED from training. | |
| CURRICULUM_SOURCES = [ | |
| ("tinystories", ["tinystories*.txt", "tiny_*.txt"], | |
| 0.70, 0.10, 0.03, 0.02), | |
| ("simplewiki", ["simplewiki*.txt", "wiki_*.txt"], | |
| 0.10, 0.70, 0.10, 0.03), | |
| ("cosmopedia", ["cosmopedia*.txt", "cosmo_*.txt"], | |
| 0.10, 0.10, 0.70, 0.10), | |
| ("finephrase", ["finephrase*.txt", "fine_*.txt"], | |
| 0.10, 0.10, 0.17, 0.85), | |
| ] | |
| # Hard floor: no source ever drops below this fraction at any step. | |
| # 0.0 disables. 0.02 = 2% minimum, prevents catastrophic forgetting. | |
| CURRICULUM_FLOOR = 0.02 | |
| mod_capacity = 0.25 | |
| use_mtp = False | |
| mtp_depth = 1 | |
| mtp_lambda = 0.1 | |
| use_nvfp4 = False | |
| use_int8 = False | |
| int8_group_size = 128 | |
| mod_budget_coeff = 0.05 | |
| mod_gate_entropy_coeff = 0.0 | |
| mod_hard_eval = False | |
| use_flexattention = True | |
| numberoflayers = 16 | |
| numberofheads = 8 | |
| D_MODEL = 384 | |
| ALPHA_COMPILE_MODE="reduce-overhead" | |
| num_kv_heads = 4 | |
| sliding_window_size = 512 | |
| use_sdpa=False | |
| optimizer_8bit = True | |
| use_wsd = True | |
| wsd_decay_steps = 700 | |
| ffn_mult = 8 // 3 | |
| use_moe = False | |
| n_experts = 8 | |
| n_shared = 1 | |
| # T4 path | |
| PRECISION = 'bf16' | |
| GLOBAL_DTYPE = PRECISION | |
| USE_ASYNC_LAYER_PREFETCH = True | |
| intra_op_threads = 2 | |
| num_workers = 2 | |
| GPU_SIDE_BATCHING = True | |
| DISABLE_MMAP = False | |
| use_flash_ops = False | |
| use_xsa = True | |
| batch_size = 256 | |
| weight_decay = 0.01 | |
| use_static_lr = False | |
| static_lr = 1e-3 | |
| learning_rate = 1e-3 | |
| use_qk_norm = True | |
| # Gated Attention (Qiu et al., NeurIPS 2025 Best Paper, arXiv:2505.06708) | |
| # Per-head sigmoid gate on the SDPA output, before out_proj. | |
| # Gate reads the first `attn_gate_window` dims of the post-norm block input. | |
| # 0 = full n_embd (dense gate). 12-32 is the speedrun-recommended range. | |
| use_attn_gate = True | |
| attn_gate_window = 16 | |
| max_steps = 4000 | |
| data_parallel_only = True | |
| use_fsdp = False | |
| dropout = 0.0 | |
| use_muon = True | |
| DATA_PATH = './training_data' | |
| data_path = DATA_PATH | |
| TOKENIZER_MODEL_PATH = './tokenizer' | |
| use_streaming_batcher = False | |