| !!python/object/new:configs.Config |
| dictitems: |
| model: !!python/object/new:configs.Config |
| dictitems: |
| __type__: configs.Transformer_Config |
| alibi: null |
| attention_type: mamba2 |
| attn_classes_path: transformers.models.qwen2.modeling_qwen2.QWEN2_ATTENTION_CLASSES |
| attn_path: rwkv6attn.RWKV6Attention |
| balance_state: 0 |
| brope: null |
| classname: qwen2 |
| cmix: x060 |
| cmix2: x060 |
| ctx_len: 4096 |
| dim_att: 0 |
| dim_ffn: 4800 |
| dropout: 0.0 |
| eos_token_id: 2 |
| gate_rank_type: 0 |
| gdn: !!python/object/new:configs.Config |
| dictitems: |
| __type__: configs.GDN_Config |
| expand_v: 2 |
| head_dim: 64 |
| num_heads: 6 |
| num_v_heads: 6 |
| groupnorm_att: 0 |
| head_size: 64 |
| head_size_divisor: 8 |
| hf_cfg: null |
| hf_path: '' |
| inv_other_layer_ratio: 1 |
| kv_cache_compression_ratio: 16 |
| lora_rank_decay: 0 |
| lora_rank_gate: 0 |
| lora_rank_iclr: 0 |
| lora_rank_tokenshift: 0 |
| lora_rank_value_residual_mix: 0 |
| mamba: !!python/object/new:configs.Config |
| dictitems: |
| __type__: configs.Mamba_Config |
| attn_layer_indices: [] |
| d_conv: 4 |
| d_inner: 1920 |
| d_state: 256 |
| expand: 1 |
| headdim: 64 |
| ngroups: 6 |
| n_embd: 1920 |
| n_layer: 56 |
| num_key_value_heads: 6 |
| parallel: 0 |
| preserve_last_n_layers: 0 |
| reinit_att: 0 |
| rms_norm_eps: 1.0e-06 |
| rope: !!python/object/new:configs.Config |
| dictitems: |
| __type__: configs.RoPE_Config |
| base: 490000.0 |
| rebase: 1.0 |
| rescale: 1.0 |
| tie_word_embeddings: false |
| tmix: qwen_mamba2 |
| tmix2: qwen2 |
| use_pos_emb: 1 |
| use_tokenshift: 1 |
| vocab_padding_idx: 102 |
| vocab_size: 99000 |
| runtime: !!python/object:configs.Runtime_Config |
| epoch_count: 999999999 |
| epoch_global_steps: 999999999 |
| global_step_bsz: 32 |
| my_pile_prev_p: 0 |
| my_timestamp: 2025-09-24-15-53-42 |
| proj_path: out/L56-D1920-qwen_mamba2_qwen2-e1-i1920-s256-hd64-gn6-A0-S4096--step1-rand2b |
| run_name: 'qwen_mamba2_qwen2 L56 D1920 ctx4096 ' |
| train: !!python/object/new:configs.Config |
| dictitems: |
| accelerator: gpu |
| accumulate_grad_batches: 1 |
| adam_eps: 1.0e-08 |
| attention_distillation_stage: 1 |
| beta1: 0.9 |
| beta2: 0.95 |
| check_val_every_n_epoch: 1 |
| data_file: ../cache/datasets/huggingface/Teaven/Selection/yulan-random2b/YuLan-Mini/data |
| data_type: binidx |
| devices: 8 |
| ds_bucket_mb: 200 |
| epoch_begin: 0 |
| epoch_save: 5 |
| grad_cp: 0 |
| gradient_clip_val: 1.0 |
| layerwise_lr: 1 |
| load_model: out/YuLanMini-2.4B.pth |
| load_partial: 1 |
| log_every_n_steps: 50 |
| lr2_final: -1 |
| lr2_init: -1 |
| lr_decay_type: cos |
| lr_endpoint: 1.0 |
| lr_final: 1.0e-05 |
| lr_init: 0.001 |
| lr_wait: 0.0 |
| magic_prime: 451961 |
| micro_bsz: 4 |
| my_exit_tokens: 1851236987 |
| num_nodes: 1 |
| optimizer: dsfusedadamw |
| precision: bf16 |
| proj_dir: out |
| proj_name: L56-D1920-qwen_mamba2_qwen2-e1-i1920-s256-hd64-gn6-A0-S4096 |
| proj_suffix: step1-rand2b |
| proj_suffix0: '' |
| seed_everything: 1337 |
| strategy: deepspeed_stage_1 |
| teacher: null |
| train_stage: 3 |
| val_check_interval: null |
| validation_data_file: '' |
| wandb: block-distill |
| warmup_steps: 50 |
| weight_decay: 0 |
| weight_decay_final: -1.0 |
|
|