!!python/object/new:configs.Config dictitems: model: !!python/object/new:configs.Config dictitems: __type__: configs.Transformer_Config alibi: null attention_type: mamba2 attn_classes_path: transformers.models.qwen2.modeling_qwen2.QWEN2_ATTENTION_CLASSES attn_path: rwkv6attn.RWKV6Attention balance_state: 0 brope: null classname: qwen2 cmix: x060 cmix2: x060 ctx_len: 4096 dim_att: 0 dim_ffn: 4800 dropout: 0.0 eos_token_id: 2 gate_rank_type: 0 groupnorm_att: 0 head_size: 64 head_size_divisor: 8 hf_cfg: null hf_path: '' inv_other_layer_ratio: 1 kv_cache_compression_ratio: 16 lora_rank_decay: 0 lora_rank_gate: 0 lora_rank_iclr: 0 lora_rank_tokenshift: 0 lora_rank_value_residual_mix: 0 mamba: !!python/object/new:configs.Config dictitems: __type__: configs.Mamba_Config attn_layer_indices: [] d_conv: 4 d_inner: 3840 d_state: 128 expand: 2 headdim: 64 ngroups: 6 n_embd: 1920 n_layer: 56 num_key_value_heads: 6 parallel: 0 preserve_last_n_layers: 0 reinit_att: 0 rms_norm_eps: 1.0e-06 rope: !!python/object/new:configs.Config dictitems: __type__: configs.RoPE_Config base: 490000.0 rebase: 1.0 rescale: 1.0 tie_word_embeddings: false tmix: qwen_mamba2 tmix2: qwen2 use_pos_emb: 1 use_tokenshift: 1 vocab_padding_idx: 102 vocab_size: 99000 runtime: !!python/object:configs.Runtime_Config epoch_count: 999999999 epoch_global_steps: 999999999 global_step_bsz: 32 my_pile_prev_p: 0 my_timestamp: 2025-09-01-00-39-16 proj_path: out/L56-D1920-qwen_mamba2_qwen2-e2-i3840-s128-hd64-gn6-A0-S4096--step1 run_name: 'qwen_mamba2_qwen2 L56 D1920 ctx4096 ' train: !!python/object/new:configs.Config dictitems: accelerator: gpu accumulate_grad_batches: 1 adam_eps: 1.0e-08 attention_distillation_stage: 1 beta1: 0.9 beta2: 0.95 check_val_every_n_epoch: 1 data_file: data/yulan-mini-2B/yulan-mini/yulan-mini data_type: binidx devices: 8 ds_bucket_mb: 200 epoch_begin: 0 epoch_save: 5 grad_cp: 0 gradient_clip_val: 1.0 layerwise_lr: 1 load_model: out/YuLanMini-2.4B.pth load_partial: 1 log_every_n_steps: 50 lr2_final: -1 lr2_init: -1 lr_decay_type: cos lr_endpoint: 1.0 lr_final: 1.0e-05 lr_init: 0.001 lr_wait: 0.0 magic_prime: 451961 micro_bsz: 4 my_exit_tokens: 1851236987 num_nodes: 1 optimizer: dsfusedadamw precision: bf16 proj_dir: out proj_name: L56-D1920-qwen_mamba2_qwen2-e2-i3840-s128-hd64-gn6-A0-S4096 proj_suffix: step1 proj_suffix0: '' seed_everything: 1337 strategy: deepspeed_stage_1 teacher: null train_stage: 3 val_check_interval: null validation_data_file: '' wandb: '' warmup_steps: 50 weight_decay: 0 weight_decay_final: -1.0