{ "allow_inexact_legacy_data_resume": false, "attention_bias": false, "attn_implementation": "flash_attention_4", "aux_adam_beta1": 0.9, "aux_adam_beta2": 0.95, "aux_adam_eps": 1e-08, "aux_adam_lr": 0.0003, "aux_adam_weight_decay": 0.1, "beta1": 0.9, "beta2": 0.95, "checkpoint_bucket_folder": "checkpoints_llama_1b_kda_3to1_h12_6b_1308", "cross_document_attention": true, "dataloader_prefetch_factor": 2, "dataloader_workers": 8, "dataset_config": "sample-10BT", "dataset_name": "HuggingFaceFW/fineweb", "eval_benchmarks": false, "eval_every_steps": 250, "eval_holdout_fraction": 0.005, "eval_max_examples": 512, "eval_streaming_buffer_size": 1, "grad_accum_steps": 24, "grad_clip": 1.0, "hf_assets_dir": "./hf_assets_llama_1b_kda_3to1_h12", "hidden_act": "silu", "hidden_size": 1536, "init_from_checkpoint": null, "initializer_range": 0.02, "intermediate_size": 5120, "kda_allow_neg_eigval": false, "kda_conv_bias": false, "kda_conv_size": 4, "kda_every": 4, "kda_expand_v": 1.0, "kda_full_attn_every": null, "kda_full_attn_layers": null, "kda_full_attn_range": null, "kda_head_dim": 128, "kda_lower_bound": null, "kda_num_heads": 12, "kda_num_v_heads": null, "kda_offset": 0, "kda_safe_gate": false, "kda_use_short_conv": true, "learning_rate": 0.0003, "liger_kernel_config": { "cross_entropy": false, "fused_linear_cross_entropy": true, "rms_norm": true, "rope": true, "swiglu": true }, "log_every_steps": 1, "lr_decay_ratio": null, "lr_decay_type": "cosine", "max_position_embeddings": 8192, "max_seq_len": 8192, "max_steps": 3053, "min_lr_factor": 0.1, "muon_lr": 0.02, "muon_momentum": 0.95, "muon_nesterov": true, "muon_ns_steps": 5, "muon_weight_decay": 0.1, "num_attention_heads": 12, "num_hidden_layers": 32, "num_key_value_heads": 6, "optimizer_eps": 1e-08, "optimizer_implementation": "fused", "optimizer_name": "muon", "output_dir": "./checkpoints_llama_1b_kda_3to1_h12_6b", "per_device_batch_size": 10, "qk_norm": true, "qk_norm_eps": 1e-06, "resume_from_checkpoint": null, "rms_norm_eps": 1e-06, "rope_theta": 1000000, "save_every_steps": 250, "seed": 42, "streaming_buffer_size": 10000, "sync_checkpoints_to_bucket": false, "target_train_tokens": null, "tie_word_embeddings": true, "tokenizer_name": "meta-llama/Llama-2-7b", "torch_compile": false, "torch_compile_mode": "default", "use_liger_kernel": true, "use_wandb": true, "vocab_size": 32000, "wandb_log_console": false, "wandb_project": "llama-1b-6b-torchtitan", "wandb_run_name": "llama-1b-kda-3to1-h12-1308", "warmup_steps": 150, "weight_decay": 0.1 }