tianzl66's picture
Release audited B300 LoRA training seed 44; actual zero warmup and complete in-domain HNS results
ffaca6a verified
Raw History Blame Contribute Delete
959 Bytes
{
"base_model": "<LOCAL_PATH>/Qwen3-8B",
"task": "magicoder",
"peft_method": "lora",
"train_profile": "none",
"num_train_epochs": 1.0,
"max_steps": null,
"stop_at_step": null,
"stop_at_epoch": null,
"resume_from_checkpoint": null,
"training_budget": "epochs",
"per_device_train_batch_size": 16,
"gradient_accumulation_steps": 2,
"global_train_batch_size": 32,
"world_size": 1,
"effective_global_batch_size": 32,
"steps_per_epoch": 1563,
"total_steps": 1563,
"lr": 2e-05,
"warmup_ratio": 0.05,
"weight_decay": 0.0,
"lr_scheduler_type": "cosine",
"min_lr_ratio": null,
"max_seq_len": 4096,
"dataset_size": 50000,
"dataset_path": "<LOCAL_PATH>/magicoder-train.parquet",
"max_train_samples": 50000,
"dataset_seed": 42,
"sft_format": "chat",
"chat_template_mode": "non_thinking",
"save_strategy": "epoch",
"save_steps": 500,
"save_total_limit": 1,
"logging_steps": 25,
"adalora_schedule": null
}