{ "method": "distillation_seqkd", "behavior_id": "mathematical", "spec_sha256": "fd0a06bd394ab5cef4e5e7ea2c56e58a379566599c133743c1eab0732d65355c", "spec_sha256_scheme": "file-v1", "spec_extends": null, "parent_spec_sha256": null, "base_model": "Qwen/Qwen2.5-7B-Instruct", "train_file": "/root/seqkd.jsonl", "buckets": null, "n_rows": 2002, "rank": 32, "lora_dropout": 0.05, "learning_rate": 0.0001, "folded_from": null, "epochs": 1.0, "effective_batch": 16, "max_len": 2048, "loss_mask": "all_turns", "grad_ckpt": true, "seed": 42, "optimizer_steps": 126, "train_loss": 0.02499645475357298 }