Download training_config.json from LakshyAAAgrawal/continuous-thought-r11_single_perstep_g1: direct link, hf CLI and curl.
- Browser
- Download file 593 Bytes
-
https://huggingface.co/LakshyAAAgrawal/continuous-thought-r11_single_perstep_g1/resolve/ac0ff862919a38222ab2b391e9db940d73d9f04c/training_config.json
- Command line
-
hf download hf://LakshyAAAgrawal/continuous-thought-r11_single_perstep_g1@ac0ff862919a38222ab2b391e9db940d73d9f04c/training_config.json
-
curl -L -o training_config.json https://huggingface.co/LakshyAAAgrawal/continuous-thought-r11_single_perstep_g1/resolve/ac0ff862919a38222ab2b391e9db940d73d9f04c/training_config.json
593 Bytes
| { | |
| "mode": "codi_single", | |
| "model": "Qwen/Qwen3-1.7B", | |
| "rollouts": "data/rollouts.jsonl", | |
| "teacher_states": "data/teacher_states_v3.pt", | |
| "output_dir": "checkpoints/r11_single_perstep_g1", | |
| "epochs": 3, | |
| "batch_size": 2, | |
| "grad_accum": 8, | |
| "lr": 0.0002, | |
| "gamma": 1.0, | |
| "warmup_ratio": 0.03, | |
| "num_latent": 6, | |
| "max_prompt_len": 512, | |
| "max_answer_len": 128, | |
| "max_len": 1024, | |
| "answer_format": "full", | |
| "per_step_distill": true, | |
| "log_every": 10, | |
| "seed": 42, | |
| "lora_rank": 32, | |
| "lora_alpha": 16, | |
| "best_loss": 0.6426934776270491, | |
| "total_time_s": 1477.4382388591766 | |
| } |