{ "seed": 20260908, "distillation_policy": "At user request, after long SFT freeze a checkpoint on validation and evaluate Approve-or-Deny. Report the result to the user before proceeding. Distill only if its benchmark accuracy does not strictly exceed the fresh auto-1b-bf16 benchmark accuracy; skip distillation if it does. A tie triggers distillation.", "sft_short": { "epochs": 2, "lr": 1.5e-05, "max_length_inclusion": 4096 }, "sft_long": { "epochs": 1, "lr": 6e-06, "short_replay_ratio": 1.0 }, "kd_short": { "epochs": 1, "lr": 5e-06, "alpha": 0.5, "temperature": 2.0 }, "kd_long": { "epochs": 1, "lr": 3e-06, "alpha": 0.5, "temperature": 2.0, "short_replay_ratio": 1.0 }, "optimizer": "Full-parameter fused AdamW; fp32 master parameters/moments; bf16 autocast; betas=(0.9,0.95); eps=1e-8; weight_decay=0.01; grad_clip=1.0", "schedule": "3% linear warmup then cosine to 10% of peak learning rate in each phase", "batching": "Length buckets; <=16384 padded tokens per short microbatch; single longer example; optimizer step at >=128 examples or >=131072 true tokens", "long_context": "Every retained row is used at its full tokenized length, with model capacity of 65536; no training input required truncation.", "distillation": "Teacher forward only; temperature-scaled KL plus hard-label CE; alpha reduced to 0.125 where teacher argmax disagrees with label", "selection": "Validation-only: 0.7 * overall balanced accuracy + 0.3 * balanced accuracy at >=16384 tokens; each phase best and last compared on the full 7824-row selection split", "publish_gate": { "benchmark_accuracy_min": 0.95, "benchmark_long_accuracy_min": 0.9, "false_approve_improvement_min": 0.015, "false_deny_improvement_min": 0.025, "validation_audit_accuracy_min": 0.95, "paired_accuracy_ci95_lower_min": 0.0 } }