{ "temperatures": { "choice": 1.2321, "noul": 1.297, "score": 0.7558 }, "applied_target": "mixed", "applied_fits": { "choice": "train", "noul": "train", "score": "hard" }, "previous": { "applied_target": "train", "temperatures": { "choice": 1.2321, "noul": 1.297, "score": 1.1423 }, "replaced": "2026-09-26" }, "why": "Refit 2026-09-26 without retraining. Both fits are unchanged and come from the calibration split only. On every evaluation set, re-tempering the stored logits: score questions calibrate better at the hard fit (T 0.76) than at the train fit (T 1.14), whose ordinal target is deliberately smoothed; choice and noul stay on the train fit, which the public sets prefer. Question-weighted ECE over the 20 sets 0.0809 -> 0.0718, macro 0.0708 -> 0.0642, NLL 0.5233 -> 0.5139; accuracy unchanged. See eval/RESULTS.md, Temperature fits.", "top_level_stats_describe": "the train fit (nll_before/nll_after/ece_before/ece_after below are its calibration-split numbers)", "calib_file": "/workspace/jeb/data-v1/calib.jsonl", "n": { "choice": 225, "noul": 147, "score": 549 }, "fits": { "hard": { "choice": { "T": 0.8349, "nll_before": 0.3342, "nll_after": 0.33 }, "noul": { "T": 0.9013, "nll_before": 0.3097, "nll_after": 0.3086 }, "score": { "T": 0.7558, "nll_before": 0.883, "nll_after": 0.8662 } }, "train": { "choice": { "T": 1.2321, "nll_before": 0.4621, "nll_after": 0.4537 }, "noul": { "T": 1.297, "nll_before": 0.3953, "nll_after": 0.387 }, "score": { "T": 1.1423, "nll_before": 1.059, "nll_after": 1.055 } }, "mixed": { "choice": { "T": 1.2321, "nll_before": 0.4621, "nll_after": 0.4537, "source": "train" }, "noul": { "T": 1.297, "nll_before": 0.3953, "nll_after": 0.387, "source": "train" }, "score": { "T": 0.7558, "nll_before": 0.883, "nll_after": 0.8662, "source": "hard" } } }, "nll_before": { "choice": 0.4621, "noul": 0.3953, "score": 1.059 }, "nll_after": { "choice": 0.4537, "noul": 0.387, "score": 1.055 }, "ece_before": { "choice": 0.1062, "noul": 0.0737, "score": 0.079 }, "ece_after": { "choice": 0.12, "noul": 0.0951, "score": 0.1082 }, "accuracy": { "choice": 0.9111, "noul": 0.898, "score": 0.6503 }, "score_targets": "ordinal", "score_ordinal_adjacent": 0.2 }