{ "scope": "Validation-only diagnostic motivating an explicitly revised checkpoint selection policy; no reserved calibration/test/holdout predictions accessed", "selection_policy": { "id": "crossfit_temperature_nll_v1", "folds": 4, "seed": 431, "grouping": "source group, nested within task family; groups shuffled within sorted families and assigned round-robin", "temperature_grid": { "count": 101, "log10_min": -1.0, "log10_max": 1.3, "arithmetic": "Python float64" }, "temperature_fit": "minimum macro-family NLL on the other three folds; smallest temperature wins ties", "score": "macro-family mean of all held-out decision NLLs", "deployment_temperature": "fit afresh on reserved calibration only after model selection" }, "selection_source_sha256": "be0a7a8496b5b830aa572ceba93606320f442063fd38180503fd6980dc1c578f", "model_results": [ { "step": 40, "predictions_path": "/home/andy/ai/opensysone/runs/20260916T183823Z-train/artifacts/validation_step_000040_predictions.json", "predictions_sha256": "e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b", "raw_macro_nll": 0.39566109237724656, "crossfit_macro_nll": 0.3595222692275388, "accuracy": 0.875, "fold_temperatures": [ 1.5703628043335522, 1.6557699634695275, 1.5703628043335522, 1.7458221529205038 ], "per_family": { "arc": { "count": 128, "score": 0.3246282279964344, "raw_macro_nll": 0.3743996537632195, "accuracy": 0.90625 }, "banking": { "count": 128, "score": 0.23241229297629554, "raw_macro_nll": 0.22994083210335414, "accuracy": 0.9140625 }, "boolq": { "count": 128, "score": 0.4348036430263749, "raw_macro_nll": 0.58309304281893, "accuracy": 0.84375 }, "snli": { "count": 128, "score": 0.4462449129110504, "raw_macro_nll": 0.39521084082348246, "accuracy": 0.8359375 } } }, { "step": 128, "predictions_path": "/home/andy/ai/opensysone/runs/20260916T192239Z-24h/training/validation_step_000128_predictions.json", "predictions_sha256": "e686d63b8f93d466dc085e4e32ab904b218839246dee8211e8b71cff512db985", "raw_macro_nll": 0.44268335003700615, "crossfit_macro_nll": 0.3185177281077473, "accuracy": 0.890625, "fold_temperatures": [ 2.2750974307720706, 2.2750974307720706, 2.39883291901949, 2.39883291901949 ], "per_family": { "arc": { "count": 128, "score": 0.29531450555418093, "raw_macro_nll": 0.34207193492789617, "accuracy": 0.90625 }, "banking": { "count": 128, "score": 0.18676987812687565, "raw_macro_nll": 0.17689879145408868, "accuracy": 0.9375 }, "boolq": { "count": 128, "score": 0.4084573236595828, "raw_macro_nll": 0.7099410978597552, "accuracy": 0.859375 }, "snli": { "count": 128, "score": 0.38352920509034977, "raw_macro_nll": 0.5418215759062844, "accuracy": 0.859375 } } } ], "bootstrap": { "replicates": 5000, "seed": 431, "method": "Paired source-group resampling stratified by family, preserving fixed group folds; refit each fold temperature within every replicate using macro-family training NLL", "step128_minus40_crossfit_nll": -0.041004541119791516, "crossfit_nll_ci95": [ -0.07982194525314908, -0.0011518414800151043 ], "step128_minus40_accuracy": 0.015625, "accuracy_ci95": [ -0.01171875, 0.04296875 ], "limitations": "Conditional on these validation decisions and fixed folds; not adjusted for prior model/checkpoint selection or this criterion revision; independent test/holdout remain decisive" }, "discordant_accuracy": { "gains": 29, "losses": 21, "mcnemar_exact_two_sided_p": 0.3222363203575469 }, "selection_cpu_seconds_for_two_512_row_candidates": 0.06691360900003929, "decision": "Adopt fixed crossfit_temperature_nll_v1 for new fleet checkpoints; retain raw NLL separately; fit final serving temperature afresh only on reserved calibration after selection" }