opensysone / source /results /20260916-fleet-setup /selection-diagnostic.json
andyshu's picture
Back up verified OpenSysOne training snapshot and pinned source
2d5c26a verified
Raw History Blame Contribute Delete
4.48 kB
{
"scope": "Validation-only diagnostic motivating an explicitly revised checkpoint selection policy; no reserved calibration/test/holdout predictions accessed",
"selection_policy": {
"id": "crossfit_temperature_nll_v1",
"folds": 4,
"seed": 431,
"grouping": "source group, nested within task family; groups shuffled within sorted families and assigned round-robin",
"temperature_grid": {
"count": 101,
"log10_min": -1.0,
"log10_max": 1.3,
"arithmetic": "Python float64"
},
"temperature_fit": "minimum macro-family NLL on the other three folds; smallest temperature wins ties",
"score": "macro-family mean of all held-out decision NLLs",
"deployment_temperature": "fit afresh on reserved calibration only after model selection"
},
"selection_source_sha256": "be0a7a8496b5b830aa572ceba93606320f442063fd38180503fd6980dc1c578f",
"model_results": [
{
"step": 40,
"predictions_path": "/home/andy/ai/opensysone/runs/20260916T183823Z-train/artifacts/validation_step_000040_predictions.json",
"predictions_sha256": "e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b",
"raw_macro_nll": 0.39566109237724656,
"crossfit_macro_nll": 0.3595222692275388,
"accuracy": 0.875,
"fold_temperatures": [
1.5703628043335522,
1.6557699634695275,
1.5703628043335522,
1.7458221529205038
],
"per_family": {
"arc": {
"count": 128,
"score": 0.3246282279964344,
"raw_macro_nll": 0.3743996537632195,
"accuracy": 0.90625
},
"banking": {
"count": 128,
"score": 0.23241229297629554,
"raw_macro_nll": 0.22994083210335414,
"accuracy": 0.9140625
},
"boolq": {
"count": 128,
"score": 0.4348036430263749,
"raw_macro_nll": 0.58309304281893,
"accuracy": 0.84375
},
"snli": {
"count": 128,
"score": 0.4462449129110504,
"raw_macro_nll": 0.39521084082348246,
"accuracy": 0.8359375
}
}
},
{
"step": 128,
"predictions_path": "/home/andy/ai/opensysone/runs/20260916T192239Z-24h/training/validation_step_000128_predictions.json",
"predictions_sha256": "e686d63b8f93d466dc085e4e32ab904b218839246dee8211e8b71cff512db985",
"raw_macro_nll": 0.44268335003700615,
"crossfit_macro_nll": 0.3185177281077473,
"accuracy": 0.890625,
"fold_temperatures": [
2.2750974307720706,
2.2750974307720706,
2.39883291901949,
2.39883291901949
],
"per_family": {
"arc": {
"count": 128,
"score": 0.29531450555418093,
"raw_macro_nll": 0.34207193492789617,
"accuracy": 0.90625
},
"banking": {
"count": 128,
"score": 0.18676987812687565,
"raw_macro_nll": 0.17689879145408868,
"accuracy": 0.9375
},
"boolq": {
"count": 128,
"score": 0.4084573236595828,
"raw_macro_nll": 0.7099410978597552,
"accuracy": 0.859375
},
"snli": {
"count": 128,
"score": 0.38352920509034977,
"raw_macro_nll": 0.5418215759062844,
"accuracy": 0.859375
}
}
}
],
"bootstrap": {
"replicates": 5000,
"seed": 431,
"method": "Paired source-group resampling stratified by family, preserving fixed group folds; refit each fold temperature within every replicate using macro-family training NLL",
"step128_minus40_crossfit_nll": -0.041004541119791516,
"crossfit_nll_ci95": [
-0.07982194525314908,
-0.0011518414800151043
],
"step128_minus40_accuracy": 0.015625,
"accuracy_ci95": [
-0.01171875,
0.04296875
],
"limitations": "Conditional on these validation decisions and fixed folds; not adjusted for prior model/checkpoint selection or this criterion revision; independent test/holdout remain decisive"
},
"discordant_accuracy": {
"gains": 29,
"losses": 21,
"mcnemar_exact_two_sided_p": 0.3222363203575469
},
"selection_cpu_seconds_for_two_512_row_candidates": 0.06691360900003929,
"decision": "Adopt fixed crossfit_temperature_nll_v1 for new fleet checkpoints; retain raw NLL separately; fit final serving temperature afresh only on reserved calibration after selection"
}