Download source/results/20260916-fleet-setup/selection-diagnostic.json from andyshu/opensysone: direct link, hf CLI and curl.
- Browser
- Download file 4.48 kB
-
https://huggingface.co/andyshu/opensysone/resolve/main/source/results/20260916-fleet-setup/selection-diagnostic.json
- Command line
-
hf download hf://andyshu/opensysone/source/results/20260916-fleet-setup/selection-diagnostic.json
-
curl -L -o selection-diagnostic.json https://huggingface.co/andyshu/opensysone/resolve/main/source/results/20260916-fleet-setup/selection-diagnostic.json
4.48 kB
| { | |
| "scope": "Validation-only diagnostic motivating an explicitly revised checkpoint selection policy; no reserved calibration/test/holdout predictions accessed", | |
| "selection_policy": { | |
| "id": "crossfit_temperature_nll_v1", | |
| "folds": 4, | |
| "seed": 431, | |
| "grouping": "source group, nested within task family; groups shuffled within sorted families and assigned round-robin", | |
| "temperature_grid": { | |
| "count": 101, | |
| "log10_min": -1.0, | |
| "log10_max": 1.3, | |
| "arithmetic": "Python float64" | |
| }, | |
| "temperature_fit": "minimum macro-family NLL on the other three folds; smallest temperature wins ties", | |
| "score": "macro-family mean of all held-out decision NLLs", | |
| "deployment_temperature": "fit afresh on reserved calibration only after model selection" | |
| }, | |
| "selection_source_sha256": "be0a7a8496b5b830aa572ceba93606320f442063fd38180503fd6980dc1c578f", | |
| "model_results": [ | |
| { | |
| "step": 40, | |
| "predictions_path": "/home/andy/ai/opensysone/runs/20260916T183823Z-train/artifacts/validation_step_000040_predictions.json", | |
| "predictions_sha256": "e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b", | |
| "raw_macro_nll": 0.39566109237724656, | |
| "crossfit_macro_nll": 0.3595222692275388, | |
| "accuracy": 0.875, | |
| "fold_temperatures": [ | |
| 1.5703628043335522, | |
| 1.6557699634695275, | |
| 1.5703628043335522, | |
| 1.7458221529205038 | |
| ], | |
| "per_family": { | |
| "arc": { | |
| "count": 128, | |
| "score": 0.3246282279964344, | |
| "raw_macro_nll": 0.3743996537632195, | |
| "accuracy": 0.90625 | |
| }, | |
| "banking": { | |
| "count": 128, | |
| "score": 0.23241229297629554, | |
| "raw_macro_nll": 0.22994083210335414, | |
| "accuracy": 0.9140625 | |
| }, | |
| "boolq": { | |
| "count": 128, | |
| "score": 0.4348036430263749, | |
| "raw_macro_nll": 0.58309304281893, | |
| "accuracy": 0.84375 | |
| }, | |
| "snli": { | |
| "count": 128, | |
| "score": 0.4462449129110504, | |
| "raw_macro_nll": 0.39521084082348246, | |
| "accuracy": 0.8359375 | |
| } | |
| } | |
| }, | |
| { | |
| "step": 128, | |
| "predictions_path": "/home/andy/ai/opensysone/runs/20260916T192239Z-24h/training/validation_step_000128_predictions.json", | |
| "predictions_sha256": "e686d63b8f93d466dc085e4e32ab904b218839246dee8211e8b71cff512db985", | |
| "raw_macro_nll": 0.44268335003700615, | |
| "crossfit_macro_nll": 0.3185177281077473, | |
| "accuracy": 0.890625, | |
| "fold_temperatures": [ | |
| 2.2750974307720706, | |
| 2.2750974307720706, | |
| 2.39883291901949, | |
| 2.39883291901949 | |
| ], | |
| "per_family": { | |
| "arc": { | |
| "count": 128, | |
| "score": 0.29531450555418093, | |
| "raw_macro_nll": 0.34207193492789617, | |
| "accuracy": 0.90625 | |
| }, | |
| "banking": { | |
| "count": 128, | |
| "score": 0.18676987812687565, | |
| "raw_macro_nll": 0.17689879145408868, | |
| "accuracy": 0.9375 | |
| }, | |
| "boolq": { | |
| "count": 128, | |
| "score": 0.4084573236595828, | |
| "raw_macro_nll": 0.7099410978597552, | |
| "accuracy": 0.859375 | |
| }, | |
| "snli": { | |
| "count": 128, | |
| "score": 0.38352920509034977, | |
| "raw_macro_nll": 0.5418215759062844, | |
| "accuracy": 0.859375 | |
| } | |
| } | |
| } | |
| ], | |
| "bootstrap": { | |
| "replicates": 5000, | |
| "seed": 431, | |
| "method": "Paired source-group resampling stratified by family, preserving fixed group folds; refit each fold temperature within every replicate using macro-family training NLL", | |
| "step128_minus40_crossfit_nll": -0.041004541119791516, | |
| "crossfit_nll_ci95": [ | |
| -0.07982194525314908, | |
| -0.0011518414800151043 | |
| ], | |
| "step128_minus40_accuracy": 0.015625, | |
| "accuracy_ci95": [ | |
| -0.01171875, | |
| 0.04296875 | |
| ], | |
| "limitations": "Conditional on these validation decisions and fixed folds; not adjusted for prior model/checkpoint selection or this criterion revision; independent test/holdout remain decisive" | |
| }, | |
| "discordant_accuracy": { | |
| "gains": 29, | |
| "losses": 21, | |
| "mcnemar_exact_two_sided_p": 0.3222363203575469 | |
| }, | |
| "selection_cpu_seconds_for_two_512_row_candidates": 0.06691360900003929, | |
| "decision": "Adopt fixed crossfit_temperature_nll_v1 for new fleet checkpoints; retain raw NLL separately; fit final serving temperature afresh only on reserved calibration after selection" | |
| } | |