Jev-Style-Qwen3.5-2B-Decision-v2-GGUF / Jev-Style-v2-Calibrated-Q8_0.calibration.json
chaoliangUNSW's picture
Rename GGUF files so the quant type ends the filename (Ollama :Q4_K_M tags)
f34035e verified
Raw History Blame Contribute Delete
1.81 kB
{
"temperature": 1.0,
"fitted_temperature_folded": 1.0403540135054734,
"temperature_folded": true,
"folded_tensor": "output_norm.weight",
"source_calibration": {
"temperature": 1.0403540135054734,
"calibration_n": 3100,
"objective": "sample_mean_soft_cross_entropy",
"bounds": [
0.05,
20
],
"nll_before": 0.5142612871187658,
"nll_after": 0.5139652439657095,
"backend": "gguf",
"model": "results/Jev-Style-v2-Q8_0.gguf",
"source_sha256": "2fde7f45dce3440abfde145bb30ad61e2643b1f853866b5760b235685328dc1c"
},
"input_sha256": "f659d164d0fed4e645f711cbf56c177ee63c1856e6875f48a27ebac2cfb11175",
"output_sha256": "5c2aa0d35b24a27f03228b2c62ebaaebd9b5b785844634d4217278d822751494",
"validation_required": false,
"validation": {
"n": 500,
"argmax_agreement": 0.992,
"cuda_same_subset": {
"accuracy": 0.7910177949703642,
"macro_f1": 0.7760857891780776,
"nll": 0.6000106706289864,
"brier": 0.317876961372947,
"ece": 0.16696329399611096
},
"q8_same_subset": {
"accuracy": 0.7868855635654054,
"macro_f1": 0.773615445189482,
"nll": 0.6015417570028033,
"brier": 0.318930434743987,
"ece": 0.16630794459650552
},
"accuracy_difference": -0.004132231404958775,
"nll_difference": 0.0015310863738169367,
"temperature_folded": true,
"passed": true,
"practical_deployment_gate_passed": true,
"initial_strict_accuracy_target_met_on_subset": false,
"initial_accuracy_loss_target": 0.003,
"note": "500-example subset check: 99.2% agreement. Observed task-macro loss is 0.413 pp, so the initial 0.3 pp accuracy goal is not met on this subset. Prefer BF16/MLX when accuracy is the priority; this is not a bound on population degradation."
}
}