{ "n": 500, "argmax_agreement": 0.992, "cuda_same_subset": { "accuracy": 0.7910177949703642, "macro_f1": 0.7760857891780776, "nll": 0.6000106706289864, "brier": 0.317876961372947, "ece": 0.16696329399611096 }, "q8_same_subset": { "accuracy": 0.7868855635654054, "macro_f1": 0.773615445189482, "nll": 0.6015417570028033, "brier": 0.318930434743987, "ece": 0.16630794459650552 }, "accuracy_difference": -0.004132231404958775, "nll_difference": 0.0015310863738169367, "temperature_folded": true, "passed": true, "practical_deployment_gate_passed": true, "initial_strict_accuracy_target_met_on_subset": false, "initial_accuracy_loss_target": 0.003, "note": "500-example subset check: 99.2% agreement. Observed task-macro loss is 0.413 pp, so the initial 0.3 pp accuracy goal is not met on this subset. Prefer BF16/MLX when accuracy is the priority; this is not a bound on population degradation." }