{ "application": { "by_head_key": { "boolean/state4": [ "boolean", "multilabel" ], "ordered/choiceS": "ordered", "single/choiceR": "single" }, "choice": { "branches": "|R single choice (single/choiceR), |S ordered choice (ordered/choiceS)", "confidence": "the listed top-1 probability", "expression": "probabilities = softmax(x[:n] / T)", "note": "slice to the n listed options first, then divide by T. Reserved slots are never scored.", "rule": "SLICE THEN TEMPER" }, "granularity": "per answer type. The merged yes/no head serves two types, so by_head_key maps it to both and the served type selects the scalar.", "idempotence": "apply exactly once; the returned probability and the listed top-1 derive from the same tempered read.", "note": "Nouls and choices are tempered DIFFERENTLY. Implement exactly as written.", "noul": { "branches": "yes/no and every multi-label candidate (head_key boolean/state4, the merged head)", "confidence": "max(P(yes), 1 - P(yes))", "expression": "z = x[0] - logsumexp(x[1:]); P(yes) = sigmoid(z / T)", "note": "x is the full four-letter logit vector. Do NOT compute softmax(x / T)[0].", "rule": "COLLAPSE THEN TEMPER" } }, "fit": { "calibration_file_sha256": "ea069d224501af950caaae6914d6fdf14ce6d4d2bda566c41549b1e8c30a6222", "decision": "served at T = 1.0 for every type: the fitted scalars did not improve held-out calibration (test ECE worse in 8 of 10 type x modality cells; n-weighted 0.0212 fitted vs 0.0199 unscaled)", "modality": "image rows use the same per-type scalar (no modality key in serving)", "scored_heads_note": "scores were produced with a heads file whose two extra (entity/multilabel) slots were never read; its other 16 arrays are byte-identical to the shipped heads file, so the served logits are the fitted logits", "scored_heads_sha256": "96ea51416bbeb32d991b7b38d7f0c22ff3e82539c8910284c4fb1961f2869ace", "source": "real development documents (held out from test), this model's BF16 scores, one NLL-minimising scalar per answer type" }, "fitted_on_model": { "adapter_sha256": "d122466d430a058bb6457d919f811160e97fbd20149f4f24ca455c5d83e360a0", "trained_heads_sha256": "f766d752d7768a419a9657155cf27f042834d9de29392cf7470d8725130e67ab" }, "frozen": true, "models": { "boolean": { "applied": false, "fit_units": 192, "fitted_temperature": 0.8175095705097734, "head_key": "boolean/state4", "kind": "noul", "task": "boolean", "temperature": 1.0, "unit": "question" }, "multilabel": { "applied": false, "fit_units": 1862, "fitted_temperature": 0.842297230286191, "head_key": "boolean/state4", "kind": "noul", "task": "multilabel", "temperature": 1.0, "unit": "candidate noul" }, "ordered": { "applied": false, "fit_units": 134, "fitted_temperature": 1.2561869742268443, "head_key": "ordered/choiceS", "kind": "choice", "task": "ordered", "temperature": 1.0, "unit": "question" }, "single": { "applied": false, "fit_units": 134, "fitted_temperature": 1.107722547236206, "head_key": "single/choiceR", "kind": "choice", "task": "single", "temperature": 1.0, "unit": "question" } }, "schema": "solomon-readout-temperature-v3", "sha256": "945bad449b7f5ffc88e597277d632fbab81c3c8729e22c8babd3f4a45fe1378b" }