botp
/

Solomon / serving /readout-temperature-v3.json
orz99's picture ArcherHume's picture
Duplicate from DoccyHealth/Solomon
1d2de8a
Raw History Blame Contribute Delete
3.37 kB
{
"application": {
"by_head_key": {
"boolean/state4": [
"boolean",
"multilabel"
],
"ordered/choiceS": "ordered",
"single/choiceR": "single"
},
"choice": {
"branches": "|R single choice (single/choiceR), |S ordered choice (ordered/choiceS)",
"confidence": "the listed top-1 probability",
"expression": "probabilities = softmax(x[:n] / T)",
"note": "slice to the n listed options first, then divide by T. Reserved slots are never scored.",
"rule": "SLICE THEN TEMPER"
},
"granularity": "per answer type. The merged yes/no head serves two types, so by_head_key maps it to both and the served type selects the scalar.",
"idempotence": "apply exactly once; the returned probability and the listed top-1 derive from the same tempered read.",
"note": "Nouls and choices are tempered DIFFERENTLY. Implement exactly as written.",
"noul": {
"branches": "yes/no and every multi-label candidate (head_key boolean/state4, the merged head)",
"confidence": "max(P(yes), 1 - P(yes))",
"expression": "z = x[0] - logsumexp(x[1:]); P(yes) = sigmoid(z / T)",
"note": "x is the full four-letter logit vector. Do NOT compute softmax(x / T)[0].",
"rule": "COLLAPSE THEN TEMPER"
}
},
"fit": {
"calibration_file_sha256": "ea069d224501af950caaae6914d6fdf14ce6d4d2bda566c41549b1e8c30a6222",
"decision": "served at T = 1.0 for every type: the fitted scalars did not improve held-out calibration (test ECE worse in 8 of 10 type x modality cells; n-weighted 0.0212 fitted vs 0.0199 unscaled)",
"modality": "image rows use the same per-type scalar (no modality key in serving)",
"scored_heads_note": "scores were produced with a heads file whose two extra (entity/multilabel) slots were never read; its other 16 arrays are byte-identical to the shipped heads file, so the served logits are the fitted logits",
"scored_heads_sha256": "96ea51416bbeb32d991b7b38d7f0c22ff3e82539c8910284c4fb1961f2869ace",
"source": "real development documents (held out from test), this model's BF16 scores, one NLL-minimising scalar per answer type"
},
"fitted_on_model": {
"adapter_sha256": "d122466d430a058bb6457d919f811160e97fbd20149f4f24ca455c5d83e360a0",
"trained_heads_sha256": "f766d752d7768a419a9657155cf27f042834d9de29392cf7470d8725130e67ab"
},
"frozen": true,
"models": {
"boolean": {
"applied": false,
"fit_units": 192,
"fitted_temperature": 0.8175095705097734,
"head_key": "boolean/state4",
"kind": "noul",
"task": "boolean",
"temperature": 1.0,
"unit": "question"
},
"multilabel": {
"applied": false,
"fit_units": 1862,
"fitted_temperature": 0.842297230286191,
"head_key": "boolean/state4",
"kind": "noul",
"task": "multilabel",
"temperature": 1.0,
"unit": "candidate noul"
},
"ordered": {
"applied": false,
"fit_units": 134,
"fitted_temperature": 1.2561869742268443,
"head_key": "ordered/choiceS",
"kind": "choice",
"task": "ordered",
"temperature": 1.0,
"unit": "question"
},
"single": {
"applied": false,
"fit_units": 134,
"fitted_temperature": 1.107722547236206,
"head_key": "single/choiceR",
"kind": "choice",
"task": "single",
"temperature": 1.0,
"unit": "question"
}
},
"schema": "solomon-readout-temperature-v3",
"sha256": "945bad449b7f5ffc88e597277d632fbab81c3c8729e22c8babd3f4a45fe1378b"
}