jbrashear commited on
Commit
e4b70f7
·
verified ·
1 Parent(s): e653e65

Temperatures: take the 27B's refit (score 0.7558) from frontier-infra/jebadiah-27b@1c0d794f

Browse files
Files changed (2) hide show
  1. README.md +1 -1
  2. temperatures.json +38 -2
README.md CHANGED
@@ -42,7 +42,7 @@ needs about its own size in GPU or unified memory, plus about 1 GB for a 4k cont
42
  The answer is the log probability of each option label ("A", "B", ...) at the answer position, which
43
  llama-server's `/completion` returns. The script renders the prompt exactly as AINode does, sends the raw
44
  text (so the server's own chat template is never used), renormalises over the labels and applies
45
- `temperatures.json` (choice 1.2321, noul 1.297, score 1.1423). You need a llama.cpp that knows the `qwen35` architecture: we checked
46
  v0.5.0 (older builds refuse the file).
47
 
48
  ```bash
 
42
  The answer is the log probability of each option label ("A", "B", ...) at the answer position, which
43
  llama-server's `/completion` returns. The script renders the prompt exactly as AINode does, sends the raw
44
  text (so the server's own chat template is never used), renormalises over the labels and applies
45
+ `temperatures.json` (choice 1.2321, noul 1.297, score 0.7558). You need a llama.cpp that knows the `qwen35` architecture: we checked
46
  v0.5.0 (older builds refuse the file).
47
 
48
  ```bash
temperatures.json CHANGED
@@ -2,9 +2,25 @@
2
  "temperatures": {
3
  "choice": 1.2321,
4
  "noul": 1.297,
5
- "score": 1.1423
6
  },
7
- "applied_target": "train",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8
  "calib_file": "/workspace/jeb/data-v1/calib.jsonl",
9
  "n": {
10
  "choice": 225,
@@ -45,6 +61,26 @@
45
  "nll_before": 1.059,
46
  "nll_after": 1.055
47
  }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
48
  }
49
  },
50
  "nll_before": {
 
2
  "temperatures": {
3
  "choice": 1.2321,
4
  "noul": 1.297,
5
+ "score": 0.7558
6
  },
7
+ "applied_target": "mixed",
8
+ "applied_fits": {
9
+ "choice": "train",
10
+ "noul": "train",
11
+ "score": "hard"
12
+ },
13
+ "previous": {
14
+ "applied_target": "train",
15
+ "temperatures": {
16
+ "choice": 1.2321,
17
+ "noul": 1.297,
18
+ "score": 1.1423
19
+ },
20
+ "replaced": "2026-09-26"
21
+ },
22
+ "why": "Refit 2026-09-26 without retraining. Both fits are unchanged and come from the calibration split only. On every evaluation set, re-tempering the stored logits: score questions calibrate better at the hard fit (T 0.76) than at the train fit (T 1.14), whose ordinal target is deliberately smoothed; choice and noul stay on the train fit, which the public sets prefer. Question-weighted ECE over the 20 sets 0.0809 -> 0.0718, macro 0.0708 -> 0.0642, NLL 0.5233 -> 0.5139; accuracy unchanged. See eval/RESULTS.md, Temperature fits.",
23
+ "top_level_stats_describe": "the train fit (nll_before/nll_after/ece_before/ece_after below are its calibration-split numbers)",
24
  "calib_file": "/workspace/jeb/data-v1/calib.jsonl",
25
  "n": {
26
  "choice": 225,
 
61
  "nll_before": 1.059,
62
  "nll_after": 1.055
63
  }
64
+ },
65
+ "mixed": {
66
+ "choice": {
67
+ "T": 1.2321,
68
+ "nll_before": 0.4621,
69
+ "nll_after": 0.4537,
70
+ "source": "train"
71
+ },
72
+ "noul": {
73
+ "T": 1.297,
74
+ "nll_before": 0.3953,
75
+ "nll_after": 0.387,
76
+ "source": "train"
77
+ },
78
+ "score": {
79
+ "T": 0.7558,
80
+ "nll_before": 0.883,
81
+ "nll_after": 0.8662,
82
+ "source": "hard"
83
+ }
84
  }
85
  },
86
  "nll_before": {