chaoliangUNSW commited on
Commit
aa6d27a
·
verified ·
1 Parent(s): dca731a

Add calibrated Q4_K_M and BF16 GGUF download options

Browse files
.gitattributes CHANGED
@@ -37,3 +37,5 @@ Jev-Style-v2-Q8_0-Calibrated.gguf filter=lfs diff=lfs merge=lfs -text
37
  figures/benchmark.png filter=lfs diff=lfs merge=lfs -text
38
  figures/calibration.png filter=lfs diff=lfs merge=lfs -text
39
  figures/robustness.png filter=lfs diff=lfs merge=lfs -text
 
 
 
37
  figures/benchmark.png filter=lfs diff=lfs merge=lfs -text
38
  figures/calibration.png filter=lfs diff=lfs merge=lfs -text
39
  figures/robustness.png filter=lfs diff=lfs merge=lfs -text
40
+ Jev-Style-v2-Q4_K_M-Calibrated.gguf filter=lfs diff=lfs merge=lfs -text
41
+ Jev-Style-v2-BF16-Calibrated.gguf filter=lfs diff=lfs merge=lfs -text
Jev-Style-v2-BF16-Calibrated.calibration.json ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "temperature": 1.0,
3
+ "fitted_temperature_folded": 1.0408715111841746,
4
+ "temperature_folded": true,
5
+ "folded_tensor": "output_norm.weight",
6
+ "source_calibration": {
7
+ "temperature": 1.0408715111841746,
8
+ "calibration_n": 3100,
9
+ "objective": "sample_mean_soft_cross_entropy",
10
+ "bounds": [
11
+ 0.05,
12
+ 20
13
+ ],
14
+ "nll_before": 0.5144028513612269,
15
+ "nll_after": 0.5140996476734587,
16
+ "backend": "gguf",
17
+ "model": "source_project/h100_v2/results/Jev-Style-v2-BF16-uncalibrated.gguf",
18
+ "source_sha256": "2fde7f45dce3440abfde145bb30ad61e2643b1f853866b5760b235685328dc1c"
19
+ },
20
+ "input_sha256": "79fb883ffc5831f165c11ca0119c7702b3c5209c916d939f2555fb1c7d45413c",
21
+ "output_sha256": "8baa111eec6e30a5c9e97d559b53127a19ef30171e8aed26673153b4f77adcbb",
22
+ "validation_required": false,
23
+ "validation": {
24
+ "format": "BF16",
25
+ "n": 500,
26
+ "argmax_agreement": 0.996,
27
+ "cuda_same_subset": {
28
+ "accuracy": 0.7910177949703642,
29
+ "macro_f1": 0.7760857891780776,
30
+ "nll": 0.6000106706289864,
31
+ "brier": 0.317876961372947,
32
+ "ece": 0.16696329399611096
33
+ },
34
+ "gguf_same_subset": {
35
+ "accuracy": 0.7936151975677668,
36
+ "macro_f1": 0.77850284511292,
37
+ "nll": 0.6007551256850553,
38
+ "brier": 0.31813150017828207,
39
+ "ece": 0.16639447171288832
40
+ },
41
+ "accuracy_difference": 0.0025974025974025983,
42
+ "nll_difference": 0.0007444550560689045,
43
+ "temperature_folded": true,
44
+ "runtime_temperature": 1.0,
45
+ "calibration_n": 3100,
46
+ "criteria": {
47
+ "agreement_min": 0.99,
48
+ "accuracy_loss_max": 0.005,
49
+ "nll_increase_max": 0.015
50
+ },
51
+ "calibration_backend": "llama.cpp batched, checked against native readout",
52
+ "evaluation_backend": "llama.cpp native single-sequence exact option logits",
53
+ "weight_bytes": 3775700896,
54
+ "passed": true
55
+ }
56
+ }
Jev-Style-v2-BF16-Calibrated.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8baa111eec6e30a5c9e97d559b53127a19ef30171e8aed26673153b4f77adcbb
3
+ size 3775700896
Jev-Style-v2-Q4_K_M-Calibrated.calibration.json ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "temperature": 1.0,
3
+ "fitted_temperature_folded": 1.0123069568523906,
4
+ "temperature_folded": true,
5
+ "folded_tensor": "output_norm.weight",
6
+ "source_calibration": {
7
+ "temperature": 1.0123069568523906,
8
+ "calibration_n": 3100,
9
+ "objective": "sample_mean_soft_cross_entropy",
10
+ "bounds": [
11
+ 0.05,
12
+ 20
13
+ ],
14
+ "nll_before": 0.5255198219301063,
15
+ "nll_after": 0.5254917547798398,
16
+ "backend": "gguf",
17
+ "model": "source_project/h100_v2/results/Jev-Style-v2-Q4_K_M.gguf",
18
+ "source_sha256": "2fde7f45dce3440abfde145bb30ad61e2643b1f853866b5760b235685328dc1c"
19
+ },
20
+ "input_sha256": "88982768b42a345708c93347ed49b1efebf1be366efe61c202d7ca1e931822fd",
21
+ "output_sha256": "c697d3b29d07fdd37b6ebeb5c98066f4c31632162adca6258db23f75184eb0c4",
22
+ "validation_required": false,
23
+ "validation": {
24
+ "format": "Q4_K_M",
25
+ "n": 500,
26
+ "argmax_agreement": 0.914,
27
+ "cuda_same_subset": {
28
+ "accuracy": 0.7910177949703642,
29
+ "macro_f1": 0.7760857891780776,
30
+ "nll": 0.6000106706289864,
31
+ "brier": 0.317876961372947,
32
+ "ece": 0.16696329399611096
33
+ },
34
+ "gguf_same_subset": {
35
+ "accuracy": 0.7818348025857907,
36
+ "macro_f1": 0.764930756852403,
37
+ "nll": 0.6474576399658714,
38
+ "brier": 0.3418843970562367,
39
+ "ece": 0.17704328656854143
40
+ },
41
+ "accuracy_difference": -0.00918299238457343,
42
+ "nll_difference": 0.04744696933688497,
43
+ "temperature_folded": true,
44
+ "runtime_temperature": 1.0,
45
+ "calibration_n": 3100,
46
+ "criteria": {
47
+ "agreement_min": 0.9,
48
+ "accuracy_loss_max": 0.03,
49
+ "nll_increase_max": 0.1
50
+ },
51
+ "calibration_backend": "llama.cpp batched, checked against native readout",
52
+ "evaluation_backend": "llama.cpp native single-sequence exact option logits",
53
+ "weight_bytes": 1274388384,
54
+ "passed": true
55
+ }
56
+ }
Jev-Style-v2-Q4_K_M-Calibrated.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c697d3b29d07fdd37b6ebeb5c98066f4c31632162adca6258db23f75184eb0c4
3
+ size 1274388384
README.md CHANGED
@@ -14,17 +14,26 @@ tags:
14
  - jev-style
15
  - single-prefill
16
  ---
17
- # Jev-Style-Qwen3.5-2B-Decision v2 (GGUF Q8_0)
18
 
19
  A **Jev-style decision model** for classification, routing and typed choices. Give it a state, a question and a list of options; one prefill returns a selected option **with calibrated probabilities**.
20
 
21
  | Build | Weight size | Inference |
22
  |---|---:|---|
23
  | [HF BF16](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2) | 3.76 GB | Transformers + decision client |
24
- | **GGUF Q8_0 · this repository** | 2.01 GB | llama.cpp + decision client |
25
  | [MLX BF16](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-MLX-bf16) | 3.76 GB | Apple Silicon + native MLX client |
26
 
27
- **Download this build:** [Jev-Style-v2-Q8_0-Calibrated.gguf](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/resolve/main/Jev-Style-v2-Q8_0-Calibrated.gguf?download=true). The repository also includes its calibration, inference client and evaluation records.
 
 
 
 
 
 
 
 
 
28
 
29
  ## Results
30
 
@@ -91,11 +100,15 @@ The separate typed-decisions group contains 2,000 teacher-reference decisions fr
91
 
92
  ## Deployment validation
93
 
 
 
94
  | Released format | Weight size | Validated result | Evaluation set |
95
  |---|---:|---|---|
96
  | HF BF16 | 3.76 GB | **81.27%** real-label macro accuracy | Full 3,277 real-label decisions |
97
  | Native MLX BF16 | 3.76 GB | **99.6%** choice agreement with CUDA BF16 | Frozen 500-decision deployment subset |
98
  | Calibrated GGUF Q8_0 | 2.01 GB | **99.2%** choice agreement with CUDA BF16 | Same 500-decision deployment subset |
 
 
99
 
100
  Each deployment format has its own validation record. Native MLX packaging reproduces the verified MLX client's logits exactly on all 500 deployment cases. The Q8_0 model is approximately **46.7% smaller** than the BF16 GGUF export.
101
 
@@ -133,6 +146,8 @@ python jev_decision_client.py --url http://127.0.0.1:8080 \
133
  --options negative positive
134
  ```
135
 
 
 
136
  Use a llama.cpp build with Qwen3.5 support. Conversion and native evaluation used commit `b29c606e28a01b1bc8c1351026a0fa6e616bf6c4`. The client uses the native `/completion` endpoint, requests complete declared-option log-probabilities and increases the candidate count as needed. The supplied `gguf_logits.cpp` reads all declared-option logits directly through the C API.
137
 
138
  **Runtime calibration temperature is 1.0** for this file: its fitted temperature has already been incorporated. The accompanying calibration JSON records the exact settings and checksum. Serve the raw decision prompt shown below, with the full declared option list.
 
14
  - jev-style
15
  - single-prefill
16
  ---
17
+ # Jev-Style-Qwen3.5-2B-Decision v2 (GGUF)
18
 
19
  A **Jev-style decision model** for classification, routing and typed choices. Give it a state, a question and a list of options; one prefill returns a selected option **with calibrated probabilities**.
20
 
21
  | Build | Weight size | Inference |
22
  |---|---:|---|
23
  | [HF BF16](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2) | 3.76 GB | Transformers + decision client |
24
+ | **GGUF · this repository** | 1.27–3.78 GB | Q4_K_M / Q8_0 / BF16 · llama.cpp |
25
  | [MLX BF16](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-MLX-bf16) | 3.76 GB | Apple Silicon + native MLX client |
26
 
27
+ **Choose a precision:**
28
+
29
+ | Precision | File size | Choice agreement | Macro accuracy |
30
+ |---|---:|---:|---:|
31
+ | [Q4_K_M](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/resolve/main/Jev-Style-v2-Q4_K_M-Calibrated.gguf?download=true) | 1.27 GB | 91.4% | 78.18% |
32
+ | [Q8_0](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/resolve/main/Jev-Style-v2-Q8_0-Calibrated.gguf?download=true) | 2.01 GB | 99.2% | 78.69% |
33
+ | [BF16](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/resolve/main/Jev-Style-v2-BF16-Calibrated.gguf?download=true) | 3.78 GB | 99.6% | 79.36% |
34
+
35
+ All three files include independently fitted calibration; use **runtime temperature 1.0**. Agreement is against CUDA merged BF16 on the same frozen 500-decision subset; accuracy is the task-macro average over its real-label examples. [Full precision comparison](evaluation/quantization_summary.json).
36
+
37
 
38
  ## Results
39
 
 
100
 
101
  ## Deployment validation
102
 
103
+ GGUF is available in **Q4_K_M, Q8_0 and BF16**, each with its own calibration and verification record.
104
+
105
  | Released format | Weight size | Validated result | Evaluation set |
106
  |---|---:|---|---|
107
  | HF BF16 | 3.76 GB | **81.27%** real-label macro accuracy | Full 3,277 real-label decisions |
108
  | Native MLX BF16 | 3.76 GB | **99.6%** choice agreement with CUDA BF16 | Frozen 500-decision deployment subset |
109
  | Calibrated GGUF Q8_0 | 2.01 GB | **99.2%** choice agreement with CUDA BF16 | Same 500-decision deployment subset |
110
+ | Calibrated GGUF Q4_K_M | 1.27 GB | **91.4%** choice agreement with CUDA BF16 | Same 500-decision deployment subset |
111
+ | Calibrated GGUF BF16 | 3.78 GB | **99.6%** choice agreement with CUDA BF16 | Same 500-decision deployment subset |
112
 
113
  Each deployment format has its own validation record. Native MLX packaging reproduces the verified MLX client's logits exactly on all 500 deployment cases. The Q8_0 model is approximately **46.7% smaller** than the BF16 GGUF export.
114
 
 
146
  --options negative positive
147
  ```
148
 
149
+ The quick-start command selects Q8_0. To use Q4_K_M or BF16, replace its model filename with `Jev-Style-v2-Q4_K_M-Calibrated.gguf` or `Jev-Style-v2-BF16-Calibrated.gguf`.
150
+
151
  Use a llama.cpp build with Qwen3.5 support. Conversion and native evaluation used commit `b29c606e28a01b1bc8c1351026a0fa6e616bf6c4`. The client uses the native `/completion` endpoint, requests complete declared-option log-probabilities and increases the candidate count as needed. The supplied `gguf_logits.cpp` reads all declared-option logits directly through the C API.
152
 
153
  **Runtime calibration temperature is 1.0** for this file: its fitted temperature has already been incorporated. The accompanying calibration JSON records the exact settings and checksum. Serve the raw decision prompt shown below, with the full declared option list.
SHA256SUMS.json CHANGED
@@ -12,8 +12,8 @@
12
  "sha256": "50cbab8a892c5f2993b8c7351a99182507472def3b1374558308605d99b86b32"
13
  },
14
  "README.md": {
15
- "bytes": 10788,
16
- "sha256": "e2fc4983b02dec5266c219b9333e0ef59dcdb88a4d9393eb686a734036c73963"
17
  },
18
  "evaluation/baseline_sensitivity.json": {
19
  "bytes": 2015,
@@ -82,5 +82,45 @@
82
  "figures/calibration.png": {
83
  "bytes": 183790,
84
  "sha256": "2a4e163354b0d0d31bf5fcdfcfbcf726fb152fb2f6419432ddddaec3b475a921"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
85
  }
86
  }
 
12
  "sha256": "50cbab8a892c5f2993b8c7351a99182507472def3b1374558308605d99b86b32"
13
  },
14
  "README.md": {
15
+ "bytes": 11957,
16
+ "sha256": "37b0885c870ef69cf315d4e8bdae3e09c81972f0fd70d4417284fef50a83ddda"
17
  },
18
  "evaluation/baseline_sensitivity.json": {
19
  "bytes": 2015,
 
82
  "figures/calibration.png": {
83
  "bytes": 183790,
84
  "sha256": "2a4e163354b0d0d31bf5fcdfcfbcf726fb152fb2f6419432ddddaec3b475a921"
85
+ },
86
+ "evaluation/quantization_summary.json": {
87
+ "bytes": 5465,
88
+ "sha256": "45c350f225f9a403d036d96a80ef14ca9cd89bb9c25af82617071833846eefd6"
89
+ },
90
+ "evaluation/calibration_batch_parity.json": {
91
+ "bytes": 147,
92
+ "sha256": "1f819d59b9ebf6ffb79953832c3f7ae4564d7bae7fa68209b6583a9d92db2bf5"
93
+ },
94
+ "Jev-Style-v2-Q4_K_M-Calibrated.gguf": {
95
+ "bytes": 1274388384,
96
+ "sha256": "c697d3b29d07fdd37b6ebeb5c98066f4c31632162adca6258db23f75184eb0c4"
97
+ },
98
+ "Jev-Style-v2-Q4_K_M-Calibrated.calibration.json": {
99
+ "bytes": 1824,
100
+ "sha256": "4a5abb23d86ddd44b8b6b1054b48a0a9c40560cc0ee3edd24ae6be8e623ad5f1"
101
+ },
102
+ "evaluation/q4_k_m_deployment.json": {
103
+ "bytes": 941,
104
+ "sha256": "b67e128b7e46287a84ce487171882b4b8ece888c28cfa92d09e72fefec298bcc"
105
+ },
106
+ "evaluation/q4_k_m_http_smoke.json": {
107
+ "bytes": 694,
108
+ "sha256": "5a0e2b0696e3aa918aff6e8ad2ebf1532091edf81a10815bf3add221c36932d3"
109
+ },
110
+ "Jev-Style-v2-BF16-Calibrated.gguf": {
111
+ "bytes": 3775700896,
112
+ "sha256": "8baa111eec6e30a5c9e97d559b53127a19ef30171e8aed26673153b4f77adcbb"
113
+ },
114
+ "Jev-Style-v2-BF16-Calibrated.calibration.json": {
115
+ "bytes": 1840,
116
+ "sha256": "e5eb54d405d213c751ba61b334a8370d03332c915d9c49725da3b7a6ee295109"
117
+ },
118
+ "evaluation/bf16_deployment.json": {
119
+ "bytes": 946,
120
+ "sha256": "19cbc98b2c892509f804b4b758ab5009dbcd5fef9807b4cbedb4f7114a4efab2"
121
+ },
122
+ "evaluation/bf16_http_smoke.json": {
123
+ "bytes": 690,
124
+ "sha256": "d729e4dd7ab5d425e5eb031ca1dc92d74302a7971d9aeeba782c15e4904a5f59"
125
  }
126
  }
evaluation/bf16_deployment.json ADDED
@@ -0,0 +1,33 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "BF16",
3
+ "n": 500,
4
+ "argmax_agreement": 0.996,
5
+ "cuda_same_subset": {
6
+ "accuracy": 0.7910177949703642,
7
+ "macro_f1": 0.7760857891780776,
8
+ "nll": 0.6000106706289864,
9
+ "brier": 0.317876961372947,
10
+ "ece": 0.16696329399611096
11
+ },
12
+ "gguf_same_subset": {
13
+ "accuracy": 0.7936151975677668,
14
+ "macro_f1": 0.77850284511292,
15
+ "nll": 0.6007551256850553,
16
+ "brier": 0.31813150017828207,
17
+ "ece": 0.16639447171288832
18
+ },
19
+ "accuracy_difference": 0.0025974025974025983,
20
+ "nll_difference": 0.0007444550560689045,
21
+ "temperature_folded": true,
22
+ "runtime_temperature": 1.0,
23
+ "calibration_n": 3100,
24
+ "criteria": {
25
+ "agreement_min": 0.99,
26
+ "accuracy_loss_max": 0.005,
27
+ "nll_increase_max": 0.015
28
+ },
29
+ "calibration_backend": "llama.cpp batched, checked against native readout",
30
+ "evaluation_backend": "llama.cpp native single-sequence exact option logits",
31
+ "weight_bytes": 3775700896,
32
+ "passed": true
33
+ }
evaluation/bf16_http_smoke.json ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "passed": true,
3
+ "model": "Jev-Style-v2-BF16-Calibrated.gguf",
4
+ "cases": [
5
+ {
6
+ "options": 2,
7
+ "choice": "positive",
8
+ "candidate_count": 64,
9
+ "max_probability_difference_vs_native": 5.900205922393376e-06
10
+ },
11
+ {
12
+ "options": 4,
13
+ "choice": "science and technology",
14
+ "candidate_count": 64,
15
+ "max_probability_difference_vs_native": 0.0006067566569580851
16
+ },
17
+ {
18
+ "options": 5,
19
+ "choice": "neutral",
20
+ "candidate_count": 64,
21
+ "max_probability_difference_vs_native": 0.001473201814370717
22
+ }
23
+ ],
24
+ "server": "version: 0.4.1 (build 10964, commit b29c606e2)\nbuilt with AppleClang 17.0.0.17000604 for Darwin arm64"
25
+ }
evaluation/calibration_batch_parity.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "n": 32,
3
+ "max_probability_difference": 3.980138511150422e-05,
4
+ "argmax_agreement": 1.0,
5
+ "seconds": 13.387850046157837,
6
+ "passed": true
7
+ }
evaluation/q4_k_m_deployment.json ADDED
@@ -0,0 +1,33 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "Q4_K_M",
3
+ "n": 500,
4
+ "argmax_agreement": 0.914,
5
+ "cuda_same_subset": {
6
+ "accuracy": 0.7910177949703642,
7
+ "macro_f1": 0.7760857891780776,
8
+ "nll": 0.6000106706289864,
9
+ "brier": 0.317876961372947,
10
+ "ece": 0.16696329399611096
11
+ },
12
+ "gguf_same_subset": {
13
+ "accuracy": 0.7818348025857907,
14
+ "macro_f1": 0.764930756852403,
15
+ "nll": 0.6474576399658714,
16
+ "brier": 0.3418843970562367,
17
+ "ece": 0.17704328656854143
18
+ },
19
+ "accuracy_difference": -0.00918299238457343,
20
+ "nll_difference": 0.04744696933688497,
21
+ "temperature_folded": true,
22
+ "runtime_temperature": 1.0,
23
+ "calibration_n": 3100,
24
+ "criteria": {
25
+ "agreement_min": 0.9,
26
+ "accuracy_loss_max": 0.03,
27
+ "nll_increase_max": 0.1
28
+ },
29
+ "calibration_backend": "llama.cpp batched, checked against native readout",
30
+ "evaluation_backend": "llama.cpp native single-sequence exact option logits",
31
+ "weight_bytes": 1274388384,
32
+ "passed": true
33
+ }
evaluation/q4_k_m_http_smoke.json ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "passed": true,
3
+ "model": "Jev-Style-v2-Q4_K_M-Calibrated.gguf",
4
+ "cases": [
5
+ {
6
+ "options": 2,
7
+ "choice": "positive",
8
+ "candidate_count": 64,
9
+ "max_probability_difference_vs_native": 8.535655050662637e-07
10
+ },
11
+ {
12
+ "options": 4,
13
+ "choice": "science and technology",
14
+ "candidate_count": 64,
15
+ "max_probability_difference_vs_native": 1.7518853354103747e-05
16
+ },
17
+ {
18
+ "options": 5,
19
+ "choice": "neutral",
20
+ "candidate_count": 64,
21
+ "max_probability_difference_vs_native": 9.285758503529973e-05
22
+ }
23
+ ],
24
+ "server": "version: 0.4.1 (build 10964, commit b29c606e2)\nbuilt with AppleClang 17.0.0.17000604 for Darwin arm64"
25
+ }
evaluation/quantization_summary.json ADDED
@@ -0,0 +1,150 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "evaluation_decisions": 500,
3
+ "comparison": "same frozen subset vs CUDA merged BF16",
4
+ "metric": "real-label task-macro accuracy; teacher-reference decisions excluded from this macro",
5
+ "variants": {
6
+ "Q4_K_M": {
7
+ "filename": "Jev-Style-v2-Q4_K_M-Calibrated.gguf",
8
+ "weight_bytes": 1274388384,
9
+ "sha256": "c697d3b29d07fdd37b6ebeb5c98066f4c31632162adca6258db23f75184eb0c4",
10
+ "argmax_agreement": 0.914,
11
+ "real_label_macro_accuracy": 0.7818348025857907,
12
+ "real_label_macro": {
13
+ "accuracy": 0.7818348025857907,
14
+ "macro_f1": 0.764930756852403,
15
+ "nll": 0.6474576399658714,
16
+ "brier": 0.3418843970562367,
17
+ "ece": 0.17704328656854143
18
+ },
19
+ "calibration_n": 3100,
20
+ "temperature_folded": true,
21
+ "runtime_temperature": 1.0,
22
+ "validation": {
23
+ "format": "Q4_K_M",
24
+ "n": 500,
25
+ "argmax_agreement": 0.914,
26
+ "cuda_same_subset": {
27
+ "accuracy": 0.7910177949703642,
28
+ "macro_f1": 0.7760857891780776,
29
+ "nll": 0.6000106706289864,
30
+ "brier": 0.317876961372947,
31
+ "ece": 0.16696329399611096
32
+ },
33
+ "gguf_same_subset": {
34
+ "accuracy": 0.7818348025857907,
35
+ "macro_f1": 0.764930756852403,
36
+ "nll": 0.6474576399658714,
37
+ "brier": 0.3418843970562367,
38
+ "ece": 0.17704328656854143
39
+ },
40
+ "accuracy_difference": -0.00918299238457343,
41
+ "nll_difference": 0.04744696933688497,
42
+ "temperature_folded": true,
43
+ "runtime_temperature": 1.0,
44
+ "calibration_n": 3100,
45
+ "criteria": {
46
+ "agreement_min": 0.9,
47
+ "accuracy_loss_max": 0.03,
48
+ "nll_increase_max": 0.1
49
+ },
50
+ "calibration_backend": "llama.cpp batched, checked against native readout",
51
+ "evaluation_backend": "llama.cpp native single-sequence exact option logits",
52
+ "weight_bytes": 1274388384,
53
+ "passed": true
54
+ }
55
+ },
56
+ "Q8_0": {
57
+ "filename": "Jev-Style-v2-Q8_0-Calibrated.gguf",
58
+ "weight_bytes": 2012004256,
59
+ "sha256": "5c2aa0d35b24a27f03228b2c62ebaaebd9b5b785844634d4217278d822751494",
60
+ "argmax_agreement": 0.992,
61
+ "real_label_macro_accuracy": 0.7868855635654054,
62
+ "real_label_macro": {
63
+ "accuracy": 0.7868855635654054,
64
+ "macro_f1": 0.773615445189482,
65
+ "nll": 0.6015417570028033,
66
+ "brier": 0.318930434743987,
67
+ "ece": 0.16630794459650552
68
+ },
69
+ "calibration_n": 3100,
70
+ "temperature_folded": true,
71
+ "runtime_temperature": 1.0,
72
+ "validation": {
73
+ "n": 500,
74
+ "argmax_agreement": 0.992,
75
+ "cuda_same_subset": {
76
+ "accuracy": 0.7910177949703642,
77
+ "macro_f1": 0.7760857891780776,
78
+ "nll": 0.6000106706289864,
79
+ "brier": 0.317876961372947,
80
+ "ece": 0.16696329399611096
81
+ },
82
+ "q8_same_subset": {
83
+ "accuracy": 0.7868855635654054,
84
+ "macro_f1": 0.773615445189482,
85
+ "nll": 0.6015417570028033,
86
+ "brier": 0.318930434743987,
87
+ "ece": 0.16630794459650552
88
+ },
89
+ "accuracy_difference": -0.004132231404958775,
90
+ "nll_difference": 0.0015310863738169367,
91
+ "temperature_folded": true,
92
+ "passed": true,
93
+ "practical_deployment_gate_passed": true,
94
+ "initial_strict_accuracy_target_met_on_subset": false,
95
+ "initial_accuracy_loss_target": 0.003,
96
+ "note": "500-example subset check: 99.2% agreement. Observed task-macro loss is 0.413 pp, so the initial 0.3 pp accuracy goal is not met on this subset. Prefer BF16/MLX when accuracy is the priority; this is not a bound on population degradation."
97
+ }
98
+ },
99
+ "BF16": {
100
+ "filename": "Jev-Style-v2-BF16-Calibrated.gguf",
101
+ "weight_bytes": 3775700896,
102
+ "sha256": "8baa111eec6e30a5c9e97d559b53127a19ef30171e8aed26673153b4f77adcbb",
103
+ "argmax_agreement": 0.996,
104
+ "real_label_macro_accuracy": 0.7936151975677668,
105
+ "real_label_macro": {
106
+ "accuracy": 0.7936151975677668,
107
+ "macro_f1": 0.77850284511292,
108
+ "nll": 0.6007551256850553,
109
+ "brier": 0.31813150017828207,
110
+ "ece": 0.16639447171288832
111
+ },
112
+ "calibration_n": 3100,
113
+ "temperature_folded": true,
114
+ "runtime_temperature": 1.0,
115
+ "validation": {
116
+ "format": "BF16",
117
+ "n": 500,
118
+ "argmax_agreement": 0.996,
119
+ "cuda_same_subset": {
120
+ "accuracy": 0.7910177949703642,
121
+ "macro_f1": 0.7760857891780776,
122
+ "nll": 0.6000106706289864,
123
+ "brier": 0.317876961372947,
124
+ "ece": 0.16696329399611096
125
+ },
126
+ "gguf_same_subset": {
127
+ "accuracy": 0.7936151975677668,
128
+ "macro_f1": 0.77850284511292,
129
+ "nll": 0.6007551256850553,
130
+ "brier": 0.31813150017828207,
131
+ "ece": 0.16639447171288832
132
+ },
133
+ "accuracy_difference": 0.0025974025974025983,
134
+ "nll_difference": 0.0007444550560689045,
135
+ "temperature_folded": true,
136
+ "runtime_temperature": 1.0,
137
+ "calibration_n": 3100,
138
+ "criteria": {
139
+ "agreement_min": 0.99,
140
+ "accuracy_loss_max": 0.005,
141
+ "nll_increase_max": 0.015
142
+ },
143
+ "calibration_backend": "llama.cpp batched, checked against native readout",
144
+ "evaluation_backend": "llama.cpp native single-sequence exact option logits",
145
+ "weight_bytes": 3775700896,
146
+ "passed": true
147
+ }
148
+ }
149
+ }
150
+ }