chaoliangUNSW commited on
Commit
656ca59
·
verified ·
1 Parent(s): b96f211

Release Jev-Style-0.8B-Decision-v3

Browse files
.gitattributes CHANGED
@@ -33,3 +33,14 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ figures/beyond_laya.png filter=lfs diff=lfs merge=lfs -text
37
+ figures/calibration.png filter=lfs diff=lfs merge=lfs -text
38
+ figures/design_table.png filter=lfs diff=lfs merge=lfs -text
39
+ figures/headline_typed.png filter=lfs diff=lfs merge=lfs -text
40
+ figures/jevbench.png filter=lfs diff=lfs merge=lfs -text
41
+ figures/latency.png filter=lfs diff=lfs merge=lfs -text
42
+ figures/long_context.png filter=lfs diff=lfs merge=lfs -text
43
+ figures/multilingual.png filter=lfs diff=lfs merge=lfs -text
44
+ figures/quantization.png filter=lfs diff=lfs merge=lfs -text
45
+ figures/zeroshot.png filter=lfs diff=lfs merge=lfs -text
46
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
NOTICE ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ chaoliangUNSW/Jev-Style-0.8B-Decision-v3
2
+ Copyright 2026 chaoliangUNSW. Licensed under the Apache License, Version 2.0 (see LICENSE).
3
+
4
+ This model is a fine-tuned derivative of Qwen3.5-0.8B (https://huggingface.co/Qwen/Qwen3.5-0.8B,
5
+ revision 2fc06364715b967f1860aea9cf38778875588b17), Copyright 2026 Alibaba Cloud, licensed under the Apache
6
+ License, Version 2.0. The LICENSE file in this repository is the license file distributed with Qwen3.5-0.8B.
7
+
8
+ Modifications relative to Qwen3.5-0.8B:
9
+ - all text-model weights were fine-tuned (full fine-tuning, bf16 training) to score typed decision questions
10
+ (choice / score / true-false) with a verdict readout: logit(" yes") - logit(" no") at one " ->" slot per
11
+ option; calibration temperatures were fitted afterwards (readout_config.json);
12
+ - the vision tower (model.visual.*) and the multi-token-prediction head (mtp.*) were removed; the checkpoint
13
+ is a text-only Qwen3_5ForCausalLM with tied input/output embeddings;
14
+ - added the runtime script, readout/release configuration files, the integrity manifest and this NOTICE.
15
+
16
+ The question types (choice / score / noul) follow the typed-decision convention of Laya
17
+ (https://github.com/NandhaKishorM/laya, Apache-2.0) so both models can be evaluated on the same
18
+ inputs. No Laya code or weights are included.
19
+
20
+ Third generation (v3) of the Jev-Style decision series. Earlier generations: v1 =
21
+ chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision (public GGUF release: chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF) and
22
+ v2 = chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2, both 2B models built on Qwen3.5-2B-Base. v3 is fine-tuned from
23
+ Qwen3.5-0.8B; no weights of v1 or v2 were reused.
24
+
25
+ Not affiliated with, endorsed by or connected to TypeSafe or Jev. "Jev-Style" only describes the kind
26
+ of model (a small typed-decision model in a similar style); no Jev weights, code or outputs are included.
27
+ Not affiliated with or endorsed by Alibaba Cloud / the Qwen team or the Laya authors.
README.md ADDED
@@ -0,0 +1,448 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ base_model: Qwen/Qwen3.5-0.8B
4
+ base_model_relation: finetune
5
+ library_name: transformers
6
+ pipeline_tag: text-classification
7
+ language:
8
+ - en
9
+ - zh
10
+ - ar
11
+ - bg
12
+ - de
13
+ - el
14
+ - es
15
+ - fr
16
+ - hi
17
+ - ja
18
+ - ko
19
+ - pt
20
+ - ru
21
+ - sw
22
+ - ta
23
+ - th
24
+ - tr
25
+ - ur
26
+ - vi
27
+ tags:
28
+ - decision-model
29
+ - jev-style
30
+ - system-one
31
+ - calibration
32
+ - classification
33
+ - long-context
34
+ - multilingual
35
+ - qwen3.5
36
+ ---
37
+
38
+ # Jev-Style-0.8B-Decision-v3
39
+
40
+ **Jev-Style decision series:** [v1 · 2B](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF) → [v2 · 2B](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2) → **v3 · 0.8B (this model)** · **Website:** [jevstyle.com](https://jevstyle.com)
41
+
42
+ **One state. One pass. Every option scored.**
43
+
44
+ Jev-Style v3 is a 0.8B decision model. It takes inputs of up to 25,600 tokens, reads the state once and returns a calibrated
45
+ probability for every option of every question you ask about it. There is no letter cap on options, and it was
46
+ evaluated in 51 languages.
47
+
48
+ | **0.8B** | **25,600 tokens** | **77 options** | **51 languages** | **0.53 GB** |
49
+ |:---:|:---:|:---:|:---:|:---:|
50
+ | parameters, full fine-tune | input, preregistered 25K claim passed | scored in one pass (largest tested) | evaluated on MASSIVE intent | 4-bit Q4_K_M; same top-1 as FP32 on 240/240 parity rows |
51
+
52
+ | Build | Size | Runtime |
53
+ |---|---:|---|
54
+ | **Transformers safetensors (bf16) · this repository** | 1.50 GB | PyTorch on CUDA, Apple MPS or CPU (`jev_style_decision.py`) |
55
+ | [GGUF F16 / Q8_0 / Q4_K_M](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-GGUF) | 1.52 / 0.81 / 0.53 GB | llama.cpp + the bundled `jev-score` scorer |
56
+ | [MLX bf16](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-bf16) | 1.50 GB | Apple silicon, mlx-lm |
57
+ | [MLX 8-bit](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-8bit) | 0.80 GB | Apple silicon, mlx-lm |
58
+
59
+ ## Highlights
60
+
61
+ - **79.2% on 2,000 typed decisions: +6.4 points over Jev, +5.7 over our 2B v2, +2.6 over Laya's typed
62
+ checkpoint.** Its Brier score is **3.2× lower than Jev's** (0.046 vs 0.148). v3 and Laya typed trained on
63
+ this dataset's train split; Jev's number is zero-shot, from the dataset card.
64
+ - **Up to +30.3 points over the best official Laya checkpoint on five decision tasks**, on identical rows: +30.3 on model routing,
65
+ +29.5 macro-F1 on toxicity, +29.4 on the 37 locales held out of MASSIVE training, +19.0 on 77-way Banking77
66
+ and +7.2 balanced accuracy on jailbreak detection. Every paired 95% CI excludes zero.
67
+ - **Ahead of Laya multilingual in 51 of 51 languages.** MASSIVE intent macro accuracy is 71.7% vs 40.1% (+31.7 points),
68
+ and v3 stays above 3× chance in every language.
69
+ - **4.5× lower NLL, 3.0× lower Brier and 5.5× lower ECE** than the best Laya checkpoint as deployed (with its shipped
70
+ temperatures), across 49 suites and 17,416 paired rows.
71
+ - **25,600-token inputs, with the preregistered 25K claim passed.** At 24K tokens v3 answers **98.3%** of 1,280
72
+ real items correctly. The question-only and state-swap controls stay at chance, and the preregistered controlled
73
+ accuracy is within 1.1 points of the 2K–4K reference.
74
+ - **64.1% zero-shot on JevBench v1.4.1**, ahead of Laya (58.4%) and every Qwen3.5-0.8B-based system on the board
75
+ (point estimates on 231 public items). On two topic sets it never trained on, v3 leads English Laya by +12.3 and
76
+ +12.5 points, and it comes within 4 points of Jev on tweet_topic.
77
+ - **Up to 4.6× faster than a Laya-architecture engine when 10 questions share one 4K-token state** (1,381 ms vs
78
+ 6,364 ms with the GGUF runtime's `many_mode="batched"`; the engine is our round-1 MacLaya-4K, one call per question,
79
+ not an official Laya checkpoint). The
80
+ **0.53 GB Q4_K_M file matched full precision on 240 of 240 parity rows** (plus 6 of 6 at 16K and 25.6K tokens).
81
+
82
+ Each result below states its protocol and source under the figure.
83
+
84
+ ## Results
85
+
86
+ ### Typed decisions: 0.8B beats the 2B models and Jev
87
+
88
+ ![Typed-decisions accuracy and Brier score: Jev-Style 0.8B v3 vs Jev, Laya typed and the 2B v1/v2](figures/headline_typed.png)
89
+
90
+ At 0.8B parameters, v3 scores **79.2%** on the 2,000 typed decisions. That is **+6.4 points over Jev**, +5.7 over
91
+ our 2B v2 and +2.6 over Laya's typed checkpoint, and the Brier score is **3.2× lower than Jev's** (0.046 vs 0.148).
92
+
93
+ <sub>Typed-decisions test set (LocalLLaMA/typed-decisions), 2,000 decisions from 400 states. In-domain for v3 and Laya typed (both trained on its train split); zero-shot for Jev (numbers from the dataset card, measured through the Jev API on all 2,000 decisions). Laya: official typed-decisions checkpoint re-run by us on identical rows with its shipped temperature. 2B v1/v2: teacher agreement as reported on the v2 card (same 2,000 decisions, scored by that card's harness; v1 was not trained on typed decisions, v2's training pool included typed workflow decisions). Jev's accuracy is published as 0.727, so the gap is 6.40–6.50 points. v3: 1,583 / 2,000 correct, 95% CI 77.3–80.9% (Wilson); v3 minus Laya typed, paired bootstrap 95% CI +1.0 to +4.2 points.</sub>
94
+
95
+ **Head-to-head against Laya's typed checkpoint.** Both models trained on this dataset's train split, and v3 wins on all four
96
+ metrics, each with a paired 95% CI that excludes zero:
97
+
98
+ | Metric (2,000 decisions) | Laya typed-decisions checkpoint | **Jev-Style 0.8B v3** | Difference, paired 95% CI |
99
+ |---|---:|---:|---|
100
+ | Accuracy ↑ | 76.6% | **79.2%** | +2.6 pts [+1.0, +4.2] |
101
+ | Soft accuracy ↑ | 47.1% | **52.4%** | +5.4 pts [+5.0, +5.7] |
102
+ | Brier vs soft labels ↓ | 0.061 | **0.046** | −0.016 [−0.019, −0.012] |
103
+ | Score-question MAE ↓ | 0.242 | **0.195** | −0.047 [−0.060, −0.035] |
104
+
105
+ <sub>Both in-domain; v3 also trained on 27,300 synthetic typed items from other workflows. Laya: official checkpoint re-run by us on identical rows with its shipped temperature. Paired case-cluster bootstrap within suites, 2,000 resamples.</sub>
106
+
107
+ ### Beyond Laya: up to +30 points
108
+
109
+ ![v3 vs the best official Laya checkpoint on five decision tasks](figures/beyond_laya.png)
110
+
111
+ **On five decision tasks scored on identical rows, the 0.8B v3 beats the best official Laya checkpoint on every
112
+ one:** +19.0 points on 77-way Banking77, +7.2 balanced accuracy on jailbreak detection, +29.5 macro-F1 on
113
+ toxicity, +30.3 on model routing and +29.4 across the 37 locales held out of MASSIVE training.
114
+
115
+ <sub>Laya numbers: official checkpoints (English, typed-decisions, multilingual) re-run by us on identical rows with their shipped temperatures and default token budgets; the best of the three is shown per task. v3 trained on tasks of the same kind from other datasets, never on these evaluation rows: intent (CLINC150/HWU64; Banking77 never trained), jailbreak (other permissive sets plus teacher data), toxicity (civil_comments plus teacher data; toxic-chat is evaluation-only), routing (teacher-written; the gsm8k/mbpp/AG rows are evaluation-only), MASSIVE in 14 other locales (no MASSIVE rows in these 37). n = 400 / 400 / 400 / 399 / 3,700 (37 × 100). Every gap's paired 95% bootstrap CI excludes zero.</sub>
116
+
117
+ ### 51 languages, 51 wins
118
+
119
+ ![Per-language MASSIVE intent accuracy, v3 vs Laya multilingual, 51 languages](figures/multilingual.png)
120
+
121
+ **One 0.8B model, 51 languages, 51 wins over Laya.** On MASSIVE intent (20 options per question) v3 averages
122
+ **71.7%** across 51 languages, against 40.1% for the official Laya multilingual checkpoint (+31.7 points). It
123
+ beats Laya multilingual in every one of the 51 languages, by at least 11 points, and stays above 3× chance in all
124
+ of them. That includes the 37 locales held out of MASSIVE training (65.5% vs 36.1%), 32 of them outside the 19
125
+ fine-tuning languages.
126
+
127
+ <sub>MASSIVE intent (mteb/amazon_massive_intent) test rows, 100 per language, 20 candidate intents per row (chance 5%, 3× chance 15%); accuracy = top-scored option. v3: in-domain for the 14 trained locales, held out for the other 37 (vi/th/el/ur had about 1.3K translated-NLI training rows each; zh-TW shares Chinese with zh-CN; a 69-row multilingual jailbreak set in training may include a few prompts in other held-out languages). Laya: official multilingual checkpoint re-run by us on identical rows with its shipped temperature and default token budget (held-out for Laya). v3 is also ahead of the best of the three official Laya checkpoints in all 51 languages (per-language point estimates on 100 rows each, smallest gap 10 points). Paired 95% CI of the 51-language macro difference: +30.1 to +33.1 points.</sub>
128
+
129
+ ### Probabilities you can act on
130
+
131
+ ![Macro NLL, Brier and ECE over 49 suites: v3 vs the best Laya checkpoint](figures/calibration.png)
132
+
133
+ **Across 49 suites and 17,416 identical rows, v3's probabilities beat the best official Laya checkpoint on all three
134
+ probability-quality metrics:** **4.5× lower NLL** (0.493 vs 2.213), **3.0× lower Brier** (0.239 vs 0.712) and
135
+ **5.5× lower ECE** (0.054 vs 0.299).
136
+
137
+ <sub>Macro average over 49 suites, 17,416 identical rows for both models. Mixed protocol for both v3 and Laya (in-domain on some suites, held-out on others). As deployed: v3 with its supplied temperatures; Laya with the shipped temperatures of its official checkpoints, re-run by us on identical rows. Best Laya = best of the three official checkpoints per metric (multilingual on all three). Paired case-cluster bootstrap 95% CIs (2,000 resamples) of every difference exclude zero.</sub>
138
+
139
+ **Stable under option shuffling.** When the options are presented in a different order, v3 changed its answer on
140
+ **1 of 200** option-order pairs (0.5%), against 22 of 200 (11.0%) for the best Laya checkpoint.
141
+
142
+ <sub>MASSIVE intent English, 200 option-permutation pairs, identical rows. In-domain for v3, held-out for Laya (typed-decisions checkpoint, the best of the three here). Paired 95% CI of the difference: −15.0 to −6.5 points.</sub>
143
+
144
+ ### Long context: flat from 1K to 24K tokens
145
+
146
+ ![Controlled accuracy by input length, 1K to 24K tokens](figures/long_context.png)
147
+
148
+ **v3 reads documents far past Laya's 512 / 1,024-token default budgets.** Controlled accuracy stays within 3.8
149
+ points across all seven length bins. At 24K it is 55.3%, 1.1 points from the 2K–4K reference (56.4%) and well
150
+ inside the preregistered ±5-point limit, so **the 25K claim passed**. In plain accuracy, v3 answers **98.3%**
151
+ (1,258 of 1,280) of the real 24K-token items correctly and at least 96.9% in every length bin. The same questions
152
+ with the state removed or swapped for another item's state fall to chance (28.9% and 28.5%, against 28.4% chance
153
+ at 24K), so the answers cannot be recovered from the question alone.
154
+
155
+ <sub>v3 only. Laya's default input budget is 512 tokens (English) / 1,024 (multilingual, typed) per the Laya README, so Laya is not plotted. Suite long_grid_plus, English and Chinese documents: preregistered 2026-09-24 and amended before any model was scored (+96 items per 24K depth decile, thresholds unchanged); 320 items per bin, 1,280 at 24K. Controlled accuracy = the real item is correct AND its question-only and state-swap controls pass; both controls are at chance in every length bin. 25K claim rule: |24K − 2K–4K reference| ≤ 5 points and every 24K evidence-depth decile within 10 points of it.</sub>
156
+
157
+ ### JevBench: ahead of Laya and every Qwen3.5-0.8B-based system
158
+
159
+ ![JevBench v1.4.1 public accuracy: v3 vs Laya and the Qwen3.5-0.8B-based systems](figures/jevbench.png)
160
+
161
+ **On the 231 public JevBench v1.4.1 items, v3 scores 64.1% zero-shot**: 5.6 points above Laya, and ahead of every
162
+ Qwen3.5-0.8B-based system on the board, including a dedicated 0.8B decision fine-tune (+4.8 points) and
163
+ SimpleJev on the same base (+9.5 points). Every answer is a valid option (231 of 231), because v3 can only score
164
+ the options it is given.
165
+
166
+ <sub>JevBench v1.4.1, public items only (231). v3: self-run zero-shot with the vendored official harness (commit 24b9b5c), 148 / 231 correct, 95% CI 57.7–70.0% (Wilson); training-pool contamination scan: 0 hits; not an official leaderboard entry. Other rows: public accuracy as published in the board's [v1.4.1 results file](https://github.com/fstandhartinger/jevbench). Shown: Laya plus every Qwen3.5-0.8B-based system on the board; other board systems are not shown. Laya's and M. Ghafiri's scores lie inside v3's 95% CI, so those two leads are point estimates, not significant at n = 231.</sub>
167
+
168
+ ### Zero-shot topics: +12 points over English Laya
169
+
170
+ ![Zero-shot tweet_topic and fin_topic accuracy: v3 vs English Laya, with Jev on tweet_topic](figures/zeroshot.png)
171
+
172
+ **On two topic sets it never trained on, v3 leads English Laya by +12.3 points on tweet_topic** (75.5% vs 63.2%)
173
+ **and +12.5 points on the 20-way fin_topic** (46.7% vs 34.2%). On tweet_topic it lands **within 4 points of Jev**
174
+ (75.5% vs 79.3%). Macro-F1 leads over English Laya are +13.8 points (59.9% vs 46.1%) and +8.9 points (45.2% vs 36.2%).
175
+ With its shipped temperature, its probabilities are also better calibrated than Jev's on both sets: ECE 0.027 vs
176
+ 0.063 on tweet_topic and 0.046 vs 0.166 on fin_topic.
177
+
178
+ <sub>Zero-shot for every system: neither set is in v3's training pool; accuracy over every row of the pinned test files (n = 1,693 and 4,117). Jev (1.13, API) and English Laya: numbers published by the [elcronos jev-vs-open-decision-models study](https://github.com/elcronos/jev-vs-open-decision-models) with its own prompt (results/cross_dataset_summary.json @ a1901bc), not re-run by us. v3: scored by us on the identical rows, label sets and instruction, in v3's own input format; tweet_topic accuracy 95% CI 73.4–77.5%. ECE: 15 equal-width bins as in the study; v3 with its shipped global temperature (0.880, fitted on v3's own calibration split, never on these sets), Jev's ECE as published (raw API probabilities).</sub>
179
+
180
+ ### Speed: many questions, one read
181
+
182
+ ![Warm p50 latency: v3 vs a Laya-architecture engine, 1K to 8K-token states](figures/latency.png)
183
+
184
+ **Ask many questions about one long state and v3 pulls away.** v3 reads the state once and scores every question
185
+ in a single call (GGUF runtime, `many_mode="batched"`; the default exact mode shares whole 1,024-token chunks of the
186
+ state and gives results identical to one call per question). With 5 to 10 questions per state, that makes it **1.4× to 1.9× faster** than a Laya-architecture
187
+ engine (our round-1 MacLaya-4K) on 1K-token states and **2.6× to 4.6× faster** on 4K-token states (4K tokens with
188
+ 10 questions: 1,381 ms vs 6,364 ms). It also answers
189
+ questions about 8K-token states in 2.3 to 2.6 s, which the 4K-budget engine cannot run at all.
190
+
191
+ <sub>Identical-architecture timing: untrained Qwen3.5-0.8B export (latency does not depend on the weights). v3 = llama.cpp GGUF F16, one call per state with all questions scored together (`many_mode="batched"` in the GGUF runtime). Comparison engine = round-1 MacLaya-4K, our own fine-tune of the Laya multilingual architecture (4,096-token budget, FP32 on Apple MPS), one call per question; it is not an official Laya checkpoint. Compared at 5 and 10 questions per state. Apple M1 Max 64 GB, warm end-to-end p50, idle run 2026-09-23, prefix reuse off.</sub>
192
+
193
+ ### Quantization: 4-bit, 0.53 GB, same calls
194
+
195
+ ![Top-1 agreement with full precision and file size for every format of v1, v2 and v3](figures/quantization.png)
196
+
197
+ **Quantize it to 4-bit and it still makes the same call.** Every shipped v3 format (GGUF F16, Q8_0 and Q4_K_M;
198
+ MLX bf16 and 8-bit) matches the PyTorch FP32 model on **240 of 240 parity rows**, plus 6 of 6 prompts of about 16K
199
+ and 25.6K tokens. The 0.53 GB Q4_K_M file is about **2.4× smaller** than the 2B v2's Q4_K_M (1.27 GB).
200
+
201
+ <sub>v3: top-1 agreement with the PyTorch FP32 reference on 240 parity rows (a mixed fixture drawn from the training pool, 22 categories, English and Chinese), plus 6 extra rows at about 16K and 25.6K tokens (3 each), where every format also agrees 6/6. v3 sizes are the exported weight files (GB = 10^9 bytes). 2B v1/v2 numbers and sizes are as reported on their public Hugging Face GGUF cards: 500 held-out decisions each, against bf16 for v1 and CUDA merged BF16 for v2. The fixtures (training-pool rows for v3, held-out rows for v1/v2) and references differ, so the rows are not a paired comparison and no agreement gap is claimed.</sub>
202
+
203
+ ## Third generation: what changed
204
+
205
+ ![Design comparison of Jev-Style v1, v2 and v3](figures/design_table.png)
206
+
207
+ <sub>v1/v2: as reported on their public Hugging Face cards (v1 GGUF card; v2 and v2-GGUF cards; v1's typed-decisions number is reported on the v2 card). v3: release manifest, training config and evaluation files; Q4_K_M size = exported file (GB = 10^9 bytes), parity rows drawn from the training pool. Typed decisions: same 2,000 decisions from 400 states; v1/v2 scored by the v2 card's harness, v3 by ours. In-domain for v3; v1 was not trained on typed decisions; v2's pool included typed workflow decisions. "51 evaluated" = MASSIVE locales, 14 trained + 37 held out; the fine-tuning pool covers 19 languages. 25,600 tokens = the runtime's whole-input limit, 25× v2's 1,024-token prompt.</sub>
208
+
209
+ <details>
210
+ <summary><strong>The same comparison as a text table</strong></summary>
211
+
212
+ | | Jev-Style 2B v1 | Jev-Style 2B v2 | **Jev-Style 0.8B v3** |
213
+ |---|---|---|---|
214
+ | Parameters | 2B (Qwen3.5-2B-Base) | 2B (continued from v1) | **0.8B** (752M text-model parameters) |
215
+ | Training | LoRA rank 16 (all linear layers) | LoRA rank 32 (33.6M trainable parameters) | **Full fine-tune** (every weight trained) |
216
+ | Readout | Option-letter token (one letter per option) | Option-letter token (' A' ... ' Z') | **Verdict slot per option** (every option scored, one pass) |
217
+ | Options per decision | Up to 26 (20 via top_logprobs) | 2–26 (letter-capped) | **No letter cap** (tested with 77 options) |
218
+ | Context | Not stated (quickstart: server default) | 1,024-token prompt (quickstart runs -c 2048) | **25,600 tokens** (preregistered 25K claim passed) |
219
+ | Languages | English (five English task families) | English (English state required) | **51 evaluated** (MASSIVE locales; 19 languages in fine-tuning) |
220
+ | Questions per state read | 1 (one question per prompt) | 1 (one question per prompt) | **Many** (all questions in one call) |
221
+ | Q4_K_M file | 1.3 GB (as reported on the v1 GGUF card) | 1.27 GB (as reported on the v2 GGUF card) | **0.53 GB** (matches FP32 on 240 / 240 parity rows) |
222
+ | Typed decisions, teacher agreement | 53.35% (2,000 decisions / 400 states) | 73.45% (same 2,000 decisions) | **79.15%** (same 2,000; 1,583 correct) |
223
+
224
+ </details>
225
+
226
+ Three ceilings of v1 and v2 are gone in v3:
227
+
228
+ - **No letter cap.** v1 and v2 read one option-letter token, so a question could have at most 26 options. v3
229
+ scores a verdict slot per option, so the options are whatever you pass. Banking77 was run with all 77 intents
230
+ in one pass.
231
+ - **25× the prompt budget.** v2's interface is a 1,024-token prompt. v3 takes 25,600 tokens, and its long-context
232
+ claim was preregistered and passed.
233
+ - **Read once, ask many.** v1 and v2 put one question in each prompt. v3 renders the state once and answers any
234
+ number of questions about it. With 10 questions on a 4K-token state this is 4.6× faster (GGUF, `many_mode="batched"`) than a
235
+ Laya-architecture engine (our round-1 MacLaya-4K) that calls once per question.
236
+
237
+ ## What "Jev-style" means
238
+
239
+ [Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev) (TypeSafe AI, 2026) introduced *System One*
240
+ decision models. Instead of generating text, the model takes a state and a typed question and returns a probability
241
+ for each allowed answer in a single pass. Jev-Style models follow that pattern with open weights:
242
+
243
+ - **choice**: pick one of N named options, with a probability for each;
244
+ - **noul** (yes/no): the probability that a statement about the state is true;
245
+ - **score**: a distribution over 2 to 10 ordered levels.
246
+
247
+ The model cannot answer outside the options it is given, and it never decodes text.
248
+
249
+ > **Independent project.** Jev-Style is not affiliated with, endorsed by or connected to TypeSafe AI or Jev, and no
250
+ > Jev weights, code or outputs are used. It is also not affiliated with the Laya authors or the Qwen team. Jev and
251
+ > Laya numbers on this card come from the sources named under each result.
252
+
253
+ ## Quick start (transformers)
254
+
255
+ ```bash
256
+ pip install -U huggingface_hub
257
+ hf download chaoliangUNSW/Jev-Style-0.8B-Decision-v3 --local-dir jev-v3
258
+ cd jev-v3
259
+ pip install -r requirements.txt # torch, transformers>=5.0 (Qwen3.5 support), tokenizers, numpy
260
+ ```
261
+
262
+ The repository ships `jev_style_decision.py`, a self-contained runtime. It handles input rendering, the verdict
263
+ readout, the fitted temperatures and budget checks.
264
+
265
+ ```python
266
+ from jev_style_decision import JevStyleDecision
267
+
268
+ m = JevStyleDecision(".") # float32 on CUDA, Apple MPS or CPU (device="cpu" to force)
269
+ r = m.decide(
270
+ {"ticket": "I was charged twice for my subscription this month.", "customer_tier": "pro"},
271
+ "Which team should handle this ticket?",
272
+ options={"billing": "payments, invoices, refunds",
273
+ "technical": "bugs and outages",
274
+ "sales": "new purchases"},
275
+ category="theme_routing",
276
+ )
277
+ print(r["answer"], r["probabilities"])
278
+ # billing (probabilities ≈ billing 0.978, sales 0.017, technical 0.004 on CPU, float32)
279
+ ```
280
+
281
+ Other question types, and several questions about one state:
282
+
283
+ ```python
284
+ state = "Order #1182: paid, packed, handed to the courier on Monday. Tracking shows 'delivered' on Wednesday."
285
+ m.decide(state, "Has the order been delivered?", qtype="noul") # {"false": p, "true": p}
286
+ m.decide(state, "How urgent is a follow-up?", qtype="score",
287
+ options=["not urgent", "somewhat urgent", "urgent", "critical"]) # levels "0".."3"
288
+ m.decide_many(state, [
289
+ {"t": "noul", "ins": "Was the order paid?", "crit": None},
290
+ {"t": "choice", "ins": "Which step is the order at?",
291
+ "crit": {"packing": None, "in transit": None, "delivered": None}},
292
+ ])
293
+ ```
294
+
295
+ From the command line:
296
+
297
+ ```bash
298
+ python jev_style_decision.py --state "The film was excellent." \
299
+ --question "What is the sentiment of this review?" \
300
+ --options '["negative", "positive"]' --category general_sentiment
301
+ # -> "answer": "positive", probability 0.989
302
+ ```
303
+
304
+ `decide` returns a dict with `answer`, `probabilities`, the raw `scores`, the `temperature` used,
305
+ `top_probability`, `entropy_concentration`, `input_tokens`, `head_tokens`, `model` and `backend`. Batch mode reads JSON lines (`--jsonl file|-`), and
306
+ `--verify` checks every file against the sha256 manifest before loading.
307
+
308
+ **GGUF (llama.cpp):** [Jev-Style-0.8B-Decision-v3-GGUF](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-GGUF).
309
+ It includes `jev_score.cpp`, a small libllama scorer that reads logits only at the verdict slots and scores many
310
+ questions on one decoded state, plus `jev_style_decision_gguf.py` with the same API.
311
+
312
+ **MLX (Apple silicon):** [bf16](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-bf16) and
313
+ [8-bit](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-8bit), each with
314
+ `jev_style_decision_mlx.py` and the same API.
315
+
316
+ ## Input format and readout
317
+
318
+ Each segment is tokenised on its own and the pieces are concatenated. Text inside the state or the options is
319
+ tokenised with special tokens disabled, so a string such as `<|im_end|>` in user data stays plain text.
320
+
321
+ ```text
322
+ State:
323
+ <state: plain text, or any JSON value serialised with ensure_ascii=False>
324
+
325
+ Question [<choice|noul|score>]: <question>
326
+ Options:
327
+ - <option 1>
328
+ - <option 2>
329
+ Judge each option:
330
+ <option 1> ->
331
+ <option 2> ->
332
+ ```
333
+
334
+ - **Verdict slot.** The hidden state at the ` ->` token that ends option k's line is option k's verdict slot. Its
335
+ score is `logit(" yes") − logit(" no")` at that position, computed as `h_k · (w_yes − w_no)` from the final
336
+ normalised hidden state and the tied embedding rows, in float32. No parameters are added, so the weights stay a
337
+ standard Qwen3.5 text model.
338
+ - **Probabilities.** `softmax(scores / T)`. `T` is looked up by calibration family × question type × option-count
339
+ bucket in `readout_config.json` (20 fitted groups), and the global value 0.880 is the fallback. Pass
340
+ `category=` (for example `theme_routing`, `general_topic`, `intent`, `typed_official`) to pick the family. With
341
+ no `category`, the runtime uses the global temperature (0.880), and `temperature=1.0` gives the raw scores.
342
+ - **Option text.** Choice options render as `name` or `name: description`. Score levels render as
343
+ `level i: description`. Yes/no questions render as `false: …` / `true: …`, with default descriptions when none
344
+ are given.
345
+
346
+ ## Usage notes
347
+
348
+ - **Input budget.** The whole rendered input, meaning state, question, options and readout, may be up to 25,600
349
+ tokens. The head (question, options and readout) may be up to 2,048 tokens. Over-budget inputs raise
350
+ `InputBudgetError`, and nothing is ever truncated silently.
351
+ - **Decisions only.** The model scores the options you give it and returns probabilities. It does not generate
352
+ text, and it takes no actions on its own.
353
+ - **Options.** `choice` takes any number of named options within the 2,048-token head (77 is the largest set we
354
+ evaluated; the GGUF runtime accepts up to 256 options per question). `score` takes 2 to 10 ordered levels, lowest first. `noul` needs no options.
355
+ - **Several questions about one state.** Use `decide_many`. In the GGUF runtime it sends all questions to the
356
+ bundled scorer in one request. By default the results are identical to one `decide` call per question; to keep
357
+ them identical, the state is shared only in whole 1,024-token blocks, so the time saved starts at 1,024-token
358
+ states and grows with the state length. `JevStyleDecisionGGUF(..., many_mode="batched")` reads the whole state
359
+ once and scores all questions together, as in the latency chart; its probabilities differed from `decide` by at
360
+ most 0.002 in our tests, and a near-tied top answer can change.
361
+ - **Precision.** The evaluation numbers on this card were computed with the PyTorch weights. The GGUF and MLX
362
+ builds were checked for top-1 parity with the PyTorch FP32 reference (see Quantization).
363
+ - **Weights.** This is a text-only `Qwen3_5ForCausalLM`: 752,393,024 parameters, 24 layers (18 Gated DeltaNet +
364
+ 6 full attention), hidden size 1,024, tied embeddings. The vision tower and the multi-token-prediction head
365
+ were removed.
366
+
367
+ ## Training
368
+
369
+ - **Base:** [Qwen/Qwen3.5-0.8B](https://huggingface.co/Qwen/Qwen3.5-0.8B) (revision `2fc06364`).
370
+ - **Run:** full fine-tune in bf16 on one NVIDIA H100 80GB. It took 994 optimizer steps over 131.4M tokens, with
371
+ a learning rate of 2e-5 and about 96 minutes of training steps (5,780 s). Checkpoint step 981 was selected by
372
+ the preregistered development score.
373
+ - **Mixture:** a 321,756-row training pool in 19 languages, with inputs up to 25,600 tokens for long documents and agent
374
+ histories:
375
+ - typed decisions: the LocalLLaMA/typed-decisions train split plus 27,300 synthetic typed items;
376
+ - general classification and QA: MNLI and translated NLI, AG News, GoEmotions, DAIR Emotion, SQuAD v2, SST-5
377
+ and BoolQ;
378
+ - intents: MASSIVE intent and scenario in 14 locales, CLINC150 and HWU64;
379
+ - application themes: spam, phishing, jailbreak and prompt injection, toxicity, ticket triage and model routing;
380
+ - Mac agent step-gate and goal-done checks;
381
+ - long-context retrieval, tables and QA.
382
+ - **Calibration:** 20 group temperatures plus a global one, fitted on 15,655 held-out calibration rows (never
383
+ test rows).
384
+
385
+ ## Training data and licences
386
+
387
+ - **Base model:** Qwen/Qwen3.5-0.8B by the Qwen team (Alibaba Cloud), Apache-2.0. The Apache License 2.0 text is
388
+ in `LICENSE`, and `NOTICE` lists the modifications. This model is released under Apache-2.0.
389
+ - **Datasets with restrictive or unclear terms** (kept in training by the author's decision):
390
+ - DAIR Emotion (14,757 training rows): its dataset card says it should be used for educational and research
391
+ purposes only.
392
+ - AG News (30,000 training rows): its licence is listed as unknown, and its card describes it as provided by the
393
+ academic community for research and non-commercial use.
394
+ - SST-5, MNLI, an HWU64 mirror, QuALITY, Enron spam and a phishing-email dataset also carry their own terms.
395
+ Check each source before commercial use.
396
+ - **Outputs of other models** (kept by the author's decision): an OpenAI GPT model wrote the theme data for
397
+ routing, triage and jailbreak, the teacher-translated NLI data and the Chinese filler text for long documents.
398
+ OpenAI GPT and Anthropic Claude models designed the synthetic typed-decision workflows, and Anthropic Claude
399
+ models labelled them. The providers' terms of use may restrict how models trained on such outputs may be used,
400
+ so check them for your use case.
401
+ - **Evaluation-only data** (0 training rows): Banking77, toxic-chat, XNLI, ContractNLI, MS MARCO, tweet_topic,
402
+ fin_topic, the JevBench items and the support-ticket set.
403
+
404
+ <details>
405
+ <summary><strong>Evaluation records and vector charts</strong></summary>
406
+
407
+ - Every chart is also provided as SVG: [headline_typed](figures/headline_typed.svg),
408
+ [beyond_laya](figures/beyond_laya.svg), [multilingual](figures/multilingual.svg),
409
+ [calibration](figures/calibration.svg), [long_context](figures/long_context.svg),
410
+ [jevbench](figures/jevbench.svg), [zeroshot](figures/zeroshot.svg), [latency](figures/latency.svg),
411
+ [quantization](figures/quantization.svg), [design_table](figures/design_table.svg).
412
+ - Plotted values, sources and protocol labels for each chart:
413
+ [headline_typed](figures/headline_typed.data.json), [beyond_laya](figures/beyond_laya.json),
414
+ [multilingual](figures/multilingual.data.json), [calibration](figures/calibration.data.json),
415
+ [long_context](figures/long_context.json), [jevbench](figures/jevbench.data.json),
416
+ [zeroshot](figures/zeroshot.json), [latency](figures/latency.data.json),
417
+ [quantization](figures/quantization.data.json), [design_table](figures/design_table.data.json).
418
+ - Every v3 and re-run Laya number comes from prediction files that were each scored once. Paired differences use
419
+ a case-cluster bootstrap within suites (2,000 resamples). A win is only claimed when the 95% CI excludes zero,
420
+ except where a chart or note says otherwise (JevBench leads over Laya and M. Ghafiri, and per-language MASSIVE
421
+ gaps against the best of three Laya checkpoints, are point estimates).
422
+ - Public sources: [Laya](https://huggingface.co/convaiinnovations/laya) (official checkpoints and README;
423
+ [BENCHMARKS.md](https://github.com/NandhaKishorM/laya/blob/main/BENCHMARKS.md)),
424
+ [LocalLLaMA/typed-decisions](https://huggingface.co/datasets/LocalLLaMA/typed-decisions) (dataset card with Jev's
425
+ numbers), [JevBench](https://github.com/fstandhartinger/jevbench) (v1.4.1 results file),
426
+ [elcronos/jev-vs-open-decision-models](https://github.com/elcronos/jev-vs-open-decision-models) (zero-shot topic
427
+ study), and the [2B v1](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF) and
428
+ [2B v2](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2) cards.
429
+
430
+ </details>
431
+
432
+ ## Citation
433
+
434
+ ```bibtex
435
+ @misc{jevstyle2026v3,
436
+ title = {Jev-Style-0.8B-Decision-v3: a long-context, multilingual 0.8B decision model},
437
+ author = {chaoliangUNSW},
438
+ year = {2026},
439
+ howpublished = {\url{https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3}},
440
+ note = {Fine-tuned from Qwen/Qwen3.5-0.8B}
441
+ }
442
+ ```
443
+
444
+ ## Contact
445
+
446
+ I welcome internship, employment, and research collaboration opportunities. Please contact me at [**yanchaoliang369@gmail.com**](mailto:yanchaoliang369@gmail.com).
447
+
448
+ 欢迎提供实习、工作及科研合作机会,请邮件联系:[yanchaoliang369@gmail.com](mailto:yanchaoliang369@gmail.com)。
chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is true %}
150
+ {{- '<think>\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n\n</think>\n\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
config.json ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForCausalLM"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "attn_output_gate": true,
8
+ "bos_token_id": null,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 248044,
11
+ "full_attention_interval": 4,
12
+ "head_dim": 256,
13
+ "hidden_act": "silu",
14
+ "hidden_size": 1024,
15
+ "initializer_range": 0.02,
16
+ "intermediate_size": 3584,
17
+ "layer_types": [
18
+ "linear_attention",
19
+ "linear_attention",
20
+ "linear_attention",
21
+ "full_attention",
22
+ "linear_attention",
23
+ "linear_attention",
24
+ "linear_attention",
25
+ "full_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "linear_attention",
29
+ "full_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "linear_attention",
33
+ "full_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "linear_attention",
37
+ "full_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "linear_attention",
41
+ "full_attention"
42
+ ],
43
+ "linear_conv_kernel_dim": 4,
44
+ "linear_key_head_dim": 128,
45
+ "linear_num_key_heads": 16,
46
+ "linear_num_value_heads": 16,
47
+ "linear_value_head_dim": 128,
48
+ "mamba_ssm_dtype": "float32",
49
+ "max_position_embeddings": 262144,
50
+ "mlp_only_layers": [],
51
+ "model_type": "qwen3_5_text",
52
+ "mtp_num_hidden_layers": 1,
53
+ "mtp_use_dedicated_embeddings": false,
54
+ "num_attention_heads": 8,
55
+ "num_hidden_layers": 24,
56
+ "num_key_value_heads": 2,
57
+ "pad_token_id": null,
58
+ "partial_rotary_factor": 0.25,
59
+ "rms_norm_eps": 1e-06,
60
+ "rope_parameters": {
61
+ "mrope_interleaved": true,
62
+ "mrope_section": [
63
+ 11,
64
+ 11,
65
+ 10
66
+ ],
67
+ "partial_rotary_factor": 0.25,
68
+ "rope_theta": 10000000,
69
+ "rope_type": "default"
70
+ },
71
+ "tie_word_embeddings": true,
72
+ "transformers_version": "5.17.0",
73
+ "use_cache": true,
74
+ "vocab_size": 248320
75
+ }
figures/beyond_laya.json ADDED
@@ -0,0 +1,94 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "beyond_laya",
3
+ "unit": "percent (value x 100)",
4
+ "source": "runs/macjev/report_r2/scoreboard/scoreboard.json via hf_staging/v3_card/chart_data.json",
5
+ "delta_ci_method": "case-cluster bootstrap within suites, 2000 resamples (scoreboard n_boot)",
6
+ "rows": [
7
+ {
8
+ "task": "Banking77",
9
+ "metric": "accuracy, 77 intents",
10
+ "v3": 0.6825,
11
+ "best_laya": 0.4925,
12
+ "best_laya_checkpoint": "typed",
13
+ "delta_pts": 19.0,
14
+ "delta_ci95_pts": [
15
+ 13.99,
16
+ 24.0
17
+ ],
18
+ "n": 400,
19
+ "v3_entry": "banking77.v3",
20
+ "laya_entry": "banking77.laya_best",
21
+ "scoreboard_field": "metrics['banking77']",
22
+ "protocol_label": "v3: held-out (banking77 never trained; CLINC150/HWU64 intents are) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
23
+ },
24
+ {
25
+ "task": "Jailbreak",
26
+ "metric": "balanced accuracy",
27
+ "v3": 0.9036238842064085,
28
+ "best_laya": 0.8311462000782389,
29
+ "best_laya_checkpoint": "multilingual",
30
+ "delta_pts": 7.25,
31
+ "delta_ci95_pts": [
32
+ 2.99,
33
+ 11.43
34
+ ],
35
+ "n": 400,
36
+ "v3_entry": "theme.guardrails_jailbreak.balanced_accuracy.v3",
37
+ "laya_entry": "theme.guardrails_jailbreak.balanced_accuracy.laya_best",
38
+ "scoreboard_field": "metrics['theme.guardrails_jailbreak.balanced_accuracy']",
39
+ "protocol_label": "v3: held-out source (other permissive jailbreak sets + teacher) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
40
+ },
41
+ {
42
+ "task": "Toxicity",
43
+ "metric": "macro-F1",
44
+ "v3": 0.7089254055198327,
45
+ "best_laya": 0.41360068097985436,
46
+ "best_laya_checkpoint": "multilingual",
47
+ "delta_pts": 29.53,
48
+ "delta_ci95_pts": [
49
+ 24.65,
50
+ 34.37
51
+ ],
52
+ "n": 400,
53
+ "v3_entry": "theme.moderation_toxicity.macro_f1.v3",
54
+ "laya_entry": "theme.moderation_toxicity.macro_f1.laya_best",
55
+ "scoreboard_field": "metrics['theme.moderation_toxicity.macro_f1']",
56
+ "protocol_label": "v3: held-out source (civil_comments + teacher; toxic-chat eval-only) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
57
+ },
58
+ {
59
+ "task": "Model routing",
60
+ "metric": "accuracy",
61
+ "v3": 0.9624060150375939,
62
+ "best_laya": 0.6591478696741855,
63
+ "best_laya_checkpoint": "typed",
64
+ "delta_pts": 30.33,
65
+ "delta_ci95_pts": [
66
+ 26.06,
67
+ 35.09
68
+ ],
69
+ "n": 399,
70
+ "v3_entry": "theme.model_routing_domain.v3",
71
+ "laya_entry": "theme.model_routing_domain.laya_best",
72
+ "scoreboard_field": "metrics['theme.model_routing_domain']",
73
+ "protocol_label": "v3: near-domain (teacher-written; gsm8k/mbpp/AG rows eval-only) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
74
+ },
75
+ {
76
+ "task": "MASSIVE intent",
77
+ "metric": "37 held-out locales",
78
+ "v3": 0.6551351351351351,
79
+ "best_laya": 0.36108108108108106,
80
+ "best_laya_checkpoint": "multilingual",
81
+ "delta_pts": 29.41,
82
+ "delta_ci95_pts": [
83
+ 27.54,
84
+ 31.3
85
+ ],
86
+ "n": 3700,
87
+ "v3_entry": "massive51.unseen37.macro_accuracy.v3",
88
+ "laya_entry": "massive51.unseen37.macro_accuracy.laya_best",
89
+ "scoreboard_field": "metrics['massive51.unseen37.macro_accuracy']",
90
+ "protocol_label": "v3: held-out (37 languages never trained) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
91
+ }
92
+ ],
93
+ "footnote": "Laya numbers: official checkpoints (English, typed-decisions, multilingual) re-run by us on identical rows with their shipped temperatures and\ndefault token budgets; the best of the three is shown per task. v3 trained on same-kind tasks from other datasets, never on these eval rows:\nintent (CLINC150/HWU64; Banking77 never trained), jailbreak (other permissive sets + teacher), toxicity (civil_comments + teacher; toxic-chat\neval-only), routing (teacher-written; gsm8k/mbpp/AG rows eval-only), MASSIVE in 14 other locales (no MASSIVE rows in these 37).\nn = 400 / 400 / 400 / 399 / 3,700 (37 x 100). Every gap's paired 95% bootstrap CI excludes zero. Plotted values: figures/beyond_laya.json."
94
+ }
figures/beyond_laya.png ADDED

Git LFS Details

  • SHA256: 0dbd24837759acf05ed3a668016ecaea75a41a093b64588417d2fd750b95779b
  • Pointer size: 131 Bytes
  • Size of remote file: 186 kB
figures/beyond_laya.svg ADDED
figures/calibration.data.json ADDED
@@ -0,0 +1,99 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "calibration",
3
+ "metric": "49-suite macro NLL / Brier / ECE, as deployed (lower is better)",
4
+ "protocol_label": "v3: mixed (49 suites) / Laya: mixed (49 suites). Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures.",
5
+ "rows": [
6
+ {
7
+ "panel": "NLL",
8
+ "metric": "49-suite macro NLL, as deployed",
9
+ "n_rows": 17416,
10
+ "v3": 0.4928839178581892,
11
+ "best_laya": 2.212565130608199,
12
+ "best_laya_checkpoint": "multilingual",
13
+ "all_laya_checkpoints": {
14
+ "english": 9.669232280968933,
15
+ "typed": 7.344998151307391,
16
+ "multilingual": 2.212565130608199
17
+ },
18
+ "ratio_best_laya_over_v3": 4.48901871301224,
19
+ "diff_v3_minus_best": -1.7196812127500098,
20
+ "diff_ci95": [
21
+ -1.7632826815264786,
22
+ -1.6751892907984907
23
+ ],
24
+ "diff_ci_method": "case-cluster bootstrap within suites, 2000 resamples (scoreboard n_boot)",
25
+ "source_path": "runs/macjev/report_r2/scoreboard/scoreboard.json",
26
+ "fields": [
27
+ "metrics['t4.macro_nll'].ours_value",
28
+ "metrics['t4.macro_nll'].laya_best"
29
+ ],
30
+ "entries": [
31
+ "t4.macro_nll.v3",
32
+ "t4.macro_nll.laya_best"
33
+ ],
34
+ "claim": "t4.macro_nll_vs_best_laya"
35
+ },
36
+ {
37
+ "panel": "Brier score",
38
+ "metric": "49-suite macro Brier, as deployed",
39
+ "n_rows": 17416,
40
+ "v3": 0.23930092306086748,
41
+ "best_laya": 0.7120664110846202,
42
+ "best_laya_checkpoint": "multilingual",
43
+ "all_laya_checkpoints": {
44
+ "english": 1.0124090901595464,
45
+ "typed": 0.9474950671291875,
46
+ "multilingual": 0.7120664110846202
47
+ },
48
+ "ratio_best_laya_over_v3": 2.975610799894417,
49
+ "diff_v3_minus_best": -0.47276548802375273,
50
+ "diff_ci95": [
51
+ -0.4843787573639837,
52
+ -0.46074418926883626
53
+ ],
54
+ "diff_ci_method": "case-cluster bootstrap within suites, 2000 resamples (scoreboard n_boot)",
55
+ "source_path": "runs/macjev/report_r2/scoreboard/scoreboard.json",
56
+ "fields": [
57
+ "metrics['t4.macro_brier'].ours_value",
58
+ "metrics['t4.macro_brier'].laya_best"
59
+ ],
60
+ "entries": [
61
+ "t4.macro_brier.v3",
62
+ "t4.macro_brier.laya_best"
63
+ ],
64
+ "claim": "t4.macro_brier_vs_best_laya"
65
+ },
66
+ {
67
+ "panel": "ECE",
68
+ "metric": "49-suite macro ECE, as deployed",
69
+ "n_rows": 17416,
70
+ "v3": 0.05416919804670487,
71
+ "best_laya": 0.2994167384382702,
72
+ "best_laya_checkpoint": "multilingual",
73
+ "all_laya_checkpoints": {
74
+ "english": 0.4647782759946152,
75
+ "typed": 0.40828649732151606,
76
+ "multilingual": 0.2994167384382702
77
+ },
78
+ "ratio_best_laya_over_v3": 5.527435318132494,
79
+ "diff_v3_minus_best": -0.24524754039156535,
80
+ "diff_ci95": [
81
+ -0.24448168599481393,
82
+ -0.22893121774401398
83
+ ],
84
+ "diff_ci_method": "case-cluster bootstrap within suites, 2000 resamples (scoreboard n_boot)",
85
+ "source_path": "runs/macjev/report_r2/scoreboard/scoreboard.json",
86
+ "fields": [
87
+ "metrics['t4.macro_ece_as_deployed'].ours_value",
88
+ "metrics['t4.macro_ece_as_deployed'].laya_best"
89
+ ],
90
+ "entries": [
91
+ "t4.macro_ece_as_deployed.v3",
92
+ "t4.macro_ece_as_deployed.laya_best"
93
+ ],
94
+ "claim": "t4.macro_ece_as_deployed_vs_best_laya"
95
+ }
96
+ ],
97
+ "footnote": "Macro average over 49 suites, 17,416 identical rows for both models. Protocol: mixed for both v3 and Laya (in-domain on some suites, held-out on others). As deployed: v3 with its supplied deployment temperatures; Laya numbers: official checkpoints re-run by us on identical rows with their shipped temperatures. Best Laya = best of the three official checkpoints (English, typed-decisions, multilingual) per metric; multilingual is best on all three. Paired case-cluster bootstrap 95% CIs (2,000 resamples) of every v3 minus Laya difference exclude zero. Plotted values: figures/calibration.data.json.",
98
+ "not_plotted": "optional typed-decisions reliability diagram omitted (see agent report)"
99
+ }
figures/calibration.png ADDED

Git LFS Details

  • SHA256: 3176339f6e4f58c9efa583b101f5862141a326468cf177f2c50c3ce2d1adf06f
  • Pointer size: 131 Bytes
  • Size of remote file: 149 kB
figures/calibration.svg ADDED
figures/design_table.data.json ADDED
@@ -0,0 +1,175 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "design_table",
3
+ "columns": [
4
+ "Jev-Style 2B v1",
5
+ "Jev-Style 2B v2",
6
+ "Jev-Style 0.8B v3"
7
+ ],
8
+ "rows": [
9
+ {
10
+ "row": "Parameters",
11
+ "Jev-Style 2B v1": {
12
+ "main": "2B",
13
+ "sub": "Qwen3.5-2B-Base"
14
+ },
15
+ "Jev-Style 2B v2": {
16
+ "main": "2B",
17
+ "sub": "continued from v1"
18
+ },
19
+ "Jev-Style 0.8B v3": {
20
+ "main": "0.8B",
21
+ "sub": "752M text-model params"
22
+ }
23
+ },
24
+ {
25
+ "row": "Training",
26
+ "Jev-Style 2B v1": {
27
+ "main": "LoRA rank 16",
28
+ "sub": "all linear layers"
29
+ },
30
+ "Jev-Style 2B v2": {
31
+ "main": "LoRA rank 32",
32
+ "sub": "33.6M trainable params"
33
+ },
34
+ "Jev-Style 0.8B v3": {
35
+ "main": "Full fine-tune",
36
+ "sub": "every weight trained"
37
+ }
38
+ },
39
+ {
40
+ "row": "Readout",
41
+ "Jev-Style 2B v1": {
42
+ "main": "Option-letter token",
43
+ "sub": "one letter per option"
44
+ },
45
+ "Jev-Style 2B v2": {
46
+ "main": "Option-letter token",
47
+ "sub": "' A' ... ' Z'"
48
+ },
49
+ "Jev-Style 0.8B v3": {
50
+ "main": "Verdict slot per option",
51
+ "sub": "every option scored, one pass"
52
+ }
53
+ },
54
+ {
55
+ "row": "Options per decision",
56
+ "Jev-Style 2B v1": {
57
+ "main": "Up to 26",
58
+ "sub": "20 via top_logprobs"
59
+ },
60
+ "Jev-Style 2B v2": {
61
+ "main": "2-26",
62
+ "sub": "letter-capped"
63
+ },
64
+ "Jev-Style 0.8B v3": {
65
+ "main": "No letter cap",
66
+ "sub": "tested with 77 options"
67
+ }
68
+ },
69
+ {
70
+ "row": "Context",
71
+ "Jev-Style 2B v1": {
72
+ "main": "Not stated",
73
+ "sub": "quickstart: server default"
74
+ },
75
+ "Jev-Style 2B v2": {
76
+ "main": "1,024-token prompt",
77
+ "sub": "quickstart runs -c 2048"
78
+ },
79
+ "Jev-Style 0.8B v3": {
80
+ "main": "25,600 tokens",
81
+ "sub": "preregistered 25K claim passed"
82
+ }
83
+ },
84
+ {
85
+ "row": "Languages",
86
+ "Jev-Style 2B v1": {
87
+ "main": "English",
88
+ "sub": "five English task families"
89
+ },
90
+ "Jev-Style 2B v2": {
91
+ "main": "English",
92
+ "sub": "English state required"
93
+ },
94
+ "Jev-Style 0.8B v3": {
95
+ "main": "51 evaluated",
96
+ "sub": "MASSIVE locales; 19 in fine-tuning"
97
+ }
98
+ },
99
+ {
100
+ "row": "Questions per state read",
101
+ "Jev-Style 2B v1": {
102
+ "main": "1",
103
+ "sub": "one question per prompt"
104
+ },
105
+ "Jev-Style 2B v2": {
106
+ "main": "1",
107
+ "sub": "one question per prompt"
108
+ },
109
+ "Jev-Style 0.8B v3": {
110
+ "main": "Many",
111
+ "sub": "all questions in one call"
112
+ }
113
+ },
114
+ {
115
+ "row": "Q4_K_M file",
116
+ "Jev-Style 2B v1": {
117
+ "main": "1.3 GB",
118
+ "sub": "as reported on the v1 GGUF card"
119
+ },
120
+ "Jev-Style 2B v2": {
121
+ "main": "1.27 GB",
122
+ "sub": "as reported on the v2 GGUF card"
123
+ },
124
+ "Jev-Style 0.8B v3": {
125
+ "main": "0.53 GB",
126
+ "sub": "matches FP32 on 240/240 parity rows"
127
+ }
128
+ },
129
+ {
130
+ "row": "Typed decisions, teacher agreement",
131
+ "Jev-Style 2B v1": {
132
+ "main": "53.35%",
133
+ "sub": "2,000 decisions / 400 states"
134
+ },
135
+ "Jev-Style 2B v2": {
136
+ "main": "73.45%",
137
+ "sub": "same 2,000 decisions"
138
+ },
139
+ "Jev-Style 0.8B v3": {
140
+ "main": "79.15%",
141
+ "sub": "same 2,000 \u00b7 1,583 correct"
142
+ }
143
+ }
144
+ ],
145
+ "numbers": {
146
+ "v1_q4_k_m_agreement": 0.944,
147
+ "v2_q4_k_m_agreement": 0.914,
148
+ "v3_q4_k_m_agreement": {
149
+ "agree": 240,
150
+ "n": 240,
151
+ "value": 1.0
152
+ },
153
+ "typed_teacher_agreement": {
154
+ "v1": 0.5335,
155
+ "v2": 0.7345,
156
+ "v3": 0.7915,
157
+ "v3_correct": 1583,
158
+ "n": 2000,
159
+ "states": 400
160
+ },
161
+ "params": {
162
+ "v1": "2B",
163
+ "v2": "2B",
164
+ "v3_text_model": 752393024
165
+ },
166
+ "size_ratio_v3_over_2b": 0.4,
167
+ "context_ratio_v3_over_v2_prompt": 25.0
168
+ },
169
+ "sources": {
170
+ "v1": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF/raw/main/README.md (file table 'Same decision as bf16' Q4_K_M 94.4%; 'LoRA rank 16 on all linear layers'; 'Up to 26 options (20 when ... top_logprobs)'; 'Trained on five English task families'; quickstart llama-server without -c; prompt has one [Question])",
171
+ "v2": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/raw/main/README.md (Q4_K_M 91.4% choice agreement vs CUDA BF16 on frozen 500-decision subset; '2-26 unique options, within a 1,024-token prompt'; 'English state'; quickstart -c 2048; rank-32 LoRA, 33,638,400 trainable; continued from v1; ' A' through ' Z' readout; typed-decisions 53.35% v1 / 73.45% v2, 2,000 decisions from 400 states)",
172
+ "v3": "chart_data.json: design.generations (export_manifest.json parameters_text_model=752393024; config.resolved.json readout=verdict, no LoRA keys; jevbench/results.json scorer.max_len=25600; apps.jsonl banking77_full 400 rows x 77 options; scoreboard long_grid_plus claim_25k_ok=true), quant.v3 gguf-q4_k_m 240/240 (export_r2/main/parity/report.json), typed.accuracy.v3 1583/2000 (typed_test.jsonl, 400 group_ids)"
173
+ },
174
+ "footnote": "v1/v2: as reported on their public Hugging Face cards (v1 GGUF card; v2 and v2-GGUF cards; v1's typed-decisions number is reported on the v2 card). v3: release manifest, training config and eval files; Q4_K_M size = exported file (GB = 10^9 bytes), parity rows drawn from the training pool. Typed decisions: same 2,000 decisions from 400 states; v1/v2 scored by the v2 card's harness, v3 by ours. In-domain for v3; v1 not trained on typed decisions; v2's pool included typed workflow decisions. '51 evaluated' = MASSIVE locales (14 trained, 37 held out); fine-tuning covers 19 languages. 25,600 tokens = the runtime's whole-input limit; 25x = vs v2's 1,024-token prompt."
175
+ }
figures/design_table.png ADDED

Git LFS Details

  • SHA256: 5b907319a5a4f6ee8eea5800fbc732b95d15e941b1d0cd8d774d3705a1341744
  • Pointer size: 131 Bytes
  • Size of remote file: 332 kB
figures/design_table.svg ADDED
figures/headline_typed.data.json ADDED
@@ -0,0 +1,90 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "figure": "headline_typed",
3
+ "panels": {
4
+ "accuracy_pct": [
5
+ {
6
+ "label": "Jev-Style 2B v1",
7
+ "entry": "typed.teacher_agreement.v1",
8
+ "raw": 0.5335,
9
+ "plotted": 53.4,
10
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2/raw/main/README.md",
11
+ "field": "card text: 'The separate typed-decisions group contains 2,000 teacher-reference decisions from 400 states. Teacher agreement is 53.35% for v1, 37.55% for English Laya and 73.45% for v2'"
12
+ },
13
+ {
14
+ "label": "Jev",
15
+ "entry": "typed.accuracy.jev",
16
+ "raw": 0.727,
17
+ "plotted": 72.7,
18
+ "source": "docs/round2_audit/track_jev_results.json",
19
+ "field": "string 'Jev typed-decisions (zero-shot)': '0.727 acc; KL 1.442; Brier 0.148; ECE 0.144; soft-acc 0.580' (from LocalLLaMA/typed-decisions dataset card / Laya HF card)"
20
+ },
21
+ {
22
+ "label": "Jev-Style 2B v2",
23
+ "entry": "typed.teacher_agreement.v2",
24
+ "raw": 0.7345,
25
+ "plotted": 73.5,
26
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2/raw/main/README.md",
27
+ "field": "card text: 'The separate typed-decisions group contains 2,000 teacher-reference decisions from 400 states. Teacher agreement is 53.35% for v1, 37.55% for English Laya and 73.45% for v2'"
28
+ },
29
+ {
30
+ "label": "Laya (typed ckpt)",
31
+ "entry": "typed.accuracy.laya_typed",
32
+ "raw": 0.766,
33
+ "plotted": 76.6,
34
+ "source": "runs/macjev/report_r2/scoreboard/scoreboard.json",
35
+ "field": "metrics['typed.accuracy'].laya.typed.value"
36
+ },
37
+ {
38
+ "label": "Jev-Style 0.8B v3",
39
+ "entry": "typed.accuracy.v3",
40
+ "raw": 0.7915,
41
+ "plotted": 79.2,
42
+ "source": "runs/macjev/report_r2/scoreboard/scoreboard.json",
43
+ "field": "metrics['typed.accuracy'].ours_value"
44
+ }
45
+ ],
46
+ "brier_vs_soft": [
47
+ {
48
+ "label": "Jev",
49
+ "entry": "typed.Brier_vs_soft_labels.jev",
50
+ "raw": 0.148,
51
+ "plotted": 0.148,
52
+ "source": "docs/round2_audit/track_jev_results.json",
53
+ "field": "string 'Jev typed-decisions (zero-shot)': '0.727 acc; KL 1.442; Brier 0.148; ECE 0.144; soft-acc 0.580' (from LocalLLaMA/typed-decisions dataset card / Laya HF card)"
54
+ },
55
+ {
56
+ "label": "Laya (typed ckpt)",
57
+ "entry": "typed.brier_vs_soft.laya_typed",
58
+ "raw": 0.06146485330135357,
59
+ "plotted": 0.061,
60
+ "source": "runs/macjev/report_r2/scoreboard/scoreboard.json",
61
+ "field": "metrics['typed.brier_vs_soft'].laya.typed.value"
62
+ },
63
+ {
64
+ "label": "Jev-Style 0.8B v3",
65
+ "entry": "typed.brier_vs_soft.v3",
66
+ "raw": 0.04583493309263009,
67
+ "plotted": 0.046,
68
+ "source": "runs/macjev/report_r2/scoreboard/scoreboard.json",
69
+ "field": "metrics['typed.brier_vs_soft'].ours_value"
70
+ }
71
+ ]
72
+ },
73
+ "annotations": {
74
+ "delta_vs_jev_pts": 6.4,
75
+ "delta_vs_v2_pts": 5.7,
76
+ "delta_vs_laya_typed_pts": 2.6,
77
+ "brier_ratio_jev_over_v3": 3.229,
78
+ "brier_pct_lower_than_laya_typed": 25.4,
79
+ "v3_acc_ci95_wilson": [
80
+ 0.7731,
81
+ 0.8087
82
+ ],
83
+ "v3_minus_laya_typed_paired_ci95": [
84
+ 0.010499999999999954,
85
+ 0.04249999999999998
86
+ ]
87
+ },
88
+ "protocol": "in-domain for v3 and Laya typed; zero-shot for Jev (dataset card); 2B v1/v2 as reported on the v2 card",
89
+ "footnote": "Typed-decisions test set (LocalLLaMA/typed-decisions), 2,000 decisions from 400 states. In-domain for v3 and Laya typed (both trained on its\ntrain split); zero-shot for Jev (dataset-card numbers, Jev API, all 2,000 decisions). Laya: official typed-decisions checkpoint re-run by us on\nidentical rows with its shipped temperature. 2B v1/v2: teacher agreement as reported on the v2 card (same 2,000 decisions, that card's harness;\nv1 was not trained on typed decisions, v2's pool included typed workflow decisions). v3 95% CI 77.3\u201380.9% (Wilson);\nv3 minus Laya typed, paired bootstrap 95% CI +1.0 to +4.2 pts. Plotted values: figures/headline_typed.data.json."
90
+ }
figures/headline_typed.png ADDED

Git LFS Details

  • SHA256: 9e763e4cbd86a25299785b94a89945f2c3a7cab4b1c7d5000f9456c4d220a1f4
  • Pointer size: 131 Bytes
  • Size of remote file: 176 kB
figures/headline_typed.svg ADDED
figures/jevbench.data.json ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "jevbench",
3
+ "metric": "JevBench v1.4.1 public accuracy (231 items)",
4
+ "rows": [
5
+ {
6
+ "label": "Jev-Style 0.8B v3",
7
+ "key": "v3",
8
+ "accuracy": 0.6406926406926406,
9
+ "correct": 148,
10
+ "n": 231,
11
+ "colour": "#3456F0",
12
+ "source": "runs/macjev/received/ext_evals/main/jevbench/results.json :: public_accuracy"
13
+ },
14
+ {
15
+ "label": "Qwen3.5-0.8B Decision Model (M. Ghafiri)",
16
+ "key": "mghafiri-qwen3.5-0.8b-decision-model",
17
+ "accuracy": 0.5930735930735931,
18
+ "correct": 137,
19
+ "n": 231,
20
+ "colour": "#C4C4CB",
21
+ "source": "data/external/jevbench_v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[key=mghafiri-qwen3.5-0.8b-decision-model].public_accuracy",
22
+ "v3_lead_pts": 4.76
23
+ },
24
+ {
25
+ "label": "Laya (ModernBERT-large, 421M)",
26
+ "key": "laya",
27
+ "accuracy": 0.5844155844155844,
28
+ "correct": 135,
29
+ "n": 231,
30
+ "colour": "#D0D0D6",
31
+ "source": "data/external/jevbench_v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[key=laya].public_accuracy",
32
+ "v3_lead_pts": 5.63
33
+ },
34
+ {
35
+ "label": "SimpleJev (Qwen3.5-0.8B)",
36
+ "key": "simplejev-qwen3.5-0.8b",
37
+ "accuracy": 0.5454545454545454,
38
+ "correct": 126,
39
+ "n": 231,
40
+ "colour": "#C4C4CB",
41
+ "source": "data/external/jevbench_v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[key=simplejev-qwen3.5-0.8b].public_accuracy",
42
+ "v3_lead_pts": 9.52
43
+ }
44
+ ],
45
+ "footnote": "JevBench v1.4.1, public items only (231). v3: self-run zero-shot with the vendored official harness (commit 24b9b5c); training-pool contamination scan: 0 hits; not an official leaderboard entry. Other rows: public accuracy as published in the board's v1.4.1 results file (github.com/fstandhartinger/jevbench). Shown: Laya plus every Qwen3.5-0.8B-based system on the board. Laya and M. Ghafiri lie inside v3's 95% CI (57.7\u201370.0%, Wilson): those two leads are point estimates, not significant at n = 231.",
46
+ "v3_run": {
47
+ "source_path": "runs/macjev/received/ext_evals/main/jevbench/results.json",
48
+ "field": "public_accuracy (148/231; independent check equal)",
49
+ "protocol_label": "v3 self-run with the vendored official harness on the 231 public items, zero-shot (contamination scan of the training pool: 0 hits, runs/macjev/external/jevbench/contamination.json); other rows as published in jevbench v1.4.1 results.json (full runs by the board); not an official leaderboard entry",
50
+ "ci95": [
51
+ 0.577,
52
+ 0.6998
53
+ ],
54
+ "ci_method": "Wilson 95%"
55
+ },
56
+ "board": {
57
+ "source_path": "data/external/jevbench_v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json",
58
+ "protocol_label": "as published by the board (commit 24b9b5c1609a, tag v1.4.1)"
59
+ }
60
+ }
figures/jevbench.png ADDED

Git LFS Details

  • SHA256: b75ab6864b4487d0a323c1019f7ce405fedcdb5a4d8dc910c86a29b3141b5388
  • Pointer size: 131 Bytes
  • Size of remote file: 143 kB
figures/jevbench.svg ADDED
figures/latency.data.json ADDED
@@ -0,0 +1,51 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "latency",
3
+ "from_chart_data_entry": "latency.idle",
4
+ "source_path": "runs/macjev/latency/m1max_untrained_base_2026-09-23_review_idle/latency.json",
5
+ "field": "results['llamacpp-f16'|'laya-multilingual-mps-fp32'][cell].p50_ms",
6
+ "protocol_label": "M1 Max 64 GB, warm end-to-end p50 ms, idle run 2026-09-23 (load avg 3-8); untrained identical-architecture Qwen3.5-0.8B export (latency does not depend on weights); v3 = llama.cpp GGUF F16, one call per state; comparison engine = round-1 MacLaya-4K (our Laya-multilingual fine-tune, FP32/MPS, 4,096-token budget), one decide() per question; prefix reuse off",
7
+ "plotted": [
8
+ {
9
+ "cell": "1024x5",
10
+ "v3_p50_ms": 394,
11
+ "maclaya4k_p50_ms": 543,
12
+ "speedup_label": "1.4x"
13
+ },
14
+ {
15
+ "cell": "1024x10",
16
+ "v3_p50_ms": 525,
17
+ "maclaya4k_p50_ms": 1019,
18
+ "speedup_label": "1.9x"
19
+ },
20
+ {
21
+ "cell": "4096x5",
22
+ "v3_p50_ms": 1212,
23
+ "maclaya4k_p50_ms": 3204,
24
+ "speedup_label": "2.6x"
25
+ },
26
+ {
27
+ "cell": "4096x10",
28
+ "v3_p50_ms": 1381,
29
+ "maclaya4k_p50_ms": 6364,
30
+ "speedup_label": "4.6x"
31
+ },
32
+ {
33
+ "cell": "8192x1",
34
+ "v3_p50_ms": 2295,
35
+ "maclaya4k_p50_ms": null,
36
+ "speedup_label": "v3 only"
37
+ },
38
+ {
39
+ "cell": "8192x5",
40
+ "v3_p50_ms": 2424,
41
+ "maclaya4k_p50_ms": null,
42
+ "speedup_label": "v3 only"
43
+ },
44
+ {
45
+ "cell": "8192x10",
46
+ "v3_p50_ms": 2613,
47
+ "maclaya4k_p50_ms": null,
48
+ "speedup_label": "v3 only"
49
+ }
50
+ ]
51
+ }
figures/latency.png ADDED

Git LFS Details

  • SHA256: d6d0d34b875997b8a733d4776dc34a1b564ddc64b48cea9d0def7f72a34af79e
  • Pointer size: 131 Bytes
  • Size of remote file: 203 kB
figures/latency.svg ADDED
figures/long_context.json ADDED
@@ -0,0 +1,216 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "figure": "long_context",
3
+ "plotted": [
4
+ {
5
+ "length_bin": 1024,
6
+ "items": 320,
7
+ "controlled_pct": 57.81,
8
+ "ci95_pct": [
9
+ 52.34,
10
+ 63.1
11
+ ]
12
+ },
13
+ {
14
+ "length_bin": 2048,
15
+ "items": 320,
16
+ "controlled_pct": 54.69,
17
+ "ci95_pct": [
18
+ 49.21,
19
+ 60.05
20
+ ]
21
+ },
22
+ {
23
+ "length_bin": 4096,
24
+ "items": 320,
25
+ "controlled_pct": 58.13,
26
+ "ci95_pct": [
27
+ 52.65,
28
+ 63.4
29
+ ]
30
+ },
31
+ {
32
+ "length_bin": 8192,
33
+ "items": 320,
34
+ "controlled_pct": 54.37,
35
+ "ci95_pct": [
36
+ 48.9,
37
+ 59.75
38
+ ]
39
+ },
40
+ {
41
+ "length_bin": 12288,
42
+ "items": 320,
43
+ "controlled_pct": 54.37,
44
+ "ci95_pct": [
45
+ 48.9,
46
+ 59.75
47
+ ]
48
+ },
49
+ {
50
+ "length_bin": 16384,
51
+ "items": 320,
52
+ "controlled_pct": 54.69,
53
+ "ci95_pct": [
54
+ 49.21,
55
+ 60.05
56
+ ]
57
+ },
58
+ {
59
+ "length_bin": 24576,
60
+ "items": 1280,
61
+ "controlled_pct": 55.31,
62
+ "ci95_pct": [
63
+ 52.58,
64
+ 58.02
65
+ ]
66
+ }
67
+ ],
68
+ "reference_controlled_2k_4k_pct": 56.41,
69
+ "gap_24k_vs_ref_pts": -1.09,
70
+ "spread_all_bins_pts": 3.75,
71
+ "claim_25k_ok": true,
72
+ "a_gap_ok": true,
73
+ "b_deciles_ok": true,
74
+ "c_controls_ok": true,
75
+ "verified_length_bin": 4096,
76
+ "caution": "verified_length_bin = 4096: the stricter every-bin (8K-16K) decile criterion fails in some deciles; the card may say 'preregistered 25K claim passed' but not 'verified at every length'",
77
+ "controlled_24k_by_decile": {
78
+ "0": 0.5234375,
79
+ "1": 0.515625,
80
+ "2": 0.546875,
81
+ "3": 0.5390625,
82
+ "4": 0.5546875,
83
+ "5": 0.5546875,
84
+ "6": 0.5703125,
85
+ "7": 0.5625,
86
+ "8": 0.5390625,
87
+ "9": 0.625
88
+ },
89
+ "controls_pooled_by_length": {
90
+ "1024": {
91
+ "question_only": {
92
+ "n": 320,
93
+ "accuracy": 0.290625,
94
+ "chance": 0.28003348214285717,
95
+ "p_above_chance": 0.3511374281963401,
96
+ "at_chance": true
97
+ },
98
+ "state_swap": {
99
+ "n": 320,
100
+ "accuracy": 0.315625,
101
+ "chance": 0.28003348214285717,
102
+ "p_above_chance": 0.07484199516389486,
103
+ "at_chance": true
104
+ }
105
+ },
106
+ "2048": {
107
+ "question_only": {
108
+ "n": 320,
109
+ "accuracy": 0.2875,
110
+ "chance": 0.28029017857142857,
111
+ "p_above_chance": 0.40558180434986946,
112
+ "at_chance": true
113
+ },
114
+ "state_swap": {
115
+ "n": 320,
116
+ "accuracy": 0.303125,
117
+ "chance": 0.28029017857142857,
118
+ "p_above_chance": 0.18406466252955953,
119
+ "at_chance": true
120
+ }
121
+ },
122
+ "4096": {
123
+ "question_only": {
124
+ "n": 320,
125
+ "accuracy": 0.253125,
126
+ "chance": 0.28010044642857146,
127
+ "p_above_chance": 0.8864307901965104,
128
+ "at_chance": true
129
+ },
130
+ "state_swap": {
131
+ "n": 320,
132
+ "accuracy": 0.29375,
133
+ "chance": 0.28010044642857146,
134
+ "p_above_chance": 0.304486548332512,
135
+ "at_chance": true
136
+ }
137
+ },
138
+ "8192": {
139
+ "question_only": {
140
+ "n": 320,
141
+ "accuracy": 0.28125,
142
+ "chance": 0.28131324404761904,
143
+ "p_above_chance": 0.5273741085290076,
144
+ "at_chance": true
145
+ },
146
+ "state_swap": {
147
+ "n": 320,
148
+ "accuracy": 0.284375,
149
+ "chance": 0.28131324404761904,
150
+ "p_above_chance": 0.4747527191420883,
151
+ "at_chance": true
152
+ }
153
+ },
154
+ "12288": {
155
+ "question_only": {
156
+ "n": 320,
157
+ "accuracy": 0.278125,
158
+ "chance": 0.2804017857142857,
159
+ "p_above_chance": 0.5644923218252806,
160
+ "at_chance": true
161
+ },
162
+ "state_swap": {
163
+ "n": 320,
164
+ "accuracy": 0.284375,
165
+ "chance": 0.2804017857142857,
166
+ "p_above_chance": 0.4593971620159274,
167
+ "at_chance": true
168
+ }
169
+ },
170
+ "16384": {
171
+ "question_only": {
172
+ "n": 320,
173
+ "accuracy": 0.265625,
174
+ "chance": 0.28533854166666667,
175
+ "p_above_chance": 0.8139517721332229,
176
+ "at_chance": true
177
+ },
178
+ "state_swap": {
179
+ "n": 320,
180
+ "accuracy": 0.28125,
181
+ "chance": 0.28533854166666667,
182
+ "p_above_chance": 0.5936977453471026,
183
+ "at_chance": true
184
+ }
185
+ },
186
+ "24576": {
187
+ "question_only": {
188
+ "n": 1280,
189
+ "accuracy": 0.2890625,
190
+ "chance": 0.2837481398809524,
191
+ "p_above_chance": 0.33937799270227736,
192
+ "at_chance": true
193
+ },
194
+ "state_swap": {
195
+ "n": 1280,
196
+ "accuracy": 0.28515625,
197
+ "chance": 0.2837481398809524,
198
+ "p_above_chance": 0.4658977510199284,
199
+ "at_chance": true
200
+ }
201
+ }
202
+ },
203
+ "laya_budgets": {
204
+ "laya_english": 512,
205
+ "laya_multilingual": 1024,
206
+ "laya_typed_decisions": 1024,
207
+ "source": "third_party/laya/README.md lines 366-368 (defaults); models/laya-official/.../multilingual/rl_agent_config.json max_len 1024"
208
+ },
209
+ "v3_context_tokens": 25600,
210
+ "sources": [
211
+ "runs/macjev/received/own_evals/main/test.own/long_grid_plus.jsonl + runs/macjev/received/own_evals/main/temperatures.json via macjev.eval.aggregate.suite_metrics('long_grid_plus') (recomputed); matches runs/macjev/report_r2/scoreboard/scoreboard.json long_grid_plus",
212
+ "chart_data.json#long_grid_plus.bins"
213
+ ],
214
+ "protocol_label": "v3 only (Laya cannot run >4K; not applicable). Controlled = real row correct AND question_only + state_swap controls pass; Wilson CI; 2K-4K reference; preregistered 2026-09-24, amended pre-data (+96 items per 24K decile, thresholds unchanged)",
215
+ "footnote": "v3 only. Laya's default input budget is 512 tokens (English) / 1,024 (multilingual, typed) per the Laya README;\nlonger rows exceed Laya's default budgets, so Laya is not plotted. Suite long_grid_plus: preregistered 2026-09-24, amended pre-data\n(+96 items per 24K depth decile, thresholds unchanged); 320 items per bin, 1,280 at 24K. Controlled accuracy = real-state\nanswer correct AND its question-only and state-swap controls pass; both controls are at chance in every length bin.\n25K claim rule: |24K \u2212 2K\u20134K reference| \u2264 5 pts and every 24K evidence-depth decile within 10 pts of it.\nPlotted values: figures/long_context.json (bin-level controlled accuracy and CIs)."
216
+ }
figures/long_context.png ADDED

Git LFS Details

  • SHA256: 957fb443f67d1ed91ef61fa83b0496347299823799bc97692d25893060b1e26d
  • Pointer size: 131 Bytes
  • Size of remote file: 225 kB
figures/long_context.svg ADDED
figures/multilingual.data.json ADDED
@@ -0,0 +1,486 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "multilingual",
3
+ "metric": "per-language MASSIVE intent accuracy, argmax over 20 scored options, 100 rows/language",
4
+ "sources": {
5
+ "v3": "runs/macjev/received/20260924-0313/evals/main/test/massive51.jsonl",
6
+ "laya_multilingual": "runs/macjev/laya_baselines/multilingual/massive51.jsonl",
7
+ "chart_data_entry": "chart_data.json#massive51.per_language (re-verified against raw files)"
8
+ },
9
+ "protocol": "14 trained + 37 locales held out of MASSIVE training for v3; held-out for all Laya checkpoints. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures. Wilson 95% CI per language (n=100)",
10
+ "chance": 0.05,
11
+ "three_x_chance": 0.15000000000000002,
12
+ "summary": {
13
+ "languages": 51,
14
+ "v3_beats_laya_multilingual": 51,
15
+ "v3_beats_best_laya_checkpoint": 51,
16
+ "v3_beats_laya_multilingual_never_trained": 37,
17
+ "v3_above_3x_chance": 51,
18
+ "laya_multilingual_above_3x_chance": 48,
19
+ "macro_v3": 0.7174509803921569,
20
+ "macro_laya_multilingual": 0.4007843137254902,
21
+ "macro_v3_never_trained37": 0.6551351351351351,
22
+ "macro_laya_multilingual_never_trained37": 0.36108108108108106,
23
+ "min_margin_vs_laya_multilingual": 0.11
24
+ },
25
+ "rows_plotted_sorted": [
26
+ {
27
+ "lang": "zh-CN",
28
+ "name": "Chinese (CN)",
29
+ "trained": true,
30
+ "v3": 0.95,
31
+ "laya_multilingual": 0.65,
32
+ "best_laya": 0.65,
33
+ "best_laya_ckpt": "multilingual"
34
+ },
35
+ {
36
+ "lang": "ja",
37
+ "name": "Japanese",
38
+ "trained": true,
39
+ "v3": 0.95,
40
+ "laya_multilingual": 0.64,
41
+ "best_laya": 0.64,
42
+ "best_laya_ckpt": "multilingual"
43
+ },
44
+ {
45
+ "lang": "es",
46
+ "name": "Spanish",
47
+ "trained": true,
48
+ "v3": 0.95,
49
+ "laya_multilingual": 0.58,
50
+ "best_laya": 0.58,
51
+ "best_laya_ckpt": "multilingual"
52
+ },
53
+ {
54
+ "lang": "ru",
55
+ "name": "Russian",
56
+ "trained": true,
57
+ "v3": 0.94,
58
+ "laya_multilingual": 0.57,
59
+ "best_laya": 0.57,
60
+ "best_laya_ckpt": "multilingual"
61
+ },
62
+ {
63
+ "lang": "en",
64
+ "name": "English",
65
+ "trained": true,
66
+ "v3": 0.93,
67
+ "laya_multilingual": 0.71,
68
+ "best_laya": 0.82,
69
+ "best_laya_ckpt": "english"
70
+ },
71
+ {
72
+ "lang": "zh-TW",
73
+ "name": "Chinese (TW)",
74
+ "trained": false,
75
+ "v3": 0.93,
76
+ "laya_multilingual": 0.61,
77
+ "best_laya": 0.61,
78
+ "best_laya_ckpt": "multilingual"
79
+ },
80
+ {
81
+ "lang": "de",
82
+ "name": "German",
83
+ "trained": true,
84
+ "v3": 0.93,
85
+ "laya_multilingual": 0.5,
86
+ "best_laya": 0.5,
87
+ "best_laya_ckpt": "multilingual"
88
+ },
89
+ {
90
+ "lang": "fr",
91
+ "name": "French",
92
+ "trained": true,
93
+ "v3": 0.9,
94
+ "laya_multilingual": 0.6,
95
+ "best_laya": 0.6,
96
+ "best_laya_ckpt": "typed"
97
+ },
98
+ {
99
+ "lang": "it",
100
+ "name": "Italian",
101
+ "trained": false,
102
+ "v3": 0.89,
103
+ "laya_multilingual": 0.52,
104
+ "best_laya": 0.52,
105
+ "best_laya_ckpt": "multilingual"
106
+ },
107
+ {
108
+ "lang": "pl",
109
+ "name": "Polish",
110
+ "trained": false,
111
+ "v3": 0.89,
112
+ "laya_multilingual": 0.5,
113
+ "best_laya": 0.5,
114
+ "best_laya_ckpt": "multilingual"
115
+ },
116
+ {
117
+ "lang": "ko",
118
+ "name": "Korean",
119
+ "trained": true,
120
+ "v3": 0.89,
121
+ "laya_multilingual": 0.47,
122
+ "best_laya": 0.47,
123
+ "best_laya_ckpt": "multilingual"
124
+ },
125
+ {
126
+ "lang": "pt",
127
+ "name": "Portuguese",
128
+ "trained": true,
129
+ "v3": 0.88,
130
+ "laya_multilingual": 0.5,
131
+ "best_laya": 0.5,
132
+ "best_laya_ckpt": "typed"
133
+ },
134
+ {
135
+ "lang": "hi",
136
+ "name": "Hindi",
137
+ "trained": true,
138
+ "v3": 0.87,
139
+ "laya_multilingual": 0.46,
140
+ "best_laya": 0.46,
141
+ "best_laya_ckpt": "multilingual"
142
+ },
143
+ {
144
+ "lang": "sv",
145
+ "name": "Swedish",
146
+ "trained": false,
147
+ "v3": 0.85,
148
+ "laya_multilingual": 0.49,
149
+ "best_laya": 0.49,
150
+ "best_laya_ckpt": "multilingual"
151
+ },
152
+ {
153
+ "lang": "ar",
154
+ "name": "Arabic",
155
+ "trained": true,
156
+ "v3": 0.85,
157
+ "laya_multilingual": 0.46,
158
+ "best_laya": 0.46,
159
+ "best_laya_ckpt": "multilingual"
160
+ },
161
+ {
162
+ "lang": "vi",
163
+ "name": "Vietnamese",
164
+ "trained": false,
165
+ "v3": 0.85,
166
+ "laya_multilingual": 0.34,
167
+ "best_laya": 0.34,
168
+ "best_laya_ckpt": "multilingual"
169
+ },
170
+ {
171
+ "lang": "fa",
172
+ "name": "Persian",
173
+ "trained": false,
174
+ "v3": 0.84,
175
+ "laya_multilingual": 0.51,
176
+ "best_laya": 0.51,
177
+ "best_laya_ckpt": "multilingual"
178
+ },
179
+ {
180
+ "lang": "id",
181
+ "name": "Indonesian",
182
+ "trained": false,
183
+ "v3": 0.84,
184
+ "laya_multilingual": 0.51,
185
+ "best_laya": 0.51,
186
+ "best_laya_ckpt": "multilingual"
187
+ },
188
+ {
189
+ "lang": "tr",
190
+ "name": "Turkish",
191
+ "trained": true,
192
+ "v3": 0.83,
193
+ "laya_multilingual": 0.4,
194
+ "best_laya": 0.4,
195
+ "best_laya_ckpt": "multilingual"
196
+ },
197
+ {
198
+ "lang": "nl",
199
+ "name": "Dutch",
200
+ "trained": false,
201
+ "v3": 0.82,
202
+ "laya_multilingual": 0.47,
203
+ "best_laya": 0.47,
204
+ "best_laya_ckpt": "multilingual"
205
+ },
206
+ {
207
+ "lang": "nb",
208
+ "name": "Norwegian",
209
+ "trained": false,
210
+ "v3": 0.81,
211
+ "laya_multilingual": 0.56,
212
+ "best_laya": 0.56,
213
+ "best_laya_ckpt": "multilingual"
214
+ },
215
+ {
216
+ "lang": "da",
217
+ "name": "Danish",
218
+ "trained": false,
219
+ "v3": 0.81,
220
+ "laya_multilingual": 0.52,
221
+ "best_laya": 0.52,
222
+ "best_laya_ckpt": "multilingual"
223
+ },
224
+ {
225
+ "lang": "ro",
226
+ "name": "Romanian",
227
+ "trained": false,
228
+ "v3": 0.77,
229
+ "laya_multilingual": 0.37,
230
+ "best_laya": 0.37,
231
+ "best_laya_ckpt": "multilingual"
232
+ },
233
+ {
234
+ "lang": "af",
235
+ "name": "Afrikaans",
236
+ "trained": false,
237
+ "v3": 0.76,
238
+ "laya_multilingual": 0.35,
239
+ "best_laya": 0.35,
240
+ "best_laya_ckpt": "multilingual"
241
+ },
242
+ {
243
+ "lang": "ta",
244
+ "name": "Tamil",
245
+ "trained": true,
246
+ "v3": 0.74,
247
+ "laya_multilingual": 0.31,
248
+ "best_laya": 0.31,
249
+ "best_laya_ckpt": "multilingual"
250
+ },
251
+ {
252
+ "lang": "sw",
253
+ "name": "Swahili",
254
+ "trained": true,
255
+ "v3": 0.74,
256
+ "laya_multilingual": 0.23,
257
+ "best_laya": 0.23,
258
+ "best_laya_ckpt": "multilingual"
259
+ },
260
+ {
261
+ "lang": "ms",
262
+ "name": "Malay",
263
+ "trained": false,
264
+ "v3": 0.73,
265
+ "laya_multilingual": 0.41,
266
+ "best_laya": 0.41,
267
+ "best_laya_ckpt": "multilingual"
268
+ },
269
+ {
270
+ "lang": "el",
271
+ "name": "Greek",
272
+ "trained": false,
273
+ "v3": 0.72,
274
+ "laya_multilingual": 0.44,
275
+ "best_laya": 0.44,
276
+ "best_laya_ckpt": "multilingual"
277
+ },
278
+ {
279
+ "lang": "he",
280
+ "name": "Hebrew",
281
+ "trained": false,
282
+ "v3": 0.72,
283
+ "laya_multilingual": 0.37,
284
+ "best_laya": 0.37,
285
+ "best_laya_ckpt": "multilingual"
286
+ },
287
+ {
288
+ "lang": "ur",
289
+ "name": "Urdu",
290
+ "trained": false,
291
+ "v3": 0.7,
292
+ "laya_multilingual": 0.42,
293
+ "best_laya": 0.42,
294
+ "best_laya_ckpt": "multilingual"
295
+ },
296
+ {
297
+ "lang": "az",
298
+ "name": "Azerbaijani",
299
+ "trained": false,
300
+ "v3": 0.66,
301
+ "laya_multilingual": 0.36,
302
+ "best_laya": 0.36,
303
+ "best_laya_ckpt": "multilingual"
304
+ },
305
+ {
306
+ "lang": "bn",
307
+ "name": "Bengali",
308
+ "trained": false,
309
+ "v3": 0.65,
310
+ "laya_multilingual": 0.45,
311
+ "best_laya": 0.45,
312
+ "best_laya_ckpt": "multilingual"
313
+ },
314
+ {
315
+ "lang": "sl",
316
+ "name": "Slovenian",
317
+ "trained": false,
318
+ "v3": 0.65,
319
+ "laya_multilingual": 0.37,
320
+ "best_laya": 0.37,
321
+ "best_laya_ckpt": "multilingual"
322
+ },
323
+ {
324
+ "lang": "th",
325
+ "name": "Thai",
326
+ "trained": false,
327
+ "v3": 0.64,
328
+ "laya_multilingual": 0.48,
329
+ "best_laya": 0.48,
330
+ "best_laya_ckpt": "multilingual"
331
+ },
332
+ {
333
+ "lang": "hu",
334
+ "name": "Hungarian",
335
+ "trained": false,
336
+ "v3": 0.63,
337
+ "laya_multilingual": 0.36,
338
+ "best_laya": 0.36,
339
+ "best_laya_ckpt": "multilingual"
340
+ },
341
+ {
342
+ "lang": "te",
343
+ "name": "Telugu",
344
+ "trained": false,
345
+ "v3": 0.62,
346
+ "laya_multilingual": 0.22,
347
+ "best_laya": 0.22,
348
+ "best_laya_ckpt": "multilingual"
349
+ },
350
+ {
351
+ "lang": "kn",
352
+ "name": "Kannada",
353
+ "trained": false,
354
+ "v3": 0.61,
355
+ "laya_multilingual": 0.3,
356
+ "best_laya": 0.3,
357
+ "best_laya_ckpt": "multilingual"
358
+ },
359
+ {
360
+ "lang": "fi",
361
+ "name": "Finnish",
362
+ "trained": false,
363
+ "v3": 0.6,
364
+ "laya_multilingual": 0.34,
365
+ "best_laya": 0.34,
366
+ "best_laya_ckpt": "multilingual"
367
+ },
368
+ {
369
+ "lang": "jv",
370
+ "name": "Javanese",
371
+ "trained": false,
372
+ "v3": 0.6,
373
+ "laya_multilingual": 0.3,
374
+ "best_laya": 0.3,
375
+ "best_laya_ckpt": "multilingual"
376
+ },
377
+ {
378
+ "lang": "hy",
379
+ "name": "Armenian",
380
+ "trained": false,
381
+ "v3": 0.59,
382
+ "laya_multilingual": 0.25,
383
+ "best_laya": 0.25,
384
+ "best_laya_ckpt": "multilingual"
385
+ },
386
+ {
387
+ "lang": "ka",
388
+ "name": "Georgian",
389
+ "trained": false,
390
+ "v3": 0.56,
391
+ "laya_multilingual": 0.15,
392
+ "best_laya": 0.15,
393
+ "best_laya_ckpt": "multilingual"
394
+ },
395
+ {
396
+ "lang": "tl",
397
+ "name": "Tagalog",
398
+ "trained": false,
399
+ "v3": 0.54,
400
+ "laya_multilingual": 0.35,
401
+ "best_laya": 0.35,
402
+ "best_laya_ckpt": "multilingual"
403
+ },
404
+ {
405
+ "lang": "km",
406
+ "name": "Khmer",
407
+ "trained": false,
408
+ "v3": 0.54,
409
+ "laya_multilingual": 0.2,
410
+ "best_laya": 0.2,
411
+ "best_laya_ckpt": "multilingual"
412
+ },
413
+ {
414
+ "lang": "sq",
415
+ "name": "Albanian",
416
+ "trained": false,
417
+ "v3": 0.52,
418
+ "laya_multilingual": 0.3,
419
+ "best_laya": 0.3,
420
+ "best_laya_ckpt": "multilingual"
421
+ },
422
+ {
423
+ "lang": "ml",
424
+ "name": "Malayalam",
425
+ "trained": false,
426
+ "v3": 0.52,
427
+ "laya_multilingual": 0.28,
428
+ "best_laya": 0.28,
429
+ "best_laya_ckpt": "multilingual"
430
+ },
431
+ {
432
+ "lang": "lv",
433
+ "name": "Latvian",
434
+ "trained": false,
435
+ "v3": 0.51,
436
+ "laya_multilingual": 0.31,
437
+ "best_laya": 0.31,
438
+ "best_laya_ckpt": "multilingual"
439
+ },
440
+ {
441
+ "lang": "is",
442
+ "name": "Icelandic",
443
+ "trained": false,
444
+ "v3": 0.5,
445
+ "laya_multilingual": 0.35,
446
+ "best_laya": 0.35,
447
+ "best_laya_ckpt": "multilingual"
448
+ },
449
+ {
450
+ "lang": "my",
451
+ "name": "Burmese",
452
+ "trained": false,
453
+ "v3": 0.46,
454
+ "laya_multilingual": 0.16,
455
+ "best_laya": 0.16,
456
+ "best_laya_ckpt": "multilingual"
457
+ },
458
+ {
459
+ "lang": "cy",
460
+ "name": "Welsh",
461
+ "trained": false,
462
+ "v3": 0.37,
463
+ "laya_multilingual": 0.13,
464
+ "best_laya": 0.17,
465
+ "best_laya_ckpt": "typed"
466
+ },
467
+ {
468
+ "lang": "mn",
469
+ "name": "Mongolian",
470
+ "trained": false,
471
+ "v3": 0.28,
472
+ "laya_multilingual": 0.16,
473
+ "best_laya": 0.16,
474
+ "best_laya_ckpt": "multilingual"
475
+ },
476
+ {
477
+ "lang": "am",
478
+ "name": "Amharic",
479
+ "trained": false,
480
+ "v3": 0.26,
481
+ "laya_multilingual": 0.15,
482
+ "best_laya": 0.16,
483
+ "best_laya_ckpt": "typed"
484
+ }
485
+ ]
486
+ }
figures/multilingual.png ADDED

Git LFS Details

  • SHA256: 544a3982019f7d0d8f788f4931ba88bde317956c87f7a7138fcd5e4ec22013d4
  • Pointer size: 131 Bytes
  • Size of remote file: 301 kB
figures/multilingual.svg ADDED
figures/quantization.data.json ADDED
@@ -0,0 +1,151 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "quantization",
3
+ "metric": "top-1 agreement with the full-precision reference (%), plus file size (GB = bytes/1e9)",
4
+ "rows": [
5
+ {
6
+ "tier": "16-bit",
7
+ "model": "v3",
8
+ "label": "Jev-Style 0.8B v3 \u00b7 GGUF F16",
9
+ "agree": 100.0,
10
+ "n": 240,
11
+ "agree_count": 240,
12
+ "size_gb": 1.516744128,
13
+ "size_note": "local export file (bytes / 1e9)",
14
+ "reference": "torch FP32",
15
+ "source": "runs/macjev/export_r2/main/parity/report.json; sizes: chart_data.json export.sizes"
16
+ },
17
+ {
18
+ "tier": "16-bit",
19
+ "model": "v3",
20
+ "label": "Jev-Style 0.8B v3 \u00b7 MLX bf16",
21
+ "agree": 100.0,
22
+ "n": 240,
23
+ "agree_count": 240,
24
+ "size_gb": 1.504827355,
25
+ "size_note": "local export file (bytes / 1e9)",
26
+ "reference": "torch FP32",
27
+ "source": "runs/macjev/export_r2/main/parity/report.json; sizes: chart_data.json export.sizes"
28
+ },
29
+ {
30
+ "tier": "16-bit",
31
+ "model": "v2",
32
+ "label": "Jev-Style 2B v2 \u00b7 GGUF BF16",
33
+ "agree": 99.6,
34
+ "n": 500,
35
+ "size_gb": 3.78,
36
+ "size_note": "as reported on card",
37
+ "reference": "CUDA merged BF16",
38
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/raw/main/README.md"
39
+ },
40
+ {
41
+ "tier": "16-bit",
42
+ "model": "v2",
43
+ "label": "Jev-Style 2B v2 \u00b7 MLX BF16",
44
+ "agree": 99.6,
45
+ "n": 500,
46
+ "size_gb": 3.76,
47
+ "size_note": "as reported on card",
48
+ "reference": "CUDA merged BF16",
49
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/raw/main/README.md (table: 'Native MLX BF16 | 3.76 GB | 99.6% choice agreement')"
50
+ },
51
+ {
52
+ "tier": "16-bit",
53
+ "model": "v1",
54
+ "label": "Jev-Style 2B v1 \u00b7 GGUF BF16",
55
+ "agree": 99.8,
56
+ "n": 500,
57
+ "size_gb": 3.9,
58
+ "size_note": "as reported on card",
59
+ "reference": "bf16",
60
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF/raw/main/README.md"
61
+ },
62
+ {
63
+ "tier": "8-bit",
64
+ "model": "v3",
65
+ "label": "Jev-Style 0.8B v3 \u00b7 GGUF Q8_0",
66
+ "agree": 100.0,
67
+ "n": 240,
68
+ "agree_count": 240,
69
+ "size_gb": 0.811843008,
70
+ "size_note": "local export file (bytes / 1e9)",
71
+ "reference": "torch FP32",
72
+ "source": "runs/macjev/export_r2/main/parity/report.json; sizes: chart_data.json export.sizes"
73
+ },
74
+ {
75
+ "tier": "8-bit",
76
+ "model": "v3",
77
+ "label": "Jev-Style 0.8B v3 \u00b7 MLX 8-bit",
78
+ "agree": 100.0,
79
+ "n": 240,
80
+ "agree_count": 240,
81
+ "size_gb": 0.799973748,
82
+ "size_note": "local export file (bytes / 1e9)",
83
+ "reference": "torch FP32",
84
+ "source": "runs/macjev/export_r2/main/parity/report.json; sizes: chart_data.json export.sizes"
85
+ },
86
+ {
87
+ "tier": "8-bit",
88
+ "model": "v2",
89
+ "label": "Jev-Style 2B v2 \u00b7 GGUF Q8_0",
90
+ "agree": 99.2,
91
+ "n": 500,
92
+ "size_gb": 2.01,
93
+ "size_note": "as reported on card",
94
+ "reference": "CUDA merged BF16",
95
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/raw/main/README.md"
96
+ },
97
+ {
98
+ "tier": "8-bit",
99
+ "model": "v1",
100
+ "label": "Jev-Style 2B v1 \u00b7 GGUF Q8_0",
101
+ "agree": 99.4,
102
+ "n": 500,
103
+ "size_gb": 2.1,
104
+ "size_note": "as reported on card",
105
+ "reference": "bf16",
106
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF/raw/main/README.md"
107
+ },
108
+ {
109
+ "tier": "4-bit",
110
+ "model": "v3",
111
+ "label": "Jev-Style 0.8B v3 \u00b7 GGUF Q4_K_M",
112
+ "agree": 100.0,
113
+ "n": 240,
114
+ "agree_count": 240,
115
+ "size_gb": 0.529296832,
116
+ "size_note": "local export file (bytes / 1e9)",
117
+ "reference": "torch FP32",
118
+ "source": "runs/macjev/export_r2/main/parity/report.json; sizes: chart_data.json export.sizes"
119
+ },
120
+ {
121
+ "tier": "4-bit",
122
+ "model": "v2",
123
+ "label": "Jev-Style 2B v2 \u00b7 GGUF Q4_K_M",
124
+ "agree": 91.4,
125
+ "n": 500,
126
+ "size_gb": 1.27,
127
+ "size_note": "as reported on card",
128
+ "reference": "CUDA merged BF16",
129
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/raw/main/README.md"
130
+ },
131
+ {
132
+ "tier": "4-bit",
133
+ "model": "v1",
134
+ "label": "Jev-Style 2B v1 \u00b7 GGUF Q4_K_M",
135
+ "agree": 94.39999999999999,
136
+ "n": 500,
137
+ "size_gb": 1.3,
138
+ "size_note": "as reported on card",
139
+ "reference": "bf16",
140
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF/raw/main/README.md"
141
+ }
142
+ ],
143
+ "deltas": {
144
+ "v2 Q4_K_M size / v3 Q4_K_M size": 2.4
145
+ },
146
+ "no_agreement_delta_claimed": "v3 parity rows are training-pool rows; v1/v2 used 500 held-out decisions, so no agreement gap is claimed",
147
+ "long_points": "v3 formats also 100% top-1 at 16,384 and 25,600 tokens (3 rows each), parity report long_points",
148
+ "not_plotted": "mlx-4bit (98.75%, 237/240) is not a published format and is omitted",
149
+ "footnote": "v3: top-1 agreement with the PyTorch FP32 reference on a 240-row parity fixture drawn from the training pool (22 categories, en+zh), plus 6 extra rows at about 16K and 25.6K tokens (6/6 agree). v3 sizes = exported weight files (GB = 10^9 bytes). 2B v1/v2: as reported on their public HF GGUF cards (500 held-out decisions each; vs bf16 for v1, vs CUDA merged BF16 for v2; card sizes). Different fixtures (training-pool rows for v3, held-out rows for v1/v2) and references: rows are not a paired comparison. x-axis starts at 88%.",
150
+ "protocol_label": "v3 G5 parity: 240-row mixed fixture (drawn from training-pool rows, 22 categories, en+zh) vs torch FP32; plus 16K and 25.6K long points (3 rows each). 2B numbers are from their cards on different fixtures (500 decisions vs bf16)"
151
+ }
figures/quantization.png ADDED

Git LFS Details

  • SHA256: bc75c212caa11008ca929a685d3085040c2b4698afb58d04a0ad692b815be230
  • Pointer size: 131 Bytes
  • Size of remote file: 254 kB
figures/quantization.svg ADDED
figures/zeroshot.json ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "zeroshot",
3
+ "metric": "accuracy (argmax over the set's labels; all rows, none unsupported)",
4
+ "plotted": {
5
+ "tweet_topic": {
6
+ "v3": 75.49,
7
+ "laya_en": 63.2,
8
+ "delta_pts_vs_laya_en": 12.29
9
+ },
10
+ "fin_topic": {
11
+ "v3": 46.71,
12
+ "laya_en": 34.2,
13
+ "delta_pts_vs_laya_en": 12.51
14
+ }
15
+ },
16
+ "jev_marker": {
17
+ "set": "tweet_topic",
18
+ "jev": 79.33,
19
+ "gap_pts": 3.84
20
+ },
21
+ "v3_recomputed": {
22
+ "tweet_topic": {
23
+ "correct": 1278,
24
+ "n": 1693,
25
+ "ci95": [
26
+ 0.7341996455995274,
27
+ 0.7749556999409333
28
+ ]
29
+ },
30
+ "fin_topic": {
31
+ "correct": 1923,
32
+ "n": 4117,
33
+ "ci95": [
34
+ 0.45202817585620597,
35
+ 0.48239008987126547
36
+ ]
37
+ }
38
+ },
39
+ "entries": [
40
+ "zeroshot.tweet_topic.accuracy.v3",
41
+ "zeroshot.tweet_topic.accuracy.laya_en",
42
+ "zeroshot.tweet_topic.accuracy.jev",
43
+ "zeroshot.fin_topic.accuracy.v3",
44
+ "zeroshot.fin_topic.accuracy.laya_en",
45
+ "zeroshot.fin_topic.accuracy.jev"
46
+ ],
47
+ "claims": [
48
+ "tweet_topic_accuracy_vs_laya_en",
49
+ "fin_topic_accuracy_vs_laya_en",
50
+ "tweet_topic_accuracy_vs_jev"
51
+ ],
52
+ "sources": {
53
+ "v3": "runs/macjev/received/ext_evals/main/zeroshot_topics/metrics.json comparison.clean[*].ours.accuracy; recomputed from ext.zeroshot.<set>.jsonl",
54
+ "jev_laya_en": "src/macjev/eval/external/zeroshot_topics.py PUBLISHED = elcronos results/cross_dataset_summary.json @ a1901bc3d520e73936de8d4326545c0cdcf742fb"
55
+ },
56
+ "not_plotted_on_purpose": "Jev on fin_topic (not a win, not within 5 pts); macro-F1 vs Jev",
57
+ "footnote": "Zero-shot for every system: neither set is in v3's training pool; accuracy over every row of the pinned test files. Jev (1.13, API) and English Laya: numbers published by the elcronos jev-vs-open-decision-models study with its own prompt (results/cross_dataset_summary.json @ a1901bc), not re-run by us. v3: scored by us on the identical rows, label sets and instruction, in v3's own input format."
58
+ }
figures/zeroshot.png ADDED

Git LFS Details

  • SHA256: 255e196d449c086130d0f06c493c64a7cc21730a38d21c1c2188641e4c08c002
  • Pointer size: 131 Bytes
  • Size of remote file: 129 kB
figures/zeroshot.svg ADDED
generation_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "eos_token_id": 248044,
4
+ "transformers_version": "5.17.0",
5
+ "use_cache": true
6
+ }
jev_style_decision.py ADDED
@@ -0,0 +1,517 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Jev-Style-0.8B-Decision-v3: typed decisions with transformers / PyTorch (CUDA, MPS, CPU).
2
+
3
+ Self-contained runtime for chaoliangUNSW/Jev-Style-0.8B-Decision-v3 (Apache-2.0). No dependency on any training code:
4
+ rendering, verdict readout and calibration are implemented below and reproduce the reference
5
+ implementation used for evaluation (see release_config.json -> "runtime_parity").
6
+
7
+ Calibration temperature: with no category (the default) probabilities use the global temperature of
8
+ readout_config.json (temperatures.global = 0.880); pass category=... (CLI --category, JSONL "category")
9
+ for the fitted group temperature of that category's family x question type x option-count bucket, or
10
+ temperature=... to override (1.0 = uncalibrated scores).
11
+ """
12
+ # ----------------------------------------------------------------------------------------------
13
+ # Shared core (identical in jev_style_decision.py, jev_style_decision_gguf.py and
14
+ # jev_style_decision_mlx.py): input rendering, verdict readout, calibrated probabilities.
15
+ #
16
+ # Input layout ("macjev-render-v1"; token segments are encoded separately and concatenated):
17
+ #
18
+ # State:\n<state>\n\n
19
+ # Question [<type>]: <question>\nOptions:\n
20
+ # - <option 1>\n ... - <option K>\n
21
+ # Judge each option:\n
22
+ # <option 1> ->\n ... <option K> ->\n
23
+ #
24
+ # Score of option k = logit(" yes") - logit(" no") at the k-th " ->" token (computed from the final
25
+ # hidden state and the tied embedding rows, float32). Probabilities = softmax(scores / T), where T is
26
+ # the calibration temperature shipped in readout_config.json:
27
+ # * no category given (the default): T = temperatures.global (the file's global temperature);
28
+ # * category="..." given: T = the fitted group temperature of (family of that category x question
29
+ # type x option-count bucket), or temperatures.global when that group was not fitted;
30
+ # * temperature=... given: that value (1.0 = uncalibrated scores).
31
+ # T is clamped to temperatures.clamp. Text inside the state or options is tokenised with special
32
+ # tokens disabled, so e.g. "<|im_end|>" in user text can never act as a control token.
33
+ #
34
+ # Budgets: whole input <= 25,600 tokens; question + options + readout ("head") <= 2,048 tokens.
35
+ # Larger inputs raise InputBudgetError. Nothing is ever truncated.
36
+ # ----------------------------------------------------------------------------------------------
37
+ import argparse
38
+ import hashlib
39
+ import json
40
+ import math
41
+ import sys
42
+ from pathlib import Path
43
+
44
+ import numpy as np
45
+
46
+ MODEL_NAME = "Jev-Style-0.8B-Decision-v3"
47
+ TEMPLATE_VERSION = "macjev-render-v1"
48
+ READOUT_FORMAT = "macjev-readout-v1"
49
+ CONTEXT_LIMIT = 25_600 # state + question + options + readout
50
+ HARD_HEAD_MAX = 2048 # question + options + readout
51
+ QTYPES = ("choice", "score", "noul")
52
+ HERE = Path(__file__).resolve().parent
53
+
54
+ # calibration families (category prefix -> family), same table the temperatures were fitted with
55
+ FAMILY_BY_CATEGORY_PREFIX = (("typed_official", "typed"), ("typed_synthetic", "typed_synth"), ("general_", "general"),
56
+ ("intent", "intent"), ("nli", "nli"), ("theme_", "theme"), ("mac_", "mac"),
57
+ ("long_", "long"))
58
+
59
+
60
+ class InputBudgetError(ValueError):
61
+ """The rendered input exceeds a token budget. Nothing was truncated."""
62
+
63
+
64
+ class QuestionError(ValueError):
65
+ """The question/options are malformed."""
66
+
67
+
68
+ # -- questions ------------------------------------------------------------------------------------
69
+ def option_names(question):
70
+ """Canonical option identifiers, in the order the probabilities are returned."""
71
+ if not isinstance(question, dict):
72
+ raise QuestionError("question must be a dict {'t', 'ins', 'crit'}")
73
+ t, crit = question.get("t"), question.get("crit")
74
+ if not isinstance(question.get("ins"), str) or not question["ins"].strip():
75
+ raise QuestionError("question text ('ins') must be a non-empty string")
76
+ if t == "choice":
77
+ if not isinstance(crit, dict) or not crit:
78
+ raise QuestionError("choice needs a non-empty dict {option name: description or None}")
79
+ return [str(k) for k in crit]
80
+ if t == "score":
81
+ if not isinstance(crit, list) or not 2 <= len(crit) <= 10:
82
+ raise QuestionError("score needs a list of 2..10 level descriptions")
83
+ return [str(i) for i in range(len(crit))]
84
+ if t == "noul":
85
+ if crit is not None and not isinstance(crit, dict):
86
+ raise QuestionError("noul criteria must be None or {'false': ..., 'true': ...}")
87
+ return ["false", "true"]
88
+ raise QuestionError(f"unknown question type {t!r} (expected one of {QTYPES})")
89
+
90
+
91
+ def make_question(question, options=None, qtype=None):
92
+ """Build a typed question.
93
+
94
+ * ``question`` already a dict {"t", "ins", "crit"}: validated and returned.
95
+ * ``qtype="choice"`` (default when ``options`` is given): ``options`` = {name: description or None}
96
+ or a list of names.
97
+ * ``qtype="score"``: ``options`` = list of 2..10 level descriptions (level 0 first).
98
+ * ``qtype="noul"`` (default when no options): a true/false statement; ``options`` may be
99
+ {"false": "...", "true": "..."} to describe the two outcomes.
100
+ """
101
+ if isinstance(question, dict):
102
+ q = dict(question)
103
+ else:
104
+ if qtype is None:
105
+ qtype = "choice" if options is not None else "noul"
106
+ if qtype == "choice":
107
+ if isinstance(options, (list, tuple)):
108
+ if len(set(map(str, options))) != len(options):
109
+ raise QuestionError("duplicate option names")
110
+ crit = {str(o): None for o in options}
111
+ else:
112
+ crit = options
113
+ elif qtype == "score":
114
+ crit = list(options) if options is not None else None
115
+ else:
116
+ crit = options
117
+ q = {"t": qtype, "ins": question, "crit": crit}
118
+ option_names(q)
119
+ return q
120
+
121
+
122
+ def serialize_state(state):
123
+ """Strings pass through unchanged; any other JSON value is serialised (ensure_ascii=False)."""
124
+ if isinstance(state, str):
125
+ return state
126
+ return json.dumps(state, ensure_ascii=False)
127
+
128
+
129
+ def _criterion(value):
130
+ if isinstance(value, str):
131
+ return value
132
+ return json.dumps(value, ensure_ascii=False, separators=(", ", ": "), default=str)
133
+
134
+
135
+ def render_options(question):
136
+ t, crit = question["t"], question.get("crit")
137
+ if t == "choice":
138
+ return [k if v is None or v == "" else f"{k}: {_criterion(v)}" for k, v in crit.items()]
139
+ if t == "score":
140
+ return [f"level {i}: {_criterion(c)}" for i, c in enumerate(crit)]
141
+ crit = crit or {}
142
+ false_c, true_c = crit.get("false"), crit.get("true")
143
+ return ["false: " + (_criterion(false_c) if false_c not in (None, "") else "no, the statement does not hold"),
144
+ "true: " + (_criterion(true_c) if true_c not in (None, "") else "yes, the statement holds")]
145
+
146
+
147
+ # -- tokenizer + renderer -----------------------------------------------------------------------
148
+ class TextEncoder:
149
+ """HF ``tokenizers`` tokenizer.json; no BOS/EOS, special tokens in text are split (never control tokens)."""
150
+
151
+ def __init__(self, tokenizer_json):
152
+ from tokenizers import Tokenizer
153
+ self.tk = Tokenizer.from_file(str(tokenizer_json))
154
+ self.tk.encode_special_tokens = True
155
+
156
+ def __call__(self, text):
157
+ return self.tk.encode(text, add_special_tokens=False).ids
158
+
159
+
160
+ class Rendered:
161
+ __slots__ = ("ids", "prefix_len", "slots", "names", "head_tokens")
162
+
163
+ def __init__(self, ids, prefix_len, slots, names, head_tokens):
164
+ self.ids, self.prefix_len, self.slots, self.names, self.head_tokens = ids, prefix_len, slots, names, head_tokens
165
+
166
+
167
+ class Renderer:
168
+ def __init__(self, encode, readout_cfg, max_len=CONTEXT_LIMIT, head_max=HARD_HEAD_MAX):
169
+ if readout_cfg.get("format") != READOUT_FORMAT or readout_cfg.get("template") != TEMPLATE_VERSION:
170
+ raise ValueError("readout_config.json is not a macjev-readout-v1 / macjev-render-v1 config")
171
+ if readout_cfg.get("readout") != "verdict":
172
+ raise ValueError("this runtime implements the verdict readout only")
173
+ if not 0 < int(max_len) <= CONTEXT_LIMIT:
174
+ raise ValueError(f"max_len must be in 1..{CONTEXT_LIMIT}")
175
+ if not 0 < int(head_max) <= HARD_HEAD_MAX:
176
+ raise ValueError(f"head_max must be in 1..{HARD_HEAD_MAX}")
177
+ self.enc, self.max_len, self.head_max = encode, int(max_len), int(head_max)
178
+ st = readout_cfg["slot_tokens"]
179
+ self.yes, self.no, arrow = int(st["yes"]["id"]), int(st["no"]["id"]), int(st["verdict_slot"]["id"])
180
+ for text, want in ((" yes", self.yes), (" no", self.no), (" ->", arrow)):
181
+ got = self.enc(text)
182
+ if got != [want]:
183
+ raise ValueError(f"tokenizer mismatch: {text!r} -> {got}, readout_config expects [{want}]")
184
+ self.arrow = [arrow]
185
+ self.newline = self.enc("\n")
186
+ self.dash = self.enc("- ")
187
+ self.judge = self.enc("Judge each option:\n")
188
+
189
+ def prefix_ids(self, state):
190
+ return self.enc("State:\n") + self.enc(serialize_state(state)) + self.enc("\n\n")
191
+
192
+ def render(self, state, question, head_max=None, max_len=None):
193
+ head_max = self.head_max if head_max is None else int(head_max)
194
+ max_len = self.max_len if max_len is None else min(int(max_len), self.max_len)
195
+ if head_max > HARD_HEAD_MAX:
196
+ raise InputBudgetError(f"head_max may not exceed {HARD_HEAD_MAX}")
197
+ names = option_names(question)
198
+ opts = [self.enc(o) for o in render_options(question)]
199
+ suffix = self.enc(f"Question [{question['t']}]: {question['ins']}\nOptions:\n")
200
+ for o in opts:
201
+ suffix += self.dash + o + self.newline
202
+ suffix += self.judge
203
+ rel = []
204
+ for o in opts:
205
+ suffix += o + self.arrow
206
+ rel.append(len(suffix) - 1)
207
+ suffix += self.newline
208
+ if len(suffix) > head_max:
209
+ raise InputBudgetError(f"question + options + readout need {len(suffix)} tokens; the head budget is "
210
+ f"{head_max} (hard cap {HARD_HEAD_MAX}). Nothing was truncated: shorten the "
211
+ f"question/options or split the options over several questions.")
212
+ prefix = self.prefix_ids(state)
213
+ ids = prefix + suffix
214
+ if len(ids) > max_len:
215
+ raise InputBudgetError(f"input needs {len(ids)} tokens (state {len(prefix)} + head {len(suffix)}); the "
216
+ f"limit is {max_len} (model maximum {CONTEXT_LIMIT}). Nothing was truncated: "
217
+ f"shorten the state.")
218
+ return Rendered(ids, len(prefix), [len(prefix) + s for s in rel], names, len(suffix))
219
+
220
+
221
+ # -- calibration ----------------------------------------------------------------------------------
222
+ def family(category):
223
+ for prefix, fam in FAMILY_BY_CATEGORY_PREFIX:
224
+ if category.startswith(prefix):
225
+ return fam
226
+ return "other"
227
+
228
+
229
+ def option_bucket(k):
230
+ return "2" if k <= 2 else "3-5" if k <= 5 else "6-10" if k <= 10 else "11-20" if k <= 20 else "21+"
231
+
232
+
233
+ def lookup_temperature(temps, category, qtype, n_options):
234
+ """Calibration temperature. ``category`` None/"" -> the global temperature; otherwise the fitted
235
+ group (family(category) x qtype x option bucket), falling back to the global temperature."""
236
+ g = None
237
+ if category:
238
+ g = (temps.get("groups") or {}).get(f"{family(category)}|{qtype}|{option_bucket(n_options)}")
239
+ t = g["T"] if g else temps.get("global", 1.0)
240
+ lo, hi = temps.get("clamp", [0.3, 5.0])
241
+ return float(min(hi, max(lo, t)))
242
+
243
+
244
+ def concentration(p):
245
+ k = len(p)
246
+ if k < 2:
247
+ return 1.0
248
+ ent = -(p * np.log(np.clip(p, 1e-12, 1.0))).sum()
249
+ return float(np.clip(1.0 - ent / math.log(k), 0.0, 1.0))
250
+
251
+
252
+ def _sha256(path):
253
+ h = hashlib.sha256()
254
+ with open(path, "rb") as f:
255
+ for b in iter(lambda: f.read(1 << 22), b""):
256
+ h.update(b)
257
+ return h.hexdigest()
258
+
259
+
260
+ def verify_manifest(model_dir, only=None):
261
+ """Re-hash the files listed in manifest.json (all, or those whose path starts with one of ``only``).
262
+ Documentation (README.md, figures/, assets/) is recorded in the manifest but not checked here, so
263
+ a card edit never makes the runtime refuse to load."""
264
+ model_dir = Path(model_dir)
265
+ man = json.loads((model_dir / "manifest.json").read_text())
266
+ bad, missing, checked = [], [], 0
267
+ for name, rec in man["files"].items():
268
+ if name == "README.md" or name.startswith(("assets/", "figures/")):
269
+ continue
270
+ if only and not any(name == o or name.startswith(o.rstrip("/") + "/") for o in only):
271
+ continue
272
+ p = model_dir / name
273
+ if not p.exists():
274
+ missing.append(name)
275
+ elif _sha256(p) != rec["sha256"]:
276
+ bad.append(name)
277
+ checked += 1
278
+ return {"ok": not bad and not missing, "checked": checked, "bad": bad, "missing": missing}
279
+
280
+
281
+ class DecisionBase:
282
+ """Backend-independent part. Subclasses implement ``_scores(rendered) -> list[float]`` and may
283
+ override ``_scores_many(list of rendered) -> list of list[float]`` (several questions, one state).
284
+
285
+ Calibration: ``category`` (constructor default or per call) selects the fitted group temperature
286
+ of that category's family; with no category anywhere, the global temperature of
287
+ readout_config.json (temperatures.global) is used."""
288
+ backend = "base"
289
+
290
+ def _setup(self, model_dir, tokenizer_json, category=None, head_max=HARD_HEAD_MAX, max_len=CONTEXT_LIMIT):
291
+ self.model_dir = Path(model_dir)
292
+ self.readout_config = json.loads((self.model_dir / "readout_config.json").read_text())
293
+ self.temperatures = self.readout_config["temperatures"]
294
+ self.default_category = category or None # None -> temperatures.global
295
+ self.encode = TextEncoder(tokenizer_json)
296
+ self.renderer = Renderer(self.encode, self.readout_config, max_len=max_len, head_max=head_max)
297
+
298
+ def temperature(self, question, category=None):
299
+ """T for ``question``: group temperature of ``category`` (or the constructor's default category);
300
+ the global temperature when neither is given."""
301
+ return lookup_temperature(self.temperatures, category or self.default_category, question["t"],
302
+ len(option_names(question)))
303
+
304
+ def _scores_many(self, rendered):
305
+ return [self._scores(r) for r in rendered]
306
+
307
+ def _result(self, r, q, scores, category=None, temperature=None):
308
+ scores = [float(x) for x in scores]
309
+ t = float(temperature) if temperature is not None else self.temperature(q, category)
310
+ z = np.asarray(scores, float) / t
311
+ if not np.all(np.isfinite(z)):
312
+ raise FloatingPointError("non-finite decision scores")
313
+ p = np.exp(z - z.max())
314
+ p /= p.sum()
315
+ i = int(p.argmax())
316
+ return {"answer": r.names[i], "probabilities": dict(zip(r.names, p.tolist())),
317
+ "scores": dict(zip(r.names, scores)), "temperature": t, "top_probability": float(p[i]),
318
+ "entropy_concentration": concentration(p), "input_tokens": len(r.ids), "head_tokens": r.head_tokens,
319
+ "model": MODEL_NAME, "backend": self.backend}
320
+
321
+ def decide(self, state, question, options=None, qtype=None, category=None, temperature=None, head_max=None):
322
+ """Score one question about ``state``.
323
+
324
+ Returns {"answer", "probabilities" {option: p}, "scores" {option: logit(yes)-logit(no)},
325
+ "temperature", "top_probability", "entropy_concentration", "input_tokens", "head_tokens"}.
326
+ Temperature: with no ``category`` (here or in the constructor) the global temperature of
327
+ readout_config.json is used; ``category`` picks the fitted group temperature of its family
328
+ (e.g. "mac_gate", "general_topic", "theme_routing", "intent", "typed_official");
329
+ ``temperature`` overrides both (1.0 = uncalibrated scores).
330
+ Raises InputBudgetError (never truncates) or QuestionError.
331
+ """
332
+ q = make_question(question, options, qtype)
333
+ r = self.renderer.render(state, q, head_max=head_max)
334
+ return self._result(r, q, self._scores(r), category, temperature)
335
+
336
+ def decide_many(self, state, questions, category=None, temperature=None, head_max=None):
337
+ """Several questions about the same state (each a dict {"t","ins","crit"}); results in order.
338
+ Same outputs as calling decide() per question. The llama.cpp runtime sends all questions in one
339
+ request and shares the state in whole 1,024-token ubatches (see JevStyleDecisionGGUF, also for
340
+ its faster, not bit-identical many_mode="batched"). All questions are rendered and
341
+ budget-checked before any scoring."""
342
+ qs = [make_question(q) for q in questions]
343
+ rs = [self.renderer.render(state, q, head_max=head_max) for q in qs]
344
+ if not rs:
345
+ return []
346
+ return [self._result(r, q, sc, category, temperature) for r, q, sc in zip(rs, qs, self._scores_many(rs))]
347
+
348
+
349
+ def base_arg_parser(description):
350
+ ap = argparse.ArgumentParser(description=description)
351
+ ap.add_argument("--model-dir", default=str(HERE), help="folder with the weights and readout_config.json")
352
+ ap.add_argument("--state", help="state as plain text")
353
+ ap.add_argument("--state-json", help="state as a JSON value")
354
+ ap.add_argument("--question", help="question text (or a JSON question {'t','ins','crit'})")
355
+ ap.add_argument("--options", help="JSON: {name: description} or [names] (choice); [levels] (score)")
356
+ ap.add_argument("--qtype", choices=QTYPES)
357
+ ap.add_argument("--category", help="calibration family key, e.g. mac_gate, general_topic, theme_routing, intent "
358
+ "(default: none -> the global temperature of readout_config.json)")
359
+ ap.add_argument("--temperature", type=float, help="override the calibrated temperature")
360
+ ap.add_argument("--head-max", type=int, default=HARD_HEAD_MAX)
361
+ ap.add_argument("--max-len", type=int, default=CONTEXT_LIMIT)
362
+ ap.add_argument("--jsonl", help="batch mode: input JSON lines {id?, state, question, options?, qtype?, "
363
+ "category?}; one JSON result per line on stdout")
364
+ ap.add_argument("--verify", action="store_true", help="check sha256 of the files in manifest.json first")
365
+ return ap
366
+
367
+
368
+ def run_cli(args, engine):
369
+ def one(rec):
370
+ q = rec["question"]
371
+ return engine.decide(rec.get("state", ""), q, options=rec.get("options"), qtype=rec.get("qtype"),
372
+ category=rec.get("category"), temperature=rec.get("temperature", args.temperature))
373
+ if args.jsonl:
374
+ src = sys.stdin if args.jsonl == "-" else open(args.jsonl, encoding="utf-8")
375
+ for n, line in enumerate(src):
376
+ if not line.strip():
377
+ continue
378
+ rec = json.loads(line)
379
+ rid = rec.get("id", n)
380
+ try:
381
+ out = {"id": rid, **one(rec)}
382
+ except (InputBudgetError, QuestionError) as e:
383
+ out = {"id": rid, "error": f"{type(e).__name__}: {e}"}
384
+ print(json.dumps(out, ensure_ascii=False), flush=True)
385
+ return 0
386
+ if args.question is None:
387
+ raise SystemExit("--question (or --jsonl) is required")
388
+ state = json.loads(args.state_json) if args.state_json is not None else (args.state or "")
389
+ question = args.question
390
+ if question.lstrip().startswith("{"):
391
+ question = json.loads(question)
392
+ rec = {"state": state, "question": question, "options": json.loads(args.options) if args.options else None,
393
+ "qtype": args.qtype, "category": args.category}
394
+ print(json.dumps(one(rec), ensure_ascii=False, indent=2))
395
+ return 0
396
+ # ---------------------------------------------------------------------------- end of shared core
397
+
398
+
399
+ # ------------------------------------------------------------------------------ PyTorch backend
400
+ ATTN_CHUNK = 1024
401
+ _CHUNKED_NAME = "jev_chunked_sdpa"
402
+ _CHUNKED_REGISTERED = False
403
+
404
+
405
+ def _chunked_sdpa_forward(module, query, key, value, attention_mask, dropout=0.0, scaling=None, is_causal=None,
406
+ **kwargs):
407
+ """Query-chunked SDPA (MPS / CPU): the same computation as transformers' sdpa, but the
408
+ [heads x queries x keys] score matrix is built ATTN_CHUNK queries at a time, so a 25,600-token
409
+ input does not need ~21 GB for one attention call. Inputs of <= ATTN_CHUNK tokens take the
410
+ unchanged sdpa path."""
411
+ import torch
412
+ from transformers.integrations.sdpa_attention import sdpa_attention_forward
413
+ q_len, kv_len = query.shape[2], key.shape[2]
414
+ chunk = int(getattr(module, "jev_attn_chunk", ATTN_CHUNK) or ATTN_CHUNK)
415
+ if q_len <= chunk:
416
+ return sdpa_attention_forward(module, query, key, value, attention_mask, dropout=dropout, scaling=scaling,
417
+ is_causal=is_causal, **kwargs)
418
+ causal = is_causal if is_causal is not None else getattr(module, "is_causal", True)
419
+ past = kv_len - q_len
420
+ outs = []
421
+ for s in range(0, q_len, chunk):
422
+ e = min(q_len, s + chunk)
423
+ if attention_mask is not None:
424
+ m = attention_mask if attention_mask.shape[-2] == 1 else attention_mask[:, :, s:e, :]
425
+ elif causal:
426
+ qpos = torch.arange(s + past, e + past, device=query.device)
427
+ kpos = torch.arange(kv_len, device=query.device)
428
+ m = (kpos[None, :] <= qpos[:, None])[None, None]
429
+ else:
430
+ m = None
431
+ out, _ = sdpa_attention_forward(module, query[:, :, s:e], key, value, m, dropout=dropout, scaling=scaling,
432
+ is_causal=False, **kwargs)
433
+ outs.append(out)
434
+ return torch.cat(outs, dim=1), None
435
+
436
+
437
+ def _enable_chunked_attention(model, chunk=ATTN_CHUNK):
438
+ global _CHUNKED_REGISTERED
439
+ if getattr(model.config, "_attn_implementation", None) not in ("sdpa", _CHUNKED_NAME):
440
+ return False
441
+ try:
442
+ from transformers import AttentionInterface
443
+ from transformers.masking_utils import ALL_MASK_ATTENTION_FUNCTIONS, AttentionMaskInterface
444
+ except ImportError:
445
+ return False
446
+ if not _CHUNKED_REGISTERED:
447
+ AttentionInterface.register(_CHUNKED_NAME, _chunked_sdpa_forward)
448
+ AttentionMaskInterface.register(_CHUNKED_NAME, ALL_MASK_ATTENTION_FUNCTIONS["sdpa"])
449
+ _CHUNKED_REGISTERED = True
450
+ model.set_attn_implementation(_CHUNKED_NAME)
451
+ for m in model.modules():
452
+ if hasattr(m, "is_causal"):
453
+ m.jev_attn_chunk = int(chunk)
454
+ return True
455
+
456
+
457
+ class JevStyleDecision(DecisionBase):
458
+ """Transformers / PyTorch runtime (CUDA, Apple MPS or CPU).
459
+
460
+ >>> m = JevStyleDecision(".") # float32 on the best available device
461
+ >>> m.decide({"messages": ["Refund still missing after 3 weeks"]},
462
+ ... "Which team should handle this ticket?",
463
+ ... options={"billing": "payments, refunds", "tech": "bugs, crashes", "sales": "pricing, plans"},
464
+ ... category="theme_routing")["probabilities"]
465
+ """
466
+ backend = "torch"
467
+
468
+ def __init__(self, model_dir=HERE, device=None, dtype="float32", category=None, head_max=HARD_HEAD_MAX,
469
+ max_len=CONTEXT_LIMIT, attn_chunk=ATTN_CHUNK, verify=False):
470
+ import torch
471
+ self.torch = torch
472
+ model_dir = Path(model_dir)
473
+ if verify:
474
+ res = verify_manifest(model_dir)
475
+ if not res["ok"]:
476
+ raise RuntimeError(f"integrity check failed: {res}")
477
+ self._setup(model_dir, model_dir / "tokenizer.json", category, head_max, max_len)
478
+ if device is None:
479
+ device = ("cuda" if torch.cuda.is_available() else
480
+ "mps" if getattr(torch.backends, "mps", None) and torch.backends.mps.is_available() else "cpu")
481
+ dt = getattr(torch, dtype) if isinstance(dtype, str) else dtype
482
+ try:
483
+ from transformers.models.qwen3_5.modeling_qwen3_5 import Qwen3_5ForCausalLM as cls
484
+ except ImportError as e:
485
+ raise ImportError("this model needs a transformers version with Qwen3.5 support "
486
+ "(transformers.models.qwen3_5)") from e
487
+ try:
488
+ model = cls.from_pretrained(str(model_dir), dtype=dt)
489
+ except TypeError: # transformers 4.x keyword
490
+ model = cls.from_pretrained(str(model_dir), torch_dtype=dt)
491
+ self.model = model.to(device).eval()
492
+ self.device, self.dtype = device, dt
493
+ self.chunked_attention = _enable_chunked_attention(self.model, attn_chunk) if device != "cuda" else False
494
+ w = self.model.get_output_embeddings().weight
495
+ self.direction = (w[self.renderer.yes].float() - w[self.renderer.no].float()).detach()
496
+
497
+ def _scores(self, r):
498
+ torch = self.torch
499
+ with torch.no_grad():
500
+ ids = torch.tensor([r.ids], device=self.device)
501
+ h = self.model.model(input_ids=ids, use_cache=False).last_hidden_state # final normed hidden states
502
+ hs = h[0, torch.tensor(r.slots, device=self.device)].float()
503
+ return (hs @ self.direction).cpu().tolist()
504
+
505
+
506
+ def main(argv=None):
507
+ ap = base_arg_parser(f"{MODEL_NAME}: typed decisions with transformers / PyTorch")
508
+ ap.add_argument("--device", choices=["cuda", "mps", "cpu"])
509
+ ap.add_argument("--dtype", default="float32", choices=["float32", "bfloat16", "float16"])
510
+ args = ap.parse_args(argv)
511
+ engine = JevStyleDecision(args.model_dir, device=args.device, dtype=args.dtype, category=args.category,
512
+ head_max=args.head_max, max_len=args.max_len, verify=args.verify)
513
+ return run_cli(args, engine)
514
+
515
+
516
+ if __name__ == "__main__":
517
+ raise SystemExit(main())
manifest.json ADDED
@@ -0,0 +1,181 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "jev-style-manifest-v1",
3
+ "repo": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3",
4
+ "created_unix": 1790261965.8405728,
5
+ "files": {
6
+ "LICENSE": {
7
+ "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a",
8
+ "bytes": 11544
9
+ },
10
+ "NOTICE": {
11
+ "sha256": "3c1bee42827901d754cf64e437cf8837ea94241348f5f4be9e6c4e14fbfc5716",
12
+ "bytes": 1966
13
+ },
14
+ "README.md": {
15
+ "sha256": "ccf12d32e307f97aa6818c098fc874aec0f3666c79af3834fee799f98eb6871b",
16
+ "bytes": 31905
17
+ },
18
+ "chat_template.jinja": {
19
+ "sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80",
20
+ "bytes": 7755
21
+ },
22
+ "config.json": {
23
+ "sha256": "3f56b6210db3c52e0bafb50da005eb52b6d340ab4ad980091879b397743675fc",
24
+ "bytes": 1790
25
+ },
26
+ "figures/beyond_laya.json": {
27
+ "sha256": "a6dfd582fe3aca32bcd67cf01b85190d92084a0790f7744dd03180df5923f946",
28
+ "bytes": 4357
29
+ },
30
+ "figures/beyond_laya.png": {
31
+ "sha256": "0dbd24837759acf05ed3a668016ecaea75a41a093b64588417d2fd750b95779b",
32
+ "bytes": 185526
33
+ },
34
+ "figures/beyond_laya.svg": {
35
+ "sha256": "4b64a4e71d3c9c05c9a785003bbfbd7269c5341bcc8abd560a2b156cfb771d6b",
36
+ "bytes": 22000
37
+ },
38
+ "figures/calibration.data.json": {
39
+ "sha256": "0396f5d48ba839c65718a7f1b449e7fa074503cbd1b9db48a2d40096c2d6142a",
40
+ "bytes": 3765
41
+ },
42
+ "figures/calibration.png": {
43
+ "sha256": "3176339f6e4f58c9efa583b101f5862141a326468cf177f2c50c3ce2d1adf06f",
44
+ "bytes": 148880
45
+ },
46
+ "figures/calibration.svg": {
47
+ "sha256": "7cdb177072406c736a03bcec696adacf037cf810f156dd8f94ab6cf24e1d76ec",
48
+ "bytes": 15047
49
+ },
50
+ "figures/design_table.data.json": {
51
+ "sha256": "84fd672a2c69d6881c199dc09ccbe7b1313b73bf7f56f54f574cf4eefe052003",
52
+ "bytes": 5313
53
+ },
54
+ "figures/design_table.png": {
55
+ "sha256": "5b907319a5a4f6ee8eea5800fbc732b95d15e941b1d0cd8d774d3705a1341744",
56
+ "bytes": 331936
57
+ },
58
+ "figures/design_table.svg": {
59
+ "sha256": "a0a23216766b235c950a071a614691b010dcea9f3a0c5bb8c6f8cc2a7ecae0bc",
60
+ "bytes": 26655
61
+ },
62
+ "figures/headline_typed.data.json": {
63
+ "sha256": "04380aa3823f80489ae37f893de0bda67ae2dae38fd07e37f88700f7c173f0ec",
64
+ "bytes": 3819
65
+ },
66
+ "figures/headline_typed.png": {
67
+ "sha256": "9e763e4cbd86a25299785b94a89945f2c3a7cab4b1c7d5000f9456c4d220a1f4",
68
+ "bytes": 175884
69
+ },
70
+ "figures/headline_typed.svg": {
71
+ "sha256": "a49ec30b28319d682d9465145bf6435ac4ec4ad11c46a28c552c9b851fb27f37",
72
+ "bytes": 20359
73
+ },
74
+ "figures/jevbench.data.json": {
75
+ "sha256": "f9173e65f5cd1fe9fadad8c93a8d00dbe5181e7314d49f18c6f547a0ed168543",
76
+ "bytes": 2645
77
+ },
78
+ "figures/jevbench.png": {
79
+ "sha256": "b75ab6864b4487d0a323c1019f7ce405fedcdb5a4d8dc910c86a29b3141b5388",
80
+ "bytes": 142720
81
+ },
82
+ "figures/jevbench.svg": {
83
+ "sha256": "2014d7664f268c2621e68226d4ec92eec6dd8dc7a04029024b5110a9341b812f",
84
+ "bytes": 12181
85
+ },
86
+ "figures/latency.data.json": {
87
+ "sha256": "c0eb510a69dfe1b1cd85beca6ba3c354be4984d588b4aae2672331606a119cbc",
88
+ "bytes": 1402
89
+ },
90
+ "figures/latency.png": {
91
+ "sha256": "d6d0d34b875997b8a733d4776dc34a1b564ddc64b48cea9d0def7f72a34af79e",
92
+ "bytes": 203081
93
+ },
94
+ "figures/latency.svg": {
95
+ "sha256": "c7bdf011eec488aae7197d653cb8486dfd8ccbd1a58f5ba222873de7410aca30",
96
+ "bytes": 24079
97
+ },
98
+ "figures/long_context.json": {
99
+ "sha256": "e6289e61da3bf4117ed4ed05820b99620e541c25cdeb5262e3a580fa9febff62",
100
+ "bytes": 5553
101
+ },
102
+ "figures/long_context.png": {
103
+ "sha256": "957fb443f67d1ed91ef61fa83b0496347299823799bc97692d25893060b1e26d",
104
+ "bytes": 224537
105
+ },
106
+ "figures/long_context.svg": {
107
+ "sha256": "ad7be559f77b11b56df84b4f286f9cea9c076d4caf47a2f8fb0f67ab46f7968b",
108
+ "bytes": 21403
109
+ },
110
+ "figures/multilingual.data.json": {
111
+ "sha256": "21018e6ed6ec35f30b2739a5fd0adfae10eee6aa6d24e0e877651317d7cc76e7",
112
+ "bytes": 9921
113
+ },
114
+ "figures/multilingual.png": {
115
+ "sha256": "544a3982019f7d0d8f788f4931ba88bde317956c87f7a7138fcd5e4ec22013d4",
116
+ "bytes": 300658
117
+ },
118
+ "figures/multilingual.svg": {
119
+ "sha256": "c4442d53ba20f1a3bfa3b5764db211a7a18db95c48ec834f78d454ca13efd7e4",
120
+ "bytes": 63997
121
+ },
122
+ "figures/quantization.data.json": {
123
+ "sha256": "a140816ef145a6dd096ef471e4afbff1cecd33adb663c6cb856e19845349fc2d",
124
+ "bytes": 5473
125
+ },
126
+ "figures/quantization.png": {
127
+ "sha256": "bc75c212caa11008ca929a685d3085040c2b4698afb58d04a0ad692b815be230",
128
+ "bytes": 253992
129
+ },
130
+ "figures/quantization.svg": {
131
+ "sha256": "70720bb46ec1e718dcbfac9a83d3b5d11d50b3a5cc3356915465d70d95f3d43b",
132
+ "bytes": 28756
133
+ },
134
+ "figures/zeroshot.json": {
135
+ "sha256": "003c087074b876efab789fadf94f51d134c604d5f2efce417557732de2272e9d",
136
+ "bytes": 1867
137
+ },
138
+ "figures/zeroshot.png": {
139
+ "sha256": "255e196d449c086130d0f06c493c64a7cc21730a38d21c1c2188641e4c08c002",
140
+ "bytes": 128831
141
+ },
142
+ "figures/zeroshot.svg": {
143
+ "sha256": "a148fa7cb8f4e0b6e529996441fadfa883595bfeb05dc93c4048609071655790",
144
+ "bytes": 14474
145
+ },
146
+ "generation_config.json": {
147
+ "sha256": "62153eb6c69f2e1f426beaa8002b7186437e949c7588167085df14e10e9c0a73",
148
+ "bytes": 116
149
+ },
150
+ "jev_style_decision.py": {
151
+ "sha256": "be38df723d83ef5fdffb8f7dfaee63b3c31e632b233fc3c7d5f6eb8c83fd994f",
152
+ "bytes": 25917
153
+ },
154
+ "model.safetensors": {
155
+ "sha256": "0f8c861605dcdfb356a63e056baa2106e81042e8981e8d0d26fe50b34599541e",
156
+ "bytes": 1504827608
157
+ },
158
+ "readout_config.json": {
159
+ "sha256": "01de9bcce7effbfd1ae0a3fa13e52d7fa9aa4dd1cd1e47977d6bbdcd5e63054a",
160
+ "bytes": 6236
161
+ },
162
+ "release_config.json": {
163
+ "sha256": "7ee881089c1ca5b019f98a26e60da4b0a454100d8cd557e5e352df3813aa31c0",
164
+ "bytes": 7088
165
+ },
166
+ "requirements.txt": {
167
+ "sha256": "855306c7cd8db9fcea2f865d61a0c53330c530c9dcf07e955a9250aaae2993f1",
168
+ "bytes": 201
169
+ },
170
+ "tokenizer.json": {
171
+ "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
172
+ "bytes": 19989325
173
+ },
174
+ "tokenizer_config.json": {
175
+ "sha256": "66e427c470fe580fe8c7b5725d857af23d8417e37fae62667ec698306a19987b",
176
+ "bytes": 1124
177
+ }
178
+ },
179
+ "readme_hashed": true,
180
+ "note": "manifest.json hashes every file of the repo except itself, README.md and figures/ included. The runtime --verify check skips the documentation (README.md, figures/, assets/) and checks every other file."
181
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0f8c861605dcdfb356a63e056baa2106e81042e8981e8d0d26fe50b34599541e
3
+ size 1504827608
readout_config.json ADDED
@@ -0,0 +1,247 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "macjev-readout-v1",
3
+ "model_name": "Jev-Style-0.8B-Decision-v3",
4
+ "readout": "verdict",
5
+ "template": "macjev-render-v1",
6
+ "slot_tokens": {
7
+ "yes": {
8
+ "text": " yes",
9
+ "id": 9542
10
+ },
11
+ "no": {
12
+ "text": " no",
13
+ "id": 874
14
+ },
15
+ "verdict_slot": {
16
+ "text": " ->",
17
+ "id": 1411
18
+ },
19
+ "letters": {
20
+ "A": 357,
21
+ "B": 417,
22
+ "C": 351,
23
+ "D": 414,
24
+ "E": 458,
25
+ "F": 426,
26
+ "G": 469,
27
+ "H": 462,
28
+ "I": 353,
29
+ "J": 604,
30
+ "K": 710,
31
+ "L": 436,
32
+ "M": 380,
33
+ "N": 443,
34
+ "O": 496,
35
+ "P": 387,
36
+ "Q": 1167,
37
+ "R": 423,
38
+ "S": 326,
39
+ "T": 345,
40
+ "U": 533,
41
+ "V": 629,
42
+ "W": 457,
43
+ "X": 1543,
44
+ "Y": 783,
45
+ "Z": 1799,
46
+ "a": 264,
47
+ "b": 292,
48
+ "c": 272,
49
+ "d": 293,
50
+ "e": 378,
51
+ "f": 281,
52
+ "g": 338,
53
+ "h": 304,
54
+ "i": 585,
55
+ "j": 492,
56
+ "k": 580,
57
+ "l": 324,
58
+ "m": 295,
59
+ "n": 307,
60
+ "o": 296,
61
+ "p": 280,
62
+ "q": 2715,
63
+ "r": 427,
64
+ "s": 274,
65
+ "t": 259,
66
+ "u": 560,
67
+ "v": 343,
68
+ "w": 288,
69
+ "x": 830,
70
+ "y": 374,
71
+ "z": 1110
72
+ }
73
+ },
74
+ "score": "per option k: logit[' yes'] - logit[' no'] at the k-th ' ->' slot, computed as h_slot . (w_yes - w_no) from the final normed hidden state and the tied embedding rows (float32)",
75
+ "probabilities": "softmax(scores / T); T = temperatures.groups['<family>|<qtype>|<option bucket>'].T (else temperatures.global), clamped to temperatures.clamp",
76
+ "families": {
77
+ "typed_official*": "typed",
78
+ "typed_synthetic*": "typed_synth",
79
+ "general_*": "general",
80
+ "intent*": "intent",
81
+ "nli*": "nli",
82
+ "theme_*": "theme",
83
+ "mac_*": "mac",
84
+ "long_*": "long",
85
+ "anything else": "other (global T)"
86
+ },
87
+ "option_buckets": [
88
+ "2",
89
+ "3-5",
90
+ "6-10",
91
+ "11-20",
92
+ "21+"
93
+ ],
94
+ "default_category": null,
95
+ "default_category_note": "no category given -> temperatures.global (the global temperature); pass category=... for the fitted group temperature of that category's family",
96
+ "tokenizer_sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
97
+ "budgets": {
98
+ "max_len": 25600,
99
+ "head_max": 2048,
100
+ "hard_head_max": 2048,
101
+ "note": "whole input <= max_len tokens, question+options+readout <= head_max; larger inputs raise InputBudgetError, nothing is truncated"
102
+ },
103
+ "temperatures": {
104
+ "version": "macjev-temperatures-v1",
105
+ "global": 0.8800546821789332,
106
+ "groups": {
107
+ "general|choice|3-5": {
108
+ "T": 0.8563796906101172,
109
+ "T_raw": 0.8524962478467872,
110
+ "n": 600,
111
+ "weight": 0.8571428571428571
112
+ },
113
+ "general|choice|6-10": {
114
+ "T": 0.7883347591845132,
115
+ "T_raw": 0.7797058230459033,
116
+ "n": 1000,
117
+ "weight": 0.9090909090909091
118
+ },
119
+ "general|noul|2": {
120
+ "T": 0.9702474579038656,
121
+ "T_raw": 1.0023209290239254,
122
+ "n": 300,
123
+ "weight": 0.75
124
+ },
125
+ "general|score|3-5": {
126
+ "T": 0.9583320444900012,
127
+ "T_raw": 0.9748039508642105,
128
+ "n": 500,
129
+ "weight": 0.8333333333333334
130
+ },
131
+ "intent|choice|11-20": {
132
+ "T": 0.8536586990515279,
133
+ "T_raw": 0.8530447902066461,
134
+ "n": 4233,
135
+ "weight": 0.9769213016385876
136
+ },
137
+ "intent|choice|21+": {
138
+ "T": 0.7508751181814466,
139
+ "T_raw": 0.737974406022733,
140
+ "n": 916,
141
+ "weight": 0.9015748031496063
142
+ },
143
+ "long|choice|2": {
144
+ "T": 1.0101601686032944,
145
+ "T_raw": 2.5327604766936602,
146
+ "n": 15,
147
+ "weight": 0.13043478260869565
148
+ },
149
+ "long|choice|3-5": {
150
+ "T": 0.9789458925825579,
151
+ "T_raw": 0.9984431087771811,
152
+ "n": 540,
153
+ "weight": 0.84375
154
+ },
155
+ "long|choice|6-10": {
156
+ "T": 0.6918332634850016,
157
+ "T_raw": 0.6260903832047418,
158
+ "n": 241,
159
+ "weight": 0.7067448680351907
160
+ },
161
+ "long|noul|2": {
162
+ "T": 0.8183260460233314,
163
+ "T_raw": 0.7857458244621078,
164
+ "n": 179,
165
+ "weight": 0.6415770609318996
166
+ },
167
+ "long|score|3-5": {
168
+ "T": 0.8412851094005789,
169
+ "T_raw": 0.7952160414763221,
170
+ "n": 80,
171
+ "weight": 0.4444444444444444
172
+ },
173
+ "mac|choice|3-5": {
174
+ "T": 0.7640449061346866,
175
+ "T_raw": 0.753267027600698,
176
+ "n": 995,
177
+ "weight": 0.908675799086758
178
+ },
179
+ "mac|noul|2": {
180
+ "T": 0.922463080251143,
181
+ "T_raw": 0.9300179680535603,
182
+ "n": 577,
183
+ "weight": 0.8522895125553914
184
+ },
185
+ "mac|score|3-5": {
186
+ "T": 2.729450780811877,
187
+ "T_raw": 4.999707266277221,
188
+ "n": 187,
189
+ "weight": 0.6515679442508711
190
+ },
191
+ "nli|choice|3-5": {
192
+ "T": 1.003611674961589,
193
+ "T_raw": 1.0108425661039413,
194
+ "n": 1830,
195
+ "weight": 0.9481865284974094
196
+ },
197
+ "theme|choice|6-10": {
198
+ "T": 0.9848729122522978,
199
+ "T_raw": 0.9981552921230235,
200
+ "n": 840,
201
+ "weight": 0.8936170212765957
202
+ },
203
+ "theme|noul|2": {
204
+ "T": 0.896799495386032,
205
+ "T_raw": 0.89763584528285,
206
+ "n": 2022,
207
+ "weight": 0.9528746465598492
208
+ },
209
+ "typed|choice|3-5": {
210
+ "T": 0.9751541851571508,
211
+ "T_raw": 1.0336976619219436,
212
+ "n": 176,
213
+ "weight": 0.6376811594202898
214
+ },
215
+ "typed|noul|2": {
216
+ "T": 0.9892589014450378,
217
+ "T_raw": 1.054189849787923,
218
+ "n": 184,
219
+ "weight": 0.647887323943662
220
+ },
221
+ "typed|score|3-5": {
222
+ "T": 1.0056628114581752,
223
+ "T_raw": 1.0631515969968885,
224
+ "n": 240,
225
+ "weight": 0.7058823529411765
226
+ }
227
+ },
228
+ "clamp": [
229
+ 0.3,
230
+ 5.0
231
+ ],
232
+ "shrinkage_k": 100.0,
233
+ "key": "family|qtype|option_bucket",
234
+ "n_rows": 15655,
235
+ "fitted_on": [
236
+ "pool_cal.jsonl:dfb7e9beff96c6c5"
237
+ ],
238
+ "fit_quality": {
239
+ "nll_before": 0.37752055301970966,
240
+ "nll_after": 0.36671188108641545,
241
+ "ece_before": 0.03289307373502533,
242
+ "ece_after": 0.011376234101433989,
243
+ "n": 15655
244
+ }
245
+ },
246
+ "temperatures_sha256": "39ad8f6633934ff67770725993d7354ebebd6e7a08f73e50cab04df24715f2ce"
247
+ }
release_config.json ADDED
@@ -0,0 +1,186 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "jev-style-release-v1",
3
+ "model_name": "Jev-Style-0.8B-Decision-v3",
4
+ "repo": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3",
5
+ "generation": "v3 (third generation of the Jev-Style decision series)",
6
+ "lineage": {
7
+ "v1": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision",
8
+ "v1_public_gguf": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF",
9
+ "v2": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2",
10
+ "v3": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3"
11
+ },
12
+ "base_model": "Qwen/Qwen3.5-0.8B",
13
+ "base_model_revision": "2fc06364715b967f1860aea9cf38778875588b17",
14
+ "base_model_relation": "finetune",
15
+ "architecture": "Qwen3_5ForCausalLM (text only, 24 layers: 18 Gated DeltaNet + 6 full attention, hidden 1024, tied embeddings, 752,393,024 parameters)",
16
+ "readout": "verdict",
17
+ "template": "macjev-render-v1",
18
+ "readout_config": "readout_config.json",
19
+ "budgets": {
20
+ "max_len": 25600,
21
+ "head_max": 2048
22
+ },
23
+ "source": {
24
+ "checkpoint_sha256": {
25
+ "best-0.safetensors": "b10d249adfa3e1475f0f633ab961130c06f16c898db23c846b035d24bdc5c945"
26
+ },
27
+ "text_only_model_safetensors_sha256": "0f8c861605dcdfb356a63e056baa2106e81042e8981e8d0d26fe50b34599541e",
28
+ "release_manifest_sha256": "4bc9f89795ccf2d749008de7739bc5294f3ec543404ca2db9cac6a3914cfb84a",
29
+ "llama_cpp_commit": "441df11f65ea0b6d0c72965aaf70c8241070ddcb",
30
+ "mlx": "0.32.2",
31
+ "mlx_lm": "0.31.3"
32
+ },
33
+ "calibration": {
34
+ "version": "macjev-temperatures-v1",
35
+ "global_T": 0.8800546821789332,
36
+ "groups": 20,
37
+ "n_rows": 15655,
38
+ "fit_quality": {
39
+ "nll_before": 0.37752055301970966,
40
+ "nll_after": 0.36671188108641545,
41
+ "ece_before": 0.03289307373502533,
42
+ "ece_after": 0.011376234101433989,
43
+ "n": 15655
44
+ },
45
+ "fitted_on": "calibration pool (dev/cal rows, never test rows)"
46
+ },
47
+ "g5_parity": {
48
+ "reference": "PyTorch float32 (same weights, same rendered inputs)",
49
+ "rows": 240,
50
+ "rows_note": "non-sealed training-distribution rows, 22 categories, en 215 / zh 25",
51
+ "report_sha256": "056dc3a3d1de5616f97dab3f8b5ec1f7e2c6c85258aa03f2a57b9fd65a71022b",
52
+ "g5_pass": true,
53
+ "verdict_readout": {
54
+ "gguf-f16": {
55
+ "top1_agreement": 1.0,
56
+ "dnll": 2.436279701012456e-05,
57
+ "n": 240,
58
+ "nan_rows": 0,
59
+ "gate": "top1>=0.99, |dNLL|<=0.02, no NaN, every reference row scored",
60
+ "pass": true
61
+ },
62
+ "gguf-q8_0": {
63
+ "top1_agreement": 1.0,
64
+ "dnll": -3.093173930673876e-05,
65
+ "n": 240,
66
+ "nan_rows": 0,
67
+ "gate": "top1>=0.98, |dNLL|<=0.02, no NaN, every reference row scored",
68
+ "pass": true
69
+ },
70
+ "gguf-q4_k_m": {
71
+ "top1_agreement": 1.0,
72
+ "dnll": 0.006187511546346225,
73
+ "n": 240,
74
+ "nan_rows": 0,
75
+ "gate": "report_only",
76
+ "pass": true
77
+ },
78
+ "mlx-bf16": {
79
+ "top1_agreement": 1.0,
80
+ "dnll": 0.00031018251516545803,
81
+ "n": 240,
82
+ "nan_rows": 0,
83
+ "gate": "top1>=0.99, |dNLL|<=0.02, no NaN, every reference row scored",
84
+ "pass": true
85
+ },
86
+ "mlx-bf16-f32act": {
87
+ "top1_agreement": 1.0,
88
+ "dnll": 0.00015659911501769708,
89
+ "n": 240,
90
+ "nan_rows": 0,
91
+ "gate": "top1>=0.99, |dNLL|<=0.02, no NaN, every reference row scored",
92
+ "pass": true
93
+ },
94
+ "mlx-8bit": {
95
+ "top1_agreement": 1.0,
96
+ "dnll": 0.00023178691602621093,
97
+ "n": 240,
98
+ "nan_rows": 0,
99
+ "gate": "top1>=0.98, |dNLL|<=0.02, no NaN, every reference row scored",
100
+ "pass": true
101
+ },
102
+ "mlx-4bit": {
103
+ "top1_agreement": 0.9875,
104
+ "dnll": 0.010326217052955111,
105
+ "n": 240,
106
+ "nan_rows": 0,
107
+ "gate": "report_only",
108
+ "pass": true
109
+ }
110
+ },
111
+ "gates": "top-1 >= 0.99 (16-bit) / >= 0.98 (8-bit), |dNLL| <= 0.02, no NaN; 4-bit report-only"
112
+ },
113
+ "decision_index": "not run for this release (optional follow-up)",
114
+ "related_repos": {
115
+ "main": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3",
116
+ "gguf": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3-GGUF",
117
+ "mlx-bf16": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-bf16",
118
+ "mlx-8bit": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-8bit"
119
+ },
120
+ "not_released": {
121
+ "mlx-4bit": "affine4-g64 reported only (G5 top-1 0.9875, report-only gate); not released"
122
+ },
123
+ "tested_with": {
124
+ "python": "3.12",
125
+ "torch": "2.14.0",
126
+ "transformers": "5.17.0",
127
+ "tokenizers": "0.23.2",
128
+ "numpy": "2.5.3",
129
+ "mlx": "0.32.2",
130
+ "mlx-lm": "0.31.3",
131
+ "llama.cpp": "441df11f65ea0b6d0c72965aaf70c8241070ddcb"
132
+ },
133
+ "weights": {
134
+ "model.safetensors": "bfloat16 (text-only export of the trained checkpoint)"
135
+ },
136
+ "runtime": {
137
+ "script": "jev_style_decision.py",
138
+ "class": "JevStyleDecision",
139
+ "default_dtype": "float32",
140
+ "devices": [
141
+ "cuda",
142
+ "mps",
143
+ "cpu"
144
+ ],
145
+ "long_inputs": "query-chunked SDPA on MPS/CPU (1024 queries per chunk)"
146
+ },
147
+ "runtime_parity": {
148
+ "protocol": "2026-09-24: each runtime script run in a clean subprocess (cwd = repo folder, empty PYTHONPATH, HF_HUB_OFFLINE=1, --verify) on 24 fixed parity-fixture rows (every 10th of the 240 G5 rows; 12 categories families, choice/score/noul, 92-8156 tokens) and compared with the reference (training-code) scorer of the same format on the same inputs, probabilities with the same calibration temperature; plus 2 synthetic long states (16,381 and 25,582 tokens). Rendered token ids identical to the reference renderer on 244/244 rows (240 fixture rows + 4 adversarial special-token/unicode rows). Re-run 2026-09-24 after the rename to Jev-Style-0.8B-Decision-v3, the no-category = global-temperature default and the GGUF general.name metadata edit: all six formats again identical to the reference scorer run on the original (pre-edit) files (max |prob diff| 0.0 same backend, 0 top-1 changes). Re-run 2026-09-25 after the GGUF decide_many fix (runtime code change in decide_many only, docstrings in all scripts): all six formats gave outputs identical (0.0) to the 2026-09-24 run on the 24 rows.",
149
+ "results": {
150
+ "torch-fp32-cpu": {
151
+ "rows": 24,
152
+ "errors": 0,
153
+ "vs_reference_same_backend": {
154
+ "max_abs_prob_diff": 0.0,
155
+ "max_abs_score_diff": 0.0,
156
+ "top1_changes": 0,
157
+ "top1_agreement": 1.0
158
+ },
159
+ "vs_reference_torch_fp32": {
160
+ "max_abs_prob_diff": 0.0,
161
+ "max_abs_score_diff": 0.0,
162
+ "top1_changes": 0,
163
+ "top1_agreement": 1.0
164
+ }
165
+ },
166
+ "torch-fp32-mps-long": {
167
+ "long16384:torch-mps-fp32": {
168
+ "tokens": 16381,
169
+ "answer_matches_gguf_mlx": true,
170
+ "max_abs_prob_diff_vs_gguf_f16": 0.0004853327842702093
171
+ },
172
+ "long25600:torch-mps-fp32": {
173
+ "tokens": 25582,
174
+ "answer_matches_gguf_mlx": true,
175
+ "max_abs_prob_diff_vs_gguf_f16": 0.0002457350394169111
176
+ }
177
+ },
178
+ "render_token_identity": {
179
+ "rows": 244,
180
+ "identical": 244,
181
+ "different": 0,
182
+ "bad": []
183
+ }
184
+ }
185
+ }
186
+ }
requirements.txt ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ # tested with: python 3.12, torch 2.14.0, transformers 5.17.0, tokenizers 0.23.2, numpy 2.5.3
2
+ torch>=2.4
3
+ transformers>=5.0 # needs Qwen3.5 support (transformers.models.qwen3_5)
4
+ tokenizers>=0.21
5
+ numpy
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523
3
+ size 19989325
tokenizer_config.json ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|im_end|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": true,
13
+ "local_files_only": false,
14
+ "model_max_length": 262144,
15
+ "model_specific_special_tokens": {
16
+ "audio_bos_token": "<|audio_start|>",
17
+ "audio_eos_token": "<|audio_end|>",
18
+ "audio_token": "<|audio_pad|>",
19
+ "image_token": "<|image_pad|>",
20
+ "video_token": "<|video_pad|>",
21
+ "vision_bos_token": "<|vision_start|>",
22
+ "vision_eos_token": "<|vision_end|>"
23
+ },
24
+ "pad_token": "<|endoftext|>",
25
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
26
+ "split_special_tokens": false,
27
+ "tokenizer_class": "Qwen2Tokenizer",
28
+ "unk_token": null,
29
+ "video_token": "<|video_pad|>",
30
+ "vision_bos_token": "<|vision_start|>",
31
+ "vision_eos_token": "<|vision_end|>"
32
+ }