chaoliangUNSW commited on
Commit
1b7b039
·
verified ·
1 Parent(s): adbabb3

Jev-Style-2B-Decision-v3: release

Browse files
.gitattributes CHANGED
@@ -33,3 +33,7 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ figures/banner.png filter=lfs diff=lfs merge=lfs -text
37
+ figures/jevbench.png filter=lfs diff=lfs merge=lfs -text
38
+ figures/zeroshot.png filter=lfs diff=lfs merge=lfs -text
39
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
NOTICE ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ chaoliangUNSW/Jev-Style-2B-Decision-v3
2
+ Copyright 2026 chaoliangUNSW. Licensed under the Apache License, Version 2.0 (see LICENSE).
3
+
4
+ This model is a fine-tuned derivative of Qwen3.5-2B (https://huggingface.co/Qwen/Qwen3.5-2B,
5
+ revision 15852e8c16360a2fea060d615a32b45270f8a8fc), Copyright 2026 Alibaba Cloud, licensed under the Apache
6
+ License, Version 2.0. The LICENSE file in this repository is the license file distributed with Qwen3.5-2B.
7
+
8
+ Modifications relative to Qwen3.5-2B:
9
+ - all text-model weights were fine-tuned (full fine-tuning, bf16 training; the EMA weights were selected) to
10
+ score typed decision questions (choice / score / true-false) with a verdict readout: logit(" yes") -
11
+ logit(" no") at one " ->" slot per option, using the v2 input protocol (render v2 with long-option
12
+ catalogues; block-causal attention in the full-attention layers over 2,048-token blocks); one global
13
+ calibration temperature was fitted afterwards (readout_config.json);
14
+ - the vision tower and the multi-token-prediction head were removed; the checkpoint is a text-only
15
+ Qwen3_5ForCausalLM with tied input/output embeddings;
16
+ - added the runtime script, readout/release configuration files, the integrity manifest and this NOTICE.
17
+
18
+ The question types (choice / score / noul) follow the typed-decision convention of Laya
19
+ (https://github.com/NandhaKishorM/laya, Apache-2.0) so both models can be evaluated on the same
20
+ inputs. No Laya code or weights are included.
21
+
22
+ Part of the third generation (v3) of the Jev-Style decision series. Earlier generations: v1 =
23
+ chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision (public GGUF release:
24
+ chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF) and v2 = chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2
25
+ (both built on Qwen3.5-2B-Base); the v3 series also contains chaoliangUNSW/Jev-Style-0.8B-Decision-v3 (fine-tuned from
26
+ Qwen3.5-0.8B, v1 input protocol). This model was fine-tuned from Qwen/Qwen3.5-2B; no weights of the earlier
27
+ models were used. Its input protocol (render v2 + block attention) differs from the 0.8B v3 models: use the
28
+ runtime of this repository (jev_style_decision.py).
29
+
30
+ Not affiliated with, endorsed by or connected to TypeSafe or Jev. "Jev-Style" only describes the kind
31
+ of model (a small typed-decision model in a similar style); no Jev weights, code or outputs are included.
32
+ Not affiliated with or endorsed by Alibaba Cloud / the Qwen team or the Laya authors.
README.md ADDED
@@ -0,0 +1,371 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ base_model: Qwen/Qwen3.5-2B
4
+ base_model_relation: finetune
5
+ library_name: jev-style
6
+ pipeline_tag: text-classification
7
+ tags:
8
+ - decision-model
9
+ - decision-making
10
+ - transformers
11
+ - jev-style
12
+ - system-one
13
+ - calibration
14
+ - classification
15
+ - long-context
16
+ - qwen3.5
17
+ - on-device
18
+ - llm-routing
19
+ - guardrails
20
+ ---
21
+
22
+ # Jev-Style-2B-Decision-v3
23
+
24
+ **[Try it in your browser →](https://huggingface.co/spaces/chaoliangUNSW/jev-style-2b)**
25
+
26
+ **Website:** [jevstyle.com](https://jevstyle.com/#v3-2b) · **GitHub:** [jev-style](https://github.com/lawrence3699/jev-style) · **Collection:** [all v3 builds and demos](https://huggingface.co/collections/chaoliangUNSW/jev-style-decision-v3-08b-2b-6ab87f32380cbd8c03b608b9)
27
+
28
+ **Jev-Style decision series:** [v1 · 2B](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF) → [v2 · 2B](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2) → [v3 · 0.8B](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3) → **v3 · 2B (this model)**
29
+
30
+ <!-- PIP_SNIPPET_AFTER_0.3.0 -->
31
+
32
+ **Jev-style decisions, now at 2B.** Give it a state and typed questions; it returns a calibrated probability for
33
+ every option in one pass. 1.88B parameters (text-only Qwen3.5-2B), open weights, Apache-2.0.
34
+
35
+ ![Jev-Style 2B Decision v3: 73.6% on JevBench v1.4.1 public items, the highest among the Qwen3.5-2B-family systems on the board; Jev is ahead at 86.6%; shown are the Qwen3.5-2B-family systems, our 0.8B v3, Laya and Jev, and 42 of the 82 board systems score higher; 25,600 tokens per call with no option cap](figures/banner.png)
36
+
37
+ | Public benchmark | **Jev-Style v3 · 2B** | Jev-Style v3 · 0.8B | Jev 1.13 (API) |
38
+ |---|:---:|:---:|:---:|
39
+ | JevBench v1.4.1, 231 public items ↑ | 73.6% | 64.1% | 86.6% |
40
+ | tweet_topic, zero-shot, accuracy ↑ | 82.2% | 75.5% | 79.3%¹ |
41
+ | fin_topic, zero-shot, accuracy ↑ | 61.1% | 46.7% | 67.0%¹ |
42
+ | Longest input per call | 25,600 tokens, no option cap | 25,600 tokens | |
43
+
44
+ <sub>2B v3: GGUF F16 engine, one global temperature, each benchmark run once as pre-declared. JevBench: self-run with the official harness, not an official board entry; 95% CI 67.6–78.9% (Wilson). Jev is well ahead of the 2B on JevBench and ahead on fin_topic. ¹ Jev numbers from the elcronos study (raw API), not re-run by us. On tweet_topic the 2B's macro-F1 (67.8%) is below Jev's (69.4%). Details: [Results](#results).</sub>
45
+
46
+ **73.6% on JevBench.** On the 231 public items of JevBench v1.4.1 this is the highest JevBench public accuracy
47
+ among the Qwen3.5-2B-family systems on the v1.4.1 board (decider-2b 71.0%, open-jev-zefan-2b 64.5%), and
48
+ +9.5 points over our 0.8B v3. The 95% CI (67.6–78.9%) includes decider-2b's 71.0%, so that lead is a point
49
+ estimate. Jev (86.6%) is well ahead.
50
+
51
+ **25,600 tokens, no option cap.** State, questions and every option share one 25,600-token budget. There is no
52
+ separate question/options limit: when the question and its options exceed 2,048 tokens, the runtime switches to a
53
+ numbered-option catalogue. Nothing is ever truncated; an input over budget raises `InputBudgetError`.
54
+
55
+ ## What it does
56
+
57
+ A state (text or JSON) and a typed question go in; a probability for every option comes out. The model never
58
+ generates text and cannot answer outside the options it is given.
59
+
60
+ - **choice**: pick one of N named options;
61
+ - **noul** (yes/no): the probability that a statement about the state is true;
62
+ - **score**: a distribution over 2 to 10 ordered levels.
63
+
64
+ One example, run with this repository's runtime (`jev_style_decision.py`) on the released weights, CPU, float32:
65
+
66
+ ```python
67
+ from jev_style_decision import JevStyleDecision
68
+
69
+ m = JevStyleDecision(".", device="cpu", threads=4)
70
+ r = m.decide({"ticket": "I was charged twice for my subscription this month.", "customer_tier": "pro"},
71
+ "Which team should handle this ticket?",
72
+ options={"billing": "payments, invoices, refunds", "technical": "bugs and outages", "sales": "new purchases"})
73
+ print(r["answer"], r["probabilities"])
74
+ # billing {'billing': 0.976, 'technical': 0.007, 'sales': 0.017} (rounded)
75
+ ```
76
+
77
+ Several questions about one state are scored in one call; the state is computed once and reused:
78
+
79
+ ```python
80
+ state = "Order #1182: paid, packed, handed to the courier on Monday. Tracking shows 'delivered' on Wednesday."
81
+ m.decide_many(state, [
82
+ {"t": "noul", "ins": "Has the order been delivered?", "crit": None},
83
+ {"t": "choice", "ins": "Which step is the order at?", "crit": {"packing": None, "in transit": None, "delivered": None}},
84
+ {"t": "score", "ins": "How urgent is a follow-up?", "crit": ["not urgent", "somewhat urgent", "urgent", "critical"]},
85
+ ])
86
+ # -> true 0.976 · delivered 0.686 (in transit 0.294) · level "1" 0.439 (level "0" 0.412) (rounded)
87
+ ```
88
+
89
+ ## Quick start
90
+
91
+ ```bash
92
+ pip install -U huggingface_hub
93
+ hf download chaoliangUNSW/Jev-Style-2B-Decision-v3 --local-dir jev-v3-2b && cd jev-v3-2b
94
+ pip install -r requirements.txt # torch 2.14.0, transformers 5.17.0, tokenizers 0.23.2, numpy 2.5.3
95
+
96
+ python jev_style_decision.py --model-dir . --device cpu --threads 4 \
97
+ --state "The user asked to cancel the order" \
98
+ --question "What should happen?" \
99
+ --options '{"cancel": "cancel the order", "ship": "ship it"}'
100
+ # -> "answer": "cancel", probability 0.993 (CPU, float32)
101
+ ```
102
+
103
+ `decide` returns `answer`, `probabilities`, the raw `scores`, the `temperature` used, `top_probability`,
104
+ `entropy_concentration`, token counts (`input_tokens`, `state_tokens`, `head_tokens`), `blocks`,
105
+ `catalogue_overflow`, `model` and `backend`. Batch mode reads JSON lines (`--jsonl file|-`); consecutive rows with
106
+ the same state share one state computation. `--verify` first checks the weights, tokenizer, configs and runtime against the sha256
107
+ manifest (documentation and evaluation records — `README.md`, `figures/`, `validation/` and `eval_results.json` —
108
+ are listed there but not checked).
109
+ A true/false question returns `{"false": p, "true": p}`; a score question returns the level indices `"0"`,
110
+ `"1"`, ... as option names.
111
+
112
+ ```bash
113
+ python jev_style_decision.py --model-dir . --device cpu --threads 4 --jsonl rows.jsonl
114
+ ```
115
+
116
+ Devices (`--device`): CPU and Apple MPS were run for this card; CUDA is supported by the runtime but was not run
117
+ for this release. float32 is the default (the format checks below used float32 on CPU); `--dtype bfloat16` (CUDA
118
+ only) was not parity-checked. `category=` is accepted for compatibility with the 0.8B v3 runtime and ignored: this model has one
119
+ global temperature.
120
+
121
+ ### Other builds
122
+
123
+ | Build | Size | Runtime |
124
+ |---|---:|---|
125
+ | **Transformers safetensors (bf16) · this repository** | 3.76 GB (2 shards) | PyTorch on CPU, Apple MPS or CUDA (`jev_style_decision.py`) |
126
+ | [GGUF F16 / Q8_0 / Q4_K_M](https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3-GGUF) | 3.78 / 2.01 / 1.27 GB | llama.cpp (libllama) + the bundled `jev-score-v2` scorer |
127
+ | [MLX bf16 / 8-bit](https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3-MLX) (one repository) | 3.76 / 2.00 GB | Apple silicon, mlx-lm 0.31.3 (`--model-dir bf16\|8bit`) |
128
+
129
+ ## Which file to pick
130
+
131
+ Every format was checked against the PyTorch FP32 reference on the released checkpoint, with gates declared before
132
+ any format was scored. All five pass.
133
+
134
+ | Format | Size | Same top-1 as FP32 (1,000 rows) | Max abs Δp (1,000 rows) | Accuracy (FP32: 80.8%) | Long fixture (43 questions, up to 25,600 tokens) | Gate |
135
+ |---|---:|---:|---:|---:|---:|:---:|
136
+ | PyTorch (this repository) | 3.76 GB | reference | | 80.8% | reference | |
137
+ | GGUF F16 | 3.78 GB | 100% | 0.0014 | 80.8% | 43 / 43 | PASS |
138
+ | GGUF Q8_0 | 2.01 GB | 99.7% | 0.033 | 80.7% | 43 / 43 | PASS |
139
+ | GGUF Q4_K_M | 1.27 GB | 95.7% | 0.346 | 81.1% | 42 / 43 | PASS¹ |
140
+ | MLX bf16 | 3.76 GB | 99.7% | 0.035 | 80.5% | 43 / 43 | PASS |
141
+ | MLX 8-bit (affine, group 64) | 2.00 GB | 99.6% | 0.162 | 80.8% | 43 / 43 | PASS |
142
+
143
+ - **Long documents: use GGUF Q8_0 or F16.** Q4_K_M is noticeably noisier on long inputs.
144
+ - **Apple silicon:** MLX bf16 (3.76 GB) for closeness to FP32: its scores are the closer of the two MLX builds to the
145
+ FP32 reference (max abs Δp 0.035 vs 0.162 for 8-bit). MLX 8-bit (2.00 GB) when memory is tight.
146
+ - **Smallest file:** GGUF Q4_K_M, 1.27 GB.
147
+
148
+ <sub>Reference: HF FP32 on CPU, exact block attention, on the released bf16 checkpoint. Gate fixture: 1,000 real development rows (≤4,096 tokens); these rows test agreement between formats. Long fixture: 35 requests / 43 questions up to 25,600 tokens, including catalogue-overflow questions and up to 151 options. ¹ For 4-bit the pre-declared gate is the accuracy drop (≤1.0 point); top-1 agreement is reported only. Sizes are the weight files (GB = 10^9 bytes); MLX adds a 0.42 MB FP32 norm file.</sub>
149
+
150
+ <!-- LATENCY:BEGIN generated from latency_2b.json (validation/latency_2b.json in the main repository); do not edit by hand -->
151
+
152
+ ## Speed
153
+
154
+ **Read once, then ask.** On an Apple M1 Max (GGUF F16), the first question about a 24,501-token input took 16.2 s; a further question about the same state took 0.17 s, because the state is computed once and reused (medians). Ten questions about that state in one call took 16.8 s.
155
+
156
+ | State | Questions per call | GGUF F16 | MLX bf16 | PyTorch (MPS, float32) |
157
+ |---|---:|---:|---:|---:|
158
+ | 878 tokens | 1 | 0.52 s | 0.57 s | 1.82 s |
159
+ | 878 tokens | 10 | 0.97 s | 1.07 s | 3.13 s |
160
+ | 3,950 tokens | 1 | 2.18 s | 2.26 s | 7.74 s |
161
+ | 3,950 tokens | 10 | 2.66 s | 2.80 s | 9.23 s |
162
+ | 24,436 tokens | 1 | 16.2 s | 15.5 s | 61.9 s |
163
+ | 24,436 tokens | 10 | 16.8 s | 16.2 s | 72.3 s |
164
+ | 24,436 tokens, already computed | 1 | 0.17 s | 0.15 s | 0.60 s |
165
+
166
+ <sub>Apple M1 Max, 64 GB, macOS 15.7.5. Wall time around one `decide` / `score_many` call (tokenisation included), median of 3 calls with the state recomputed each time; the model was loaded beforehand (loading took 1.2–8.0 s here, not included). States: English documentation and source code of 878, 3,950, 24,436 tokens plus the question; 10 questions = 4 choice, 4 true/false and 2 score questions about the same state in one call; with the question and options each input was up to 943, 4,015 and 24,501 tokens. GGUF: `jev-score-v2` on llama.cpp 441df11f, Metal, all layers on the GPU. MLX: mlx 0.32.2 / mlx-lm 0.31.3. PyTorch: this repository's runtime, float32 on Apple MPS. Results were identical with and without a precomputed state. Other jobs shared the machine during these runs (1-minute load average 5.5–11.1 at the end of each run), so treat the numbers as indicative. The PyTorch rows were measured in an earlier session (2026-09-27 00:45–02:15 AEST, load average 7.3–11.1); the GGUF and MLX rows were re-measured later (03:11–03:24 AEST, load average 5.5–10.0) because other jobs had slowed the earlier session (the re-measured GGUF and MLX rows shown here were up to 2.1× faster), so the PyTorch times may be pessimistic.</sub>
167
+
168
+ <!-- LATENCY:END -->
169
+
170
+ ## Results
171
+
172
+ All public benchmarks were pre-declared: GGUF F16 engine, each benchmark run once, one global temperature fitted
173
+ on our own calibration rows (never on benchmark items).
174
+
175
+ ### JevBench v1.4.1: 73.6% on the public items
176
+
177
+ ![JevBench v1.4.1 public accuracy: 2B v3 vs the Qwen3.5-2B-family systems, 0.8B v3 and Laya, with Jev as a reference line](figures/jevbench.png)
178
+
179
+ **73.6% (170 / 231)**, the highest JevBench public accuracy among the Qwen3.5-2B-family systems on the v1.4.1
180
+ board (decider-2b 71.0%, open-jev-zefan-2b 64.5%), and +9.5 points over the 0.8B v3. Jev (86.6%) is well ahead.
181
+
182
+ | System | Public accuracy (231) | Easy (48) | Standard (72) | Hard (111) | Hard-tier ECE |
183
+ |---|---:|---:|---:|---:|---:|
184
+ | **Jev-Style 2B v3** (this model, self-run) | 73.6% (170) | 100% | 95.8% | 47.7% | 0.153² |
185
+ | Jev-Style 0.8B v3 (self-run) | 64.1% (148) | 100% | 81.9% | 36.9% | 0.200² |
186
+ | Jev 1.13.0 (TypeSafe AI, API) | 86.6% | | | | |
187
+ | Decision 2B (FlyMy.AI, MiniCPM5-2B + LoRA, evaluation-only weights) | 75.3% | | | | |
188
+ | decider-2b (Mapika, Qwen3.5-2B-Base) | 71.0% | | | | |
189
+ | Raw Qwen3-4B-Instruct-2507 logits | 69.7% | | | | |
190
+ | kev 0.6B | 66.7% | | | | |
191
+ | Open-Jev 2B (Zefan Cai, Qwen3.5-2B + LoRA) | 64.5% | | | | |
192
+ | Laya (ModernBERT-large, 421M) | 58.4% | | | | |
193
+
194
+ <sub>JevBench v1.4.1 (github.com/fstandhartinger/jevbench, tag v1.4.1, commit 24b9b5c), public items only. 2B v3: self-run once with the vendored official harness, GGUF F16, zero-shot, one global temperature; 95% CI 67.6–78.9% (Wilson), which includes decider-2b (164 / 231), so that lead is a point estimate; not an official leaderboard entry. Other rows: public accuracy as published in the board's v1.4.1 results file; 42 of the 82 board systems score higher than 73.6%, almost all of them 4B or larger, or large API models. ² ECE over the 111 public hard items; the board's ECE uses all 220 hard items, so it is only an approximate comparison. Training-pool contamination scan (state hash, instruction hash, text fields of 32+ characters, 13-word spans): 0 hits.</sub>
195
+
196
+ ### Zero-shot topics: 82.2% on tweet_topic
197
+
198
+ ![Zero-shot tweet_topic and fin_topic accuracy: 2B v3 vs 0.8B v3 and Jev](figures/zeroshot.png)
199
+
200
+ **On tweet_topic the 2B scores 82.2% accuracy, above Jev's 79.3%**, which lies outside the 2B's 95% CI
201
+ (80.4–84.0%). Its macro-F1 is below Jev's (67.8% vs 69.4%). **On the 20-way fin_topic it scores 61.1%, +14.4
202
+ points over the 0.8B v3**; Jev is higher there (67.0%).
203
+
204
+ | Set | n | **2B v3 accuracy** [95% CI] | 2B v3 macro-F1 | 2B v3 ECE (calibrated) | Jev accuracy / macro-F1 / ECE (raw API) | 0.8B v3 accuracy |
205
+ |---|---:|---:|---:|---:|---:|---:|
206
+ | tweet_topic | 1,693 | 82.2% [80.4, 84.0] | 67.8% | 0.028 | 79.3% / 69.4% / 0.063 | 75.5% |
207
+ | fin_topic | 4,117 | 61.1% [59.6, 62.6] | 59.0% | 0.065 | 67.0% / 63.0% / 0.166 | 46.7% |
208
+
209
+ <sub>Zero-shot: none of these test sets is in the training pool (0 exact overlaps, 0 near-duplicates). Accuracy over every row of the pinned test files; CIs are percentile bootstrap (10,000 resamples). Jev (1.13, API): numbers published by the [elcronos jev-vs-open-decision-models study](https://github.com/elcronos/jev-vs-open-decision-models) (results/cross_dataset_summary.json @ a1901bc), not re-run by us. ECE: 15 equal-width bins; ours uses the model's one global temperature (fitted on our own calibration rows, never on these sets), Jev's is from its raw API probabilities, so the two ECE columns are not like-for-like. The temperature is fitted once for all tasks, not per test set: without it (T = 1) the 2B's ECE is 0.087 on tweet_topic and 0.038 on fin_topic.</sub>
210
+
211
+ ## How it differs from the 0.8B v3
212
+
213
+ Both v3 models read 25,600 tokens and score every option at its own verdict slot. The 2B is a separate full
214
+ fine-tune with a different input protocol, so each size needs its own runtime.
215
+
216
+ | | [Jev-Style 0.8B v3](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3) | **Jev-Style 2B v3 (this model)** |
217
+ |---|---|---|
218
+ | Base | Qwen/Qwen3.5-0.8B | **Qwen/Qwen3.5-2B** (post-trained, not -Base) |
219
+ | Parameters | 752,393,024 | **1,881,825,088** (24 layers: 18 Gated DeltaNet + 6 full attention, hidden 2,048) |
220
+ | Input protocol | v1: causal attention | **v2: block attention**. Each 2,048-token block sees everything before it plus itself |
221
+ | Question/options limit | 2,048-token head; longer option lists are split into option chunks | **None**: one 25,600-token budget; a numbered catalogue when question + options exceed 2,048 tokens |
222
+ | Calibration | 20 group temperatures + a global one | **One global temperature** (T = 0.828) |
223
+ | JevBench v1.4.1 public | 64.1% | **73.6%** |
224
+ | tweet_topic / fin_topic accuracy | 75.5% / 46.7% | **82.2% / 61.1%** |
225
+ | Smallest file | 0.53 GB (Q4_K_M) | 1.27 GB (Q4_K_M) |
226
+
227
+ ### Input format and readout
228
+
229
+ <details>
230
+ <summary><strong>Prompt layout, block attention and the score</strong></summary>
231
+
232
+ Each segment is tokenised on its own and the pieces are concatenated, so slot positions are exact. Text inside the
233
+ state, question and options is tokenised with special tokens disabled: `<|im_end|>` in user text stays plain text.
234
+ Non-special added tokens such as `<think>` encode as their single ids, as in training.
235
+
236
+ ```text
237
+ State:
238
+ <state: plain text, or any JSON value>
239
+
240
+ Question [<choice|noul|score>]: <question>
241
+ Options:
242
+ - <option 1> ->
243
+ - <option 2> ->
244
+ ```
245
+
246
+ When the question, options and slots together exceed 2,048 tokens, the options are written once as a numbered
247
+ catalogue (`Option 1: <option 1>` ...), followed by `Judge each numbered option in the complete catalogue above:` and
248
+ one `Option k ->` slot per option.
249
+
250
+ - **Block attention.** The input is cut into blocks of at most 2,048 tokens (state blocks, then the question block,
251
+ or catalogue and rubric blocks). In the 6 full-attention layers each block attends to everything before it and to
252
+ itself, with no causal mask inside the block. The Gated DeltaNet layers are ordinary recurrent layers. This is
253
+ why the shipped runtimes are required.
254
+ - **Score.** Option k's score is `logit(" yes") − logit(" no")` at its ` ->` slot, computed in float32 from the
255
+ final normalised hidden state and the tied embedding rows. No parameters are added.
256
+ - **Probabilities.** `softmax(scores / T)` with one global `T = 0.8278650621`, fitted on 2,000 independent
257
+ calibration rows (NLL 0.3003 → 0.2901). `temperature=1.0` gives the raw scores.
258
+
259
+ </details>
260
+
261
+ ## Scope and limits
262
+
263
+ - **Runtime required.** Decisions come from the runtimes shipped in the three repositories (PyTorch, GGUF +
264
+ `jev-score-v2`, MLX). Stock llama.cpp, Ollama, LM Studio or `mlx_lm.generate` can load the weights but cannot
265
+ produce the decision scores, and they would run causal attention. The 0.8B v3 runtimes are not valid for this model.
266
+ - **25,600 tokens** is the limit for the whole input (state + question + options + readout).
267
+ - **Reduced-data training.** This model was trained on a reduced data pool (60M tokens); the planned full recipe
268
+ was not run.
269
+ - **Decision Index.** We have not run Decision Index 0.2 and report no score for it. The training pool includes the
270
+ train splits of BANKING77, CLINC150 (+OOS), SGD, HellaSwag, GSM8K, ARC-Easy and ARC-Challenge (licences under
271
+ [Training data and licences](#training-data-and-licences)), and format-imitating data for SATA-Bench, BRIGHT,
272
+ NLI4CT, CRUXEval, CLadder, PhishNChips and BBH, so results on these 14 benchmarks are not zero-shot.
273
+ - **Decisions only.** The model scores the options you give it. It does not generate text and takes no actions.
274
+
275
+ ## Training
276
+
277
+ - **Base:** [Qwen/Qwen3.5-2B](https://huggingface.co/Qwen/Qwen3.5-2B) (revision `15852e8c`), Apache-2.0. The
278
+ vision tower and the multi-token-prediction head were removed; the checkpoint is a text-only
279
+ `Qwen3_5ForCausalLM` with tied embeddings.
280
+ - **Run:** full fine-tune on one Colab A100 40GB, 458 steps, 1 epoch, 60,032,377 training tokens (181,449
281
+ logical rows). Trained on a reduced data pool (60M tokens).
282
+ - **Checkpoint:** EMA vs raw weights chosen by a pre-declared rule (lowest component-macro NLL on a fixed
283
+ development sample of 1,556 rows): EMA.
284
+ - **Calibration:** one global temperature fitted on 2,000 independent calibration rows (NLL 0.3003 → 0.2901).
285
+
286
+ <!-- BEGIN DATA_LICENCES -->
287
+ ## Training data and licences
288
+
289
+ - **Base model:** Qwen/Qwen3.5-2B (Qwen team, Alibaba Cloud), Apache-2.0; see `LICENSE` and `NOTICE`.
290
+ - **Mixture:** 181,449 training rows (60,032,377 tokens; repeats counted) from 58 sources. English is 77.7% of the
291
+ tokens and Chinese 17.0%; MASSIVE adds nine more languages.
292
+
293
+ | Component | Rows | Share of tokens | Sources |
294
+ |---|---:|---:|---|
295
+ | Typed decisions | 34,008 | 19.2% | model-written business workflows (27,300 unique items) |
296
+ | Intents and yes/no QA | 32,123 | 16.2% | MASSIVE, CLINC150, BoolQ |
297
+ | Knowledge and reasoning | 32,724 | 10.6% | ARC, CommonsenseQA, MedMCQA, QASC, GSM8K, MBPP, chess puzzles; code-generated items |
298
+ | Themes | 38,290 | 10.3% | Civil Comments, SQuAD v2, five jailbreak and prompt-injection sets, model-written prompts |
299
+ | Long tables | 3,292 | 8.6% | code-generated tables |
300
+ | Retrieval and routing | 13,697 | 7.6% | BANKING77, SGD; code-generated link-safety and relevance items |
301
+ | Mac agent checks | 7,787 | 7.5% | the project's own simulators |
302
+ | Public reading tasks | 5,597 | 7.0% | 14 sets, incl. WANLI, TabFact, DROP, HelpSteer2, FinQA, MAUD |
303
+ | Hard cases | 2,281 | 6.8% | five generated families (policy, multi-hop, numeric, abstention, judging) |
304
+ | Language | 11,012 | 3.6% | HellaSwag, SNLI, code-generated clinical-trial reports |
305
+ | Option-format views | 638 | 2.6% | alternative option layouts of rows above |
306
+
307
+ - **Benchmark train splits** (official train splits only, no test split; results on these benchmarks are not
308
+ zero-shot, see Benchmarks below): BANKING77 by PolyAI (CC BY 4.0 upstream at PolyAI/banking77; rows taken from the MTEB mirror mteb/banking77, whose card says MIT), CLINC150 with its out-of-scope queries (CC BY 3.0),
309
+ Schema-Guided Dialogue (SGD; CC BY-SA 4.0), GSM8K (MIT), ARC-Easy and ARC-Challenge (CC BY-SA 4.0), and HellaSwag
310
+ (MIT; see below). Each one's repository is linked in the per-source list.
311
+ - **Full per-source list:** [validation/data_sources.json](validation/data_sources.json) (all 58 sources with the
312
+ licence recorded for each, the mixture rows they feed and a link where available).
313
+ - **Datasets with restrictive or unclear terms** (kept in the pool; check each source's terms before commercial use):
314
+ - Jailbreak prompts from the jailbreak_llms collection, which states it is for research purposes only:
315
+ In-the-Wild Jailbreak Prompts (TrustAIRLab, MIT; 1,604 rows) and the jailbreak prompts in
316
+ jackhhao/jailbreak-classification (Apache-2.0; 1,410 rows in total).
317
+ - HellaSwag: its card gives MIT only in the text (no licence field), and the original GitHub repository is blocked
318
+ after a DMCA notice from wikiHow. Only the ActivityNet-caption items were used.
319
+ - neuralchemy Prompt-injection-dataset (2,133 rows): the upstream rows its card marks research-only were removed.
320
+ - Share-alike (CC BY-SA): SQuAD v2, BoolQ, DROP, SNLI, ARC, SGD, ShARC, TempReason.
321
+ - **Outputs of other models:**
322
+ - OpenAI GPT and Anthropic Claude models designed the typed-decision workflows (34,008 rows); Claude models
323
+ labelled them. A Claude model wrote and labelled 1,243 jailbreak and toxicity prompt rows.
324
+ - OpenAI GPT models wrote two hard-case families (1,132 rows) and paraphrased the goal wording of 2,012 of the
325
+ 3,999 unique Mac goal-done items. Hard-case labels are computed by code.
326
+ - Third-party data with model-written text: WANLI (506 rows; GPT-3, revised by crowdworkers), part of the
327
+ jackhhao benign prompts (from GPTeacher, generated by GPT-4) and the HelpSteer2 responses (490 rows; mostly
328
+ NVIDIA Nemotron models and Mixtral-8x7B-Instruct).
329
+ - The providers' terms of use may restrict how models trained on such outputs may be used, so check them for
330
+ your use case. No outputs of Jev or any other TypeSafe model were used.
331
+ - **Benchmarks:** no test split was used. The train splits and format-imitating generators named under
332
+ [Scope and limits](#scope-and-limits) are in the mixture, so results on those 14 benchmarks are not zero-shot.
333
+ - **Evaluation-only data** (0 training rows): the JevBench items, tweet_topic, fin_topic and daily_dialog. The
334
+ contamination scan found no JevBench hit and no exact or near-duplicate overlap with tweet_topic and fin_topic;
335
+ daily_dialog has no exact overlap apart from one generic short phrase ("Thanks a lot"), plus three
336
+ near-duplicate-only matches. The scan ([validation/benchmarks/contamination.json](validation/benchmarks/contamination.json))
337
+ covers the full reduced pool (463,208 train rows, plus the format, development and calibration files), a superset
338
+ of the rows actually trained on (181,449 drawn, repeats counted).
339
+ <!-- END DATA_LICENCES -->
340
+
341
+ ## Disclaimers
342
+
343
+ > **Independent project.** Jev-Style is not affiliated with, endorsed by or connected to TypeSafe AI or Jev, and no
344
+ > Jev weights, code or outputs are used. It is also not affiliated with the Laya authors or the Qwen team. Jev and
345
+ > other board numbers on this card come from the sources named under each result.
346
+
347
+ **AI disclosure:** code written with AI coding assistants (Claude Code) under my direction; I designed the project,
348
+ trained the models and verified the results.
349
+
350
+ ## Licence
351
+
352
+ Apache-2.0. Built on Qwen/Qwen3.5-2B (Apache-2.0); the Apache License 2.0 text is in `LICENSE`, and `NOTICE` lists
353
+ the modifications.
354
+
355
+ ## Citation
356
+
357
+ ```bibtex
358
+ @misc{jevstyle2026v3_2b,
359
+ title = {Jev-Style-2B-Decision-v3: a 2B decision model with calibrated probabilities for every option},
360
+ author = {chaoliangUNSW},
361
+ year = {2026},
362
+ howpublished = {\url{https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3}},
363
+ note = {Fine-tuned from Qwen/Qwen3.5-2B}
364
+ }
365
+ ```
366
+
367
+ ## Contact
368
+
369
+ I welcome internship, employment, and research collaboration opportunities. Please contact me at [**yanchaoliang369@gmail.com**](mailto:yanchaoliang369@gmail.com).
370
+
371
+ 欢迎提供实习、工作及科研合作机会,请邮件联系:[yanchaoliang369@gmail.com](mailto:yanchaoliang369@gmail.com)。
chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is true %}
150
+ {{- '<think>\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n\n</think>\n\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
config.json ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForCausalLM"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "attn_output_gate": true,
8
+ "bos_token_id": null,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 248044,
11
+ "full_attention_interval": 4,
12
+ "head_dim": 256,
13
+ "hidden_act": "silu",
14
+ "hidden_size": 2048,
15
+ "initializer_range": 0.02,
16
+ "intermediate_size": 6144,
17
+ "layer_types": [
18
+ "linear_attention",
19
+ "linear_attention",
20
+ "linear_attention",
21
+ "full_attention",
22
+ "linear_attention",
23
+ "linear_attention",
24
+ "linear_attention",
25
+ "full_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "linear_attention",
29
+ "full_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "linear_attention",
33
+ "full_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "linear_attention",
37
+ "full_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "linear_attention",
41
+ "full_attention"
42
+ ],
43
+ "linear_conv_kernel_dim": 4,
44
+ "linear_key_head_dim": 128,
45
+ "linear_num_key_heads": 16,
46
+ "linear_num_value_heads": 16,
47
+ "linear_value_head_dim": 128,
48
+ "mamba_ssm_dtype": "float32",
49
+ "max_position_embeddings": 262144,
50
+ "mlp_only_layers": [],
51
+ "model_type": "qwen3_5_text",
52
+ "mtp_num_hidden_layers": 1,
53
+ "mtp_use_dedicated_embeddings": false,
54
+ "num_attention_heads": 8,
55
+ "num_hidden_layers": 24,
56
+ "num_key_value_heads": 2,
57
+ "pad_token_id": null,
58
+ "partial_rotary_factor": 0.25,
59
+ "rms_norm_eps": 1e-06,
60
+ "rope_parameters": {
61
+ "mrope_interleaved": true,
62
+ "mrope_section": [
63
+ 11,
64
+ 11,
65
+ 10
66
+ ],
67
+ "partial_rotary_factor": 0.25,
68
+ "rope_theta": 10000000,
69
+ "rope_type": "default"
70
+ },
71
+ "tie_word_embeddings": true,
72
+ "transformers_version": "5.17.0",
73
+ "use_cache": false,
74
+ "vocab_size": 248320
75
+ }
eval_results.json ADDED
@@ -0,0 +1,262 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "jev-style-eval-results-v1",
3
+ "model": "chaoliangUNSW/Jev-Style-2B-Decision-v3",
4
+ "evaluated_format": {
5
+ "engine": "GGUF F16 via jev-score-v2 (llama.cpp 441df11f, Metal), render v2 + block attention",
6
+ "weights": {
7
+ "file": "release_2b/gguf/model-f16.gguf (= Jev-Style-2B-Decision-v3-F16.gguf, tensor data identical)",
8
+ "sha256": "fe18cf4f524e0d59e92a11ee3c012c0b4b03b4e2109e8a35497737e2e55d33dd"
9
+ },
10
+ "temperature": {
11
+ "mode": "global",
12
+ "value": 0.8278650620942867,
13
+ "fitted_on_benchmark_items": false
14
+ },
15
+ "truncation": "never",
16
+ "run_policy": "pre-declared, run once (runtime_v2_dev/PREDECLARED_RELEASE_GATES_2B.md)"
17
+ },
18
+ "jevbench": {
19
+ "benchmark": "jevbench-v1.4.1",
20
+ "repo": "https://github.com/fstandhartinger/jevbench",
21
+ "commit": "24b9b5c1609a7a9e8fa14f49e5985a836c9dc842",
22
+ "split": "public",
23
+ "n_items": 231,
24
+ "correct": 170,
25
+ "public_accuracy": 0.7359307359307359,
26
+ "public_accuracy_wilson_ci95": [
27
+ 0.6755578394149232,
28
+ 0.7885850787366094
29
+ ],
30
+ "tiers": {
31
+ "easy": {
32
+ "n": 48,
33
+ "correct": 48,
34
+ "accuracy": 1.0,
35
+ "ece_top_label_10bin": 0.01884318271512942
36
+ },
37
+ "standard": {
38
+ "n": 72,
39
+ "correct": 69,
40
+ "accuracy": 0.9583333333333334,
41
+ "ece_top_label_10bin": 0.09887602156622965
42
+ },
43
+ "hard": {
44
+ "n": 111,
45
+ "correct": 53,
46
+ "accuracy": 0.4774774774774775,
47
+ "ece_top_label_10bin": 0.15324274416807362
48
+ }
49
+ },
50
+ "hard_tier_ece_public111": 0.15324274416807362,
51
+ "hard_tier_ece_note": "board ECE uses all 220 hard items; ours uses the 111 public hard items (approximate comparison)",
52
+ "unsupported": 0,
53
+ "protocol": {
54
+ "template": "macjev-render-v2-long-options",
55
+ "layout": "sb",
56
+ "block": 2048,
57
+ "total_budget_tokens": 25600,
58
+ "truncation": "never",
59
+ "question_options_cap": null,
60
+ "over_budget_rule": "complete rendering > 25,600 tokens -> status unsupported (InputBudgetError), counted per benchmark and scored as wrong",
61
+ "catalogue_overflow_rule": "question+options over one 2,048-token block -> numbered catalogue in prefix blocks + numbered-option rubric (RendererV2); counted per benchmark",
62
+ "readout": "FP32 h_slot . (W_yes - W_no) at each option slot; p = softmax(scores / T_global)"
63
+ },
64
+ "board": {
65
+ "file": "results/v1.4.1/jevbench-v1.4.1-results.json",
66
+ "sha256": "e6754863056503fe2b010410fc7111df884ac1f9ce4449aa369aab61d98092cd",
67
+ "commit": "24b9b5c1609a7a9e8fa14f49e5985a836c9dc842",
68
+ "revision": "v1.4.1",
69
+ "n_systems": 82,
70
+ "systems_with_higher_public_accuracy": 42,
71
+ "selected_rows": [
72
+ {
73
+ "key": "jev-1.13.0",
74
+ "name": "Jev 1.13.0 (TypeSafe AI)",
75
+ "public_accuracy": 0.8658008658008658,
76
+ "underlying": "closed"
77
+ },
78
+ {
79
+ "key": "raw-qwen3-4b-instruct-2507",
80
+ "name": "Raw Qwen3 4B Instruct 2507 direct logits",
81
+ "public_accuracy": 0.696969696969697,
82
+ "underlying": "Qwen3-4B-Instruct-2507 BF16"
83
+ },
84
+ {
85
+ "key": "decision-2b",
86
+ "name": "Decision 2B (FlyMy.AI, v59)",
87
+ "public_accuracy": 0.7532467532467533,
88
+ "underlying": "openbmb/MiniCPM5-2B with a trained LoRA adapter and pointer head (26.2M trainable parameters)"
89
+ },
90
+ {
91
+ "key": "decider-2b",
92
+ "name": "decider-2b (Mapika)",
93
+ "public_accuracy": 0.70995670995671,
94
+ "underlying": "Qwen3.5-2B-Base with a trained decision readout, 1.9B"
95
+ },
96
+ {
97
+ "key": "laya",
98
+ "name": "Laya (Convai Innovations, ModernBERT-large 421M)",
99
+ "public_accuracy": 0.5844155844155844,
100
+ "underlying": "ModernBERT-large encoder + option-marker decision head, 421M, RLCD-trained"
101
+ },
102
+ {
103
+ "key": "kev-0.6b",
104
+ "name": "kev 0.6B (research preview)",
105
+ "public_accuracy": 0.6666666666666666,
106
+ "underlying": "Qwen3-0.6B-Base + LoRA + learned pointer head; jaredpalmer/kev-0.6b"
107
+ },
108
+ {
109
+ "key": "open-jev-zefan-2b",
110
+ "name": "Open-Jev 2B (Zefan Cai)",
111
+ "public_accuracy": 0.645021645021645,
112
+ "underlying": "Qwen3.5-2B plus rank-8 LoRA and trained scalar decision head"
113
+ }
114
+ ]
115
+ },
116
+ "previous_v3_0.8b": {
117
+ "public_accuracy": 0.6406926406926406,
118
+ "correct": 148,
119
+ "source_sha256": "6c1f94220dc8197e504827213b3567e00919740bb705929bed88535812f2a033"
120
+ },
121
+ "source": {
122
+ "file": "runs/macjev/bench_2b/jevbench/gguf_f16/results.json",
123
+ "sha256": "9a467fb4ddcc9a42f9d82ff8f8cf1ce5cfafd51d27b8d4dca2b9c1a681f67a9c"
124
+ }
125
+ },
126
+ "zero_shot": {
127
+ "benchmark": "elcronos zero-shot sets (tweet_topic, fin_topic, daily_dialog)",
128
+ "elcronos_commit": "a1901bc3d520e73936de8d4326545c0cdcf742fb",
129
+ "sets": {
130
+ "tweet_topic": {
131
+ "n": 1693,
132
+ "n_unsupported": 0,
133
+ "n_classes": 6,
134
+ "accuracy": 0.822209096278795,
135
+ "accuracy_ci95": [
136
+ 0.8038984051978736,
137
+ 0.8399438865918486
138
+ ],
139
+ "macro_f1": 0.677897069437572,
140
+ "macro_f1_ci95": [
141
+ 0.6455020683998111,
142
+ 0.7085891187789015
143
+ ],
144
+ "ece15_temperature_calibrated": 0.027919883880813873,
145
+ "ece15_raw_T1": 0.08661185749227743,
146
+ "nll": 0.5282712172517265,
147
+ "brier": 0.262576013332696,
148
+ "majority_class_accuracy": 0.3963378617838157,
149
+ "ci_method": "percentile bootstrap, 10000 resamples, default_rng(0), rows resampled i.i.d.",
150
+ "in_training_pool": false
151
+ },
152
+ "fin_topic": {
153
+ "n": 4117,
154
+ "n_unsupported": 0,
155
+ "n_classes": 20,
156
+ "accuracy": 0.6111246052951178,
157
+ "accuracy_ci95": [
158
+ 0.5960650959436483,
159
+ 0.6259412193344669
160
+ ],
161
+ "macro_f1": 0.5897964463354406,
162
+ "macro_f1_ci95": [
163
+ 0.5694510128995812,
164
+ 0.6071880523494645
165
+ ],
166
+ "ece15_temperature_calibrated": 0.06460657860002232,
167
+ "ece15_raw_T1": 0.03822893519578447,
168
+ "nll": 1.159706614583174,
169
+ "brier": 0.5285731205072254,
170
+ "majority_class_accuracy": 0.2069468059266456,
171
+ "ci_method": "percentile bootstrap, 10000 resamples, default_rng(0), rows resampled i.i.d.",
172
+ "in_training_pool": false
173
+ },
174
+ "daily_dialog": {
175
+ "n": 7740,
176
+ "n_unsupported": 0,
177
+ "n_classes": 7,
178
+ "accuracy": 0.7744186046511627,
179
+ "accuracy_ci95": [
180
+ 0.7649870801033591,
181
+ 0.7835917312661499
182
+ ],
183
+ "macro_f1": 0.3724097968832693,
184
+ "macro_f1_ci95": [
185
+ 0.3480292808769088,
186
+ 0.3955194464258293
187
+ ],
188
+ "ece15_temperature_calibrated": 0.03934861337502408,
189
+ "ece15_raw_T1": 0.027480647988479354,
190
+ "nll": 0.6778540478231467,
191
+ "brier": 0.3382397818519505,
192
+ "majority_class_accuracy": 0.8166666666666667,
193
+ "ci_method": "percentile bootstrap, 10000 resamples, default_rng(0), rows resampled i.i.d.",
194
+ "in_training_pool": false
195
+ }
196
+ },
197
+ "jev_reference": {
198
+ "values": {
199
+ "tweet_topic": {
200
+ "accuracy": 0.7932663910218547,
201
+ "macro_f1": 0.6936,
202
+ "ece15": 0.0631
203
+ },
204
+ "fin_topic": {
205
+ "accuracy": 0.669905270828273,
206
+ "macro_f1": 0.6298,
207
+ "ece15": 0.1664
208
+ },
209
+ "daily_dialog": {
210
+ "accuracy": 0.7099483204134367,
211
+ "macro_f1": 0.3847,
212
+ "ece15": 0.1563
213
+ }
214
+ },
215
+ "source": "https://github.com/elcronos/jev-vs-open-decision-models/blob/a1901bc3d520e73936de8d4326545c0cdcf742fb/results/cross_dataset_summary.json",
216
+ "source_sha256": "5380d3a45395bfdf5340d75e7e18ecdb4b636734295d234839c7113efae608d0",
217
+ "note": "Jev numbers as published by elcronos (raw API probabilities, no temperature); sha256 recorded by the 0.8B adapter, not re-verified locally"
218
+ },
219
+ "previous_v3_0.8b": {
220
+ "tweet_topic": {
221
+ "accuracy": 0.754873006497342,
222
+ "macro_f1": 0.5993905130101003
223
+ },
224
+ "fin_topic": {
225
+ "accuracy": 0.4670876852076755,
226
+ "macro_f1": 0.45174530293977544
227
+ },
228
+ "daily_dialog": {
229
+ "accuracy": 0.32493540051679587,
230
+ "macro_f1": 0.2331496717992918
231
+ }
232
+ },
233
+ "protocol": {
234
+ "template": "macjev-render-v2-long-options",
235
+ "layout": "sb",
236
+ "block": 2048,
237
+ "total_budget_tokens": 25600,
238
+ "truncation": "never",
239
+ "question_options_cap": null,
240
+ "over_budget_rule": "complete rendering > 25,600 tokens -> status unsupported (InputBudgetError), counted per benchmark and scored as wrong",
241
+ "catalogue_overflow_rule": "question+options over one 2,048-token block -> numbered catalogue in prefix blocks + numbered-option rubric (RendererV2); counted per benchmark",
242
+ "readout": "FP32 h_slot . (W_yes - W_no) at each option slot; p = softmax(scores / T_global)"
243
+ },
244
+ "source": {
245
+ "file": "runs/macjev/bench_2b/zeroshot/gguf_f16/metrics.json",
246
+ "sha256": "f80c80ae55678cce974927df0c6a33a5bf5e2ba85e9b14a181e6e5d16bcee3c5"
247
+ },
248
+ "run": {
249
+ "file": "runs/macjev/bench_2b/zeroshot/gguf_f16/run.json",
250
+ "sha256": "0361c42dc3247387b8325aedc872de644b9c0eca80f6e6db9fd68629c27fd389"
251
+ }
252
+ },
253
+ "contamination": {
254
+ "source": {
255
+ "file": "validation/benchmarks/contamination.json",
256
+ "sha256": "87f7f645b950b4cd626ff248b1899578c078db74d4c74bc944b017badc06a8e8",
257
+ "generated_from": "runs/macjev/release_2b/cards/_build/contamination_public.json"
258
+ },
259
+ "note": "JevBench public items vs the whole 2B training pool: state hash, instruction hash, 32+ character text fields, 13-word spans"
260
+ },
261
+ "decision_index": "requested from the maintainer after release (not run by us)"
262
+ }
figures/banner.data.json ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "left": {
3
+ "big": "73.6% on JevBench",
4
+ "sub": "+9.5 points over our 0.8B v3; Jev 86.6%",
5
+ "tokens": "25,600 tokens per call, no option cap"
6
+ },
7
+ "rows": [
8
+ {
9
+ "label": "Jev 1.13 (API)",
10
+ "accuracy": 0.8658008658008658,
11
+ "pct_1dp": 86.6
12
+ },
13
+ {
14
+ "label": "2B v3 \u00b7 this model",
15
+ "accuracy": 0.7359307359307359,
16
+ "pct_1dp": 73.6
17
+ },
18
+ {
19
+ "label": "decider-2b",
20
+ "accuracy": 0.70995670995671,
21
+ "pct_1dp": 71.0
22
+ },
23
+ {
24
+ "label": "Open-Jev 2B",
25
+ "accuracy": 0.645021645021645,
26
+ "pct_1dp": 64.5
27
+ },
28
+ {
29
+ "label": "0.8B v3",
30
+ "accuracy": 0.6406926406926406,
31
+ "pct_1dp": 64.1
32
+ },
33
+ {
34
+ "label": "Laya",
35
+ "accuracy": 0.5844155844155844,
36
+ "pct_1dp": 58.4
37
+ }
38
+ ],
39
+ "shown": "Shown: Qwen3.5-2B-family systems, our 0.8B v3, Laya and Jev \u00b7 42 of 82 board systems score higher than 73.6%",
40
+ "note": "231 public items. 2B v3: self-run with the official harness (GGUF F16), not an official board entry; 95% CI 67.6\u201378.9%, so the lead over decider-2b is inside the CI. Other rows as published on the v1.4.1 board.",
41
+ "sources": [
42
+ "figures/jevbench.data.json"
43
+ ]
44
+ }
figures/banner.png ADDED

Git LFS Details

  • SHA256: 2623e4afad36239aaa007c34350d6f2fbacb33430ffd3c36ce2f7ee7db849b47
  • Pointer size: 131 Bytes
  • Size of remote file: 447 kB
figures/jevbench.data.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "jevbench",
3
+ "metric": "JevBench v1.4.1 public accuracy (231 items)",
4
+ "rows": [
5
+ {
6
+ "label": "Jev-Style 2B v3 (this model)",
7
+ "accuracy": 0.7359307359307359,
8
+ "accuracy_pct_1dp": 73.6,
9
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3/blob/main/validation/benchmarks/jevbench_v1.4.1_results.json :: public_accuracy (170/231)"
10
+ },
11
+ {
12
+ "label": "decider-2b (Mapika, Qwen3.5-2B-Base)",
13
+ "accuracy": 0.70995670995671,
14
+ "accuracy_pct_1dp": 71.0,
15
+ "source": "https://github.com/fstandhartinger/jevbench/blob/v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[decider-2b]"
16
+ },
17
+ {
18
+ "label": "Open-Jev 2B (Zefan Cai, Qwen3.5-2B + LoRA)",
19
+ "accuracy": 0.645021645021645,
20
+ "accuracy_pct_1dp": 64.5,
21
+ "source": "https://github.com/fstandhartinger/jevbench/blob/v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[open-jev-zefan-2b]"
22
+ },
23
+ {
24
+ "label": "Jev-Style 0.8B v3",
25
+ "accuracy": 0.6406926406926406,
26
+ "accuracy_pct_1dp": 64.1,
27
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3/blob/main/figures/jevbench.data.json (0.8B v3 card; row Jev-Style 0.8B v3, 148/231)"
28
+ },
29
+ {
30
+ "label": "Laya (ModernBERT-large, 421M)",
31
+ "accuracy": 0.5844155844155844,
32
+ "accuracy_pct_1dp": 58.4,
33
+ "source": "https://github.com/fstandhartinger/jevbench/blob/v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[laya]"
34
+ }
35
+ ],
36
+ "reference_line": {
37
+ "label": "Jev 1.13.0",
38
+ "accuracy": 0.8658008658008658,
39
+ "source": "https://github.com/fstandhartinger/jevbench/blob/v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[jev-1.13.0]"
40
+ },
41
+ "ci95_wilson_2b": [
42
+ 0.6755578394149232,
43
+ 0.7885850787366094
44
+ ],
45
+ "board_systems_higher_than_2b": 42,
46
+ "board_systems": 82,
47
+ "board_sha256": "e6754863056503fe2b010410fc7111df884ac1f9ce4449aa369aab61d98092cd",
48
+ "footnote": "JevBench v1.4.1, 231 public items. 2B v3: self-run once with the official harness (commit 24b9b5c), GGUF F16 engine, one global temperature, not an official board entry; 95% CI 67.6-78.9% (Wilson), so its lead over decider-2b (164/231) is inside the CI; training-pool contamination scan: 0 hits. Other rows: public accuracy as published in the board's v1.4.1 results file. Shown: the Qwen3.5-2B-family systems on the board, our 0.8B v3, Laya and Jev; 42 of the 82 board systems score higher than 73.6%."
49
+ }
figures/jevbench.png ADDED

Git LFS Details

  • SHA256: c8542dba9ecc1bd2a4ea3bda844cbc3fbd661c40d6eb9254098b279a92a74ce8
  • Pointer size: 131 Bytes
  • Size of remote file: 168 kB
figures/jevbench.svg ADDED
figures/zeroshot.data.json ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "zeroshot",
3
+ "metric": "accuracy over every row of the pinned test files",
4
+ "values": {
5
+ "tweet_topic": {
6
+ "2b": 0.822209096278795,
7
+ "2b_ci95": [
8
+ 0.8038984051978736,
9
+ 0.8399438865918486
10
+ ],
11
+ "2b_macro_f1": 0.677897069437572,
12
+ "2b_ece15": 0.027919883880813873,
13
+ "08b": 0.754873006497342,
14
+ "jev": 0.7932663910218547,
15
+ "jev_macro_f1": 0.6936,
16
+ "jev_ece15": 0.0631,
17
+ "n": 1693
18
+ },
19
+ "fin_topic": {
20
+ "2b": 0.6111246052951178,
21
+ "2b_ci95": [
22
+ 0.5960650959436483,
23
+ 0.6259412193344669
24
+ ],
25
+ "2b_macro_f1": 0.5897964463354406,
26
+ "2b_ece15": 0.06460657860002232,
27
+ "08b": 0.4670876852076755,
28
+ "jev": 0.669905270828273,
29
+ "jev_macro_f1": 0.6298,
30
+ "jev_ece15": 0.1664,
31
+ "n": 4117
32
+ }
33
+ },
34
+ "sources": {
35
+ "2b": "https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3/blob/main/validation/benchmarks/zeroshot_metrics.json",
36
+ "0.8b": "https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3/blob/main/figures/zeroshot.json (0.8B v3 card; v3_recomputed: tweet_topic 1278/1693, fin_topic 1923/4117)",
37
+ "jev": "https://github.com/elcronos/jev-vs-open-decision-models/blob/a1901bc3d520e73936de8d4326545c0cdcf742fb/results/cross_dataset_summary.json (as copied in https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3/blob/main/validation/benchmarks/zeroshot_metrics.json :: comparison)"
38
+ },
39
+ "footnote": "Zero-shot: none of these test sets is in the 2B or 0.8B training pool; accuracy over every row of the pinned test files (n = 1,693 and 4,117). 2B v3: GGUF F16 engine, one global temperature, run once; tweet_topic 95% CI 80.4-84.0%. Jev (1.13, API): numbers published by the elcronos jev-vs-open-decision-models study (cross_dataset_summary.json @ a1901bc), not re-run by us. Macro-F1 is below Jev on both sets (tweet_topic 67.8% vs 69.4%; fin_topic 59.0% vs 63.0%)."
40
+ }
figures/zeroshot.png ADDED

Git LFS Details

  • SHA256: d51f9a7bc46b88d8dcf5ba46f15d479c781302ca531208ff0838106da8aae755
  • Pointer size: 131 Bytes
  • Size of remote file: 138 kB
figures/zeroshot.svg ADDED
generation_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "eos_token_id": 248044,
4
+ "transformers_version": "5.17.0",
5
+ "use_cache": true
6
+ }
jev_style_decision.py ADDED
@@ -0,0 +1,806 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Jev-Style-2B-Decision-v3: typed decisions with transformers / PyTorch (CUDA, MPS, CPU).
2
+
3
+ Self-contained runtime for chaoliangUNSW/Jev-Style-2B-Decision-v3 (Apache-2.0). No dependency on any training
4
+ code: rendering (render v2 + attention blocks), block-causal attention, verdict readout and calibration are
5
+ implemented below and reproduce the reference implementation used for evaluation (see release_config.json ->
6
+ "runtime_parity"). Needs torch, transformers (with Qwen3.5 support), tokenizers and numpy.
7
+
8
+ python jev_style_decision.py --state "..." --question "Which option?" --options '["a", "b"]'
9
+
10
+ from jev_style_decision import JevStyleDecision
11
+ m = JevStyleDecision(".") # float32, best available device
12
+ m.decide(state, "Is the task finished?") # true/false question
13
+ m.decide_many(state, [q1, q2, q3]) # one state, computed once
14
+
15
+ Probabilities use the ONE calibrated global temperature of readout_config.json (temperatures.global
16
+ = 0.828) unless temperature=... is given (1.0 = uncalibrated scores). ``category=...`` is accepted
17
+ for compatibility with the 0.8B v3 runtime and ignored: this model has no category temperatures.
18
+
19
+ NOTE: this model uses a different input protocol (render v2 + block attention) from the 0.8B v3
20
+ models; the 0.8B runtimes must not be used with these weights (they would run ordinary causal
21
+ attention over a different layout and give wrong answers).
22
+ """
23
+ # ----------------------------------------------------------------------------------------------
24
+ # Shared core (byte-identical in jev_style_decision.py, jev_style_decision_gguf.py and jev_style_decision_mlx.py):
25
+ # input rendering (v2), verdict readout, calibration, budgets, errors, manifest check, CLI/JSONL.
26
+ #
27
+ # Input layout ("macjev-render-v2-long-options", layout "sb"; token segments are encoded separately
28
+ # and concatenated, so slot positions are exact):
29
+ #
30
+ # state prefix State:\n<state>\n\n
31
+ # short form Question [<type>]: <question>\nOptions:\n
32
+ # (question + - <option 1> ->\n ... - <option K> ->\n
33
+ # options + slots <= 2,048 tokens)
34
+ # overflow form Question [<type>]: <question>\nOptions:\n
35
+ # (otherwise) Option 1: <option 1>\n ... Option K: <option K>\n (the catalogue)
36
+ # Judge each numbered option in the complete catalogue above:\n
37
+ # Option 1 ->\n ... Option K ->\n (the rubric)
38
+ #
39
+ # Attention blocks ([start, stop) token ranges): the state is cut into consecutive 2,048-token
40
+ # blocks; the short form is one block; in the overflow form the catalogue and the rubric are each cut
41
+ # into 2,048-token blocks. In the 6 full-attention layers every block attends to all earlier tokens
42
+ # and to itself with NO causal mask inside the block (block-causal); the Gated-DeltaNet layers are
43
+ # ordinary recurrent layers. A backend must compute exactly these blocks (never merged, never re-split).
44
+ #
45
+ # Score of option k = logit(" yes") - logit(" no") at its " ->" slot = h_slot . (W_yes - W_no) in
46
+ # float32 (final normed hidden state, tied embedding rows). Probabilities = softmax(scores / T) in
47
+ # canonical option order. T = the ONE global temperature of readout_config.json
48
+ # (temperatures.global, fitted on 2,000 calibration rows) unless temperature=... overrides it
49
+ # (1.0 = uncalibrated scores). This model has no category / group temperatures and no separate
50
+ # question/options budget, so the 0.8B v3 runtime's --category and --head-max options do not exist here.
51
+ #
52
+ # Budget: the complete input (state + question + options + readout) <= 25,600 tokens. Larger inputs
53
+ # raise InputBudgetError; nothing is ever truncated. Text inside the state, question and options is
54
+ # tokenised with special tokens disabled, so e.g. "<|im_end|>" in user text can never act as a
55
+ # control token. A non-finite score (NaN / inf) raises NonFiniteScoreError: no probabilities are made.
56
+ # ----------------------------------------------------------------------------------------------
57
+ import argparse
58
+ import hashlib
59
+ import json
60
+ import math
61
+ import sys
62
+ from pathlib import Path
63
+
64
+ import numpy as np
65
+
66
+ MODEL_NAME = "Jev-Style-2B-Decision-v3"
67
+ TEMPLATE_VERSION = "macjev-render-v2-long-options"
68
+ READOUT_FORMAT = "macjev-readout-v2"
69
+ LAYOUT = "sb"
70
+ BLOCK = 2048 # attention block size (a processing unit, not a content limit)
71
+ CONTEXT_LIMIT = 25_600 # state + question + options + readout, all included
72
+ QTYPES = ("choice", "score", "noul")
73
+ JSONL_GROUP_MAX = 256 # consecutive JSONL rows with one state scored in one backend call
74
+ HERE = Path(__file__).resolve().parent
75
+
76
+
77
+ class InputBudgetError(ValueError):
78
+ """The rendered input exceeds the token budget. Nothing was truncated."""
79
+
80
+
81
+ class QuestionError(ValueError):
82
+ """The question/options are malformed."""
83
+
84
+
85
+ class NonFiniteScoreError(FloatingPointError):
86
+ """The model produced a non-finite decision score (NaN or inf). No probabilities are returned."""
87
+
88
+
89
+ # -- questions ------------------------------------------------------------------------------------
90
+ def option_names(question):
91
+ """Canonical option identifiers, in the order the probabilities are returned."""
92
+ if not isinstance(question, dict):
93
+ raise QuestionError("question must be a dict {'t', 'ins', 'crit'}")
94
+ t, crit = question.get("t"), question.get("crit")
95
+ if not isinstance(question.get("ins"), str) or not question["ins"].strip():
96
+ raise QuestionError("question text ('ins') must be a non-empty string")
97
+ if t == "choice":
98
+ if not isinstance(crit, dict) or not crit:
99
+ raise QuestionError("choice needs a non-empty dict {option name: description or None}")
100
+ return [str(k) for k in crit]
101
+ if t == "score":
102
+ if not isinstance(crit, list) or not 2 <= len(crit) <= 10:
103
+ raise QuestionError("score needs a list of 2..10 level descriptions")
104
+ return [str(i) for i in range(len(crit))]
105
+ if t == "noul":
106
+ if crit is not None and not isinstance(crit, dict):
107
+ raise QuestionError("noul criteria must be None or {'false': ..., 'true': ...}")
108
+ return ["false", "true"]
109
+ raise QuestionError(f"unknown question type {t!r} (expected one of {QTYPES})")
110
+
111
+
112
+ def make_question(question, options=None, qtype=None):
113
+ """Build a typed question.
114
+
115
+ * ``question`` already a dict {"t", "ins", "crit"}: validated and returned.
116
+ * ``qtype="choice"`` (default when ``options`` is given): ``options`` = {name: description or None}
117
+ or a list of names.
118
+ * ``qtype="score"``: ``options`` = list of 2..10 level descriptions (level 0 first).
119
+ * ``qtype="noul"`` (default when no options): a true/false statement; ``options`` may be
120
+ {"false": "...", "true": "..."} to describe the two outcomes.
121
+ """
122
+ if isinstance(question, dict):
123
+ q = dict(question)
124
+ else:
125
+ if qtype is None:
126
+ qtype = "choice" if options is not None else "noul"
127
+ if qtype == "choice":
128
+ if isinstance(options, (list, tuple)):
129
+ if len(set(map(str, options))) != len(options):
130
+ raise QuestionError("duplicate option names")
131
+ crit = {str(o): None for o in options}
132
+ else:
133
+ crit = options
134
+ elif qtype == "score":
135
+ crit = list(options) if options is not None else None
136
+ else:
137
+ crit = options
138
+ q = {"t": qtype, "ins": question, "crit": crit}
139
+ option_names(q)
140
+ return q
141
+
142
+
143
+ def serialize_state(state):
144
+ """Strings pass through unchanged; any other JSON value is serialised (ensure_ascii=False)."""
145
+ if isinstance(state, str):
146
+ return state
147
+ return json.dumps(state, ensure_ascii=False)
148
+
149
+
150
+ def _criterion(value):
151
+ if isinstance(value, str):
152
+ return value
153
+ return json.dumps(value, ensure_ascii=False, separators=(", ", ": "), default=str)
154
+
155
+
156
+ def render_options(question):
157
+ t, crit = question["t"], question.get("crit")
158
+ if t == "choice":
159
+ return [k if v is None or v == "" else f"{k}: {_criterion(v)}" for k, v in crit.items()]
160
+ if t == "score":
161
+ return [f"level {i}: {_criterion(c)}" for i, c in enumerate(crit)]
162
+ crit = crit or {}
163
+ false_c, true_c = crit.get("false"), crit.get("true")
164
+ return ["false: " + (_criterion(false_c) if false_c not in (None, "") else "no, the statement does not hold"),
165
+ "true: " + (_criterion(true_c) if true_c not in (None, "") else "yes, the statement holds")]
166
+
167
+
168
+ # -- tokenizer + renderer -----------------------------------------------------------------------
169
+ class TextEncoder:
170
+ """HF ``tokenizers`` tokenizer.json; no BOS/EOS, special tokens in text are split (never control tokens)."""
171
+
172
+ def __init__(self, tokenizer_json):
173
+ from tokenizers import Tokenizer
174
+ self.tk = Tokenizer.from_file(str(tokenizer_json))
175
+ self.tk.encode_special_tokens = True
176
+
177
+ def __call__(self, text):
178
+ return self.tk.encode(text, add_special_tokens=False).ids
179
+
180
+ def id_to_token(self, i):
181
+ return self.tk.id_to_token(int(i))
182
+
183
+ def vocab_size(self):
184
+ return self.tk.get_vocab_size(with_added_tokens=True)
185
+
186
+
187
+ def _blocks(start, stop):
188
+ return [(s, min(s + BLOCK, stop)) for s in range(start, stop, BLOCK)]
189
+
190
+
191
+ class Rendered:
192
+ """One rendered question: token ids, the verdict slots (one per option, canonical order), the
193
+ attention blocks ([start, stop) pairs tiling [0, len(ids))) and the state prefix length."""
194
+ __slots__ = ("ids", "prefix_len", "slots", "blocks", "names", "qtype", "catalogue_overflow")
195
+
196
+ def __init__(self, ids, prefix_len, slots, blocks, names, qtype, catalogue_overflow):
197
+ self.ids, self.prefix_len, self.slots, self.blocks = ids, prefix_len, slots, blocks
198
+ self.names, self.qtype, self.catalogue_overflow = names, qtype, catalogue_overflow
199
+
200
+ @property
201
+ def state_blocks(self):
202
+ return [b for b in self.blocks if b[1] <= self.prefix_len]
203
+
204
+ @property
205
+ def question_blocks(self):
206
+ return [b for b in self.blocks if b[0] >= self.prefix_len]
207
+
208
+ @property
209
+ def input_tokens(self):
210
+ return len(self.ids)
211
+
212
+ @property
213
+ def head_tokens(self):
214
+ """question + options + readout tokens (everything after the state prefix)"""
215
+ return len(self.ids) - self.prefix_len
216
+
217
+
218
+ class Renderer:
219
+ def __init__(self, encode, readout_cfg, max_len=CONTEXT_LIMIT):
220
+ want = {"format": READOUT_FORMAT, "template": TEMPLATE_VERSION, "layout": LAYOUT, "readout": "verdict",
221
+ "block_size": BLOCK, "total_context_limit": CONTEXT_LIMIT}
222
+ bad = {k: readout_cfg.get(k) for k, v in want.items() if readout_cfg.get(k) != v}
223
+ if bad:
224
+ raise ValueError(f"readout_config.json does not describe this runtime's protocol {want}; got {bad}")
225
+ if not 0 < int(max_len) <= CONTEXT_LIMIT:
226
+ raise ValueError(f"max_len must be in 1..{CONTEXT_LIMIT}")
227
+ self.enc, self.max_len = encode, int(max_len)
228
+ self.head_max = self.max_len # no separate question/options cap (field kept for the 0.8B / jev-style API)
229
+ st = readout_cfg["slot_tokens"]
230
+ self.yes, self.no, arrow = int(st["yes"]["id"]), int(st["no"]["id"]), int(st["verdict_slot"]["id"])
231
+ for text, want_id in ((" yes", self.yes), (" no", self.no), (" ->", arrow)):
232
+ got = self.enc(text)
233
+ if got != [want_id]:
234
+ raise ValueError(f"tokenizer mismatch: {text!r} -> {got}, readout_config expects [{want_id}]")
235
+ self.arrow = [arrow]
236
+ self.newline = self.enc("\n")
237
+ self.dash = self.enc("- ")
238
+ self.rubric_head = self.enc("Judge each numbered option in the complete catalogue above:\n")
239
+ self._numbered = {}
240
+
241
+ def _option_label(self, pos, catalogue):
242
+ key = (pos, catalogue)
243
+ if key not in self._numbered:
244
+ self._numbered[key] = self.enc(f"Option {pos + 1}: " if catalogue else f"Option {pos + 1}")
245
+ return self._numbered[key]
246
+
247
+ def prefix_ids(self, state):
248
+ return self.enc("State:\n") + self.enc(serialize_state(state)) + self.enc("\n\n")
249
+
250
+ def pieces(self, state, question, prefix=None):
251
+ """Tokenised segments of one question (``prefix``: already tokenised state, to tokenise it once)."""
252
+ names = option_names(question)
253
+ return {"prefix": self.prefix_ids(state) if prefix is None else list(prefix),
254
+ "head": self.enc(f"Question [{question['t']}]: {question['ins']}\nOptions:\n"),
255
+ "opts": [self.enc(o) for o in render_options(question)], "names": names, "qtype": question["t"]}
256
+
257
+ def assemble(self, pieces, max_len=None):
258
+ maximum = self.max_len if max_len is None else min(self.max_len, int(max_len))
259
+ prefix, head, opts = list(pieces["prefix"]), list(pieces["head"]), pieces["opts"]
260
+ k = len(opts)
261
+ if k < 1:
262
+ raise QuestionError("at least one option is required")
263
+ short, rel = list(head), []
264
+ for o in opts:
265
+ short += self.dash + list(o) + self.arrow
266
+ rel.append(len(short) - 1)
267
+ short += self.newline
268
+ overflow = len(short) > BLOCK
269
+ if not overflow:
270
+ ids = prefix + short
271
+ slots = [len(prefix) + s for s in rel]
272
+ blocks = _blocks(0, len(prefix)) + [(len(prefix), len(ids))]
273
+ else:
274
+ catalogue = list(head)
275
+ for pos, o in enumerate(opts):
276
+ catalogue += self._option_label(pos, True) + list(o) + self.newline
277
+ doc_end = len(prefix) + len(catalogue)
278
+ rubric, rel = list(self.rubric_head), []
279
+ for pos in range(k):
280
+ rubric += self._option_label(pos, False) + self.arrow
281
+ rel.append(len(rubric) - 1)
282
+ rubric += self.newline
283
+ ids = prefix + catalogue + rubric
284
+ slots = [doc_end + s for s in rel]
285
+ blocks = _blocks(0, len(prefix)) + _blocks(len(prefix), doc_end) + _blocks(doc_end, len(ids))
286
+ if len(ids) > maximum:
287
+ raise InputBudgetError(f"the complete input needs {len(ids)} tokens (state {len(prefix)} + question/"
288
+ f"options/readout {len(ids) - len(prefix)}); the limit is {maximum} tokens (model "
289
+ f"maximum {CONTEXT_LIMIT}). Nothing was truncated: shorten the state, the "
290
+ f"question or the options.")
291
+ return Rendered(ids, len(prefix), slots, blocks, list(pieces["names"]), pieces["qtype"], overflow)
292
+
293
+ def render(self, state, question, max_len=None, prefix=None):
294
+ return self.assemble(self.pieces(state, question, prefix), max_len)
295
+
296
+
297
+ # -- calibration ----------------------------------------------------------------------------------
298
+ def check_temperature(t):
299
+ try:
300
+ t = float(t)
301
+ except (TypeError, ValueError):
302
+ raise ValueError(f"temperature must be a number, got {t!r}") from None
303
+ if not (math.isfinite(t) and t > 0):
304
+ raise ValueError(f"temperature must be finite and > 0, got {t!r}")
305
+ return t
306
+
307
+
308
+ def softmax_probabilities(scores, temperature):
309
+ """softmax(scores / T) in float64 (the reference computation)."""
310
+ z = np.asarray(scores, float) / float(temperature)
311
+ z = np.exp(z - z.max())
312
+ return z / z.sum()
313
+
314
+
315
+ def concentration(p):
316
+ k = len(p)
317
+ if k < 2:
318
+ return 1.0
319
+ ent = -(p * np.log(np.clip(p, 1e-12, 1.0))).sum()
320
+ return float(np.clip(1.0 - ent / math.log(k), 0.0, 1.0))
321
+
322
+
323
+ def _sha256(path):
324
+ h = hashlib.sha256()
325
+ with open(path, "rb") as f:
326
+ for b in iter(lambda: f.read(1 << 22), b""):
327
+ h.update(b)
328
+ return h.hexdigest()
329
+
330
+
331
+ NOT_VERIFIED = ("README.md", "eval_results.json") # documentation / records: in manifest.json, not checked
332
+ NOT_VERIFIED_DIRS = ("assets/", "figures/", "validation/")
333
+
334
+
335
+ def verify_manifest(model_dir, only=None):
336
+ """Re-hash the files listed in manifest.json (all, or those whose path starts with one of ``only``).
337
+ Documentation and evaluation records (README.md, eval_results.json, assets/, figures/, validation/) are
338
+ recorded in the manifest but not checked here (not even when named in ``only``), so a card edit never
339
+ makes the runtime refuse to load and a download without them still verifies."""
340
+ model_dir = Path(model_dir)
341
+ man = json.loads((model_dir / "manifest.json").read_text())
342
+ bad, missing, checked = [], [], 0
343
+ for name, rec in man["files"].items():
344
+ if name in NOT_VERIFIED or name.startswith(NOT_VERIFIED_DIRS):
345
+ continue
346
+ if only and not any(name == o or name.startswith(o.rstrip("/") + "/") for o in only):
347
+ continue
348
+ p = model_dir / name
349
+ if not p.exists():
350
+ missing.append(name)
351
+ elif _sha256(p) != rec["sha256"]:
352
+ bad.append(name)
353
+ checked += 1
354
+ return {"ok": not bad and not missing, "checked": checked, "bad": bad, "missing": missing}
355
+
356
+
357
+ class DecisionBase:
358
+ """Backend-independent part. A backend implements ``_scores_many(rendered) -> list of list[float]``:
359
+ ``rendered`` is a non-empty list of Rendered that all share ONE state (identical prefix ids and
360
+ state blocks); it returns the raw float32 scores of every slot of every Rendered (slot order =
361
+ canonical option order), computing the state blocks once per call. Backends keep the most recent
362
+ state for the next call (exact prefix match only), so consecutive calls about the same state (e.g.
363
+ JSONL rows read from stdin) reuse it."""
364
+ backend = "base"
365
+
366
+ def _setup(self, model_dir, tokenizer_json, max_len=CONTEXT_LIMIT, temperature=None):
367
+ self.model_dir = Path(model_dir)
368
+ self.readout_config = json.loads((self.model_dir / "readout_config.json").read_text())
369
+ self.calibrated_temperature = check_temperature(self.readout_config["temperatures"]["global"])
370
+ self.default_temperature = self.calibrated_temperature if temperature is None else check_temperature(temperature)
371
+ self.encode = TextEncoder(tokenizer_json)
372
+ self.renderer = Renderer(self.encode, self.readout_config, max_len=max_len)
373
+
374
+ def _scores_many(self, rendered):
375
+ raise NotImplementedError
376
+
377
+ def close(self):
378
+ pass
379
+
380
+ def __enter__(self):
381
+ return self
382
+
383
+ def __exit__(self, *exc):
384
+ self.close()
385
+
386
+ def _score_all(self, rendered):
387
+ """Raw scores (lists of floats, canonical order) for Rendered sharing one state; one backend call."""
388
+ if not rendered:
389
+ return []
390
+ p = rendered[0].prefix_len
391
+ prefix, sblocks = rendered[0].ids[:p], rendered[0].state_blocks
392
+ for r in rendered[1:]:
393
+ if r.prefix_len != p or r.ids[:p] != prefix or r.state_blocks != sblocks:
394
+ raise ValueError("all questions of one backend call must share the same state")
395
+ scores = self._scores_many(rendered)
396
+ if len(scores) != len(rendered) or any(len(s) != len(r.slots) for s, r in zip(scores, rendered)):
397
+ raise RuntimeError("backend returned a wrong number of scores")
398
+ return [[float(x) for x in s] for s in scores]
399
+
400
+ def score_pieces(self, pieces_list, max_len=None):
401
+ """Raw scores for pre-tokenised questions sharing one state: dicts {"prefix", "head", "opts",
402
+ "names", "qtype"} of token ids (see Renderer.pieces). For parity checks; no temperature."""
403
+ return self._score_all([self.renderer.assemble(p, max_len) for p in pieces_list])
404
+
405
+ def _result(self, r, scores, temperature):
406
+ t = self.default_temperature if temperature is None else check_temperature(temperature)
407
+ bad = [n for n, x in zip(r.names, scores) if not math.isfinite(x)]
408
+ if bad:
409
+ raise NonFiniteScoreError(f"non-finite decision scores for option(s) {bad} ({len(r.ids)} input tokens); "
410
+ f"refusing to return probabilities. Check the weights file / engine build.")
411
+ p = softmax_probabilities(scores, t)
412
+ if not np.all(np.isfinite(p)):
413
+ raise NonFiniteScoreError("non-finite probabilities")
414
+ i = int(p.argmax())
415
+ return {"answer": r.names[i], "probabilities": dict(zip(r.names, p.tolist())),
416
+ "scores": dict(zip(r.names, scores)), "temperature": t, "top_probability": float(p[i]),
417
+ "entropy_concentration": concentration(p), "input_tokens": len(r.ids),
418
+ "state_tokens": r.prefix_len, "head_tokens": r.head_tokens, "blocks": len(r.blocks),
419
+ "catalogue_overflow": r.catalogue_overflow, "model": MODEL_NAME, "backend": self.backend}
420
+
421
+ def decide(self, state, question, options=None, qtype=None, category=None, temperature=None, head_max=None,
422
+ max_len=None):
423
+ """Score one question about ``state``.
424
+
425
+ Returns {"answer", "probabilities" {option: p}, "scores" {option: logit(yes)-logit(no)},
426
+ "temperature", "top_probability", "entropy_concentration", "input_tokens", "state_tokens",
427
+ "head_tokens", "blocks", "catalogue_overflow", "model", "backend"}. ``temperature`` overrides
428
+ the calibrated global temperature (1.0 = uncalibrated scores). ``category`` and ``head_max`` are
429
+ accepted for compatibility with the 0.8B v3 runtime and the jev-style package, and ignored: this
430
+ model has one global temperature and no separate question/options budget. Raises InputBudgetError
431
+ (never truncates), QuestionError or NonFiniteScoreError."""
432
+ q = make_question(question, options, qtype)
433
+ r = self.renderer.render(state, q, max_len)
434
+ return self._result(r, self._score_all([r])[0], temperature)
435
+
436
+ def score_many(self, state, questions, category=None, temperature=None, head_max=None, max_len=None):
437
+ """Several questions about ONE state: the state is tokenised and computed once and reused for
438
+ every question (one backend call). ``questions``: dicts {"t","ins","crit"} (or anything
439
+ make_question accepts). Results in order, identical to calling decide() per question. All
440
+ questions are rendered and budget-checked before any scoring. ``category`` / ``head_max``: see
441
+ decide() (accepted and ignored)."""
442
+ qs = [make_question(q) for q in questions]
443
+ if not qs:
444
+ return []
445
+ prefix = self.renderer.prefix_ids(state)
446
+ rs = [self.renderer.render(state, q, max_len, prefix=prefix) for q in qs]
447
+ return [self._result(r, sc, temperature) for r, sc in zip(rs, self._score_all(rs))]
448
+
449
+ decide_many = score_many
450
+
451
+
452
+ def base_arg_parser(description):
453
+ ap = argparse.ArgumentParser(description=description)
454
+ ap.add_argument("--model-dir", default=str(HERE), help="folder with the weights and readout_config.json")
455
+ ap.add_argument("--state", help="state as plain text")
456
+ ap.add_argument("--state-json", help="state as a JSON value")
457
+ ap.add_argument("--question", help="question text (or a JSON question {'t','ins','crit'})")
458
+ ap.add_argument("--options", help="JSON: {name: description} or [names] (choice); [levels] (score)")
459
+ ap.add_argument("--qtype", choices=QTYPES)
460
+ ap.add_argument("--temperature", type=float,
461
+ help="override the calibrated global temperature of readout_config.json (1.0 = raw scores)")
462
+ ap.add_argument("--max-len", type=int, default=CONTEXT_LIMIT,
463
+ help=f"total token budget (state + question + options + readout), at most {CONTEXT_LIMIT}")
464
+ ap.add_argument("--jsonl", help="batch mode: input JSON lines {id?, state, question, options?, qtype?, "
465
+ "temperature?} ('-' = stdin); one JSON result per line on stdout. Consecutive "
466
+ "rows with an identical state share one state computation")
467
+ ap.add_argument("--verify", action="store_true", help="check sha256 of the files in manifest.json first")
468
+ return ap
469
+
470
+
471
+ def _jsonl_groups(src, streaming):
472
+ """(line number, record or error text) grouped into runs of consecutive rows with one state. From a
473
+ file up to JSONL_GROUP_MAX rows are grouped; from stdin every row is its own group (answered at once;
474
+ the backend's kept state still makes consecutive identical states cheap)."""
475
+ group, key = [], None
476
+ for n, line in enumerate(src):
477
+ if not line.strip():
478
+ continue
479
+ try:
480
+ rec = json.loads(line)
481
+ if not isinstance(rec, dict):
482
+ raise ValueError("a JSONL row must be a JSON object")
483
+ k = serialize_state(rec.get("state", ""))
484
+ except ValueError as e:
485
+ if group:
486
+ yield group
487
+ group, key = [], None
488
+ yield [(n, f"{type(e).__name__}: {e}")]
489
+ continue
490
+ if group and (k != key or len(group) >= JSONL_GROUP_MAX):
491
+ yield group
492
+ group = []
493
+ group.append((n, rec))
494
+ key = k
495
+ if streaming:
496
+ yield group
497
+ group, key = [], None
498
+ if group:
499
+ yield group
500
+
501
+
502
+ def _run_group(engine, group, args):
503
+ """Score one group of JSONL rows sharing a state. Returns (output rows, non-finite count)."""
504
+ out, todo = {}, []
505
+ prefix = None
506
+ for n, rec in group:
507
+ if isinstance(rec, str):
508
+ out[n] = {"id": n, "error": rec}
509
+ continue
510
+ rid = rec.get("id", n)
511
+ try:
512
+ if "question" not in rec:
513
+ raise QuestionError("row has no 'question'")
514
+ q = make_question(rec["question"], options=rec.get("options"), qtype=rec.get("qtype"))
515
+ t = rec.get("temperature", args.temperature)
516
+ t = None if t is None else check_temperature(t)
517
+ if prefix is None:
518
+ prefix = engine.renderer.prefix_ids(rec.get("state", ""))
519
+ todo.append((n, rid, engine.renderer.render(rec.get("state", ""), q, prefix=prefix), t))
520
+ except (InputBudgetError, QuestionError, ValueError, TypeError, AttributeError) as e:
521
+ out[n] = {"id": rid, "error": f"{type(e).__name__}: {e}"}
522
+ nonfinite = 0
523
+ if todo:
524
+ scores = engine._score_all([r for _, _, r, _ in todo])
525
+ for (n, rid, r, t), sc in zip(todo, scores):
526
+ try:
527
+ out[n] = {"id": rid, **engine._result(r, sc, t)}
528
+ except NonFiniteScoreError as e:
529
+ nonfinite += 1
530
+ print(f"ERROR row {rid}: NonFiniteScoreError: {e}", file=sys.stderr, flush=True)
531
+ out[n] = {"id": rid, "error": f"NonFiniteScoreError: {e}"}
532
+ return [out[n] for n, _ in group], nonfinite
533
+
534
+
535
+ def run_cli(args, engine):
536
+ """Exit status: 0 = ok (JSONL rows with input errors carry an "error" field), 2 = input error
537
+ (single question), 3 = at least one non-finite score (refused, see stderr)."""
538
+ if args.jsonl:
539
+ streaming = args.jsonl == "-"
540
+ src = sys.stdin if streaming else open(args.jsonl, encoding="utf-8")
541
+ nonfinite = 0
542
+ try:
543
+ for group in _jsonl_groups(src, streaming):
544
+ rows, bad = _run_group(engine, group, args)
545
+ nonfinite += bad
546
+ for row in rows:
547
+ print(json.dumps(row, ensure_ascii=False), flush=True)
548
+ finally:
549
+ if not streaming:
550
+ src.close()
551
+ return 3 if nonfinite else 0
552
+ if args.question is None:
553
+ raise SystemExit("--question (or --jsonl) is required")
554
+ try:
555
+ state = json.loads(args.state_json) if args.state_json is not None else (args.state or "")
556
+ question = args.question
557
+ if question.lstrip().startswith("{"):
558
+ try: # a JSON question {'t','ins','crit'}; else plain text
559
+ parsed = json.loads(question)
560
+ except ValueError:
561
+ parsed = None
562
+ if isinstance(parsed, dict):
563
+ question = parsed
564
+ options = json.loads(args.options) if args.options else None
565
+ res = engine.decide(state, question, options=options, qtype=args.qtype, temperature=args.temperature)
566
+ except (InputBudgetError, QuestionError, ValueError, TypeError) as e: # NonFiniteScoreError is not a ValueError
567
+ print(f"error: {type(e).__name__}: {e}", file=sys.stderr)
568
+ return 2
569
+ except NonFiniteScoreError as e:
570
+ print(f"ERROR: NonFiniteScoreError: {e}", file=sys.stderr)
571
+ return 3
572
+ print(json.dumps(res, ensure_ascii=False, indent=2))
573
+ return 0
574
+ # ---------------------------------------------------------------------------- end of shared core
575
+
576
+
577
+ # ------------------------------------------------------------------------------ PyTorch backend
578
+ # How the block-causal attention is computed exactly:
579
+ #
580
+ # The input is fed to the model ONE renderer block per forward call, in order, with a transformers
581
+ # cache: when block [a, b) runs, the cache holds tokens [0, a). In the 6 full-attention layers the
582
+ # queries are the block's own tokens and the keys/values are the cached [0, a) plus [a, b); attention
583
+ # over all of them WITHOUT any mask is then exactly "every block attends to all earlier tokens and to
584
+ # itself, no causal mask inside the block". The Gated-DeltaNet layers continue their conv / recurrent
585
+ # state from the cache, i.e. they run as ordinary causal recurrent layers over the whole input.
586
+ #
587
+ # State reuse: the state blocks are computed once per call (and kept for the next call when the next
588
+ # state is identical); every question continues from its own copy of that cache, so a question never
589
+ # sees another question's tokens. Because nothing after the state can influence the state blocks
590
+ # (block-causal), this equals recomputing state + question for each question.
591
+ import copy
592
+
593
+ ATTN_NAME = "jev_style_block_sdpa"
594
+ MPS_ATTN_CHUNK = 1024 # MPS: queries per SDPA call (row-wise identical, bounds the score matrix)
595
+ _ATTN_REGISTERED = False
596
+
597
+
598
+ def _block_sdpa(module, query, key, value, attention_mask=None, dropout=0.0, scaling=None, **kwargs):
599
+ """Attention of ONE renderer block (see above): no mask; queries = the block, keys = all tokens so far.
600
+
601
+ The backend announces each call's geometry on the attention module (``jev_expect`` = (block length,
602
+ tokens so far)). Any other use, e.g. an ordinary whole-sequence ``model(...)`` call, is refused, so
603
+ this function can never silently act as non-causal attention over a wrong span."""
604
+ import torch
605
+ import torch.nn.functional as F
606
+ q_len, kv_len = query.shape[2], key.shape[2]
607
+ expect = getattr(module, "jev_expect", None)
608
+ if expect is None or tuple(expect) != (q_len, kv_len):
609
+ raise RuntimeError(f"block attention called with {q_len} queries / {kv_len} keys, expected {expect}: this "
610
+ f"model must be run through JevStyleDecision (one renderer block per forward call)")
611
+ if attention_mask is not None:
612
+ raise RuntimeError("block attention does not take an attention mask")
613
+ if query.shape[1] % key.shape[1]:
614
+ raise ValueError("invalid grouped-query head counts")
615
+ rep = query.shape[1] // key.shape[1]
616
+ if rep > 1:
617
+ key, value = key.repeat_interleave(rep, 1), value.repeat_interleave(rep, 1)
618
+ chunk = getattr(module, "jev_attn_chunk", None)
619
+ if not chunk or q_len <= chunk:
620
+ out = F.scaled_dot_product_attention(query, key, value, is_causal=False, scale=scaling)
621
+ else:
622
+ out = torch.cat([F.scaled_dot_product_attention(query[:, :, s:s + chunk], key, value, is_causal=False,
623
+ scale=scaling) for s in range(0, q_len, chunk)], dim=2)
624
+ return out.transpose(1, 2).contiguous(), None
625
+
626
+
627
+ def _no_mask(*args, **kwargs):
628
+ return None
629
+
630
+
631
+ def _register_attention():
632
+ global _ATTN_REGISTERED
633
+ if not _ATTN_REGISTERED:
634
+ from transformers import AttentionInterface
635
+ from transformers.masking_utils import AttentionMaskInterface
636
+ AttentionInterface.register(ATTN_NAME, _block_sdpa)
637
+ AttentionMaskInterface.register(ATTN_NAME, _no_mask)
638
+ _ATTN_REGISTERED = True
639
+ return ATTN_NAME
640
+
641
+
642
+ def _pick_device(torch, device):
643
+ if device is None:
644
+ if torch.cuda.is_available():
645
+ return "cuda"
646
+ mps = getattr(torch.backends, "mps", None)
647
+ return "mps" if mps is not None and mps.is_available() else "cpu"
648
+ kind = torch.device(device).type
649
+ if kind == "cuda" and not torch.cuda.is_available():
650
+ raise RuntimeError("device='cuda' was requested but CUDA is not available (no implicit fallback)")
651
+ if kind == "mps" and not (getattr(torch.backends, "mps", None) and torch.backends.mps.is_available()):
652
+ raise RuntimeError("device='mps' was requested but MPS is not available (no implicit fallback)")
653
+ if kind not in ("cuda", "mps", "cpu"):
654
+ raise ValueError(f"unsupported device {device!r} (cuda, mps or cpu)")
655
+ return str(device)
656
+
657
+
658
+ class JevStyleDecision(DecisionBase):
659
+ """Transformers / PyTorch runtime (CUDA, Apple MPS or CPU).
660
+
661
+ >>> m = JevStyleDecision(".") # float32 on the best available device
662
+ >>> m.decide({"messages": ["Refund still missing after 3 weeks"]},
663
+ ... "Which team should handle this ticket?",
664
+ ... options={"billing": "payments, refunds", "tech": "bugs, crashes", "sales": "pricing, plans"})
665
+
666
+ device: None = cuda > mps > cpu; "cuda" / "mps" / "cpu" explicitly (never an implicit fallback).
667
+ dtype: "float32" (default; the parity-tested setting on every device) or "bfloat16" (CUDA only).
668
+ The readout h_slot . (w_yes - w_no) is always computed in float32.
669
+ temperature: None = the calibrated global temperature of readout_config.json.
670
+ threads: torch CPU threads (torch.set_num_threads; process-wide).
671
+ attn_chunk: queries per SDPA call inside a block (default: 1,024 on MPS, whole block elsewhere).
672
+ keep_state: keep the last state's cache for the next call with the identical state (exact match).
673
+ category: accepted for compatibility with the 0.8B v3 runtime and ignored (one global temperature).
674
+
675
+ decide(state, question, options=None, qtype=None, category=None, temperature=None, head_max=None, max_len=None)
676
+ decide_many(state, questions, category=None, temperature=None, head_max=None, max_len=None) (= score_many)
677
+ (the same signatures as the 0.8B v3 runtime; category and head_max are accepted and ignored)
678
+ Every result has "answer", "probabilities" (by option name; true/false questions: "false" / "true"),
679
+ "scores", "temperature", "top_probability", "entropy_concentration", "input_tokens",
680
+ "state_tokens", "head_tokens", "blocks", "catalogue_overflow", "model", "backend".
681
+ """
682
+ backend = "torch"
683
+
684
+ def __init__(self, model_dir=HERE, device=None, dtype="float32", *, max_len=CONTEXT_LIMIT, temperature=None,
685
+ threads=None, attn_chunk=None, keep_state=True, verify=False, category=None):
686
+ import torch
687
+ self.torch = torch
688
+ model_dir = Path(model_dir)
689
+ if verify:
690
+ res = verify_manifest(model_dir)
691
+ if not res["ok"]:
692
+ raise RuntimeError(f"integrity check failed: {res}")
693
+ self.verified = res
694
+ self._setup(model_dir, model_dir / "tokenizer.json", max_len, temperature)
695
+ if threads:
696
+ torch.set_num_threads(int(threads))
697
+ self.device = _pick_device(torch, device)
698
+ kind = torch.device(self.device).type
699
+ dt = getattr(torch, dtype) if isinstance(dtype, str) else dtype
700
+ if dt not in (torch.float32, torch.bfloat16):
701
+ raise ValueError(f"dtype must be float32 or bfloat16, got {dtype!r}")
702
+ if dt == torch.bfloat16 and kind != "cuda":
703
+ raise ValueError("bfloat16 is supported on CUDA only; use float32 on MPS / CPU")
704
+ cfg = json.loads((model_dir / "config.json").read_text())
705
+ layer_types = cfg.get("layer_types") or []
706
+ if cfg.get("model_type") not in ("qwen3_5_text", "qwen3_5") or "full_attention" not in layer_types:
707
+ raise ValueError(f"{model_dir / 'config.json'} is not the text-only Qwen3.5 checkpoint of {MODEL_NAME}")
708
+ try:
709
+ from transformers.models.qwen3_5.modeling_qwen3_5 import Qwen3_5ForCausalLM as cls
710
+ except ImportError as e:
711
+ raise ImportError("this model needs a transformers version with Qwen3.5 support "
712
+ "(transformers.models.qwen3_5)") from e
713
+ from transformers import DynamicCache
714
+ self._cache_cls = DynamicCache
715
+ name = _register_attention()
716
+ try:
717
+ model = cls.from_pretrained(str(model_dir), dtype=dt, attn_implementation=name)
718
+ except TypeError: # transformers 4.x keyword
719
+ model = cls.from_pretrained(str(model_dir), torch_dtype=dt, attn_implementation=name)
720
+ if getattr(model.config, "_attn_implementation", None) != name:
721
+ model.set_attn_implementation(name)
722
+ self.model = model.to(self.device).eval()
723
+ self.dtype = dt
724
+ types = list(self.model.config.layer_types)
725
+ self._attn = [layer.self_attn for layer, t in zip(self.model.model.layers, types) if t == "full_attention"]
726
+ if len(self._attn) != types.count("full_attention") or not self._attn:
727
+ raise RuntimeError("could not find the full-attention layers")
728
+ chunk = attn_chunk if attn_chunk is not None else (MPS_ATTN_CHUNK if kind == "mps" else None)
729
+ for m in self._attn:
730
+ m.jev_expect, m.jev_attn_chunk = None, (int(chunk) if chunk else None)
731
+ w = self.model.get_output_embeddings().weight
732
+ if not getattr(self.model.config, "tie_word_embeddings", False) and w is not self.model.get_input_embeddings().weight:
733
+ raise ValueError("expected tied input/output embeddings")
734
+ self.direction = (w[self.renderer.yes].float() - w[self.renderer.no].float()).detach()
735
+ self.keep_state = bool(keep_state)
736
+ self._kept = None # (state token ids, cache after the state blocks)
737
+
738
+ # -- model calls
739
+ def _block(self, ids, start, stop, cache):
740
+ """Run renderer block [start, stop) on top of ``cache`` (which holds tokens [0, start)); returns
741
+ the final normed hidden states of the block's tokens."""
742
+ torch = self.torch
743
+ x = torch.tensor([ids[start:stop]], dtype=torch.long, device=self.device)
744
+ pos = torch.arange(start, stop, dtype=torch.long, device=self.device)[None]
745
+ for m in self._attn:
746
+ m.jev_expect = (stop - start, stop)
747
+ try:
748
+ out = self.model.model(input_ids=x, position_ids=pos, past_key_values=cache, use_cache=True)
749
+ finally:
750
+ for m in self._attn:
751
+ m.jev_expect = None
752
+ return out.last_hidden_state[0]
753
+
754
+ def _state_cache(self, r):
755
+ key = r.ids[:r.prefix_len]
756
+ if self._kept is not None and self._kept[0] == key:
757
+ return self._kept[1]
758
+ self._kept = None # free the old state first
759
+ cache = self._cache_cls(config=self.model.config)
760
+ for s, e in r.state_blocks:
761
+ self._block(r.ids, s, e, cache)
762
+ if self.keep_state:
763
+ self._kept = (key, cache)
764
+ return cache
765
+
766
+ def _scores_many(self, rendered):
767
+ torch = self.torch
768
+ out = []
769
+ with torch.inference_mode():
770
+ for r in rendered:
771
+ if r.state_blocks + r.question_blocks != r.blocks or r.slots != sorted(r.slots):
772
+ raise RuntimeError("renderer blocks do not tile the input")
773
+ base = self._state_cache(rendered[0])
774
+ for r in rendered:
775
+ cache = copy.deepcopy(base) # the shared state cache itself is never modified
776
+ hs = []
777
+ for s, e in r.question_blocks:
778
+ h = self._block(r.ids, s, e, cache)
779
+ rel = [p - s for p in r.slots if s <= p < e]
780
+ if rel:
781
+ hs.append(h[torch.tensor(rel, device=h.device)])
782
+ del cache
783
+ h = torch.cat(hs).float()
784
+ if h.shape[0] != len(r.slots):
785
+ raise RuntimeError("slot count mismatch")
786
+ out.append((h @ self.direction).cpu().tolist())
787
+ return out
788
+
789
+ def close(self):
790
+ self._kept = None
791
+
792
+
793
+ def main(argv=None):
794
+ ap = base_arg_parser(f"{MODEL_NAME}: typed decisions with transformers / PyTorch")
795
+ ap.add_argument("--device", choices=["cuda", "mps", "cpu"], help="default: cuda > mps > cpu")
796
+ ap.add_argument("--dtype", default="float32", choices=["float32", "bfloat16"],
797
+ help="bfloat16 on CUDA only; the readout is float32 either way")
798
+ ap.add_argument("--threads", type=int, help="torch CPU threads")
799
+ args = ap.parse_args(argv)
800
+ engine = JevStyleDecision(args.model_dir, device=args.device, dtype=args.dtype, max_len=args.max_len,
801
+ threads=args.threads, verify=args.verify)
802
+ return run_cli(args, engine)
803
+
804
+
805
+ if __name__ == "__main__":
806
+ raise SystemExit(main())
manifest.json ADDED
@@ -0,0 +1,170 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "jev-style-manifest-v1",
3
+ "repo": "chaoliangUNSW/Jev-Style-2B-Decision-v3",
4
+ "created_unix": 1790476262.791195,
5
+ "files": {
6
+ "LICENSE": {
7
+ "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a",
8
+ "bytes": 11544
9
+ },
10
+ "NOTICE": {
11
+ "sha256": "937fb1f4a643708c77dbe48cd483cb0998874feadbb107f1ca536a3ec654a0c4",
12
+ "bytes": 2406
13
+ },
14
+ "README.md": {
15
+ "sha256": "b3d39950f74a583ec0e4e89afd779c99ccfe93da8f4d0d453fd423e84b561f6a",
16
+ "bytes": 25217
17
+ },
18
+ "chat_template.jinja": {
19
+ "sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80",
20
+ "bytes": 7755
21
+ },
22
+ "config.json": {
23
+ "sha256": "88bf86c270d616198909ed1eefef8d8c21ac1fa13f62e947f20f8e1ebd02c211",
24
+ "bytes": 1791
25
+ },
26
+ "eval_results.json": {
27
+ "sha256": "8cecf3a823603300d6fb2c94b4d87239a6c7e88b31c89f381e1fdb581a0a36fb",
28
+ "bytes": 9901
29
+ },
30
+ "figures/banner.data.json": {
31
+ "sha256": "397cbfb222986f8d59a65143f2fec5a11170385c661a86c5e54d890a24d7d89e",
32
+ "bytes": 1109
33
+ },
34
+ "figures/banner.png": {
35
+ "sha256": "2623e4afad36239aaa007c34350d6f2fbacb33430ffd3c36ce2f7ee7db849b47",
36
+ "bytes": 447467
37
+ },
38
+ "figures/jevbench.data.json": {
39
+ "sha256": "508994afea8d8b1bc33e25c675621a9b3b55b01f900daa688c240c8356dffe3b",
40
+ "bytes": 2387
41
+ },
42
+ "figures/jevbench.png": {
43
+ "sha256": "c8542dba9ecc1bd2a4ea3bda844cbc3fbd661c40d6eb9254098b279a92a74ce8",
44
+ "bytes": 167843
45
+ },
46
+ "figures/jevbench.svg": {
47
+ "sha256": "13193f8d636ad5e6685cbc58863f27bced87c0ead94c252917839c7c6eed837e",
48
+ "bytes": 12280
49
+ },
50
+ "figures/zeroshot.data.json": {
51
+ "sha256": "3ad81c3dc926c988db5c339eec4991c2e2cce46faafb89db99cc47d6d8737538",
52
+ "bytes": 1841
53
+ },
54
+ "figures/zeroshot.png": {
55
+ "sha256": "d51f9a7bc46b88d8dcf5ba46f15d479c781302ca531208ff0838106da8aae755",
56
+ "bytes": 137714
57
+ },
58
+ "figures/zeroshot.svg": {
59
+ "sha256": "f3234ba4715b0d87e081ec9fc52c0b409010612f54db7aa39307a9eda3b7545f",
60
+ "bytes": 13874
61
+ },
62
+ "generation_config.json": {
63
+ "sha256": "62153eb6c69f2e1f426beaa8002b7186437e949c7588167085df14e10e9c0a73",
64
+ "bytes": 116
65
+ },
66
+ "jev_style_decision.py": {
67
+ "sha256": "5ecba24cdfff4f0e043c2ea48c9c7f514907c6804f265906e21dec0f1b20dbd6",
68
+ "bytes": 41304
69
+ },
70
+ "model-00001-of-00002.safetensors": {
71
+ "sha256": "df592f869cb1232226d31491b783bfa9406ffb3c795eae44a3b7452e1044a8fb",
72
+ "bytes": 1999936744
73
+ },
74
+ "model-00002-of-00002.safetensors": {
75
+ "sha256": "c6ff77a32a8cf36767e51c1282bb345a873db6509847f6c75baadd8fb17eeb70",
76
+ "bytes": 1763755048
77
+ },
78
+ "model.safetensors.index.json": {
79
+ "sha256": "f60ba6dc3c3cfb32baf76fbb5f851b1a35616ba04543c35640aa973bd3947c52",
80
+ "bytes": 31529
81
+ },
82
+ "readout_config.json": {
83
+ "sha256": "4af0d578c9126d4eb1a545b6e576e0a49a967243a47b8e99e3555f081fdaa1de",
84
+ "bytes": 2031
85
+ },
86
+ "release_config.json": {
87
+ "sha256": "1c4c6917b43773fdaa7eb5ccfe4c2ed790f0c2fba6bcc2977e20701940ebb03e",
88
+ "bytes": 30558
89
+ },
90
+ "requirements.txt": {
91
+ "sha256": "d653173d284907c39856430cf264acf73b24d3350a04b0d31da8739e7f801651",
92
+ "bytes": 272
93
+ },
94
+ "tokenizer.json": {
95
+ "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
96
+ "bytes": 19989325
97
+ },
98
+ "tokenizer_config.json": {
99
+ "sha256": "66e427c470fe580fe8c7b5725d857af23d8417e37fae62667ec698306a19987b",
100
+ "bytes": 1124
101
+ },
102
+ "validation/SOURCES.json": {
103
+ "sha256": "5e35ae9eac7fc9e081693806bbf30b0bcd701d36f0807c3d29cad54c26ed2518",
104
+ "bytes": 4161
105
+ },
106
+ "validation/benchmarks/contamination.json": {
107
+ "sha256": "87f7f645b950b4cd626ff248b1899578c078db74d4c74bc944b017badc06a8e8",
108
+ "bytes": 3345
109
+ },
110
+ "validation/benchmarks/jevbench_v1.4.1_results.json": {
111
+ "sha256": "9a467fb4ddcc9a42f9d82ff8f8cf1ce5cfafd51d27b8d4dca2b9c1a681f67a9c",
112
+ "bytes": 15051
113
+ },
114
+ "validation/benchmarks/jevbench_v1.4.1_results.md": {
115
+ "sha256": "824ffa6ae920595e9277c9d6c41770bf6b2d9fc2e71ff7de9afe6ee635b69ae8",
116
+ "bytes": 1587
117
+ },
118
+ "validation/benchmarks/zeroshot_comparison.md": {
119
+ "sha256": "34828fa9375ea717c866b1e749d3fe902f0b52303c807289a769934760f6dad0",
120
+ "bytes": 1174
121
+ },
122
+ "validation/benchmarks/zeroshot_metrics.json": {
123
+ "sha256": "479ddbaa101f843487471500061906e78ce28f9af09c65eaadc3cee4b5ebdb4a",
124
+ "bytes": 8732
125
+ },
126
+ "validation/data_sources.json": {
127
+ "sha256": "415fa9dcb893dc71d97d76652fbf03600a72681c15f7248ef46035d85ea56d7c",
128
+ "bytes": 16696
129
+ },
130
+ "validation/latency_2b.json": {
131
+ "sha256": "458d00e0c81b3165072b8547278e6039b9ae0d45d54b7421008ef24c1d24ca3d",
132
+ "bytes": 40294
133
+ },
134
+ "validation/parity/PREDECLARED_RELEASE_GATES_2B.md": {
135
+ "sha256": "783aec77300779582911cdc57c71cf8b101b0cd5ca040073f7e1fa3ac20af08e",
136
+ "bytes": 2619
137
+ },
138
+ "validation/parity/cross_format_dp.json": {
139
+ "sha256": "c7332df0f02244416b01e49ecb037c76d4bf2643a2b42505ff1dfe896879e533",
140
+ "bytes": 1550
141
+ },
142
+ "validation/runtime/parity_cpu_fp32_t4.log": {
143
+ "sha256": "616618ec2b61c472137ea4cb418dc0b79019b699e7fa6f044bf06c166cb970ba",
144
+ "bytes": 4486
145
+ },
146
+ "validation/runtime/parity_mps_fp32_long.log": {
147
+ "sha256": "04e40ead1c8694fb3bbda9714074701777ea48d8d2cb3489df738ad902f351d8",
148
+ "bytes": 1008
149
+ },
150
+ "validation/runtime/reverify_main.json": {
151
+ "sha256": "3b8e573deb626385ce469b289e8e753d60c499ce21efcdeefd20a8e44b4d6c0a",
152
+ "bytes": 968
153
+ },
154
+ "validation/runtime/reverify_main_mps_long.log": {
155
+ "sha256": "98720ecc5f12d24f7a12add39a7cf332f12c300c8177eed2665a3077d0e0f0b1",
156
+ "bytes": 1007
157
+ },
158
+ "validation/runtime/v_jevstyle_e2e.json": {
159
+ "sha256": "812cbee4d6c8b0af5cd81b6b5aaa8b24f55c07c0ec934cc81c8f61c634e734a6",
160
+ "bytes": 1622
161
+ },
162
+ "validation/runtime/v_tiny_and_render.json": {
163
+ "sha256": "d3ff75211cfac2f26f0f7763ee937dee8d1e01eceae43e37f16f9ec5d0bb02bf",
164
+ "bytes": 406
165
+ }
166
+ },
167
+ "readme_hashed": true,
168
+ "readme_placeholder": false,
169
+ "note": "manifest.json hashes every file of the repo except itself, README.md, figures/ and validation/ included. The runtime --verify check (jev_style_decision.py) skips the documentation and evaluation records (README.md, figures/, assets/, validation/, eval_results.json) and checks every other listed file."
170
+ }
model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:df592f869cb1232226d31491b783bfa9406ffb3c795eae44a3b7452e1044a8fb
3
+ size 1999936744
model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c6ff77a32a8cf36767e51c1282bb345a873db6509847f6c75baadd8fb17eeb70
3
+ size 1763755048
model.safetensors.index.json ADDED
@@ -0,0 +1,328 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_parameters": 1881825088,
4
+ "total_size": 3763650176
5
+ },
6
+ "weight_map": {
7
+ "model.language_model.embed_tokens.weight": "model-00001-of-00002.safetensors",
8
+ "model.language_model.layers.0.input_layernorm.weight": "model-00001-of-00002.safetensors",
9
+ "model.language_model.layers.0.linear_attn.A_log": "model-00001-of-00002.safetensors",
10
+ "model.language_model.layers.0.linear_attn.conv1d.weight": "model-00001-of-00002.safetensors",
11
+ "model.language_model.layers.0.linear_attn.dt_bias": "model-00001-of-00002.safetensors",
12
+ "model.language_model.layers.0.linear_attn.in_proj_a.weight": "model-00001-of-00002.safetensors",
13
+ "model.language_model.layers.0.linear_attn.in_proj_b.weight": "model-00001-of-00002.safetensors",
14
+ "model.language_model.layers.0.linear_attn.in_proj_qkv.weight": "model-00001-of-00002.safetensors",
15
+ "model.language_model.layers.0.linear_attn.in_proj_z.weight": "model-00001-of-00002.safetensors",
16
+ "model.language_model.layers.0.linear_attn.norm.weight": "model-00001-of-00002.safetensors",
17
+ "model.language_model.layers.0.linear_attn.out_proj.weight": "model-00001-of-00002.safetensors",
18
+ "model.language_model.layers.0.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
19
+ "model.language_model.layers.0.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
20
+ "model.language_model.layers.0.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
21
+ "model.language_model.layers.0.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
22
+ "model.language_model.layers.1.input_layernorm.weight": "model-00001-of-00002.safetensors",
23
+ "model.language_model.layers.1.linear_attn.A_log": "model-00001-of-00002.safetensors",
24
+ "model.language_model.layers.1.linear_attn.conv1d.weight": "model-00001-of-00002.safetensors",
25
+ "model.language_model.layers.1.linear_attn.dt_bias": "model-00001-of-00002.safetensors",
26
+ "model.language_model.layers.1.linear_attn.in_proj_a.weight": "model-00001-of-00002.safetensors",
27
+ "model.language_model.layers.1.linear_attn.in_proj_b.weight": "model-00001-of-00002.safetensors",
28
+ "model.language_model.layers.1.linear_attn.in_proj_qkv.weight": "model-00001-of-00002.safetensors",
29
+ "model.language_model.layers.1.linear_attn.in_proj_z.weight": "model-00001-of-00002.safetensors",
30
+ "model.language_model.layers.1.linear_attn.norm.weight": "model-00001-of-00002.safetensors",
31
+ "model.language_model.layers.1.linear_attn.out_proj.weight": "model-00001-of-00002.safetensors",
32
+ "model.language_model.layers.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
33
+ "model.language_model.layers.1.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
34
+ "model.language_model.layers.1.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
35
+ "model.language_model.layers.1.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
36
+ "model.language_model.layers.10.input_layernorm.weight": "model-00002-of-00002.safetensors",
37
+ "model.language_model.layers.10.linear_attn.A_log": "model-00002-of-00002.safetensors",
38
+ "model.language_model.layers.10.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
39
+ "model.language_model.layers.10.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
40
+ "model.language_model.layers.10.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
41
+ "model.language_model.layers.10.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
42
+ "model.language_model.layers.10.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
43
+ "model.language_model.layers.10.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
44
+ "model.language_model.layers.10.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
45
+ "model.language_model.layers.10.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
46
+ "model.language_model.layers.10.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
47
+ "model.language_model.layers.10.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
48
+ "model.language_model.layers.10.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
49
+ "model.language_model.layers.10.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
50
+ "model.language_model.layers.11.input_layernorm.weight": "model-00002-of-00002.safetensors",
51
+ "model.language_model.layers.11.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
52
+ "model.language_model.layers.11.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
53
+ "model.language_model.layers.11.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
54
+ "model.language_model.layers.11.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
55
+ "model.language_model.layers.11.self_attn.k_norm.weight": "model-00002-of-00002.safetensors",
56
+ "model.language_model.layers.11.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
57
+ "model.language_model.layers.11.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
58
+ "model.language_model.layers.11.self_attn.q_norm.weight": "model-00002-of-00002.safetensors",
59
+ "model.language_model.layers.11.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
60
+ "model.language_model.layers.11.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
61
+ "model.language_model.layers.12.input_layernorm.weight": "model-00002-of-00002.safetensors",
62
+ "model.language_model.layers.12.linear_attn.A_log": "model-00002-of-00002.safetensors",
63
+ "model.language_model.layers.12.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
64
+ "model.language_model.layers.12.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
65
+ "model.language_model.layers.12.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
66
+ "model.language_model.layers.12.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
67
+ "model.language_model.layers.12.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
68
+ "model.language_model.layers.12.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
69
+ "model.language_model.layers.12.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
70
+ "model.language_model.layers.12.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
71
+ "model.language_model.layers.12.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
72
+ "model.language_model.layers.12.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
73
+ "model.language_model.layers.12.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
74
+ "model.language_model.layers.12.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
75
+ "model.language_model.layers.13.input_layernorm.weight": "model-00002-of-00002.safetensors",
76
+ "model.language_model.layers.13.linear_attn.A_log": "model-00002-of-00002.safetensors",
77
+ "model.language_model.layers.13.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
78
+ "model.language_model.layers.13.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
79
+ "model.language_model.layers.13.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
80
+ "model.language_model.layers.13.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
81
+ "model.language_model.layers.13.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
82
+ "model.language_model.layers.13.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
83
+ "model.language_model.layers.13.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
84
+ "model.language_model.layers.13.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
85
+ "model.language_model.layers.13.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
86
+ "model.language_model.layers.13.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
87
+ "model.language_model.layers.13.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
88
+ "model.language_model.layers.13.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
89
+ "model.language_model.layers.14.input_layernorm.weight": "model-00002-of-00002.safetensors",
90
+ "model.language_model.layers.14.linear_attn.A_log": "model-00002-of-00002.safetensors",
91
+ "model.language_model.layers.14.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
92
+ "model.language_model.layers.14.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
93
+ "model.language_model.layers.14.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
94
+ "model.language_model.layers.14.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
95
+ "model.language_model.layers.14.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
96
+ "model.language_model.layers.14.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
97
+ "model.language_model.layers.14.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
98
+ "model.language_model.layers.14.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
99
+ "model.language_model.layers.14.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
100
+ "model.language_model.layers.14.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
101
+ "model.language_model.layers.14.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
102
+ "model.language_model.layers.14.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
103
+ "model.language_model.layers.15.input_layernorm.weight": "model-00002-of-00002.safetensors",
104
+ "model.language_model.layers.15.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
105
+ "model.language_model.layers.15.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
106
+ "model.language_model.layers.15.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
107
+ "model.language_model.layers.15.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
108
+ "model.language_model.layers.15.self_attn.k_norm.weight": "model-00002-of-00002.safetensors",
109
+ "model.language_model.layers.15.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
110
+ "model.language_model.layers.15.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
111
+ "model.language_model.layers.15.self_attn.q_norm.weight": "model-00002-of-00002.safetensors",
112
+ "model.language_model.layers.15.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
113
+ "model.language_model.layers.15.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
114
+ "model.language_model.layers.16.input_layernorm.weight": "model-00002-of-00002.safetensors",
115
+ "model.language_model.layers.16.linear_attn.A_log": "model-00002-of-00002.safetensors",
116
+ "model.language_model.layers.16.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
117
+ "model.language_model.layers.16.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
118
+ "model.language_model.layers.16.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
119
+ "model.language_model.layers.16.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
120
+ "model.language_model.layers.16.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
121
+ "model.language_model.layers.16.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
122
+ "model.language_model.layers.16.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
123
+ "model.language_model.layers.16.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
124
+ "model.language_model.layers.16.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
125
+ "model.language_model.layers.16.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
126
+ "model.language_model.layers.16.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
127
+ "model.language_model.layers.16.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
128
+ "model.language_model.layers.17.input_layernorm.weight": "model-00002-of-00002.safetensors",
129
+ "model.language_model.layers.17.linear_attn.A_log": "model-00002-of-00002.safetensors",
130
+ "model.language_model.layers.17.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
131
+ "model.language_model.layers.17.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
132
+ "model.language_model.layers.17.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
133
+ "model.language_model.layers.17.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
134
+ "model.language_model.layers.17.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
135
+ "model.language_model.layers.17.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
136
+ "model.language_model.layers.17.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
137
+ "model.language_model.layers.17.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
138
+ "model.language_model.layers.17.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
139
+ "model.language_model.layers.17.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
140
+ "model.language_model.layers.17.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
141
+ "model.language_model.layers.17.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
142
+ "model.language_model.layers.18.input_layernorm.weight": "model-00002-of-00002.safetensors",
143
+ "model.language_model.layers.18.linear_attn.A_log": "model-00002-of-00002.safetensors",
144
+ "model.language_model.layers.18.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
145
+ "model.language_model.layers.18.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
146
+ "model.language_model.layers.18.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
147
+ "model.language_model.layers.18.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
148
+ "model.language_model.layers.18.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
149
+ "model.language_model.layers.18.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
150
+ "model.language_model.layers.18.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
151
+ "model.language_model.layers.18.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
152
+ "model.language_model.layers.18.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
153
+ "model.language_model.layers.18.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
154
+ "model.language_model.layers.18.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
155
+ "model.language_model.layers.18.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
156
+ "model.language_model.layers.19.input_layernorm.weight": "model-00002-of-00002.safetensors",
157
+ "model.language_model.layers.19.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
158
+ "model.language_model.layers.19.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
159
+ "model.language_model.layers.19.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
160
+ "model.language_model.layers.19.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
161
+ "model.language_model.layers.19.self_attn.k_norm.weight": "model-00002-of-00002.safetensors",
162
+ "model.language_model.layers.19.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
163
+ "model.language_model.layers.19.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
164
+ "model.language_model.layers.19.self_attn.q_norm.weight": "model-00002-of-00002.safetensors",
165
+ "model.language_model.layers.19.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
166
+ "model.language_model.layers.19.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
167
+ "model.language_model.layers.2.input_layernorm.weight": "model-00001-of-00002.safetensors",
168
+ "model.language_model.layers.2.linear_attn.A_log": "model-00001-of-00002.safetensors",
169
+ "model.language_model.layers.2.linear_attn.conv1d.weight": "model-00001-of-00002.safetensors",
170
+ "model.language_model.layers.2.linear_attn.dt_bias": "model-00001-of-00002.safetensors",
171
+ "model.language_model.layers.2.linear_attn.in_proj_a.weight": "model-00001-of-00002.safetensors",
172
+ "model.language_model.layers.2.linear_attn.in_proj_b.weight": "model-00001-of-00002.safetensors",
173
+ "model.language_model.layers.2.linear_attn.in_proj_qkv.weight": "model-00001-of-00002.safetensors",
174
+ "model.language_model.layers.2.linear_attn.in_proj_z.weight": "model-00001-of-00002.safetensors",
175
+ "model.language_model.layers.2.linear_attn.norm.weight": "model-00001-of-00002.safetensors",
176
+ "model.language_model.layers.2.linear_attn.out_proj.weight": "model-00001-of-00002.safetensors",
177
+ "model.language_model.layers.2.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
178
+ "model.language_model.layers.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
179
+ "model.language_model.layers.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
180
+ "model.language_model.layers.2.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
181
+ "model.language_model.layers.20.input_layernorm.weight": "model-00002-of-00002.safetensors",
182
+ "model.language_model.layers.20.linear_attn.A_log": "model-00002-of-00002.safetensors",
183
+ "model.language_model.layers.20.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
184
+ "model.language_model.layers.20.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
185
+ "model.language_model.layers.20.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
186
+ "model.language_model.layers.20.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
187
+ "model.language_model.layers.20.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
188
+ "model.language_model.layers.20.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
189
+ "model.language_model.layers.20.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
190
+ "model.language_model.layers.20.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
191
+ "model.language_model.layers.20.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
192
+ "model.language_model.layers.20.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
193
+ "model.language_model.layers.20.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
194
+ "model.language_model.layers.20.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
195
+ "model.language_model.layers.21.input_layernorm.weight": "model-00002-of-00002.safetensors",
196
+ "model.language_model.layers.21.linear_attn.A_log": "model-00002-of-00002.safetensors",
197
+ "model.language_model.layers.21.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
198
+ "model.language_model.layers.21.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
199
+ "model.language_model.layers.21.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
200
+ "model.language_model.layers.21.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
201
+ "model.language_model.layers.21.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
202
+ "model.language_model.layers.21.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
203
+ "model.language_model.layers.21.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
204
+ "model.language_model.layers.21.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
205
+ "model.language_model.layers.21.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
206
+ "model.language_model.layers.21.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
207
+ "model.language_model.layers.21.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
208
+ "model.language_model.layers.21.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
209
+ "model.language_model.layers.22.input_layernorm.weight": "model-00002-of-00002.safetensors",
210
+ "model.language_model.layers.22.linear_attn.A_log": "model-00002-of-00002.safetensors",
211
+ "model.language_model.layers.22.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
212
+ "model.language_model.layers.22.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
213
+ "model.language_model.layers.22.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
214
+ "model.language_model.layers.22.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
215
+ "model.language_model.layers.22.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
216
+ "model.language_model.layers.22.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
217
+ "model.language_model.layers.22.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
218
+ "model.language_model.layers.22.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
219
+ "model.language_model.layers.22.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
220
+ "model.language_model.layers.22.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
221
+ "model.language_model.layers.22.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
222
+ "model.language_model.layers.22.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
223
+ "model.language_model.layers.23.input_layernorm.weight": "model-00002-of-00002.safetensors",
224
+ "model.language_model.layers.23.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
225
+ "model.language_model.layers.23.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
226
+ "model.language_model.layers.23.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
227
+ "model.language_model.layers.23.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
228
+ "model.language_model.layers.23.self_attn.k_norm.weight": "model-00002-of-00002.safetensors",
229
+ "model.language_model.layers.23.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
230
+ "model.language_model.layers.23.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
231
+ "model.language_model.layers.23.self_attn.q_norm.weight": "model-00002-of-00002.safetensors",
232
+ "model.language_model.layers.23.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
233
+ "model.language_model.layers.23.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
234
+ "model.language_model.layers.3.input_layernorm.weight": "model-00001-of-00002.safetensors",
235
+ "model.language_model.layers.3.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
236
+ "model.language_model.layers.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
237
+ "model.language_model.layers.3.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
238
+ "model.language_model.layers.3.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
239
+ "model.language_model.layers.3.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
240
+ "model.language_model.layers.3.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
241
+ "model.language_model.layers.3.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
242
+ "model.language_model.layers.3.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
243
+ "model.language_model.layers.3.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
244
+ "model.language_model.layers.3.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
245
+ "model.language_model.layers.4.input_layernorm.weight": "model-00001-of-00002.safetensors",
246
+ "model.language_model.layers.4.linear_attn.A_log": "model-00001-of-00002.safetensors",
247
+ "model.language_model.layers.4.linear_attn.conv1d.weight": "model-00001-of-00002.safetensors",
248
+ "model.language_model.layers.4.linear_attn.dt_bias": "model-00001-of-00002.safetensors",
249
+ "model.language_model.layers.4.linear_attn.in_proj_a.weight": "model-00001-of-00002.safetensors",
250
+ "model.language_model.layers.4.linear_attn.in_proj_b.weight": "model-00001-of-00002.safetensors",
251
+ "model.language_model.layers.4.linear_attn.in_proj_qkv.weight": "model-00001-of-00002.safetensors",
252
+ "model.language_model.layers.4.linear_attn.in_proj_z.weight": "model-00001-of-00002.safetensors",
253
+ "model.language_model.layers.4.linear_attn.norm.weight": "model-00001-of-00002.safetensors",
254
+ "model.language_model.layers.4.linear_attn.out_proj.weight": "model-00001-of-00002.safetensors",
255
+ "model.language_model.layers.4.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
256
+ "model.language_model.layers.4.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
257
+ "model.language_model.layers.4.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
258
+ "model.language_model.layers.4.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
259
+ "model.language_model.layers.5.input_layernorm.weight": "model-00001-of-00002.safetensors",
260
+ "model.language_model.layers.5.linear_attn.A_log": "model-00001-of-00002.safetensors",
261
+ "model.language_model.layers.5.linear_attn.conv1d.weight": "model-00001-of-00002.safetensors",
262
+ "model.language_model.layers.5.linear_attn.dt_bias": "model-00001-of-00002.safetensors",
263
+ "model.language_model.layers.5.linear_attn.in_proj_a.weight": "model-00001-of-00002.safetensors",
264
+ "model.language_model.layers.5.linear_attn.in_proj_b.weight": "model-00001-of-00002.safetensors",
265
+ "model.language_model.layers.5.linear_attn.in_proj_qkv.weight": "model-00001-of-00002.safetensors",
266
+ "model.language_model.layers.5.linear_attn.in_proj_z.weight": "model-00001-of-00002.safetensors",
267
+ "model.language_model.layers.5.linear_attn.norm.weight": "model-00001-of-00002.safetensors",
268
+ "model.language_model.layers.5.linear_attn.out_proj.weight": "model-00001-of-00002.safetensors",
269
+ "model.language_model.layers.5.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
270
+ "model.language_model.layers.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
271
+ "model.language_model.layers.5.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
272
+ "model.language_model.layers.5.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
273
+ "model.language_model.layers.6.input_layernorm.weight": "model-00001-of-00002.safetensors",
274
+ "model.language_model.layers.6.linear_attn.A_log": "model-00001-of-00002.safetensors",
275
+ "model.language_model.layers.6.linear_attn.conv1d.weight": "model-00001-of-00002.safetensors",
276
+ "model.language_model.layers.6.linear_attn.dt_bias": "model-00001-of-00002.safetensors",
277
+ "model.language_model.layers.6.linear_attn.in_proj_a.weight": "model-00001-of-00002.safetensors",
278
+ "model.language_model.layers.6.linear_attn.in_proj_b.weight": "model-00001-of-00002.safetensors",
279
+ "model.language_model.layers.6.linear_attn.in_proj_qkv.weight": "model-00001-of-00002.safetensors",
280
+ "model.language_model.layers.6.linear_attn.in_proj_z.weight": "model-00001-of-00002.safetensors",
281
+ "model.language_model.layers.6.linear_attn.norm.weight": "model-00001-of-00002.safetensors",
282
+ "model.language_model.layers.6.linear_attn.out_proj.weight": "model-00001-of-00002.safetensors",
283
+ "model.language_model.layers.6.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
284
+ "model.language_model.layers.6.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
285
+ "model.language_model.layers.6.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
286
+ "model.language_model.layers.6.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
287
+ "model.language_model.layers.7.input_layernorm.weight": "model-00001-of-00002.safetensors",
288
+ "model.language_model.layers.7.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
289
+ "model.language_model.layers.7.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
290
+ "model.language_model.layers.7.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
291
+ "model.language_model.layers.7.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
292
+ "model.language_model.layers.7.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
293
+ "model.language_model.layers.7.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
294
+ "model.language_model.layers.7.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
295
+ "model.language_model.layers.7.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
296
+ "model.language_model.layers.7.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
297
+ "model.language_model.layers.7.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
298
+ "model.language_model.layers.8.input_layernorm.weight": "model-00001-of-00002.safetensors",
299
+ "model.language_model.layers.8.linear_attn.A_log": "model-00001-of-00002.safetensors",
300
+ "model.language_model.layers.8.linear_attn.conv1d.weight": "model-00001-of-00002.safetensors",
301
+ "model.language_model.layers.8.linear_attn.dt_bias": "model-00001-of-00002.safetensors",
302
+ "model.language_model.layers.8.linear_attn.in_proj_a.weight": "model-00001-of-00002.safetensors",
303
+ "model.language_model.layers.8.linear_attn.in_proj_b.weight": "model-00001-of-00002.safetensors",
304
+ "model.language_model.layers.8.linear_attn.in_proj_qkv.weight": "model-00001-of-00002.safetensors",
305
+ "model.language_model.layers.8.linear_attn.in_proj_z.weight": "model-00001-of-00002.safetensors",
306
+ "model.language_model.layers.8.linear_attn.norm.weight": "model-00001-of-00002.safetensors",
307
+ "model.language_model.layers.8.linear_attn.out_proj.weight": "model-00001-of-00002.safetensors",
308
+ "model.language_model.layers.8.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
309
+ "model.language_model.layers.8.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
310
+ "model.language_model.layers.8.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
311
+ "model.language_model.layers.8.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
312
+ "model.language_model.layers.9.input_layernorm.weight": "model-00002-of-00002.safetensors",
313
+ "model.language_model.layers.9.linear_attn.A_log": "model-00002-of-00002.safetensors",
314
+ "model.language_model.layers.9.linear_attn.conv1d.weight": "model-00002-of-00002.safetensors",
315
+ "model.language_model.layers.9.linear_attn.dt_bias": "model-00002-of-00002.safetensors",
316
+ "model.language_model.layers.9.linear_attn.in_proj_a.weight": "model-00002-of-00002.safetensors",
317
+ "model.language_model.layers.9.linear_attn.in_proj_b.weight": "model-00002-of-00002.safetensors",
318
+ "model.language_model.layers.9.linear_attn.in_proj_qkv.weight": "model-00002-of-00002.safetensors",
319
+ "model.language_model.layers.9.linear_attn.in_proj_z.weight": "model-00002-of-00002.safetensors",
320
+ "model.language_model.layers.9.linear_attn.norm.weight": "model-00002-of-00002.safetensors",
321
+ "model.language_model.layers.9.linear_attn.out_proj.weight": "model-00002-of-00002.safetensors",
322
+ "model.language_model.layers.9.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
323
+ "model.language_model.layers.9.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
324
+ "model.language_model.layers.9.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
325
+ "model.language_model.layers.9.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
326
+ "model.language_model.norm.weight": "model-00002-of-00002.safetensors"
327
+ }
328
+ }
readout_config.json ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "macjev-readout-v2",
3
+ "model_name": "Jev-Style-2B-Decision-v3",
4
+ "readout": "verdict",
5
+ "template": "macjev-render-v2-long-options",
6
+ "layout": "sb",
7
+ "block_size": 2048,
8
+ "total_context_limit": 25600,
9
+ "question_option_limit": null,
10
+ "slot_tokens": {
11
+ "yes": {
12
+ "text": " yes",
13
+ "id": 9542
14
+ },
15
+ "no": {
16
+ "text": " no",
17
+ "id": 874
18
+ },
19
+ "verdict_slot": {
20
+ "text": " ->",
21
+ "id": 1411
22
+ }
23
+ },
24
+ "attention": "block-causal in the 6 full-attention layers: the input is cut into [start, stop) blocks of at most block_size tokens (state blocks; one short question block, or catalogue blocks + rubric blocks when question + options + slots exceed block_size); every block attends to all earlier tokens and to itself with no causal mask inside the block. Gated-DeltaNet layers are ordinary recurrent layers.",
25
+ "score": "per option k: logit[' yes'] - logit[' no'] at the k-th ' ->' slot, computed as h_slot . (w_yes - w_no) from the final normed hidden state and the tied embedding rows (float32)",
26
+ "probabilities": "softmax(scores / T) in canonical option order; T = temperatures.global (one global temperature)",
27
+ "tokenizer_sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
28
+ "budgets": {
29
+ "max_len": 25600,
30
+ "question_option_limit": null,
31
+ "note": "the complete input (state + question + options + readout) <= max_len tokens; there is no separate question/options cap; larger inputs raise InputBudgetError, nothing is truncated"
32
+ },
33
+ "temperatures": {
34
+ "version": "macjev-temperature-v2-global",
35
+ "global": 0.8278650620942867,
36
+ "groups": null,
37
+ "groups_note": "none: this model has one global temperature (no category / family / option-count temperatures)",
38
+ "fitted_on": "2,000 independent calibration rows (not dev, not test)",
39
+ "n_rows": 2000,
40
+ "fit_quality": {
41
+ "nll_before": 0.3002617012172898,
42
+ "nll_after": 0.2900580002426921,
43
+ "n": 2000
44
+ },
45
+ "selected_weights": "ema"
46
+ }
47
+ }
release_config.json ADDED
@@ -0,0 +1,804 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "jev-style-release-v1",
3
+ "model_name": "Jev-Style-2B-Decision-v3",
4
+ "repo": "chaoliangUNSW/Jev-Style-2B-Decision-v3",
5
+ "generation": "v3 (third generation of the Jev-Style decision series)",
6
+ "lineage": {
7
+ "v1": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision",
8
+ "v1_public_gguf": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF",
9
+ "v2": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2",
10
+ "v3_0.8b": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3",
11
+ "v3_2b": "chaoliangUNSW/Jev-Style-2B-Decision-v3"
12
+ },
13
+ "base_model": "Qwen/Qwen3.5-2B",
14
+ "base_model_revision": "15852e8c16360a2fea060d615a32b45270f8a8fc",
15
+ "base_model_relation": "finetune",
16
+ "architecture": "Qwen3_5ForCausalLM (text only, 24 layers: 18 Gated DeltaNet + 6 full attention, hidden 2048, tied embeddings, 1,881,825,088 parameters)",
17
+ "readout": "verdict",
18
+ "template": "macjev-render-v2-long-options",
19
+ "layout": "sb",
20
+ "attention": "block-causal in the 6 full-attention layers (2,048-token blocks, no causal mask inside a block); Gated DeltaNet recurrent",
21
+ "readout_config": "readout_config.json",
22
+ "budgets": {
23
+ "max_len": 25600,
24
+ "question_option_limit": null,
25
+ "block_size": 2048
26
+ },
27
+ "source": {
28
+ "checkpoint": "EMA weights of the full fine-tune (text-only Qwen3_5ForCausalLM, bf16, 2 safetensors shards)",
29
+ "checkpoint_sha256": {
30
+ "model-00001-of-00002.safetensors": "df592f869cb1232226d31491b783bfa9406ffb3c795eae44a3b7452e1044a8fb",
31
+ "model-00002-of-00002.safetensors": "c6ff77a32a8cf36767e51c1282bb345a873db6509847f6c75baadd8fb17eeb70"
32
+ },
33
+ "exports_manifest": {
34
+ "file": "runs/macjev/release_2b/exports_manifest.json",
35
+ "sha256": "11c549e479c7022c960c546730d6a3c42957c1c40b1b8901a183a893917d96a5",
36
+ "note": "sha256 / bytes of the exported GGUF and MLX files (release_2b/build_exports.py)"
37
+ },
38
+ "candidate_sha256sums": {
39
+ "file": "runs/macjev/candidate_2b/hf-candidate/SHA256SUMS.json",
40
+ "sha256": "7549ed3fe6239b2862998798d8da1457fdced25cd424aa4442557ae852d6e696"
41
+ }
42
+ },
43
+ "calibration": {
44
+ "version": "macjev-temperature-v2-global",
45
+ "global_T": 0.8278650620942867,
46
+ "groups": null,
47
+ "n_rows": 2000,
48
+ "fit_quality": {
49
+ "nll_before": 0.3002617012172898,
50
+ "nll_after": 0.2900580002426921,
51
+ "n": 2000
52
+ },
53
+ "fitted_on": "2,000 independent calibration rows (not dev, not test rows)"
54
+ },
55
+ "related_repos": {
56
+ "main": "chaoliangUNSW/Jev-Style-2B-Decision-v3",
57
+ "gguf": "chaoliangUNSW/Jev-Style-2B-Decision-v3-GGUF",
58
+ "mlx": "chaoliangUNSW/Jev-Style-2B-Decision-v3-MLX"
59
+ },
60
+ "tested_with": {
61
+ "python": "3.12.13",
62
+ "torch": "2.14.0",
63
+ "transformers": "5.17.0",
64
+ "tokenizers": "0.23.2",
65
+ "numpy": "2.5.3"
66
+ },
67
+ "weights": {
68
+ "model-00001-of-00002.safetensors": "bfloat16 (text-only export of the trained checkpoint, shard 1 of 2)",
69
+ "model-00002-of-00002.safetensors": "bfloat16 (text-only export of the trained checkpoint, shard 2 of 2)"
70
+ },
71
+ "runtime": {
72
+ "script": "jev_style_decision.py",
73
+ "class": "JevStyleDecision",
74
+ "default_dtype": "float32",
75
+ "dtypes": {
76
+ "float32": "cuda, mps, cpu (the parity-tested setting)",
77
+ "bfloat16": "cuda only"
78
+ },
79
+ "devices": [
80
+ "cuda",
81
+ "mps",
82
+ "cpu"
83
+ ],
84
+ "attention": "transformers AttentionInterface 'jev_style_block_sdpa': the input is fed one renderer block per forward call on top of a cache holding every earlier token; the full-attention layers run unmasked SDPA over cache + block (exactly block-causal); any other call is refused",
85
+ "gated_deltanet": "transformers Qwen3.5 layers (flash-linear-attention kernels when installed, otherwise the transformers torch fallback), conv / recurrent state continued from the cache",
86
+ "score_many": "state blocks computed once per call, each question continues from its own copy of the state cache; the last state is kept for the next call (exact match only)",
87
+ "mps_attn_chunk": 1024,
88
+ "temperature": "readout_config.json temperatures.global unless overridden",
89
+ "compat": {
90
+ "category": "accepted (keyword) and ignored: one global temperature",
91
+ "renderer.head_max": "= max_len (25,600): no separate question/options cap"
92
+ },
93
+ "not_supported": {
94
+ "--category": "no category/group temperatures (one global T)",
95
+ "--head-max": "no question/options cap (25,600-token total budget only)"
96
+ }
97
+ },
98
+ "format_gates": {
99
+ "predeclared": {
100
+ "file": "runs/macjev/runtime_v2_dev/PREDECLARED_RELEASE_GATES_2B.md",
101
+ "sha256": "783aec77300779582911cdc57c71cf8b101b0cd5ca040073f7e1fa3ac20af08e",
102
+ "gates": "F16/bf16 top-1 >= 0.99; 8-bit top-1 >= 0.98; |dNLL| <= 0.02; accuracy drop Q8_0 / MLX-8bit <= 0.3 pp, Q4_K_M <= 1.0 pp (4-bit top-1 / NLL report-only); no non-finite scores"
103
+ },
104
+ "reference": {
105
+ "what": "HF transformers FP32 on CPU, exact v2 block attention, released bf16 checkpoint (runs/macjev/candidate_2b/hf-candidate)",
106
+ "temperature": 0.8278650620942867
107
+ },
108
+ "fixtures": {
109
+ "gate": {
110
+ "what": "1,000 real dev rows (<= 4,096 tokens); decides the release gates",
111
+ "file": "runs/macjev/runtime_v2_dev/reference/gate_fixture_n1000.jsonl",
112
+ "sha256": "1c03b8c570b988b0c83ca1f79e7fd37e6196239a0836f348300c3f23ae667b68"
113
+ },
114
+ "long": {
115
+ "what": "35 requests / 43 questions up to 25,600 tokens incl. catalogue overflow and K <= 151; accuracy gate unresolvable on 43 questions (verdict INCONCLUSIVE means only that), top-1 / dNLL reported",
116
+ "file": "runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl",
117
+ "sha256": "2b16d25eafa70c00a1106ba51b94890821b51ecb4fa393abb12403c8739ef314"
118
+ }
119
+ },
120
+ "cross_format_dp": {
121
+ "file": "runs/macjev/release_2b/parity/cross_format_dp.json",
122
+ "sha256": "c7332df0f02244416b01e49ecb037c76d4bf2643a2b42505ff1dfe896879e533"
123
+ },
124
+ "formats": {
125
+ "gguf-f16": {
126
+ "gate_fixture": {
127
+ "n": 1000,
128
+ "top1_agreement": 1.0,
129
+ "top1_flips": 0,
130
+ "max_abs_dscore": 0.005850315093994141,
131
+ "mean_abs_dscore": 0.0006183286580890569,
132
+ "dnll": 2.447897923035791e-05,
133
+ "nll_ref": 0.4436679457518203,
134
+ "nll_cand": 0.44369242473105064,
135
+ "acc_ref": 0.808,
136
+ "acc_cand": 0.808,
137
+ "acc_drop_pp": 0.0,
138
+ "max_abs_dp": 0.0013728371250730786,
139
+ "mean_max_abs_dp": 0.00012138780447113503,
140
+ "gates": {
141
+ "coverage_finite_tokens": {
142
+ "pass": true,
143
+ "problems": {},
144
+ "reference_missing_requests": 0
145
+ },
146
+ "top1": {
147
+ "threshold": 0.99,
148
+ "value": 1.0,
149
+ "pass": true
150
+ },
151
+ "dnll": {
152
+ "threshold": 0.02,
153
+ "value": 2.447897923035791e-05,
154
+ "pass": true
155
+ }
156
+ },
157
+ "verdict": "PASS",
158
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
159
+ "source": {
160
+ "file": "runs/macjev/release_2b/parity/compare_gguf_f16_gate_vs_block.json",
161
+ "sha256": "d2db48b883161cca7d9c80d51a85b3c229a3bcfb7c00086887af697aeadd4c5f"
162
+ }
163
+ },
164
+ "long_fixture": {
165
+ "n": 43,
166
+ "top1_agreement": 1.0,
167
+ "top1_flips": 0,
168
+ "max_abs_dscore": 0.004771709442138672,
169
+ "mean_abs_dscore": 0.000623485138032805,
170
+ "dnll": -0.00013262687499793202,
171
+ "nll_ref": 0.670433307194272,
172
+ "nll_cand": 0.670300680319274,
173
+ "acc_ref": 0.7209302325581395,
174
+ "acc_cand": 0.7209302325581395,
175
+ "acc_drop_pp": 0.0,
176
+ "max_abs_dp": 0.0006297677378335198,
177
+ "mean_max_abs_dp": 0.00017737997866918594,
178
+ "gates": {
179
+ "coverage_finite_tokens": {
180
+ "pass": true,
181
+ "problems": {},
182
+ "reference_missing_requests": 0
183
+ },
184
+ "top1": {
185
+ "threshold": 0.99,
186
+ "value": 1.0,
187
+ "pass": true
188
+ },
189
+ "dnll": {
190
+ "threshold": 0.02,
191
+ "value": -0.00013262687499793202,
192
+ "pass": true
193
+ }
194
+ },
195
+ "verdict": "PASS",
196
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
197
+ "source": {
198
+ "file": "runs/macjev/release_2b/parity/compare_gguf_f16_base_vs_block.json",
199
+ "sha256": "079decac551d90c59d056de5468e038209db86d240e4b31623943cae0690e2cf"
200
+ }
201
+ },
202
+ "release_verdict": "PASS"
203
+ },
204
+ "gguf-q8_0": {
205
+ "gate_fixture": {
206
+ "n": 1000,
207
+ "top1_agreement": 0.997,
208
+ "top1_flips": 3,
209
+ "max_abs_dscore": 0.10028290748596191,
210
+ "mean_abs_dscore": 0.017663478106852353,
211
+ "dnll": -0.0005833314057068772,
212
+ "nll_ref": 0.4436679457518203,
213
+ "nll_cand": 0.4430846143461134,
214
+ "acc_ref": 0.808,
215
+ "acc_cand": 0.807,
216
+ "acc_drop_pp": 0.10000000000000009,
217
+ "max_abs_dp": 0.033479594816581415,
218
+ "mean_max_abs_dp": 0.0020247272188215755,
219
+ "gates": {
220
+ "coverage_finite_tokens": {
221
+ "pass": true,
222
+ "problems": {},
223
+ "reference_missing_requests": 0
224
+ },
225
+ "top1": {
226
+ "threshold": 0.98,
227
+ "value": 0.997,
228
+ "pass": true
229
+ },
230
+ "dnll": {
231
+ "threshold": 0.02,
232
+ "value": -0.0005833314057068772,
233
+ "pass": true
234
+ },
235
+ "acc_drop_pp": {
236
+ "threshold": 0.3,
237
+ "value": 0.10000000000000009,
238
+ "resolution_pp": 0.1,
239
+ "pass": true,
240
+ "note": null
241
+ }
242
+ },
243
+ "verdict": "PASS",
244
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
245
+ "source": {
246
+ "file": "runs/macjev/release_2b/parity/compare_gguf_q8_0_gate_vs_block.json",
247
+ "sha256": "4cd6c6f9e2e340ff327cb68bc289941801c85d5216508b2318ade2b588c18a37"
248
+ }
249
+ },
250
+ "long_fixture": {
251
+ "n": 43,
252
+ "top1_agreement": 1.0,
253
+ "top1_flips": 0,
254
+ "max_abs_dscore": 0.12648522853851318,
255
+ "mean_abs_dscore": 0.01955361688020088,
256
+ "dnll": 0.002535229353331059,
257
+ "nll_ref": 0.670433307194272,
258
+ "nll_cand": 0.672968536547603,
259
+ "acc_ref": 0.7209302325581395,
260
+ "acc_cand": 0.7209302325581395,
261
+ "acc_drop_pp": 0.0,
262
+ "max_abs_dp": 0.021289327356281362,
263
+ "mean_max_abs_dp": 0.0044768677260933745,
264
+ "gates": {
265
+ "coverage_finite_tokens": {
266
+ "pass": true,
267
+ "problems": {},
268
+ "reference_missing_requests": 0
269
+ },
270
+ "top1": {
271
+ "threshold": 0.98,
272
+ "value": 1.0,
273
+ "pass": true
274
+ },
275
+ "dnll": {
276
+ "threshold": 0.02,
277
+ "value": 0.002535229353331059,
278
+ "pass": true
279
+ },
280
+ "acc_drop_pp": {
281
+ "threshold": 0.3,
282
+ "value": 0.0,
283
+ "resolution_pp": 2.3255813953488373,
284
+ "pass": null,
285
+ "note": "UNRESOLVED: 43 gold questions -> one flip = 2.33pp > 0.3pp; this fixture cannot resolve the gate (use the gate fixture with >= 334 rows; statistical power needs far more)"
286
+ }
287
+ },
288
+ "verdict": "INCONCLUSIVE",
289
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
290
+ "source": {
291
+ "file": "runs/macjev/release_2b/parity/compare_gguf_q8_0_base_vs_block.json",
292
+ "sha256": "79c5564ff7f85af5976082ad003e7a84f96f020abe75c127dc9f31649bd168b8"
293
+ }
294
+ },
295
+ "release_verdict": "PASS"
296
+ },
297
+ "gguf-q4_k_m": {
298
+ "gate_fixture": {
299
+ "n": 1000,
300
+ "top1_agreement": 0.957,
301
+ "top1_flips": 43,
302
+ "max_abs_dscore": 1.5673165321350098,
303
+ "mean_abs_dscore": 0.1311025019278954,
304
+ "dnll": 0.007028574074388894,
305
+ "nll_ref": 0.4436679457518203,
306
+ "nll_cand": 0.4506965198262092,
307
+ "acc_ref": 0.808,
308
+ "acc_cand": 0.811,
309
+ "acc_drop_pp": -0.30000000000000027,
310
+ "max_abs_dp": 0.34599917206983777,
311
+ "mean_max_abs_dp": 0.02317308476687987,
312
+ "gates": {
313
+ "coverage_finite_tokens": {
314
+ "pass": true,
315
+ "problems": {},
316
+ "reference_missing_requests": 0
317
+ },
318
+ "top1": {
319
+ "threshold": null,
320
+ "value": 0.957,
321
+ "pass": null,
322
+ "note": "report only"
323
+ },
324
+ "dnll": {
325
+ "threshold": null,
326
+ "value": 0.007028574074388894,
327
+ "pass": null,
328
+ "note": "report only for 4-bit (as in the pre-registered G5)"
329
+ },
330
+ "acc_drop_pp": {
331
+ "threshold": 1.0,
332
+ "value": -0.30000000000000027,
333
+ "resolution_pp": 0.1,
334
+ "pass": true,
335
+ "note": null
336
+ }
337
+ },
338
+ "verdict": "PASS",
339
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
340
+ "source": {
341
+ "file": "runs/macjev/release_2b/parity/compare_gguf_q4_k_m_gate_vs_block.json",
342
+ "sha256": "fa6f4bbde3f8dc97c144d94ee604c6ea0386558ac4d03ee803de9650e815c275"
343
+ }
344
+ },
345
+ "long_fixture": {
346
+ "n": 43,
347
+ "top1_agreement": 0.9767441860465116,
348
+ "top1_flips": 1,
349
+ "max_abs_dscore": 1.6832613945007324,
350
+ "mean_abs_dscore": 0.15424594652319,
351
+ "dnll": 0.11204286860779777,
352
+ "nll_ref": 0.670433307194272,
353
+ "nll_cand": 0.7824761758020697,
354
+ "acc_ref": 0.7209302325581395,
355
+ "acc_cand": 0.6976744186046512,
356
+ "acc_drop_pp": 2.3255813953488302,
357
+ "max_abs_dp": 0.15778925110927658,
358
+ "mean_max_abs_dp": 0.047022506153486646,
359
+ "gates": {
360
+ "coverage_finite_tokens": {
361
+ "pass": true,
362
+ "problems": {},
363
+ "reference_missing_requests": 0
364
+ },
365
+ "top1": {
366
+ "threshold": null,
367
+ "value": 0.9767441860465116,
368
+ "pass": null,
369
+ "note": "report only"
370
+ },
371
+ "dnll": {
372
+ "threshold": null,
373
+ "value": 0.11204286860779777,
374
+ "pass": null,
375
+ "note": "report only for 4-bit (as in the pre-registered G5)"
376
+ },
377
+ "acc_drop_pp": {
378
+ "threshold": 1.0,
379
+ "value": 2.3255813953488302,
380
+ "resolution_pp": 2.3255813953488373,
381
+ "pass": null,
382
+ "note": "UNRESOLVED: 43 gold questions -> one flip = 2.33pp > 1.0pp; this fixture cannot resolve the gate (use the gate fixture with >= 100 rows; statistical power needs far more)"
383
+ }
384
+ },
385
+ "verdict": "INCONCLUSIVE",
386
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
387
+ "source": {
388
+ "file": "runs/macjev/release_2b/parity/compare_gguf_q4_k_m_base_vs_block.json",
389
+ "sha256": "cca6700d953c15c45ef55b443d963d6d955263a2858bf713d9ba73762d78b182"
390
+ }
391
+ },
392
+ "release_verdict": "PASS",
393
+ "note": "4-bit: top-1 / dNLL are report-only by pre-declaration; accuracy-drop gate <= 1.0 pp"
394
+ },
395
+ "mlx-bf16": {
396
+ "gate_fixture": {
397
+ "n": 1000,
398
+ "top1_agreement": 0.997,
399
+ "top1_flips": 3,
400
+ "max_abs_dscore": 0.13961565494537354,
401
+ "mean_abs_dscore": 0.012822698334419146,
402
+ "dnll": 0.0001265220422789204,
403
+ "nll_ref": 0.4436679457518203,
404
+ "nll_cand": 0.4437944677940992,
405
+ "acc_ref": 0.808,
406
+ "acc_cand": 0.805,
407
+ "acc_drop_pp": 0.30000000000000027,
408
+ "max_abs_dp": 0.03549758339084519,
409
+ "mean_max_abs_dp": 0.0025865935031305484,
410
+ "gates": {
411
+ "coverage_finite_tokens": {
412
+ "pass": true,
413
+ "problems": {},
414
+ "reference_missing_requests": 0
415
+ },
416
+ "top1": {
417
+ "threshold": 0.99,
418
+ "value": 0.997,
419
+ "pass": true
420
+ },
421
+ "dnll": {
422
+ "threshold": 0.02,
423
+ "value": 0.0001265220422789204,
424
+ "pass": true
425
+ }
426
+ },
427
+ "verdict": "PASS",
428
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
429
+ "source": {
430
+ "file": "runs/macjev/release_2b/parity/compare_mlx_bf16_gate_vs_block.json",
431
+ "sha256": "7fa30fa414a4ffc12a7aa6371e77579845c3b816b90da3a813a70fec10ccd9fb"
432
+ }
433
+ },
434
+ "long_fixture": {
435
+ "n": 43,
436
+ "top1_agreement": 1.0,
437
+ "top1_flips": 0,
438
+ "max_abs_dscore": 0.08836114406585693,
439
+ "mean_abs_dscore": 0.01365492056503762,
440
+ "dnll": -0.0026943261000204055,
441
+ "nll_ref": 0.670433307194272,
442
+ "nll_cand": 0.6677389810942516,
443
+ "acc_ref": 0.7209302325581395,
444
+ "acc_cand": 0.7209302325581395,
445
+ "acc_drop_pp": 0.0,
446
+ "max_abs_dp": 0.010765431767972844,
447
+ "mean_max_abs_dp": 0.003890125022245109,
448
+ "gates": {
449
+ "coverage_finite_tokens": {
450
+ "pass": true,
451
+ "problems": {},
452
+ "reference_missing_requests": 0
453
+ },
454
+ "top1": {
455
+ "threshold": 0.99,
456
+ "value": 1.0,
457
+ "pass": true
458
+ },
459
+ "dnll": {
460
+ "threshold": 0.02,
461
+ "value": -0.0026943261000204055,
462
+ "pass": true
463
+ }
464
+ },
465
+ "verdict": "PASS",
466
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
467
+ "source": {
468
+ "file": "runs/macjev/release_2b/parity/compare_mlx_bf16_base_vs_block.json",
469
+ "sha256": "ca16e5281debaf1bc601b4a8c93cf6c1106f7b032f35c5833baa4fc8178def7c"
470
+ }
471
+ },
472
+ "release_verdict": "PASS"
473
+ },
474
+ "mlx-8bit": {
475
+ "gate_fixture": {
476
+ "n": 1000,
477
+ "top1_agreement": 0.996,
478
+ "top1_flips": 4,
479
+ "max_abs_dscore": 0.3184394836425781,
480
+ "mean_abs_dscore": 0.016680597170951345,
481
+ "dnll": 0.0005035682350758575,
482
+ "nll_ref": 0.4436679457518203,
483
+ "nll_cand": 0.44417151398689614,
484
+ "acc_ref": 0.808,
485
+ "acc_cand": 0.808,
486
+ "acc_drop_pp": 0.0,
487
+ "max_abs_dp": 0.1623697296878569,
488
+ "mean_max_abs_dp": 0.0039011049015487734,
489
+ "gates": {
490
+ "coverage_finite_tokens": {
491
+ "pass": true,
492
+ "problems": {},
493
+ "reference_missing_requests": 0
494
+ },
495
+ "top1": {
496
+ "threshold": 0.98,
497
+ "value": 0.996,
498
+ "pass": true
499
+ },
500
+ "dnll": {
501
+ "threshold": 0.02,
502
+ "value": 0.0005035682350758575,
503
+ "pass": true
504
+ },
505
+ "acc_drop_pp": {
506
+ "threshold": 0.3,
507
+ "value": 0.0,
508
+ "resolution_pp": 0.1,
509
+ "pass": true,
510
+ "note": null
511
+ }
512
+ },
513
+ "verdict": "PASS",
514
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
515
+ "source": {
516
+ "file": "runs/macjev/release_2b/parity/compare_mlx_affine8-g64_gate_vs_block.json",
517
+ "sha256": "f0e2f44498df00945901c2a9bd5bff6f742d11ab243495c8089eaf0448fda782"
518
+ }
519
+ },
520
+ "long_fixture": {
521
+ "n": 43,
522
+ "top1_agreement": 1.0,
523
+ "top1_flips": 0,
524
+ "max_abs_dscore": 0.4653351306915283,
525
+ "mean_abs_dscore": 0.023077200647273945,
526
+ "dnll": 0.01638063705636561,
527
+ "nll_ref": 0.670433307194272,
528
+ "nll_cand": 0.6868139442506376,
529
+ "acc_ref": 0.7209302325581395,
530
+ "acc_cand": 0.7209302325581395,
531
+ "acc_drop_pp": 0.0,
532
+ "max_abs_dp": 0.04628699142360499,
533
+ "mean_max_abs_dp": 0.007026009064181663,
534
+ "gates": {
535
+ "coverage_finite_tokens": {
536
+ "pass": true,
537
+ "problems": {},
538
+ "reference_missing_requests": 0
539
+ },
540
+ "top1": {
541
+ "threshold": 0.98,
542
+ "value": 1.0,
543
+ "pass": true
544
+ },
545
+ "dnll": {
546
+ "threshold": 0.02,
547
+ "value": 0.01638063705636561,
548
+ "pass": true
549
+ },
550
+ "acc_drop_pp": {
551
+ "threshold": 0.3,
552
+ "value": 0.0,
553
+ "resolution_pp": 2.3255813953488373,
554
+ "pass": null,
555
+ "note": "UNRESOLVED: 43 gold questions -> one flip = 2.33pp > 0.3pp; this fixture cannot resolve the gate (use the gate fixture with >= 334 rows; statistical power needs far more)"
556
+ }
557
+ },
558
+ "verdict": "INCONCLUSIVE",
559
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
560
+ "source": {
561
+ "file": "runs/macjev/release_2b/parity/compare_mlx_affine8-g64_base_vs_block.json",
562
+ "sha256": "e1911f21ce73dd42d6c57e8349e2ec3221386a2b44ca461a3e50b9a69057ec34"
563
+ }
564
+ },
565
+ "release_verdict": "PASS"
566
+ }
567
+ },
568
+ "torch_runtime_vs_reference_long_fixture": {
569
+ "n": 43,
570
+ "top1_agree": 1.0,
571
+ "max_abs_dp": 4.782633115096857e-06,
572
+ "mean_max_abs_dp": 6.972375795818997e-07
573
+ }
574
+ },
575
+ "decision_index": "requested from the maintainer after release (not run by us)",
576
+ "runtime_parity": {
577
+ "protocol": "the staged runtime scores the long fixture (35 requests / 43 questions, up to 25,600 tokens) from text/ids through its own renderer; raw scores compared with the HF FP32 CPU reference (runs/macjev/runtime_v2_dev/reference_trained/ref_block_fp32.jsonl)",
578
+ "fixture": {
579
+ "file": "runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl",
580
+ "sha256": "2b16d25eafa70c00a1106ba51b94890821b51ecb4fa393abb12403c8739ef314"
581
+ },
582
+ "cpu_fp32_threads4": {
583
+ "requests": 35,
584
+ "questions": 43,
585
+ "max_abs_dscore": 1.502e-05,
586
+ "top1_flips": 0,
587
+ "meta_ok_all": true,
588
+ "per_category": {
589
+ "catalogue_overflow": {
590
+ "n": 3,
591
+ "max_abs_dscore": 1.502e-05,
592
+ "top1_flips": 0
593
+ },
594
+ "catalogue_overflow_long_state": {
595
+ "n": 1,
596
+ "max_abs_dscore": 6.437e-06,
597
+ "top1_flips": 0
598
+ },
599
+ "long_16k": {
600
+ "n": 3,
601
+ "max_abs_dscore": 6.676e-06,
602
+ "top1_flips": 0
603
+ },
604
+ "long_24k": {
605
+ "n": 2,
606
+ "max_abs_dscore": 4.053e-06,
607
+ "top1_flips": 0
608
+ },
609
+ "many_options": {
610
+ "n": 2,
611
+ "max_abs_dscore": 5.96e-06,
612
+ "top1_flips": 0
613
+ },
614
+ "multi_block_3k": {
615
+ "n": 2,
616
+ "max_abs_dscore": 1.132e-05,
617
+ "top1_flips": 0
618
+ },
619
+ "multi_block_5k": {
620
+ "n": 2,
621
+ "max_abs_dscore": 5.364e-06,
622
+ "top1_flips": 0
623
+ },
624
+ "multi_block_9k": {
625
+ "n": 2,
626
+ "max_abs_dscore": 2.384e-06,
627
+ "top1_flips": 0
628
+ },
629
+ "multi_question": {
630
+ "n": 10,
631
+ "max_abs_dscore": 7.391e-06,
632
+ "top1_flips": 0
633
+ },
634
+ "prefix_boundary": {
635
+ "n": 7,
636
+ "max_abs_dscore": 9.179e-06,
637
+ "top1_flips": 0
638
+ },
639
+ "qtype_choice": {
640
+ "n": 1,
641
+ "max_abs_dscore": 5.841e-06,
642
+ "top1_flips": 0
643
+ },
644
+ "qtype_noul": {
645
+ "n": 1,
646
+ "max_abs_dscore": 1.609e-06,
647
+ "top1_flips": 0
648
+ },
649
+ "qtype_score": {
650
+ "n": 1,
651
+ "max_abs_dscore": 3.934e-06,
652
+ "top1_flips": 0
653
+ },
654
+ "short_sb": {
655
+ "n": 6,
656
+ "max_abs_dscore": 6.02e-06,
657
+ "top1_flips": 0
658
+ }
659
+ },
660
+ "source": {
661
+ "file": "runs/macjev/runtime_v2_dev/verify_release_torch/parity_cpu_fp32_t4.log",
662
+ "sha256": "616618ec2b61c472137ea4cb418dc0b79019b699e7fa6f044bf06c166cb970ba"
663
+ }
664
+ },
665
+ "mps_fp32_long_16k_24k": {
666
+ "requests": 2,
667
+ "questions": 2,
668
+ "max_abs_dscore": 8.821e-06,
669
+ "top1_flips": 0,
670
+ "meta_ok_all": true,
671
+ "per_category": {
672
+ "long_16k": {
673
+ "n": 1,
674
+ "max_abs_dscore": 2.742e-06,
675
+ "top1_flips": 0
676
+ },
677
+ "long_24k": {
678
+ "n": 1,
679
+ "max_abs_dscore": 8.821e-06,
680
+ "top1_flips": 0
681
+ }
682
+ },
683
+ "source": {
684
+ "file": "runs/macjev/runtime_v2_dev/verify_release_torch/parity_mps_fp32_long.log",
685
+ "sha256": "04e40ead1c8694fb3bbda9714074701777ea48d8d2cb3489df738ad902f351d8"
686
+ }
687
+ },
688
+ "render_and_tiny_model": {
689
+ "render": {
690
+ "question_errors": 2,
691
+ "qerr_kinds": [
692
+ "question text ('ins') must be a non-empty string"
693
+ ],
694
+ "n_rendered_equal": 406,
695
+ "catalogue_overflow": 43,
696
+ "both_rejected": 2,
697
+ "edges": {
698
+ "short_2048": 2014,
699
+ "short_2049": [
700
+ 2015,
701
+ 2074
702
+ ]
703
+ },
704
+ "big_k_blocks": 14,
705
+ "big_k_tokens": 23102
706
+ },
707
+ "tiny": {
708
+ "cases": 24,
709
+ "max_abs_d_vs_dev_ref": 2.1457672119140625e-06
710
+ },
711
+ "source": {
712
+ "file": "runs/macjev/runtime_v2_dev/verify_release_torch/v_tiny_and_render.json",
713
+ "sha256": "d3ff75211cfac2f26f0f7763ee937dee8d1e01eceae43e37f16f9ec5d0bb02bf"
714
+ }
715
+ },
716
+ "jev_style_package_e2e": {
717
+ "source": {
718
+ "file": "runs/macjev/runtime_v2_dev/verify_release_torch/v_jevstyle_e2e.json",
719
+ "sha256": "812cbee4d6c8b0af5cd81b6b5aaa8b24f55c07c0ec934cc81c8f61c634e734a6"
720
+ },
721
+ "note": "systemone request through the jev-style package adapter (torch CPU)"
722
+ },
723
+ "runtime_file": {
724
+ "file": "jev_style_decision.py",
725
+ "sha256_now": "5ecba24cdfff4f0e043c2ea48c9c7f514907c6804f265906e21dec0f1b20dbd6",
726
+ "mtime_unix": 1790448336.7685351
727
+ },
728
+ "original_records_predate_last_runtime_edit": true,
729
+ "reverified_on_current_runtime": true,
730
+ "recorded_before_last_runtime_edit": false,
731
+ "note": "the records above were written before the last edit of the runtime file; the runtime as staged now (sha256 runtime_file.sha256_now) was re-verified on the long fixture, see 'reverified' (reverified.runtime_sha256 == runtime_file.sha256_now).",
732
+ "reverified": {
733
+ "runtime_file": "jev_style_decision.py",
734
+ "runtime_sha256": "5ecba24cdfff4f0e043c2ea48c9c7f514907c6804f265906e21dec0f1b20dbd6",
735
+ "shared_core_sha256": "56deae095206fc59b9a30d8626ac5d27cacddcd94d7c7f08f6f4bc53d326a002",
736
+ "what": "PyTorch FP32 CPU, requests with <= 6000 tokens per question",
737
+ "loaded_from": "runs/macjev/hf_staging/Jev-Style-2B-Decision-v3",
738
+ "verify_manifest": true,
739
+ "compared_with": {
740
+ "file": "runs/macjev/runtime_v2_dev/reference_trained/ref_block_fp32.jsonl",
741
+ "sha256": "09ff7d4fdd97c52c28429139a4f8d53dd224a5e0bcc69b1a318978abad5e136c"
742
+ },
743
+ "fixture": {
744
+ "file": "runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl",
745
+ "sha256": "2b16d25eafa70c00a1106ba51b94890821b51ecb4fa393abb12403c8739ef314"
746
+ },
747
+ "questions": 34,
748
+ "skipped_questions": 9,
749
+ "max_abs_score_diff": 1.5020370483398438e-05,
750
+ "top1_same": "34/34",
751
+ "token_or_overflow_mismatches": [],
752
+ "seconds": 411.6,
753
+ "finished_unix": 1790449904.6816142,
754
+ "source": {
755
+ "file": "runs/macjev/hf_staging/_2b_tools/reverify_main.json",
756
+ "sha256": "3b8e573deb626385ce469b289e8e753d60c499ce21efcdeefd20a8e44b4d6c0a"
757
+ }
758
+ },
759
+ "reverified_mps_fp32_long_16k_24k": {
760
+ "what": "the staged jev_style_decision.py on MPS float32, fixture requests long_16k-1 and long_24k-1, raw scores vs the HF FP32 CPU reference (runtime_v2_dev/verify_release_torch/vparity.py, run on this folder)",
761
+ "runtime_sha256": "5ecba24cdfff4f0e043c2ea48c9c7f514907c6804f265906e21dec0f1b20dbd6",
762
+ "runtime_edited_after_this_record": false,
763
+ "requests": 2,
764
+ "questions": 2,
765
+ "max_abs_dscore": 8.821e-06,
766
+ "top1_flips": 0,
767
+ "meta_ok_all": true,
768
+ "per_category": {
769
+ "long_16k": {
770
+ "n": 1,
771
+ "max_abs_dscore": 2.742e-06,
772
+ "top1_flips": 0
773
+ },
774
+ "long_24k": {
775
+ "n": 1,
776
+ "max_abs_dscore": 8.821e-06,
777
+ "top1_flips": 0
778
+ }
779
+ },
780
+ "source": {
781
+ "file": "runs/macjev/hf_staging/_2b_tools/reverify_main_mps_long.log",
782
+ "sha256": "98720ecc5f12d24f7a12add39a7cf332f12c300c8177eed2665a3077d0e0f0b1"
783
+ }
784
+ }
785
+ },
786
+ "latency": {
787
+ "file": "validation/latency_2b.json",
788
+ "sha256": "458d00e0c81b3165072b8547278e6039b9ae0d45d54b7421008ef24c1d24ca3d",
789
+ "machine": {
790
+ "chip": "Apple M1 Max",
791
+ "memory_bytes": 68719476736,
792
+ "macos": "15.7.5",
793
+ "python": "3.12.13"
794
+ },
795
+ "backends_shown_on_card": [
796
+ "gguf-f16",
797
+ "mlx-bf16",
798
+ "torch-mps-fp32"
799
+ ],
800
+ "rows": 18,
801
+ "rows_in_file": 36,
802
+ "card_generator": "runs/macjev/release_2b/cards/_build/latency_tables.py"
803
+ }
804
+ }
requirements.txt ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ # tested with: python 3.12.13, torch 2.14.0, transformers 5.17.0, tokenizers 0.23.2, numpy 2.5.3
2
+ torch==2.14.0
3
+ transformers==5.17.0 # needs Qwen3.5 support (transformers.models.qwen3_5) and multi-token cache continuation in Gated DeltaNet
4
+ tokenizers==0.23.2
5
+ numpy==2.5.3
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523
3
+ size 19989325
tokenizer_config.json ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|im_end|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": true,
13
+ "local_files_only": false,
14
+ "model_max_length": 262144,
15
+ "model_specific_special_tokens": {
16
+ "audio_bos_token": "<|audio_start|>",
17
+ "audio_eos_token": "<|audio_end|>",
18
+ "audio_token": "<|audio_pad|>",
19
+ "image_token": "<|image_pad|>",
20
+ "video_token": "<|video_pad|>",
21
+ "vision_bos_token": "<|vision_start|>",
22
+ "vision_eos_token": "<|vision_end|>"
23
+ },
24
+ "pad_token": "<|endoftext|>",
25
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
26
+ "split_special_tokens": false,
27
+ "tokenizer_class": "Qwen2Tokenizer",
28
+ "unk_token": null,
29
+ "video_token": "<|video_pad|>",
30
+ "vision_bos_token": "<|vision_start|>",
31
+ "vision_eos_token": "<|vision_end|>"
32
+ }
validation/SOURCES.json ADDED
@@ -0,0 +1,81 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "note": "copies of the verification / benchmark records behind release_config.json and the model card; local absolute paths replaced by repository-relative ones (local_paths_scrubbed); source_sha256 = the unmodified file (equal to the published copy when local_paths_scrubbed is false)",
3
+ "files": {
4
+ "parity/cross_format_dp.json": {
5
+ "source": "runs/macjev/release_2b/parity/cross_format_dp.json",
6
+ "source_sha256": "c7332df0f02244416b01e49ecb037c76d4bf2643a2b42505ff1dfe896879e533",
7
+ "local_paths_scrubbed": false
8
+ },
9
+ "parity/PREDECLARED_RELEASE_GATES_2B.md": {
10
+ "source": "runs/macjev/runtime_v2_dev/PREDECLARED_RELEASE_GATES_2B.md",
11
+ "source_sha256": "783aec77300779582911cdc57c71cf8b101b0cd5ca040073f7e1fa3ac20af08e",
12
+ "local_paths_scrubbed": false
13
+ },
14
+ "runtime/parity_cpu_fp32_t4.log": {
15
+ "source": "runs/macjev/runtime_v2_dev/verify_release_torch/parity_cpu_fp32_t4.log",
16
+ "source_sha256": "616618ec2b61c472137ea4cb418dc0b79019b699e7fa6f044bf06c166cb970ba",
17
+ "local_paths_scrubbed": false
18
+ },
19
+ "runtime/parity_mps_fp32_long.log": {
20
+ "source": "runs/macjev/runtime_v2_dev/verify_release_torch/parity_mps_fp32_long.log",
21
+ "source_sha256": "04e40ead1c8694fb3bbda9714074701777ea48d8d2cb3489df738ad902f351d8",
22
+ "local_paths_scrubbed": false
23
+ },
24
+ "runtime/v_tiny_and_render.json": {
25
+ "source": "runs/macjev/runtime_v2_dev/verify_release_torch/v_tiny_and_render.json",
26
+ "source_sha256": "d3ff75211cfac2f26f0f7763ee937dee8d1e01eceae43e37f16f9ec5d0bb02bf",
27
+ "local_paths_scrubbed": false
28
+ },
29
+ "runtime/v_jevstyle_e2e.json": {
30
+ "source": "runs/macjev/runtime_v2_dev/verify_release_torch/v_jevstyle_e2e.json",
31
+ "source_sha256": "812cbee4d6c8b0af5cd81b6b5aaa8b24f55c07c0ec934cc81c8f61c634e734a6",
32
+ "local_paths_scrubbed": false
33
+ },
34
+ "runtime/reverify_main.json": {
35
+ "source": "runs/macjev/hf_staging/_2b_tools/reverify_main.json",
36
+ "source_sha256": "3b8e573deb626385ce469b289e8e753d60c499ce21efcdeefd20a8e44b4d6c0a",
37
+ "local_paths_scrubbed": false
38
+ },
39
+ "runtime/reverify_main_mps_long.log": {
40
+ "source": "runs/macjev/hf_staging/_2b_tools/reverify_main_mps_long.log",
41
+ "source_sha256": "98720ecc5f12d24f7a12add39a7cf332f12c300c8177eed2665a3077d0e0f0b1",
42
+ "local_paths_scrubbed": false
43
+ },
44
+ "benchmarks/jevbench_v1.4.1_results.json": {
45
+ "source": "runs/macjev/bench_2b/jevbench/gguf_f16/results.json",
46
+ "source_sha256": "9a467fb4ddcc9a42f9d82ff8f8cf1ce5cfafd51d27b8d4dca2b9c1a681f67a9c",
47
+ "local_paths_scrubbed": false
48
+ },
49
+ "benchmarks/jevbench_v1.4.1_results.md": {
50
+ "source": "runs/macjev/bench_2b/jevbench/gguf_f16/results.md",
51
+ "source_sha256": "824ffa6ae920595e9277c9d6c41770bf6b2d9fc2e71ff7de9afe6ee635b69ae8",
52
+ "local_paths_scrubbed": false
53
+ },
54
+ "benchmarks/zeroshot_metrics.json": {
55
+ "source": "runs/macjev/bench_2b/zeroshot/gguf_f16/metrics.json",
56
+ "source_sha256": "f80c80ae55678cce974927df0c6a33a5bf5e2ba85e9b14a181e6e5d16bcee3c5",
57
+ "local_paths_scrubbed": true,
58
+ "published_sha256": "479ddbaa101f843487471500061906e78ce28f9af09c65eaadc3cee4b5ebdb4a"
59
+ },
60
+ "benchmarks/zeroshot_comparison.md": {
61
+ "source": "runs/macjev/bench_2b/zeroshot/gguf_f16/comparison.md",
62
+ "source_sha256": "34828fa9375ea717c866b1e749d3fe902f0b52303c807289a769934760f6dad0",
63
+ "local_paths_scrubbed": false
64
+ },
65
+ "benchmarks/contamination.json": {
66
+ "source": "runs/macjev/release_2b/cards/_build/contamination_public.json",
67
+ "source_sha256": "87f7f645b950b4cd626ff248b1899578c078db74d4c74bc944b017badc06a8e8",
68
+ "local_paths_scrubbed": false
69
+ },
70
+ "latency_2b.json": {
71
+ "source": "runs/macjev/release_2b/latency_2b.json",
72
+ "source_sha256": "458d00e0c81b3165072b8547278e6039b9ae0d45d54b7421008ef24c1d24ca3d",
73
+ "local_paths_scrubbed": false
74
+ },
75
+ "data_sources.json": {
76
+ "source": "runs/macjev/release_2b/cards/_build/data_sources.json",
77
+ "source_sha256": "415fa9dcb893dc71d97d76652fbf03600a72681c15f7248ef46035d85ea56d7c",
78
+ "local_paths_scrubbed": false
79
+ }
80
+ }
81
+ }
validation/benchmarks/contamination.json ADDED
@@ -0,0 +1,110 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "note": "Public copy of the contamination scan of the 2B training pool. Matched pool rows are given by their dataset id (pool_row_sources); per-source pool row counts and internal records are not included. zeroshot.sets lists the three zero-shot sets this model was evaluated on (tweet_topic, fin_topic, daily_dialog).",
3
+ "pool": "reduced-v1/main",
4
+ "files": [
5
+ "train.jsonl",
6
+ "format_train.jsonl",
7
+ "dev.jsonl",
8
+ "cal.jsonl"
9
+ ],
10
+ "rows_scanned": {
11
+ "train.jsonl": 463208,
12
+ "format_train.jsonl": 500,
13
+ "dev.jsonl": 27109,
14
+ "cal.jsonl": 3325
15
+ },
16
+ "decode": {
17
+ "prefix_exact": 494142,
18
+ "head_parsed": 493642
19
+ },
20
+ "tokenizer": "Qwen/Qwen3.5-2B@15852e8c16360a2fea060d615a32b45270f8a8fc",
21
+ "seconds": 318.62241888046265,
22
+ "jevbench": {
23
+ "n_items": 231,
24
+ "hits_state": [],
25
+ "hits_ins": [],
26
+ "hits_leaf": [],
27
+ "hits_ngram_state": {},
28
+ "hits_ngram_ins_extension": {},
29
+ "n_items_with_ngram_hits_state": 0,
30
+ "n_items_with_ngram_hits_ins": 0,
31
+ "method": {
32
+ "state": "content_hash of the whole serialized state",
33
+ "ins": "content_hash of the instruction",
34
+ "leaf": "content_hash of every state string leaf >= 32 chars",
35
+ "ngram": "13-word shingles of the normalized state (pool stride 4); extension: same on the pool instruction"
36
+ }
37
+ },
38
+ "zeroshot": {
39
+ "shingle": 8,
40
+ "near_dup_fraction": 0.5,
41
+ "sets": {
42
+ "tweet_topic": {
43
+ "n": 1693,
44
+ "exact_overlap": 0,
45
+ "exact_overlap_generic_short": 0,
46
+ "near_duplicate_only": 0,
47
+ "exact_pool_rows_by_file": {},
48
+ "near_pool_rows_by_file": {},
49
+ "examples": [],
50
+ "generic_examples": [],
51
+ "near_examples": []
52
+ },
53
+ "fin_topic": {
54
+ "n": 4117,
55
+ "exact_overlap": 0,
56
+ "exact_overlap_generic_short": 0,
57
+ "near_duplicate_only": 0,
58
+ "exact_pool_rows_by_file": {},
59
+ "near_pool_rows_by_file": {},
60
+ "examples": [],
61
+ "generic_examples": [],
62
+ "near_examples": []
63
+ },
64
+ "daily_dialog": {
65
+ "n": 7740,
66
+ "exact_overlap": 0,
67
+ "exact_overlap_generic_short": 1,
68
+ "near_duplicate_only": 3,
69
+ "exact_pool_rows_by_file": {},
70
+ "near_pool_rows_by_file": {
71
+ "train.jsonl": 5
72
+ },
73
+ "examples": [],
74
+ "generic_examples": [
75
+ {
76
+ "dataset_index": 3850,
77
+ "text": " Thanks a lot ",
78
+ "pool_row_sources": [
79
+ "train.jsonl:google-research-datasets/schema_guided_dstc8",
80
+ "train.jsonl:clinc150"
81
+ ]
82
+ }
83
+ ],
84
+ "near_examples": [
85
+ {
86
+ "dataset_index": 3522,
87
+ "text": " What do you like to do in your spare time ? ",
88
+ "pool_row_sources": [
89
+ "train.jsonl:clinc150"
90
+ ]
91
+ },
92
+ {
93
+ "dataset_index": 1034,
94
+ "text": " Is there anything else I can do for you ? ",
95
+ "pool_row_sources": [
96
+ "train.jsonl:google-research-datasets/schema_guided_dstc8"
97
+ ]
98
+ },
99
+ {
100
+ "dataset_index": 1604,
101
+ "text": " What was the score at the end of the game ? ",
102
+ "pool_row_sources": [
103
+ "train.jsonl:ucinlp/drop"
104
+ ]
105
+ }
106
+ ]
107
+ }
108
+ }
109
+ }
110
+ }
validation/benchmarks/jevbench_v1.4.1_results.json ADDED
@@ -0,0 +1,511 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "benchmark": "jevbench-v1.4.1",
3
+ "repo": "https://github.com/fstandhartinger/jevbench",
4
+ "commit": "24b9b5c1609a7a9e8fa14f49e5985a836c9dc842",
5
+ "adapter": "macjev-ext-jevbench-v2 (metrics: macjev-ext-jevbench-v1)",
6
+ "vendor_verify": {
7
+ "ok": true,
8
+ "commit": "24b9b5c1609a7a9e8fa14f49e5985a836c9dc842",
9
+ "files": 13,
10
+ "source_json_sha256": "75b86955079a063422c79f5c31e8c74203609b9a492227855e2a9399c937b693"
11
+ },
12
+ "scorer": {
13
+ "scorer": "macjev-v2-gguf",
14
+ "readout": "verdict",
15
+ "kind": "gguf",
16
+ "engine_class": "macjev.export.gguf_v2.GGUFScorerV2",
17
+ "arch": "qwen3_5",
18
+ "template": "macjev-render-v2-long-options",
19
+ "layout": "sb",
20
+ "block": 2048,
21
+ "max_len": 25600,
22
+ "head_max": null,
23
+ "dtype": "gguf",
24
+ "base": "runs/macjev/release_2b/gguf/model-f16.gguf",
25
+ "sha256": "c1948e55d3a498292b818451b7726aca540c96f70c95352554f27a965ef3304b"
26
+ },
27
+ "model_name": "Jev-Style-2B-Decision-v3-GGUF-F16",
28
+ "n_items": 231,
29
+ "seconds": 262.2442800998688,
30
+ "timing": {
31
+ "n_items": 231,
32
+ "tokens": 144044,
33
+ "seconds": 261.91108745516976,
34
+ "tokens_per_second": 549.9728988168758,
35
+ "seconds_per_item": 1.1338142314076614
36
+ },
37
+ "files_sha256": {
38
+ "predictions.jsonl": "3168be473c97b93d75fa691686623ada8a2f9540fd1aceffeaf9e206608ef8ee",
39
+ "answers.jsonl": "e3bbb96c40a48fdd28e96aa9e8c34ce1c3fc2d2715f0a777ca35849e38fe5638",
40
+ "harness_records.jsonl": "b44d4b5389082af91330dc9d328f86ba5e3e599b3c69036c0ec6f1d98b946004",
41
+ "timings.jsonl": "73795be0585930e2a1b6b24d91b8349b434e58ef2a82144f679797e9e4f3775e"
42
+ },
43
+ "protocol": {
44
+ "template": "macjev-render-v2-long-options",
45
+ "layout": "sb",
46
+ "block": 2048,
47
+ "total_budget_tokens": 25600,
48
+ "truncation": "never",
49
+ "question_options_cap": null,
50
+ "over_budget_rule": "complete rendering > 25,600 tokens -> status unsupported (InputBudgetError), counted per benchmark and scored as wrong",
51
+ "catalogue_overflow_rule": "question+options over one 2,048-token block -> numbered catalogue in prefix blocks + numbered-option rubric (RendererV2); counted per benchmark",
52
+ "readout": "FP32 h_slot . (W_yes - W_no) at each option slot; p = softmax(scores / T_global)"
53
+ },
54
+ "catalogue_overflow": {},
55
+ "utc": "2026-09-26T12:06:27Z",
56
+ "public_accuracy": 0.7359307359307359,
57
+ "public_accuracy_counts": {
58
+ "correct": 170,
59
+ "scorable": 231
60
+ },
61
+ "public_accuracy_independent_check": 0.7359307359307359,
62
+ "tiers": {
63
+ "easy": {
64
+ "n": 48,
65
+ "n_scorable": 48,
66
+ "n_correct": 48,
67
+ "accuracy": 1.0,
68
+ "schema_validity": 1.0,
69
+ "schema_validity_strict": 1.0,
70
+ "operational_success": 1.0,
71
+ "brier_mean": 0.0016946904719700839,
72
+ "ece_top_label_10bin": 0.01884318271512942,
73
+ "calibration_n": 48,
74
+ "ordinal_mae": null,
75
+ "per_family": {
76
+ "extraction": {
77
+ "n": 12,
78
+ "accuracy": 1.0
79
+ },
80
+ "fact": {
81
+ "n": 12,
82
+ "accuracy": 1.0
83
+ },
84
+ "intent": {
85
+ "n": 12,
86
+ "accuracy": 1.0
87
+ },
88
+ "tool_selection": {
89
+ "n": 12,
90
+ "accuracy": 1.0
91
+ }
92
+ }
93
+ },
94
+ "standard": {
95
+ "n": 72,
96
+ "n_scorable": 72,
97
+ "n_correct": 69,
98
+ "accuracy": 0.9583333333333334,
99
+ "schema_validity": 1.0,
100
+ "schema_validity_strict": 1.0,
101
+ "operational_success": 1.0,
102
+ "brier_mean": 0.07269926087469171,
103
+ "ece_top_label_10bin": 0.09887602156622965,
104
+ "calibration_n": 72,
105
+ "ordinal_mae": 0.10494098738063083,
106
+ "per_family": {
107
+ "adequacy": {
108
+ "n": 12,
109
+ "accuracy": 0.8333333333333334
110
+ },
111
+ "extraction": {
112
+ "n": 12,
113
+ "accuracy": 1.0
114
+ },
115
+ "intent": {
116
+ "n": 12,
117
+ "accuracy": 1.0
118
+ },
119
+ "ordinal": {
120
+ "n": 12,
121
+ "accuracy": 1.0
122
+ },
123
+ "policy": {
124
+ "n": 12,
125
+ "accuracy": 1.0
126
+ },
127
+ "routing": {
128
+ "n": 12,
129
+ "accuracy": 0.9166666666666666
130
+ }
131
+ }
132
+ },
133
+ "hard": {
134
+ "n": 111,
135
+ "n_scorable": 111,
136
+ "n_correct": 53,
137
+ "accuracy": 0.4774774774774775,
138
+ "schema_validity": 1.0,
139
+ "schema_validity_strict": 1.0,
140
+ "operational_success": 1.0,
141
+ "brier_mean": 0.6071787802170037,
142
+ "ece_top_label_10bin": 0.15324274416807362,
143
+ "calibration_n": 111,
144
+ "ordinal_mae": 0.6837138642214399,
145
+ "per_family": {
146
+ "adversarial": {
147
+ "n": 6,
148
+ "accuracy": 0.6666666666666666
149
+ },
150
+ "ambiguous": {
151
+ "n": 7,
152
+ "accuracy": 0.42857142857142855
153
+ },
154
+ "judge_hard": {
155
+ "n": 17,
156
+ "accuracy": 0.47058823529411764
157
+ },
158
+ "long_policy": {
159
+ "n": 19,
160
+ "accuracy": 0.3157894736842105
161
+ },
162
+ "multi_hop": {
163
+ "n": 18,
164
+ "accuracy": 0.7222222222222222
165
+ },
166
+ "probability": {
167
+ "n": 10,
168
+ "accuracy": 0.4
169
+ },
170
+ "routing_hard": {
171
+ "n": 5,
172
+ "accuracy": 1.0
173
+ },
174
+ "temporal_numeric": {
175
+ "n": 15,
176
+ "accuracy": 0.13333333333333333
177
+ },
178
+ "tradeoff": {
179
+ "n": 6,
180
+ "accuracy": 0.16666666666666666
181
+ },
182
+ "trap": {
183
+ "n": 8,
184
+ "accuracy": 0.875
185
+ }
186
+ }
187
+ }
188
+ },
189
+ "hard_tier_ece_public111": 0.15324274416807362,
190
+ "hard_tier_ece_bins": [
191
+ {
192
+ "lo": 0.0,
193
+ "hi": 0.1,
194
+ "n": 0,
195
+ "mean_confidence": null,
196
+ "accuracy": null
197
+ },
198
+ {
199
+ "lo": 0.1,
200
+ "hi": 0.2,
201
+ "n": 0,
202
+ "mean_confidence": null,
203
+ "accuracy": null
204
+ },
205
+ {
206
+ "lo": 0.2,
207
+ "hi": 0.3,
208
+ "n": 1,
209
+ "mean_confidence": 0.21388788145283613,
210
+ "accuracy": 0.0
211
+ },
212
+ {
213
+ "lo": 0.3,
214
+ "hi": 0.4,
215
+ "n": 13,
216
+ "mean_confidence": 0.3512620124154394,
217
+ "accuracy": 0.23076923076923078
218
+ },
219
+ {
220
+ "lo": 0.4,
221
+ "hi": 0.5,
222
+ "n": 19,
223
+ "mean_confidence": 0.4498371724402195,
224
+ "accuracy": 0.42105263157894735
225
+ },
226
+ {
227
+ "lo": 0.5,
228
+ "hi": 0.6,
229
+ "n": 18,
230
+ "mean_confidence": 0.5561807791811684,
231
+ "accuracy": 0.3888888888888889
232
+ },
233
+ {
234
+ "lo": 0.6,
235
+ "hi": 0.7,
236
+ "n": 17,
237
+ "mean_confidence": 0.6470557809575238,
238
+ "accuracy": 0.35294117647058826
239
+ },
240
+ {
241
+ "lo": 0.7,
242
+ "hi": 0.8,
243
+ "n": 18,
244
+ "mean_confidence": 0.7435771863244336,
245
+ "accuracy": 0.5
246
+ },
247
+ {
248
+ "lo": 0.8,
249
+ "hi": 0.9,
250
+ "n": 13,
251
+ "mean_confidence": 0.8404461860278016,
252
+ "accuracy": 0.6923076923076923
253
+ },
254
+ {
255
+ "lo": 0.9,
256
+ "hi": 1.0,
257
+ "n": 12,
258
+ "mean_confidence": 0.9467793508081911,
259
+ "accuracy": 0.9166666666666666
260
+ }
261
+ ],
262
+ "ordinal_mae_all_score_items": 0.29786527966090054,
263
+ "n_score_items": 18,
264
+ "brier_mean_all": 0.3147728854100423,
265
+ "probability_items_mean_tvd": 0.33663270227830244,
266
+ "probability_items_n": 10,
267
+ "calibration_axis_proxy_public_hard": 67.84409046927752,
268
+ "unsupported": {
269
+ "total": 0,
270
+ "by_tier": {},
271
+ "by_reason": {},
272
+ "rule": "over the 25,600-token TOTAL budget: counted as incorrect; never truncated"
273
+ },
274
+ "accuracy_on_supported_only": 0.7359307359307359,
275
+ "schema_validity": 1.0,
276
+ "schema_validity_strict": 1.0,
277
+ "paraphrase_consistency": {
278
+ "pairs": 36,
279
+ "both_valid": 36,
280
+ "agree": 35,
281
+ "agreement": 0.9722222222222222,
282
+ "both_correct_rate_all_pairs": 0.9444444444444444
283
+ },
284
+ "temperature": {
285
+ "mode": "global",
286
+ "file": "explicit --temperature",
287
+ "sha256": null,
288
+ "value": 0.8278650620942867,
289
+ "groups_used": {
290
+ "global": {
291
+ "T": 0.8278650620942867,
292
+ "n_items": 231
293
+ }
294
+ },
295
+ "fitted_on_benchmark_items": false
296
+ },
297
+ "harness_summary_all": {
298
+ "n_planned": 231,
299
+ "n_attempted": 231,
300
+ "n_scorable": 231,
301
+ "n_valid": 231,
302
+ "n_correct": 170,
303
+ "accuracy": 0.7359307359307359,
304
+ "coverage": 1.0,
305
+ "macro_accuracy": 0.7182687792465914
306
+ },
307
+ "comparison": {
308
+ "source_file": "results/v1.4.1/jevbench-v1.4.1-results.json",
309
+ "source_sha256": "e6754863056503fe2b010410fc7111df884ac1f9ce4449aa369aab61d98092cd",
310
+ "revision": "v1.4.1",
311
+ "n_systems": 82,
312
+ "commit": "24b9b5c1609a7a9e8fa14f49e5985a836c9dc842",
313
+ "rows": [
314
+ {
315
+ "key": "jev-1.13.0",
316
+ "label": "Jev 1.13.0",
317
+ "display": "Jev 1.13.0 (TypeSafe AI)",
318
+ "repo": "https://docs.typesafe.ai",
319
+ "underlying": "closed",
320
+ "public_accuracy": 0.8658008658008658,
321
+ "public_correct_of_231": 200,
322
+ "intelligence": 53.05904597275748,
323
+ "calibration": 76.3389831504074,
324
+ "sealed_accuracy": 0.36688311688311687,
325
+ "jevbench_score": 63.29205745601932,
326
+ "rank": 1,
327
+ "ece_hard_220": 0.06061111111111118,
328
+ "tiers_v12_full": {
329
+ "easy": 1.0,
330
+ "standard": 0.9895833333333334,
331
+ "judge": 0.9452054794520548,
332
+ "hard": 0.740909090909091
333
+ },
334
+ "source": "results/v1.4.1/jevbench-v1.4.1-results.json systems[0] (key 'jev-1.13.0') @ 24b9b5c1609a"
335
+ },
336
+ {
337
+ "key": "laya",
338
+ "label": "Laya",
339
+ "display": "Laya (Convai Innovations, ModernBERT-large 421M)",
340
+ "repo": "https://huggingface.co/convaiinnovations/laya",
341
+ "underlying": "ModernBERT-large encoder + option-marker decision head, 421M, RLCD-trained",
342
+ "public_accuracy": 0.5844155844155844,
343
+ "public_correct_of_231": 135,
344
+ "intelligence": 36.132988467190394,
345
+ "calibration": 63.67551046176046,
346
+ "sealed_accuracy": 0.30844155844155846,
347
+ "jevbench_score": 30.251281619829957,
348
+ "rank": 36,
349
+ "ece_hard_220": 0.20550909090909092,
350
+ "tiers_v12_full": {
351
+ "easy": 0.9444444444444444,
352
+ "standard": 0.7291666666666666,
353
+ "judge": 0.6917808219178082,
354
+ "hard": 0.3409090909090909
355
+ },
356
+ "source": "results/v1.4.1/jevbench-v1.4.1-results.json systems[35] (key 'laya') @ 24b9b5c1609a"
357
+ },
358
+ {
359
+ "key": "mghafiri-qwen3.5-0.8b-decision-model",
360
+ "label": "same backbone: Qwen3.5-0.8B Decision Model (mghafiri)",
361
+ "display": "Qwen3.5-0.8B Decision Model (Mourad Ghafiri)",
362
+ "repo": "https://huggingface.co/mghafiri/qwen3.5-0.8B-decision-model",
363
+ "underlying": "Qwen3.5-0.8B-Base fine-tuned as a JevLite decision model with per-question calibration",
364
+ "public_accuracy": 0.5930735930735931,
365
+ "public_correct_of_231": 137,
366
+ "intelligence": 28.073004486611502,
367
+ "calibration": 68.22744227994228,
368
+ "sealed_accuracy": 0.3474025974025974,
369
+ "jevbench_score": 14.540627482985634,
370
+ "rank": 60,
371
+ "ece_hard_220": null,
372
+ "tiers_v12_full": {
373
+ "easy": 0.9861111111111112,
374
+ "standard": 0.5520833333333334,
375
+ "judge": 0.3424657534246575,
376
+ "hard": 0.509090909090909
377
+ },
378
+ "source": "results/v1.4.1/jevbench-v1.4.1-results.json systems[59] (key 'mghafiri-qwen3.5-0.8b-decision-model') @ 24b9b5c1609a"
379
+ },
380
+ {
381
+ "key": "simplejev-qwen3.5-0.8b",
382
+ "label": "same backbone, untrained: SimpleJev Qwen3.5-0.8B",
383
+ "display": "SimpleJev (Qwen3.5-0.8B, CPU)",
384
+ "repo": "https://github.com/featherless-ai/simple-jev",
385
+ "underlying": "Qwen3.5-0.8B through SimpleJev native assistant-prefill option-logit scorer",
386
+ "public_accuracy": 0.5454545454545454,
387
+ "public_correct_of_231": 126,
388
+ "intelligence": 21.482479668016303,
389
+ "calibration": 49.07354475109724,
390
+ "sealed_accuracy": 0.3474025974025974,
391
+ "jevbench_score": 7.459762136724042,
392
+ "rank": 68,
393
+ "ece_hard_220": 0.3060342388575625,
394
+ "tiers_v12_full": {
395
+ "easy": 0.875,
396
+ "standard": 0.5729166666666666,
397
+ "judge": 0.1917808219178082,
398
+ "hard": 0.4
399
+ },
400
+ "source": "results/v1.4.1/jevbench-v1.4.1-results.json systems[67] (key 'simplejev-qwen3.5-0.8b') @ 24b9b5c1609a"
401
+ },
402
+ {
403
+ "key": "decider-2b",
404
+ "label": "decider-2b",
405
+ "display": "decider-2b (Mapika)",
406
+ "repo": "https://huggingface.co/Mapika/decider-2b",
407
+ "underlying": "Qwen3.5-2B-Base with a trained decision readout, 1.9B",
408
+ "public_accuracy": 0.70995670995671,
409
+ "public_correct_of_231": 164,
410
+ "intelligence": 38.54932735543702,
411
+ "calibration": 43.48691774891777,
412
+ "sealed_accuracy": 0.24675324675324675,
413
+ "jevbench_score": 30.7375448157181,
414
+ "rank": 34,
415
+ "ece_hard_220": 0.3221518181818179,
416
+ "tiers_v12_full": {
417
+ "easy": 1.0,
418
+ "standard": 0.8541666666666666,
419
+ "judge": 0.773972602739726,
420
+ "hard": 0.4727272727272727
421
+ },
422
+ "source": "results/v1.4.1/jevbench-v1.4.1-results.json systems[33] (key 'decider-2b') @ 24b9b5c1609a"
423
+ },
424
+ {
425
+ "key": "kev-0.6b",
426
+ "label": "kev 0.6B",
427
+ "display": "kev 0.6B (research preview)",
428
+ "repo": "https://github.com/jaredpalmer/kev",
429
+ "underlying": "Qwen3-0.6B-Base + LoRA + learned pointer head; jaredpalmer/kev-0.6b",
430
+ "public_accuracy": 0.6666666666666666,
431
+ "public_correct_of_231": 154,
432
+ "intelligence": 34.20432736698408,
433
+ "calibration": 49.957886527801605,
434
+ "sealed_accuracy": 0.24025974025974026,
435
+ "jevbench_score": 24.750427849352384,
436
+ "rank": 45,
437
+ "ece_hard_220": 0.26937623307785324,
438
+ "tiers_v12_full": {
439
+ "easy": 1.0,
440
+ "standard": 0.8125,
441
+ "judge": 0.6643835616438356,
442
+ "hard": 0.4
443
+ },
444
+ "source": "results/v1.4.1/jevbench-v1.4.1-results.json systems[44] (key 'kev-0.6b') @ 24b9b5c1609a"
445
+ },
446
+ {
447
+ "key": "lev-350m",
448
+ "label": "lev-350m",
449
+ "display": "lev-350m (Franck Verrot, LFM2.5-350M)",
450
+ "repo": "https://github.com/franckverrot/lev",
451
+ "underlying": "LiquidAI/LFM2.5-350M with a LoRA and a 6.5M-parameter pointer head (a kev clone)",
452
+ "public_accuracy": 0.5844155844155844,
453
+ "public_correct_of_231": 135,
454
+ "intelligence": 34.75234376694234,
455
+ "calibration": 70.61387806637806,
456
+ "sealed_accuracy": 0.25,
457
+ "jevbench_score": 28.501129766483672,
458
+ "rank": 37,
459
+ "ece_hard_220": 0.12270454545454548,
460
+ "tiers_v12_full": {
461
+ "easy": 0.9861111111111112,
462
+ "standard": 0.71875,
463
+ "judge": 0.6917808219178082,
464
+ "hard": 0.36818181818181817
465
+ },
466
+ "source": "results/v1.4.1/jevbench-v1.4.1-results.json systems[36] (key 'lev-350m') @ 24b9b5c1609a"
467
+ },
468
+ {
469
+ "key": "decision-fast",
470
+ "label": "Decision Fast (Qwen3-0.6B)",
471
+ "display": "Decision Fast (FlyMy.AI, v53a)",
472
+ "repo": "https://huggingface.co/flymy-ai/decision-fast-preview",
473
+ "underlying": "Qwen/Qwen3-0.6B-Base with a trained LoRA adapter and pointer head (10.6M trainable parameters)",
474
+ "public_accuracy": 0.6320346320346321,
475
+ "public_correct_of_231": 146,
476
+ "intelligence": 37.074471307080934,
477
+ "calibration": 65.30757379702415,
478
+ "sealed_accuracy": 0.2564935064935065,
479
+ "jevbench_score": 32.492203456270644,
480
+ "rank": 33,
481
+ "ece_hard_220": 0.17070899271518503,
482
+ "tiers_v12_full": {
483
+ "easy": 0.9861111111111112,
484
+ "standard": 0.78125,
485
+ "judge": 0.7465753424657534,
486
+ "hard": 0.38636363636363635
487
+ },
488
+ "source": "results/v1.4.1/jevbench-v1.4.1-results.json systems[32] (key 'decision-fast') @ 24b9b5c1609a"
489
+ }
490
+ ],
491
+ "notes": [
492
+ "public_accuracy = accuracy on the 231 public items (docs/METHOD-v1.4.md, variable p).",
493
+ "intelligence / calibration are the official v1.4 axes (0-100): they blend held-out, judge and 308 sealed items that are not public, so a public-only run cannot reproduce them.",
494
+ "tiers_v12_full are accuracies on the full v1.2 tiers (public + held-out), not the public subsets; ece_hard_220 is over all 220 hard items."
495
+ ]
496
+ },
497
+ "ours_0.8b_v3": {
498
+ "label": "Jev-Style-0.8B-Decision-v3 (ours, v1 protocol)",
499
+ "public_accuracy": 0.6406926406926406,
500
+ "correct": 148,
501
+ "tiers": {
502
+ "easy": 1.0,
503
+ "standard": 0.8194444444444444,
504
+ "hard": 0.36936936936936937
505
+ },
506
+ "hard_tier_ece_public111": 0.19980380205815876,
507
+ "ordinal_mae": 0.4283973452495316,
508
+ "source": "runs/macjev/received/ext_evals/main/jevbench/results.json",
509
+ "source_sha256": "6c1f94220dc8197e504827213b3567e00919740bb705929bed88535812f2a033"
510
+ }
511
+ }
validation/benchmarks/jevbench_v1.4.1_results.md ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ## JevBench v1.4.1 -- public items (231)
2
+
3
+ | System | Public acc. (231) | Correct | Easy (48) | Standard (72) | Hard (111) | Hard ECE |
4
+ |---|---:|---:|---:|---:|---:|---:|
5
+ | **Jev-Style-2B-Decision-v3-GGUF-F16** (this run, v2 protocol) | **73.6%** | 170 | 100.0% | 95.8% | 47.7% | 0.153 (public 111) |
6
+ | Jev-Style-0.8B-Decision-v3 (ours, v1 protocol) | 64.1% | 148 | 100.0% | 81.9% | 36.9% | 0.200 (public 111) |
7
+ | Jev 1.13.0 | 86.6% | 200 | -- | -- | -- | 0.061 (all 220) |
8
+ | Laya | 58.4% | 135 | -- | -- | -- | 0.206 (all 220) |
9
+ | same backbone: Qwen3.5-0.8B Decision Model (mghafiri) | 59.3% | 137 | -- | -- | -- | n/a (all 220) |
10
+ | same backbone, untrained: SimpleJev Qwen3.5-0.8B | 54.5% | 126 | -- | -- | -- | 0.306 (all 220) |
11
+ | decider-2b | 71.0% | 164 | -- | -- | -- | 0.322 (all 220) |
12
+ | kev 0.6B | 66.7% | 154 | -- | -- | -- | 0.269 (all 220) |
13
+ | lev-350m | 58.4% | 135 | -- | -- | -- | 0.123 (all 220) |
14
+ | Decision Fast (Qwen3-0.6B) | 63.2% | 146 | -- | -- | -- | 0.171 (all 220) |
15
+
16
+ Published rows: `results/v1.4.1/jevbench-v1.4.1-results.json` sha256 `e6754863056503fe2b010410fc7111df884ac1f9ce4449aa369aab61d98092cd` (https://github.com/fstandhartinger/jevbench @ 24b9b5c1609a, tag v1.4.1).
17
+ 0.8B v3 row: `runs/macjev/received/ext_evals/main/jevbench/results.json` sha256 `6c1f94220dc8197e504827213b3567e00919740bb705929bed88535812f2a033`.
18
+ Unsupported (over the 25,600-token total budget; counted wrong, never truncated): 0 {}. Catalogue-overflow renderings: {}. Temperature: one global T = 0.8278650620942867 from `explicit --temperature`; nothing fitted on JevBench items.
validation/benchmarks/zeroshot_comparison.md ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ **Zero-shot sets outside our training pool**
2
+
3
+ | set | n | majority | Jev-Style-2B-Decision-v3-GGUF-F16 acc | F1 | ECE | Jev acc | F1 | ECE | Laya(en) acc | F1 | ECE |
4
+ |---|---|---|---|---|---|---|---|---|---|---|---|
5
+ | tweet_topic | 1693 | 0.396 | 0.822 | 0.678 | 0.028 | 0.793 | 0.694 | 0.063 | 0.632 | 0.461 | 0.130 |
6
+ | fin_topic | 4117 | 0.207 | 0.611 | 0.590 | 0.065 | 0.670 | 0.630 | 0.166 | 0.342 | 0.362 | 0.610 |
7
+ | daily_dialog | 7740 | 0.817 | 0.774 | 0.372 | 0.039 | 0.710 | 0.385 | 0.156 | 0.614 | 0.275 | 0.208 |
8
+
9
+ Jev / Laya: published by elcronos (https://github.com/elcronos/jev-vs-open-decision-models/blob/a1901bc3d520e73936de8d4326545c0cdcf742fb/results/cross_dataset_summary.json); Jev's ECE is raw (API probabilities, no temperature), Laya's uses its own shipped per-option-count temperature (not fitted by the study); ours uses our one cal-fitted temperature (T=1 ECE in metrics.json).
10
+
11
+ 0.8B v3 (v1 protocol): tweet_topic acc 0.755 / F1 0.599, fin_topic acc 0.467 / F1 0.452, daily_dialog acc 0.325 / F1 0.233 (`runs/macjev/received/ext_evals/main/zeroshot_topics/metrics.json` sha256 `6c265318088f719a763a1919103632a247a8834c554d4b44326ea0c502cc0cf6`).
validation/benchmarks/zeroshot_metrics.json ADDED
@@ -0,0 +1,304 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "temperature": {
3
+ "mode": "global",
4
+ "source": "explicit --temperature",
5
+ "sha256": null,
6
+ "value": 0.8278650620942867,
7
+ "expected_2b_v3": 0.8278650621,
8
+ "fitted_on_benchmark_items": false
9
+ },
10
+ "sets": {
11
+ "tweet_topic": {
12
+ "n": 1693,
13
+ "n_ok": 1693,
14
+ "n_unsupported": 0,
15
+ "n_classes": 6,
16
+ "temperature": 0.8278650620942867,
17
+ "majority_class_accuracy": 0.3963378617838157,
18
+ "accuracy_ok_rows": 0.822209096278795,
19
+ "ece15": 0.027919883880813873,
20
+ "ece15_raw_T1": 0.08661185749227743,
21
+ "brier": 0.262576013332696,
22
+ "brier_raw_T1": 0.27079026328321404,
23
+ "nll": 0.5282712172517265,
24
+ "nll_raw_T1": 0.5629566855388951,
25
+ "mean_confidence": 0.7947445890391743,
26
+ "accuracy": 0.822209096278795,
27
+ "macro_f1": 0.677897069437572,
28
+ "balanced_accuracy": 0.7070231758410981,
29
+ "pred_share": {
30
+ "arts & culture": 0.045481393975191964,
31
+ "business & entrepreneurs": 0.040756054341405785,
32
+ "pop culture": 0.33786178381571175,
33
+ "daily life": 0.15416420555227406,
34
+ "sports & gaming": 0.36207914943886593,
35
+ "science & technology": 0.0596574128765505
36
+ },
37
+ "accuracy_ci95": [
38
+ 0.8038984051978736,
39
+ 0.8399438865918486
40
+ ],
41
+ "macro_f1_ci95": [
42
+ 0.6455020683998111,
43
+ 0.7085891187789015
44
+ ],
45
+ "ci": "percentile bootstrap, 10000 resamples, default_rng(0), rows resampled i.i.d.",
46
+ "tokens_mean": 106.1246308328411,
47
+ "tokens_max": 173,
48
+ "in_domain": false,
49
+ "catalogue_overflow": 0
50
+ },
51
+ "fin_topic": {
52
+ "n": 4117,
53
+ "n_ok": 4117,
54
+ "n_unsupported": 0,
55
+ "n_classes": 20,
56
+ "temperature": 0.8278650620942867,
57
+ "majority_class_accuracy": 0.2069468059266456,
58
+ "accuracy_ok_rows": 0.6111246052951178,
59
+ "ece15": 0.06460657860002232,
60
+ "ece15_raw_T1": 0.03822893519578447,
61
+ "brier": 0.5285731205072254,
62
+ "brier_raw_T1": 0.5230611042844824,
63
+ "nll": 1.159706614583174,
64
+ "nll_raw_T1": 1.1731574665843,
65
+ "mean_confidence": 0.6685232597080648,
66
+ "accuracy": 0.6111246052951178,
67
+ "macro_f1": 0.5897964463354406,
68
+ "balanced_accuracy": 0.6945435859869323,
69
+ "pred_share": {
70
+ "Analyst Update": 0.030604809327179985,
71
+ "Fed | Central Banks": 0.037891668690794265,
72
+ "Company | Product News": 0.1736701481661404,
73
+ "Treasuries | Corporate Debt": 0.014330823415108088,
74
+ "Dividend": 0.026961379645372846,
75
+ "Earnings": 0.09181442798153995,
76
+ "Energy | Oil": 0.050522224921059025,
77
+ "Financials": 0.013116346854505708,
78
+ "Currencies": 0.014087928102987613,
79
+ "General News | Opinion": 0.1034734029633228,
80
+ "Gold | Metals | Materials": 0.009230021860578091,
81
+ "IPO": 0.011416079669662375,
82
+ "Legal | Regulation": 0.030604809327179985,
83
+ "M&A | Investments": 0.04930774836045664,
84
+ "Macro": 0.056594607724070926,
85
+ "Markets": 0.060480932717998544,
86
+ "Politics": 0.05926645615739616,
87
+ "Personnel Change": 0.0456643186786495,
88
+ "Stock Commentary": 0.07189701238766091,
89
+ "Stock Movement": 0.04906485304833617
90
+ },
91
+ "accuracy_ci95": [
92
+ 0.5960650959436483,
93
+ 0.6259412193344669
94
+ ],
95
+ "macro_f1_ci95": [
96
+ 0.5694510128995812,
97
+ 0.6071880523494645
98
+ ],
99
+ "ci": "percentile bootstrap, 10000 resamples, default_rng(0), rows resampled i.i.d.",
100
+ "tokens_mean": 199.59825115375273,
101
+ "tokens_max": 301,
102
+ "in_domain": false,
103
+ "catalogue_overflow": 0
104
+ },
105
+ "daily_dialog": {
106
+ "n": 7740,
107
+ "n_ok": 7740,
108
+ "n_unsupported": 0,
109
+ "n_classes": 7,
110
+ "temperature": 0.8278650620942867,
111
+ "majority_class_accuracy": 0.8166666666666667,
112
+ "accuracy_ok_rows": 0.7744186046511627,
113
+ "ece15": 0.03934861337502408,
114
+ "ece15_raw_T1": 0.027480647988479354,
115
+ "brier": 0.3382397818519505,
116
+ "brier_raw_T1": 0.33726329383754494,
117
+ "nll": 0.6778540478231467,
118
+ "nll_raw_T1": 0.6808418072965744,
119
+ "mean_confidence": 0.807695045520859,
120
+ "accuracy": 0.7744186046511627,
121
+ "macro_f1": 0.3724097968832693,
122
+ "balanced_accuracy": 0.51014991333418,
123
+ "pred_share": {
124
+ "no emotion": 0.7905684754521963,
125
+ "anger": 0.01744186046511628,
126
+ "disgust": 0.013565891472868217,
127
+ "fear": 0.020284237726098192,
128
+ "happiness": 0.09056847545219639,
129
+ "sadness": 0.033850129198966405,
130
+ "surprise": 0.03372093023255814
131
+ },
132
+ "accuracy_ci95": [
133
+ 0.7649870801033591,
134
+ 0.7835917312661499
135
+ ],
136
+ "macro_f1_ci95": [
137
+ 0.3480292808769088,
138
+ 0.3955194464258293
139
+ ],
140
+ "ci": "percentile bootstrap, 10000 resamples, default_rng(0), rows resampled i.i.d.",
141
+ "tokens_mean": 79.59651162790698,
142
+ "tokens_max": 285,
143
+ "in_domain": false,
144
+ "catalogue_overflow": 0
145
+ }
146
+ },
147
+ "comparison": {
148
+ "source": "https://github.com/elcronos/jev-vs-open-decision-models/blob/a1901bc3d520e73936de8d4326545c0cdcf742fb/results/cross_dataset_summary.json",
149
+ "source_sha256": "5380d3a45395bfdf5340d75e7e18ecdb4b636734295d234839c7113efae608d0",
150
+ "note": "repository has no licence file: only its published numbers and the public datasets it names are used; no code was copied",
151
+ "clean": [
152
+ {
153
+ "set": "tweet_topic",
154
+ "n": 1693,
155
+ "majority": 0.3963378617838157,
156
+ "Jev-Style-2B-Decision-v3-GGUF-F16": {
157
+ "accuracy": 0.822209096278795,
158
+ "macro_f1": 0.677897069437572,
159
+ "ece15": 0.027919883880813873,
160
+ "ece15_raw_T1": 0.08661185749227743,
161
+ "brier": 0.262576013332696,
162
+ "n_unsupported": 0,
163
+ "accuracy_ci95": [
164
+ 0.8038984051978736,
165
+ 0.8399438865918486
166
+ ]
167
+ },
168
+ "jev": {
169
+ "accuracy": 0.7932663910218547,
170
+ "macro_f1": 0.6936,
171
+ "ece15": 0.0631,
172
+ "brier": 0.2935
173
+ },
174
+ "laya_en": {
175
+ "accuracy": 0.6320141760189013,
176
+ "macro_f1": 0.4611,
177
+ "ece15": 0.1295,
178
+ "brier": 0.5049
179
+ },
180
+ "prismnli": {
181
+ "accuracy": 0.6326048434731246,
182
+ "macro_f1": 0.5444,
183
+ "ece15": 0.181,
184
+ "brier": 0.5374
185
+ }
186
+ },
187
+ {
188
+ "set": "fin_topic",
189
+ "n": 4117,
190
+ "majority": 0.2069468059266456,
191
+ "Jev-Style-2B-Decision-v3-GGUF-F16": {
192
+ "accuracy": 0.6111246052951178,
193
+ "macro_f1": 0.5897964463354406,
194
+ "ece15": 0.06460657860002232,
195
+ "ece15_raw_T1": 0.03822893519578447,
196
+ "brier": 0.5285731205072254,
197
+ "n_unsupported": 0,
198
+ "accuracy_ci95": [
199
+ 0.5960650959436483,
200
+ 0.6259412193344669
201
+ ]
202
+ },
203
+ "jev": {
204
+ "accuracy": 0.669905270828273,
205
+ "macro_f1": 0.6298,
206
+ "ece15": 0.1664,
207
+ "brier": 0.5085
208
+ },
209
+ "laya_en": {
210
+ "accuracy": 0.3419965994656303,
211
+ "macro_f1": 0.3623,
212
+ "ece15": 0.6096,
213
+ "brier": 1.2494
214
+ },
215
+ "prismnli": {
216
+ "accuracy": 0.3524410978868108,
217
+ "macro_f1": 0.2558,
218
+ "ece15": 0.1986,
219
+ "brier": 0.8057
220
+ }
221
+ },
222
+ {
223
+ "set": "daily_dialog",
224
+ "n": 7740,
225
+ "majority": 0.8166666666666667,
226
+ "Jev-Style-2B-Decision-v3-GGUF-F16": {
227
+ "accuracy": 0.7744186046511627,
228
+ "macro_f1": 0.3724097968832693,
229
+ "ece15": 0.03934861337502408,
230
+ "ece15_raw_T1": 0.027480647988479354,
231
+ "brier": 0.3382397818519505,
232
+ "n_unsupported": 0,
233
+ "accuracy_ci95": [
234
+ 0.7649870801033591,
235
+ 0.7835917312661499
236
+ ]
237
+ },
238
+ "jev": {
239
+ "accuracy": 0.7099483204134367,
240
+ "macro_f1": 0.3847,
241
+ "ece15": 0.1563,
242
+ "brier": 0.4603
243
+ },
244
+ "laya_en": {
245
+ "accuracy": 0.6143410852713178,
246
+ "macro_f1": 0.2748,
247
+ "ece15": 0.208,
248
+ "brier": 0.6008
249
+ },
250
+ "prismnli": {
251
+ "accuracy": 0.7652454780361757,
252
+ "macro_f1": 0.3454,
253
+ "ece15": 0.1756,
254
+ "brier": 0.4091
255
+ }
256
+ }
257
+ ],
258
+ "in_domain": []
259
+ },
260
+ "ours_0.8b_v3": {
261
+ "label": "Jev-Style-0.8B-Decision-v3 (ours, v1 protocol)",
262
+ "sets": {
263
+ "tweet_topic": {
264
+ "accuracy": 0.754873006497342,
265
+ "macro_f1": 0.5993905130101003,
266
+ "ece15": 0.027434099256085594,
267
+ "ece15_raw_T1": 0.06193629176695822,
268
+ "brier": 0.3422167077233273,
269
+ "n": 1693,
270
+ "n_unsupported": 0
271
+ },
272
+ "fin_topic": {
273
+ "accuracy": 0.4670876852076755,
274
+ "macro_f1": 0.45174530293977544,
275
+ "ece15": 0.04596390350720937,
276
+ "ece15_raw_T1": 0.05647184359290005,
277
+ "brier": 0.6470044850625658,
278
+ "n": 4117,
279
+ "n_unsupported": 0
280
+ },
281
+ "daily_dialog": {
282
+ "accuracy": 0.32493540051679587,
283
+ "macro_f1": 0.2331496717992918,
284
+ "ece15": 0.4225022513967012,
285
+ "ece15_raw_T1": 0.3871480738216389,
286
+ "brier": 1.0441535348512228,
287
+ "n": 7740,
288
+ "n_unsupported": 0
289
+ }
290
+ },
291
+ "temperature": {
292
+ "policy": "global temperature of a family|qtype|option_bucket file",
293
+ "path": "runs/macjev/h100/evals/main/temperatures.json",
294
+ "sha256": "39ad8f6633934ff67770725993d7354ebebd6e7a08f73e50cab04df24715f2ce",
295
+ "version": "macjev-temperatures-v1",
296
+ "fitted_on": [
297
+ "pool_cal.jsonl:dfb7e9beff96c6c5"
298
+ ],
299
+ "T": 0.8800546821789332
300
+ },
301
+ "source": "runs/macjev/received/ext_evals/main/zeroshot_topics/metrics.json",
302
+ "source_sha256": "6c265318088f719a763a1919103632a247a8834c554d4b44326ea0c502cc0cf6"
303
+ }
304
+ }
validation/data_sources.json ADDED
@@ -0,0 +1,641 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "chaoliangUNSW/Jev-Style-2B-Decision-v3",
3
+ "note": "Every source in the training rows of this model (58 sources). licence_recorded = the licence field stored with the rows in the training pool (one entry per distinct value; 'generated...' = generated for this project). components = the mixture-table rows of the model card. link = the dataset repository the rows were built from, where one exists (null for the 22 sources generated for this project). totals = rows and tokens actually drawn during training, repeats counted (the run's exposure record); per-source counts are not listed in this file. Check each source's own terms before use; see 'Training data and licences' on the model card.",
4
+ "totals": {
5
+ "sources": 58,
6
+ "trained_logical_rows": 181449,
7
+ "trained_tokens": 60032377
8
+ },
9
+ "sources": [
10
+ {
11
+ "id": "alisawuffles/WANLI",
12
+ "components": [
13
+ "Public reading tasks"
14
+ ],
15
+ "licence_recorded": [
16
+ "cc-by-4.0"
17
+ ],
18
+ "link": "https://huggingface.co/datasets/alisawuffles/WANLI",
19
+ "note": "written by GPT-3 and revised by crowdworkers"
20
+ },
21
+ {
22
+ "id": "allenai/ai2_arc",
23
+ "components": [
24
+ "Knowledge and reasoning"
25
+ ],
26
+ "licence_recorded": [
27
+ "cc-by-sa-4.0"
28
+ ],
29
+ "link": "https://huggingface.co/datasets/allenai/ai2_arc",
30
+ "share_alike": true
31
+ },
32
+ {
33
+ "id": "allenai/qasc",
34
+ "components": [
35
+ "Knowledge and reasoning"
36
+ ],
37
+ "licence_recorded": [
38
+ "cc-by-4.0"
39
+ ],
40
+ "link": "https://huggingface.co/datasets/allenai/qasc"
41
+ },
42
+ {
43
+ "id": "allenai/qasper",
44
+ "components": [
45
+ "Public reading tasks"
46
+ ],
47
+ "licence_recorded": [
48
+ "cc-by-4.0"
49
+ ],
50
+ "link": "https://huggingface.co/datasets/allenai/qasper"
51
+ },
52
+ {
53
+ "id": "allenai/ropes",
54
+ "components": [
55
+ "Public reading tasks"
56
+ ],
57
+ "licence_recorded": [
58
+ "cc-by-4.0"
59
+ ],
60
+ "link": "https://huggingface.co/datasets/allenai/ropes"
61
+ },
62
+ {
63
+ "id": "clinc/clinc_oos",
64
+ "components": [
65
+ "Option-format views"
66
+ ],
67
+ "licence_recorded": [
68
+ "cc-by-3.0"
69
+ ],
70
+ "link": "https://huggingface.co/datasets/clinc/clinc_oos"
71
+ },
72
+ {
73
+ "id": "clinc150",
74
+ "components": [
75
+ "Intents and yes/no QA"
76
+ ],
77
+ "licence_recorded": [
78
+ "cc-by-3.0"
79
+ ],
80
+ "link": "https://huggingface.co/datasets/clinc/clinc_oos"
81
+ },
82
+ {
83
+ "id": "coastalcph/lex_glue",
84
+ "components": [
85
+ "Public reading tasks"
86
+ ],
87
+ "licence_recorded": [
88
+ "cc-by-4.0"
89
+ ],
90
+ "link": "https://huggingface.co/datasets/coastalcph/lex_glue"
91
+ },
92
+ {
93
+ "id": "czyssrs/FinQA",
94
+ "components": [
95
+ "Public reading tasks",
96
+ "Option-format views"
97
+ ],
98
+ "licence_recorded": [
99
+ "mit"
100
+ ],
101
+ "link": "https://github.com/czyssrs/FinQA"
102
+ },
103
+ {
104
+ "id": "deepmind/aqua_rat",
105
+ "components": [
106
+ "Public reading tasks"
107
+ ],
108
+ "licence_recorded": [
109
+ "apache-2.0"
110
+ ],
111
+ "link": "https://huggingface.co/datasets/deepmind/aqua_rat"
112
+ },
113
+ {
114
+ "id": "generated:crux_type_synth",
115
+ "components": [
116
+ "Knowledge and reasoning"
117
+ ],
118
+ "licence_recorded": [
119
+ "generated"
120
+ ],
121
+ "link": null,
122
+ "note": "generated for this project"
123
+ },
124
+ {
125
+ "id": "google-research-datasets/mbpp",
126
+ "components": [
127
+ "Knowledge and reasoning"
128
+ ],
129
+ "licence_recorded": [
130
+ "cc-by-4.0"
131
+ ],
132
+ "link": "https://huggingface.co/datasets/google-research-datasets/mbpp"
133
+ },
134
+ {
135
+ "id": "google-research-datasets/schema_guided_dstc8",
136
+ "components": [
137
+ "Retrieval and routing"
138
+ ],
139
+ "licence_recorded": [
140
+ "cc-by-sa-4.0"
141
+ ],
142
+ "link": "https://huggingface.co/datasets/google-research-datasets/schema_guided_dstc8",
143
+ "share_alike": true
144
+ },
145
+ {
146
+ "id": "google/boolq",
147
+ "components": [
148
+ "Intents and yes/no QA"
149
+ ],
150
+ "licence_recorded": [
151
+ "cc-by-sa-3.0"
152
+ ],
153
+ "link": "https://huggingface.co/datasets/google/boolq",
154
+ "share_alike": true
155
+ },
156
+ {
157
+ "id": "google/civil_comments",
158
+ "components": [
159
+ "Themes"
160
+ ],
161
+ "licence_recorded": [
162
+ "cc0-1.0"
163
+ ],
164
+ "link": "https://huggingface.co/datasets/google/civil_comments"
165
+ },
166
+ {
167
+ "id": "hardfam:F1",
168
+ "components": [
169
+ "Hard cases"
170
+ ],
171
+ "licence_recorded": [
172
+ "generated"
173
+ ],
174
+ "link": null,
175
+ "note": "hard-case family written by an OpenAI GPT model; labels computed by code"
176
+ },
177
+ {
178
+ "id": "hardfam:F2",
179
+ "components": [
180
+ "Hard cases"
181
+ ],
182
+ "licence_recorded": [
183
+ "generated"
184
+ ],
185
+ "link": null,
186
+ "note": "hard-case family generated by code"
187
+ },
188
+ {
189
+ "id": "hardfam:F3",
190
+ "components": [
191
+ "Hard cases"
192
+ ],
193
+ "licence_recorded": [
194
+ "generated"
195
+ ],
196
+ "link": null,
197
+ "note": "hard-case family generated by code"
198
+ },
199
+ {
200
+ "id": "hardfam:F6",
201
+ "components": [
202
+ "Hard cases"
203
+ ],
204
+ "licence_recorded": [
205
+ "generated"
206
+ ],
207
+ "link": null,
208
+ "note": "hard-case family written by an OpenAI GPT model; labels computed by code"
209
+ },
210
+ {
211
+ "id": "hardfam:F9",
212
+ "components": [
213
+ "Hard cases"
214
+ ],
215
+ "licence_recorded": [
216
+ "generated"
217
+ ],
218
+ "link": null,
219
+ "note": "hard-case family generated by code"
220
+ },
221
+ {
222
+ "id": "hugosousa/TimeQA",
223
+ "components": [
224
+ "Public reading tasks"
225
+ ],
226
+ "licence_recorded": [
227
+ "bsd-3-clause-clear"
228
+ ],
229
+ "link": "https://huggingface.co/datasets/hugosousa/TimeQA"
230
+ },
231
+ {
232
+ "id": "jackhhao/jailbreak-classification",
233
+ "components": [
234
+ "Themes"
235
+ ],
236
+ "licence_recorded": [
237
+ "apache-2.0"
238
+ ],
239
+ "link": "https://huggingface.co/datasets/jackhhao/jailbreak-classification",
240
+ "note": "its jailbreak prompts come from the jailbreak_llms collection (research purposes only); part of the benign prompts come from GPTeacher (generated by GPT-4)"
241
+ },
242
+ {
243
+ "id": "Lakera/gandalf_ignore_instructions",
244
+ "components": [
245
+ "Themes"
246
+ ],
247
+ "licence_recorded": [
248
+ "mit"
249
+ ],
250
+ "link": "https://huggingface.co/datasets/Lakera/gandalf_ignore_instructions"
251
+ },
252
+ {
253
+ "id": "lasha-nlp/CONDAQA",
254
+ "components": [
255
+ "Public reading tasks"
256
+ ],
257
+ "licence_recorded": [
258
+ "apache-2.0"
259
+ ],
260
+ "link": "https://huggingface.co/datasets/lasha-nlp/CONDAQA"
261
+ },
262
+ {
263
+ "id": "legacy:generated_rules:approved_earliest",
264
+ "components": [
265
+ "Mac agent checks",
266
+ "Option-format views"
267
+ ],
268
+ "licence_recorded": [
269
+ "generated:round-1 project collectors and rule generator (mac_laya.tasks / trajectory_v2); no third-party text"
270
+ ],
271
+ "link": null,
272
+ "note": "generated for this project"
273
+ },
274
+ {
275
+ "id": "legacy:generated_rules:cost_under_limit",
276
+ "components": [
277
+ "Mac agent checks",
278
+ "Option-format views"
279
+ ],
280
+ "licence_recorded": [
281
+ "generated:round-1 project collectors and rule generator (mac_laya.tasks / trajectory_v2); no third-party text"
282
+ ],
283
+ "link": null,
284
+ "note": "generated for this project"
285
+ },
286
+ {
287
+ "id": "legacy:generated_rules:priority_max",
288
+ "components": [
289
+ "Mac agent checks",
290
+ "Option-format views"
291
+ ],
292
+ "licence_recorded": [
293
+ "generated:round-1 project collectors and rule generator (mac_laya.tasks / trajectory_v2); no third-party text"
294
+ ],
295
+ "link": null,
296
+ "note": "generated for this project"
297
+ },
298
+ {
299
+ "id": "legacy:generated_rules:team_min_amount",
300
+ "components": [
301
+ "Mac agent checks",
302
+ "Option-format views"
303
+ ],
304
+ "licence_recorded": [
305
+ "generated:round-1 project collectors and rule generator (mac_laya.tasks / trajectory_v2); no third-party text"
306
+ ],
307
+ "link": null,
308
+ "note": "generated for this project"
309
+ },
310
+ {
311
+ "id": "legacy:mac_v2:filesystem_policy",
312
+ "components": [
313
+ "Mac agent checks"
314
+ ],
315
+ "licence_recorded": [
316
+ "generated:round-1 project collectors and rule generator (mac_laya.tasks / trajectory_v2); no third-party text"
317
+ ],
318
+ "link": null,
319
+ "note": "generated for this project"
320
+ },
321
+ {
322
+ "id": "Lichess/chess-puzzles",
323
+ "components": [
324
+ "Knowledge and reasoning"
325
+ ],
326
+ "licence_recorded": [
327
+ "cc0-1.0"
328
+ ],
329
+ "link": "https://huggingface.co/datasets/Lichess/chess-puzzles"
330
+ },
331
+ {
332
+ "id": "macjev-synth:gsm_templates",
333
+ "components": [
334
+ "Knowledge and reasoning"
335
+ ],
336
+ "licence_recorded": [
337
+ "generated"
338
+ ],
339
+ "link": null,
340
+ "note": "generated for this project"
341
+ },
342
+ {
343
+ "id": "macjev-synth:link_safety",
344
+ "components": [
345
+ "Retrieval and routing"
346
+ ],
347
+ "licence_recorded": [
348
+ "generated"
349
+ ],
350
+ "link": null,
351
+ "note": "generated for this project"
352
+ },
353
+ {
354
+ "id": "macjev-synth:sata_type",
355
+ "components": [
356
+ "Knowledge and reasoning"
357
+ ],
358
+ "licence_recorded": [
359
+ "generated"
360
+ ],
361
+ "link": null,
362
+ "note": "generated for this project"
363
+ },
364
+ {
365
+ "id": "macjev/long_table",
366
+ "components": [
367
+ "Long tables",
368
+ "Option-format views"
369
+ ],
370
+ "licence_recorded": [
371
+ "generated:cc0 (programmatic synthetic tables)"
372
+ ],
373
+ "link": null,
374
+ "note": "generated for this project"
375
+ },
376
+ {
377
+ "id": "macjev_sim:goal_done",
378
+ "components": [
379
+ "Mac agent checks"
380
+ ],
381
+ "licence_recorded": [
382
+ "generated:macjev-sim (synthetic states written by project code; no third-party text)"
383
+ ],
384
+ "link": null,
385
+ "note": "the project's own simulator; part of the goal wording was paraphrased by an OpenAI GPT model"
386
+ },
387
+ {
388
+ "id": "massive-1.1:intent",
389
+ "components": [
390
+ "Intents and yes/no QA"
391
+ ],
392
+ "licence_recorded": [
393
+ "cc-by-4.0"
394
+ ],
395
+ "link": "https://huggingface.co/datasets/AmazonScience/massive"
396
+ },
397
+ {
398
+ "id": "mteb/banking77",
399
+ "components": [
400
+ "Retrieval and routing"
401
+ ],
402
+ "licence_recorded": [
403
+ "cc-by-4.0"
404
+ ],
405
+ "link": "https://huggingface.co/datasets/mteb/banking77",
406
+ "note": "BANKING77 by PolyAI: CC BY 4.0 upstream (https://huggingface.co/datasets/PolyAI/banking77); rows taken from the MTEB mirror, whose card says MIT"
407
+ },
408
+ {
409
+ "id": "neuralchemy/Prompt-injection-dataset",
410
+ "components": [
411
+ "Themes",
412
+ "Option-format views"
413
+ ],
414
+ "licence_recorded": [
415
+ "apache-2.0",
416
+ "cc-by-4.0",
417
+ "mit"
418
+ ],
419
+ "link": "https://huggingface.co/datasets/neuralchemy/Prompt-injection-dataset",
420
+ "note": "the upstream rows its card marks research-only were removed"
421
+ },
422
+ {
423
+ "id": "nvidia/HelpSteer2",
424
+ "components": [
425
+ "Public reading tasks",
426
+ "Option-format views"
427
+ ],
428
+ "licence_recorded": [
429
+ "cc-by-4.0"
430
+ ],
431
+ "link": "https://huggingface.co/datasets/nvidia/HelpSteer2",
432
+ "note": "responses mostly written by NVIDIA Nemotron models and Mixtral-8x7B-Instruct"
433
+ },
434
+ {
435
+ "id": "openai/gsm8k",
436
+ "components": [
437
+ "Knowledge and reasoning"
438
+ ],
439
+ "licence_recorded": [
440
+ "mit"
441
+ ],
442
+ "link": "https://huggingface.co/datasets/openai/gsm8k"
443
+ },
444
+ {
445
+ "id": "openlifescienceai/medmcqa",
446
+ "components": [
447
+ "Knowledge and reasoning"
448
+ ],
449
+ "licence_recorded": [
450
+ "apache-2.0"
451
+ ],
452
+ "link": "https://huggingface.co/datasets/openlifescienceai/medmcqa"
453
+ },
454
+ {
455
+ "id": "rajpurkar/squad_v2",
456
+ "components": [
457
+ "Themes"
458
+ ],
459
+ "licence_recorded": [
460
+ "cc-by-sa-4.0"
461
+ ],
462
+ "link": "https://huggingface.co/datasets/rajpurkar/squad_v2",
463
+ "share_alike": true
464
+ },
465
+ {
466
+ "id": "Rowan/hellaswag",
467
+ "components": [
468
+ "Language"
469
+ ],
470
+ "licence_recorded": [
471
+ "mit"
472
+ ],
473
+ "link": "https://huggingface.co/datasets/Rowan/hellaswag",
474
+ "note": "MIT only in the card text (no licence field); the original GitHub repository is blocked after a DMCA notice from wikiHow; only the ActivityNet-caption items were used"
475
+ },
476
+ {
477
+ "id": "stanfordnlp/snli",
478
+ "components": [
479
+ "Language"
480
+ ],
481
+ "licence_recorded": [
482
+ "cc-by-sa-4.0"
483
+ ],
484
+ "link": "https://huggingface.co/datasets/stanfordnlp/snli",
485
+ "share_alike": true
486
+ },
487
+ {
488
+ "id": "synth-v4:algo_reasoning",
489
+ "components": [
490
+ "Knowledge and reasoning"
491
+ ],
492
+ "licence_recorded": [
493
+ "generated"
494
+ ],
495
+ "link": null,
496
+ "note": "generated for this project"
497
+ },
498
+ {
499
+ "id": "synth:causal_graph_scm",
500
+ "components": [
501
+ "Knowledge and reasoning"
502
+ ],
503
+ "licence_recorded": [
504
+ "generated"
505
+ ],
506
+ "link": null,
507
+ "note": "generated for this project"
508
+ },
509
+ {
510
+ "id": "synth:clinical_trial_reports",
511
+ "components": [
512
+ "Language"
513
+ ],
514
+ "licence_recorded": [
515
+ "generated"
516
+ ],
517
+ "link": null,
518
+ "note": "generated for this project"
519
+ },
520
+ {
521
+ "id": "synth:reasoning_relevance",
522
+ "components": [
523
+ "Retrieval and routing"
524
+ ],
525
+ "licence_recorded": [
526
+ "generated"
527
+ ],
528
+ "link": null,
529
+ "note": "generated for this project"
530
+ },
531
+ {
532
+ "id": "tau/commonsense_qa",
533
+ "components": [
534
+ "Knowledge and reasoning"
535
+ ],
536
+ "licence_recorded": [
537
+ "mit"
538
+ ],
539
+ "link": "https://huggingface.co/datasets/tau/commonsense_qa"
540
+ },
541
+ {
542
+ "id": "teacher:themes_jailbreak",
543
+ "components": [
544
+ "Themes"
545
+ ],
546
+ "licence_recorded": [
547
+ "generated:claude-opus-5.5"
548
+ ],
549
+ "link": null,
550
+ "note": "jailbreak and toxicity prompts written and labelled by an Anthropic Claude model"
551
+ },
552
+ {
553
+ "id": "theatticusproject/maud",
554
+ "components": [
555
+ "Public reading tasks",
556
+ "Option-format views"
557
+ ],
558
+ "licence_recorded": [
559
+ "cc-by-4.0"
560
+ ],
561
+ "link": "https://huggingface.co/datasets/theatticusproject/maud"
562
+ },
563
+ {
564
+ "id": "tonytan48/TempReason",
565
+ "components": [
566
+ "Public reading tasks"
567
+ ],
568
+ "licence_recorded": [
569
+ "cc-by-sa-3.0"
570
+ ],
571
+ "link": "https://huggingface.co/datasets/tonytan48/TempReason",
572
+ "share_alike": true
573
+ },
574
+ {
575
+ "id": "TrustAIRLab/in-the-wild-jailbreak-prompts",
576
+ "components": [
577
+ "Themes",
578
+ "Option-format views"
579
+ ],
580
+ "licence_recorded": [
581
+ "mit"
582
+ ],
583
+ "link": "https://huggingface.co/datasets/TrustAIRLab/in-the-wild-jailbreak-prompts",
584
+ "note": "prompts from the jailbreak_llms collection, which states it is for research purposes only"
585
+ },
586
+ {
587
+ "id": "typed_synthetic",
588
+ "components": [
589
+ "Typed decisions"
590
+ ],
591
+ "licence_recorded": [
592
+ "generated"
593
+ ],
594
+ "link": null,
595
+ "note": "model-written business workflows: designed by OpenAI GPT and Anthropic Claude models, labelled by Claude models"
596
+ },
597
+ {
598
+ "id": "ucinlp/drop",
599
+ "components": [
600
+ "Public reading tasks",
601
+ "Option-format views"
602
+ ],
603
+ "licence_recorded": [
604
+ "cc-by-sa-4.0"
605
+ ],
606
+ "link": "https://huggingface.co/datasets/ucinlp/drop",
607
+ "share_alike": true
608
+ },
609
+ {
610
+ "id": "UCLNLP/sharc",
611
+ "components": [
612
+ "Public reading tasks"
613
+ ],
614
+ "licence_recorded": [
615
+ "cc-by-sa-3.0"
616
+ ],
617
+ "link": "https://huggingface.co/datasets/UCLNLP/sharc",
618
+ "share_alike": true
619
+ },
620
+ {
621
+ "id": "wenhu/tab_fact",
622
+ "components": [
623
+ "Public reading tasks"
624
+ ],
625
+ "licence_recorded": [
626
+ "cc-by-4.0"
627
+ ],
628
+ "link": "https://huggingface.co/datasets/wenhu/tab_fact"
629
+ },
630
+ {
631
+ "id": "yanismiraoui/prompt_injections",
632
+ "components": [
633
+ "Themes"
634
+ ],
635
+ "licence_recorded": [
636
+ "apache-2.0"
637
+ ],
638
+ "link": "https://huggingface.co/datasets/yanismiraoui/prompt_injections"
639
+ }
640
+ ]
641
+ }
validation/latency_2b.json ADDED
@@ -0,0 +1,1568 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "jev-style-latency-v1",
3
+ "model": "Jev-Style-2B-Decision-v3",
4
+ "machine": {
5
+ "chip": "Apple M1 Max",
6
+ "memory_bytes": 68719476736,
7
+ "macos": "15.7.5",
8
+ "python": "3.12.13"
9
+ },
10
+ "shared_machine_note": "measured while other agents' jobs ran on the same Mac (among them a CPU-heavy PyTorch parity job using ~3.5 cores and ~16 GB); load averages at the end of each run are recorded per row. Treat the numbers as indicative, not as a clean benchmark.",
11
+ "protocol": {
12
+ "runtimes": "the staged runtimes of the three repos (jev_style_decision_gguf.py + jev-score-v2 built with build_jev_score.sh against llama.cpp 441df11f, Metal, all layers on the GPU; jev_style_decision_mlx.py, mlx 0.32.2 / mlx-lm 0.31.3; jev_style_decision.py on MPS, float32)",
13
+ "weights": "trained release weights: release_2b/gguf/model-*.gguf (tensor data identical to the named repo files), release_2b/mlx/{bf16,affine8-g64}, candidate_2b/hf-candidate (torch)",
14
+ "state": "plain text (repository documentation + source code, English) cut to target-150 tokens; the total input per question is recorded in input_tokens_per_question",
15
+ "questions": "n_questions=1: one 4-option choice question (decide); n_questions=10: 10 mixed questions (4 choice, 4 true/false, 2 score) about the same state in one score_many call",
16
+ "one_process_per_row": true,
17
+ "cold_s": "first scoring call after loading (state + questions; includes GPU warm-up)",
18
+ "warm_median_s": "median of 3 further calls, cached state dropped before each (state recomputed)",
19
+ "state_cached_median_s": "median of 3 calls with the state already computed (only the question blocks run)",
20
+ "load_s": "runtime construction (weights from the OS file cache in most rows)",
21
+ "wall_time": "time.perf_counter around decide()/score_many(), incl. tokenisation and rendering",
22
+ "peak_rss_bytes": "ru_maxrss of the Python process (and of the jev-score-v2 child for GGUF); includes memory-mapped weight pages",
23
+ "peak_phys_footprint_bytes": "macOS lifetime-max physical footprint (proc_pid_rusage v4): memory the process owns, incl. Metal / MLX buffers it allocates. It does NOT count clean memory-mapped file pages, so for GGUF (jev-score-v2 maps the .gguf file) it excludes the weights and is not a memory requirement; use peak_rss_bytes.jev_score_v2 (which includes the mapped weight pages) as the upper bound for GGUF"
24
+ },
25
+ "reruns": {
26
+ "gguf_and_mlx_rows": "all 18 GGUF rows and all 12 MLX rows were re-measured on 2026-09-27 03:11-03:28 (fix round, same script, same state text, same answers) because an independent re-run of the first session's GGUF Q8_0 / Q4_K_M 10-question rows was ~2.4x faster (contention during the first session). The first-session rows are kept in latency/runs_superseded_2026-09-27/ and are not used here. The 6 torch MPS rows are from the first session (not re-measured).",
27
+ "superseded_dir": "latency/runs_superseded_2026-09-27/"
28
+ },
29
+ "complete": true,
30
+ "missing_runs": [],
31
+ "rows": [
32
+ {
33
+ "backend": "gguf-f16",
34
+ "n_questions": 1,
35
+ "state_tokens": 878,
36
+ "input_tokens_per_question": [
37
+ 943
38
+ ],
39
+ "load_s": 2.6054,
40
+ "cold_s": 0.5624,
41
+ "warm_median_s": 0.5242,
42
+ "warm_s": [
43
+ 0.5203,
44
+ 0.5345,
45
+ 0.5242
46
+ ],
47
+ "state_cached_median_s": 0.0653,
48
+ "peak_rss_bytes": {
49
+ "python": 352616448,
50
+ "jev_score_v2": 4451008512
51
+ },
52
+ "peak_phys_footprint_bytes": {
53
+ "python": 252413696,
54
+ "jev_score_v2": 660384576
55
+ },
56
+ "results_identical_cold_vs_state_cached": true,
57
+ "loadavg_at_end": [
58
+ 9.59,
59
+ 9.9,
60
+ 10.98
61
+ ],
62
+ "measured_unix": 1790442718.8146281,
63
+ "settings": {
64
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
65
+ "n_gpu_layers": 999,
66
+ "flash_attn": "default (llama.cpp auto on Metal)",
67
+ "gguf": "release_2b/gguf/model-f16.gguf"
68
+ },
69
+ "raw": "latency/runs/gguf-f16_1024_1q.json"
70
+ },
71
+ {
72
+ "backend": "gguf-f16",
73
+ "n_questions": 10,
74
+ "state_tokens": 878,
75
+ "input_tokens_per_question": [
76
+ 943,
77
+ 918,
78
+ 925,
79
+ 911,
80
+ 918,
81
+ 921,
82
+ 943,
83
+ 917,
84
+ 912,
85
+ 919
86
+ ],
87
+ "load_s": 1.1941,
88
+ "cold_s": 0.994,
89
+ "warm_median_s": 0.968,
90
+ "warm_s": [
91
+ 0.968,
92
+ 0.9464,
93
+ 0.9713
94
+ ],
95
+ "state_cached_median_s": 0.4965,
96
+ "peak_rss_bytes": {
97
+ "python": 328564736,
98
+ "jev_score_v2": 4516610048
99
+ },
100
+ "peak_phys_footprint_bytes": {
101
+ "python": 254527296,
102
+ "jev_score_v2": 676899648
103
+ },
104
+ "results_identical_cold_vs_state_cached": true,
105
+ "loadavg_at_end": [
106
+ 10.0,
107
+ 9.97,
108
+ 10.99
109
+ ],
110
+ "measured_unix": 1790442726.19508,
111
+ "settings": {
112
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
113
+ "n_gpu_layers": 999,
114
+ "flash_attn": "default (llama.cpp auto on Metal)",
115
+ "gguf": "release_2b/gguf/model-f16.gguf"
116
+ },
117
+ "raw": "latency/runs/gguf-f16_1024_10q.json"
118
+ },
119
+ {
120
+ "backend": "gguf-f16",
121
+ "n_questions": 1,
122
+ "state_tokens": 3950,
123
+ "input_tokens_per_question": [
124
+ 4015
125
+ ],
126
+ "load_s": 1.1659,
127
+ "cold_s": 2.2122,
128
+ "warm_median_s": 2.1753,
129
+ "warm_s": [
130
+ 2.1909,
131
+ 2.1753,
132
+ 2.1733
133
+ ],
134
+ "state_cached_median_s": 0.078,
135
+ "peak_rss_bytes": {
136
+ "python": 354697216,
137
+ "jev_score_v2": 4568334336
138
+ },
139
+ "peak_phys_footprint_bytes": {
140
+ "python": 257705600,
141
+ "jev_score_v2": 722676736
142
+ },
143
+ "results_identical_cold_vs_state_cached": true,
144
+ "loadavg_at_end": [
145
+ 9.01,
146
+ 9.76,
147
+ 10.9
148
+ ],
149
+ "measured_unix": 1790442737.183161,
150
+ "settings": {
151
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
152
+ "n_gpu_layers": 999,
153
+ "flash_attn": "default (llama.cpp auto on Metal)",
154
+ "gguf": "release_2b/gguf/model-f16.gguf"
155
+ },
156
+ "raw": "latency/runs/gguf-f16_4096_1q.json"
157
+ },
158
+ {
159
+ "backend": "gguf-f16",
160
+ "n_questions": 10,
161
+ "state_tokens": 3950,
162
+ "input_tokens_per_question": [
163
+ 4015,
164
+ 3990,
165
+ 3997,
166
+ 3983,
167
+ 3990,
168
+ 3993,
169
+ 4015,
170
+ 3989,
171
+ 3984,
172
+ 3991
173
+ ],
174
+ "load_s": 1.2112,
175
+ "cold_s": 2.6904,
176
+ "warm_median_s": 2.6583,
177
+ "warm_s": [
178
+ 2.6583,
179
+ 2.6665,
180
+ 2.6378
181
+ ],
182
+ "state_cached_median_s": 0.5467,
183
+ "peak_rss_bytes": {
184
+ "python": 331694080,
185
+ "jev_score_v2": 4564172800
186
+ },
187
+ "peak_phys_footprint_bytes": {
188
+ "python": 248334016,
189
+ "jev_score_v2": 720382976
190
+ },
191
+ "results_identical_cold_vs_state_cached": true,
192
+ "loadavg_at_end": [
193
+ 8.62,
194
+ 9.62,
195
+ 10.83
196
+ ],
197
+ "measured_unix": 1790442751.5065348,
198
+ "settings": {
199
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
200
+ "n_gpu_layers": 999,
201
+ "flash_attn": "default (llama.cpp auto on Metal)",
202
+ "gguf": "release_2b/gguf/model-f16.gguf"
203
+ },
204
+ "raw": "latency/runs/gguf-f16_4096_10q.json"
205
+ },
206
+ {
207
+ "backend": "gguf-f16",
208
+ "n_questions": 1,
209
+ "state_tokens": 24436,
210
+ "input_tokens_per_question": [
211
+ 24501
212
+ ],
213
+ "load_s": 1.1784,
214
+ "cold_s": 16.2299,
215
+ "warm_median_s": 16.1651,
216
+ "warm_s": [
217
+ 16.1645,
218
+ 16.1651,
219
+ 16.2375
220
+ ],
221
+ "state_cached_median_s": 0.1677,
222
+ "peak_rss_bytes": {
223
+ "python": 349683712,
224
+ "jev_score_v2": 4734812160
225
+ },
226
+ "peak_phys_footprint_bytes": {
227
+ "python": 239191744,
228
+ "jev_score_v2": 882732352
229
+ },
230
+ "results_identical_cold_vs_state_cached": true,
231
+ "loadavg_at_end": [
232
+ 9.74,
233
+ 9.66,
234
+ 10.75
235
+ ],
236
+ "measured_unix": 1790442818.798799,
237
+ "settings": {
238
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
239
+ "n_gpu_layers": 999,
240
+ "flash_attn": "default (llama.cpp auto on Metal)",
241
+ "gguf": "release_2b/gguf/model-f16.gguf"
242
+ },
243
+ "raw": "latency/runs/gguf-f16_24576_1q.json"
244
+ },
245
+ {
246
+ "backend": "gguf-f16",
247
+ "n_questions": 10,
248
+ "state_tokens": 24436,
249
+ "input_tokens_per_question": [
250
+ 24501,
251
+ 24476,
252
+ 24483,
253
+ 24469,
254
+ 24476,
255
+ 24479,
256
+ 24501,
257
+ 24475,
258
+ 24470,
259
+ 24477
260
+ ],
261
+ "load_s": 1.225,
262
+ "cold_s": 16.9884,
263
+ "warm_median_s": 16.8115,
264
+ "warm_s": [
265
+ 16.8115,
266
+ 16.675,
267
+ 16.8156
268
+ ],
269
+ "state_cached_median_s": 0.8643,
270
+ "peak_rss_bytes": {
271
+ "python": 367919104,
272
+ "jev_score_v2": 4738170880
273
+ },
274
+ "peak_phys_footprint_bytes": {
275
+ "python": 242992832,
276
+ "jev_score_v2": 887975168
277
+ },
278
+ "results_identical_cold_vs_state_cached": true,
279
+ "loadavg_at_end": [
280
+ 8.02,
281
+ 9.14,
282
+ 10.46
283
+ ],
284
+ "measured_unix": 1790442890.769493,
285
+ "settings": {
286
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
287
+ "n_gpu_layers": 999,
288
+ "flash_attn": "default (llama.cpp auto on Metal)",
289
+ "gguf": "release_2b/gguf/model-f16.gguf"
290
+ },
291
+ "raw": "latency/runs/gguf-f16_24576_10q.json"
292
+ },
293
+ {
294
+ "backend": "gguf-q8_0",
295
+ "n_questions": 1,
296
+ "state_tokens": 878,
297
+ "input_tokens_per_question": [
298
+ 943
299
+ ],
300
+ "load_s": 1.7934,
301
+ "cold_s": 0.6136,
302
+ "warm_median_s": 0.5663,
303
+ "warm_s": [
304
+ 0.5678,
305
+ 0.5663,
306
+ 0.5663
307
+ ],
308
+ "state_cached_median_s": 0.0682,
309
+ "peak_rss_bytes": {
310
+ "python": 328974336,
311
+ "jev_score_v2": 2721579008
312
+ },
313
+ "peak_phys_footprint_bytes": {
314
+ "python": 248874624,
315
+ "jev_score_v2": 668163520
316
+ },
317
+ "results_identical_cold_vs_state_cached": true,
318
+ "loadavg_at_end": [
319
+ 7.69,
320
+ 9.05,
321
+ 10.42
322
+ ],
323
+ "measured_unix": 1790442895.894016,
324
+ "settings": {
325
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
326
+ "n_gpu_layers": 999,
327
+ "flash_attn": "default (llama.cpp auto on Metal)",
328
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
329
+ },
330
+ "raw": "latency/runs/gguf-q8_0_1024_1q.json"
331
+ },
332
+ {
333
+ "backend": "gguf-q8_0",
334
+ "n_questions": 10,
335
+ "state_tokens": 878,
336
+ "input_tokens_per_question": [
337
+ 943,
338
+ 918,
339
+ 925,
340
+ 911,
341
+ 918,
342
+ 921,
343
+ 943,
344
+ 917,
345
+ 912,
346
+ 919
347
+ ],
348
+ "load_s": 1.0682,
349
+ "cold_s": 1.073,
350
+ "warm_median_s": 1.0243,
351
+ "warm_s": [
352
+ 1.0244,
353
+ 1.0243,
354
+ 1.0231
355
+ ],
356
+ "state_cached_median_s": 0.5268,
357
+ "peak_rss_bytes": {
358
+ "python": 347635712,
359
+ "jev_score_v2": 2757804032
360
+ },
361
+ "peak_phys_footprint_bytes": {
362
+ "python": 258377344,
363
+ "jev_score_v2": 674569792
364
+ },
365
+ "results_identical_cold_vs_state_cached": true,
366
+ "loadavg_at_end": [
367
+ 7.24,
368
+ 8.94,
369
+ 10.37
370
+ ],
371
+ "measured_unix": 1790442903.5177228,
372
+ "settings": {
373
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
374
+ "n_gpu_layers": 999,
375
+ "flash_attn": "default (llama.cpp auto on Metal)",
376
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
377
+ },
378
+ "raw": "latency/runs/gguf-q8_0_1024_10q.json"
379
+ },
380
+ {
381
+ "backend": "gguf-q8_0",
382
+ "n_questions": 1,
383
+ "state_tokens": 3950,
384
+ "input_tokens_per_question": [
385
+ 4015
386
+ ],
387
+ "load_s": 1.0689,
388
+ "cold_s": 2.3987,
389
+ "warm_median_s": 2.3345,
390
+ "warm_s": [
391
+ 2.3333,
392
+ 2.3488,
393
+ 2.3345
394
+ ],
395
+ "state_cached_median_s": 0.082,
396
+ "peak_rss_bytes": {
397
+ "python": 354189312,
398
+ "jev_score_v2": 2807808000
399
+ },
400
+ "peak_phys_footprint_bytes": {
401
+ "python": 247596736,
402
+ "jev_score_v2": 719183680
403
+ },
404
+ "results_identical_cold_vs_state_cached": true,
405
+ "loadavg_at_end": [
406
+ 6.58,
407
+ 8.74,
408
+ 10.28
409
+ ],
410
+ "measured_unix": 1790442915.085064,
411
+ "settings": {
412
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
413
+ "n_gpu_layers": 999,
414
+ "flash_attn": "default (llama.cpp auto on Metal)",
415
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
416
+ },
417
+ "raw": "latency/runs/gguf-q8_0_4096_1q.json"
418
+ },
419
+ {
420
+ "backend": "gguf-q8_0",
421
+ "n_questions": 10,
422
+ "state_tokens": 3950,
423
+ "input_tokens_per_question": [
424
+ 4015,
425
+ 3990,
426
+ 3997,
427
+ 3983,
428
+ 3990,
429
+ 3993,
430
+ 4015,
431
+ 3989,
432
+ 3984,
433
+ 3991
434
+ ],
435
+ "load_s": 1.0684,
436
+ "cold_s": 2.8854,
437
+ "warm_median_s": 2.8178,
438
+ "warm_s": [
439
+ 2.8178,
440
+ 2.8439,
441
+ 2.816
442
+ ],
443
+ "state_cached_median_s": 0.5719,
444
+ "peak_rss_bytes": {
445
+ "python": 319045632,
446
+ "jev_score_v2": 2799091712
447
+ },
448
+ "peak_phys_footprint_bytes": {
449
+ "python": 243910336,
450
+ "jev_score_v2": 716169152
451
+ },
452
+ "results_identical_cold_vs_state_cached": true,
453
+ "loadavg_at_end": [
454
+ 7.36,
455
+ 8.8,
456
+ 10.28
457
+ ],
458
+ "measured_unix": 1790442930.063521,
459
+ "settings": {
460
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
461
+ "n_gpu_layers": 999,
462
+ "flash_attn": "default (llama.cpp auto on Metal)",
463
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
464
+ },
465
+ "raw": "latency/runs/gguf-q8_0_4096_10q.json"
466
+ },
467
+ {
468
+ "backend": "gguf-q8_0",
469
+ "n_questions": 1,
470
+ "state_tokens": 24436,
471
+ "input_tokens_per_question": [
472
+ 24501
473
+ ],
474
+ "load_s": 1.0589,
475
+ "cold_s": 17.0922,
476
+ "warm_median_s": 17.0489,
477
+ "warm_s": [
478
+ 17.0489,
479
+ 17.0376,
480
+ 17.0539
481
+ ],
482
+ "state_cached_median_s": 0.1679,
483
+ "peak_rss_bytes": {
484
+ "python": 364937216,
485
+ "jev_score_v2": 2966470656
486
+ },
487
+ "peak_phys_footprint_bytes": {
488
+ "python": 267110080,
489
+ "jev_score_v2": 875831168
490
+ },
491
+ "results_identical_cold_vs_state_cached": true,
492
+ "loadavg_at_end": [
493
+ 6.7,
494
+ 8.25,
495
+ 9.94
496
+ ],
497
+ "measured_unix": 1790443000.691691,
498
+ "settings": {
499
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
500
+ "n_gpu_layers": 999,
501
+ "flash_attn": "default (llama.cpp auto on Metal)",
502
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
503
+ },
504
+ "raw": "latency/runs/gguf-q8_0_24576_1q.json"
505
+ },
506
+ {
507
+ "backend": "gguf-q8_0",
508
+ "n_questions": 10,
509
+ "state_tokens": 24436,
510
+ "input_tokens_per_question": [
511
+ 24501,
512
+ 24476,
513
+ 24483,
514
+ 24469,
515
+ 24476,
516
+ 24479,
517
+ 24501,
518
+ 24475,
519
+ 24470,
520
+ 24477
521
+ ],
522
+ "load_s": 1.0501,
523
+ "cold_s": 17.8244,
524
+ "warm_median_s": 17.7424,
525
+ "warm_s": [
526
+ 17.7424,
527
+ 17.7312,
528
+ 17.7793
529
+ ],
530
+ "state_cached_median_s": 0.8902,
531
+ "peak_rss_bytes": {
532
+ "python": 348651520,
533
+ "jev_score_v2": 2967109632
534
+ },
535
+ "peak_phys_footprint_bytes": {
536
+ "python": 260425536,
537
+ "jev_score_v2": 878698496
538
+ },
539
+ "results_identical_cold_vs_state_cached": true,
540
+ "loadavg_at_end": [
541
+ 11.28,
542
+ 9.17,
543
+ 10.13
544
+ ],
545
+ "measured_unix": 1790443076.341974,
546
+ "settings": {
547
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
548
+ "n_gpu_layers": 999,
549
+ "flash_attn": "default (llama.cpp auto on Metal)",
550
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
551
+ },
552
+ "raw": "latency/runs/gguf-q8_0_24576_10q.json"
553
+ },
554
+ {
555
+ "backend": "gguf-q4_k_m",
556
+ "n_questions": 1,
557
+ "state_tokens": 878,
558
+ "input_tokens_per_question": [
559
+ 943
560
+ ],
561
+ "load_s": 1.4512,
562
+ "cold_s": 0.6738,
563
+ "warm_median_s": 0.6391,
564
+ "warm_s": [
565
+ 0.6418,
566
+ 0.6391,
567
+ 0.6334
568
+ ],
569
+ "state_cached_median_s": 0.0782,
570
+ "peak_rss_bytes": {
571
+ "python": 334921728,
572
+ "jev_score_v2": 1988542464
573
+ },
574
+ "peak_phys_footprint_bytes": {
575
+ "python": 245073728,
576
+ "jev_score_v2": 669308992
577
+ },
578
+ "results_identical_cold_vs_state_cached": true,
579
+ "loadavg_at_end": [
580
+ 12.77,
581
+ 9.51,
582
+ 10.25
583
+ ],
584
+ "measured_unix": 1790443081.420589,
585
+ "settings": {
586
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
587
+ "n_gpu_layers": 999,
588
+ "flash_attn": "default (llama.cpp auto on Metal)",
589
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
590
+ },
591
+ "raw": "latency/runs/gguf-q4_k_m_1024_1q.json"
592
+ },
593
+ {
594
+ "backend": "gguf-q4_k_m",
595
+ "n_questions": 10,
596
+ "state_tokens": 878,
597
+ "input_tokens_per_question": [
598
+ 943,
599
+ 918,
600
+ 925,
601
+ 911,
602
+ 918,
603
+ 921,
604
+ 943,
605
+ 917,
606
+ 912,
607
+ 919
608
+ ],
609
+ "load_s": 0.9937,
610
+ "cold_s": 1.2045,
611
+ "warm_median_s": 1.157,
612
+ "warm_s": [
613
+ 1.1544,
614
+ 1.1643,
615
+ 1.157
616
+ ],
617
+ "state_cached_median_s": 0.6009,
618
+ "peak_rss_bytes": {
619
+ "python": 352059392,
620
+ "jev_score_v2": 2018754560
621
+ },
622
+ "peak_phys_footprint_bytes": {
623
+ "python": 256673344,
624
+ "jev_score_v2": 665065728
625
+ },
626
+ "results_identical_cold_vs_state_cached": true,
627
+ "loadavg_at_end": [
628
+ 13.83,
629
+ 9.78,
630
+ 10.34
631
+ ],
632
+ "measured_unix": 1790443089.6937559,
633
+ "settings": {
634
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
635
+ "n_gpu_layers": 999,
636
+ "flash_attn": "default (llama.cpp auto on Metal)",
637
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
638
+ },
639
+ "raw": "latency/runs/gguf-q4_k_m_1024_10q.json"
640
+ },
641
+ {
642
+ "backend": "gguf-q4_k_m",
643
+ "n_questions": 1,
644
+ "state_tokens": 3950,
645
+ "input_tokens_per_question": [
646
+ 4015
647
+ ],
648
+ "load_s": 0.9922,
649
+ "cold_s": 2.6608,
650
+ "warm_median_s": 2.6006,
651
+ "warm_s": [
652
+ 2.6006,
653
+ 2.6206,
654
+ 2.5977
655
+ ],
656
+ "state_cached_median_s": 0.0908,
657
+ "peak_rss_bytes": {
658
+ "python": 327221248,
659
+ "jev_score_v2": 2062483456
660
+ },
661
+ "peak_phys_footprint_bytes": {
662
+ "python": 259393408,
663
+ "jev_score_v2": 714692992
664
+ },
665
+ "results_identical_cold_vs_state_cached": true,
666
+ "loadavg_at_end": [
667
+ 11.99,
668
+ 9.57,
669
+ 10.25
670
+ ],
671
+ "measured_unix": 1790443102.1980171,
672
+ "settings": {
673
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
674
+ "n_gpu_layers": 999,
675
+ "flash_attn": "default (llama.cpp auto on Metal)",
676
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
677
+ },
678
+ "raw": "latency/runs/gguf-q4_k_m_4096_1q.json"
679
+ },
680
+ {
681
+ "backend": "gguf-q4_k_m",
682
+ "n_questions": 10,
683
+ "state_tokens": 3950,
684
+ "input_tokens_per_question": [
685
+ 4015,
686
+ 3990,
687
+ 3997,
688
+ 3983,
689
+ 3990,
690
+ 3993,
691
+ 4015,
692
+ 3989,
693
+ 3984,
694
+ 3991
695
+ ],
696
+ "load_s": 0.9859,
697
+ "cold_s": 3.2053,
698
+ "warm_median_s": 3.1651,
699
+ "warm_s": [
700
+ 3.1651,
701
+ 3.1579,
702
+ 3.1711
703
+ ],
704
+ "state_cached_median_s": 0.6462,
705
+ "peak_rss_bytes": {
706
+ "python": 326434816,
707
+ "jev_score_v2": 2073214976
708
+ },
709
+ "peak_phys_footprint_bytes": {
710
+ "python": 245614208,
711
+ "jev_score_v2": 716806400
712
+ },
713
+ "results_identical_cold_vs_state_cached": true,
714
+ "loadavg_at_end": [
715
+ 10.28,
716
+ 9.31,
717
+ 10.15
718
+ ],
719
+ "measured_unix": 1790443118.634955,
720
+ "settings": {
721
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
722
+ "n_gpu_layers": 999,
723
+ "flash_attn": "default (llama.cpp auto on Metal)",
724
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
725
+ },
726
+ "raw": "latency/runs/gguf-q4_k_m_4096_10q.json"
727
+ },
728
+ {
729
+ "backend": "gguf-q4_k_m",
730
+ "n_questions": 1,
731
+ "state_tokens": 24436,
732
+ "input_tokens_per_question": [
733
+ 24501
734
+ ],
735
+ "load_s": 1.0041,
736
+ "cold_s": 18.6914,
737
+ "warm_median_s": 18.6165,
738
+ "warm_s": [
739
+ 18.6165,
740
+ 18.6059,
741
+ 18.624
742
+ ],
743
+ "state_cached_median_s": 0.1755,
744
+ "peak_rss_bytes": {
745
+ "python": 345686016,
746
+ "jev_score_v2": 2233663488
747
+ },
748
+ "peak_phys_footprint_bytes": {
749
+ "python": 260048576,
750
+ "jev_score_v2": 880335552
751
+ },
752
+ "results_identical_cold_vs_state_cached": true,
753
+ "loadavg_at_end": [
754
+ 8.43,
755
+ 9.22,
756
+ 10.05
757
+ ],
758
+ "measured_unix": 1790443195.534742,
759
+ "settings": {
760
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
761
+ "n_gpu_layers": 999,
762
+ "flash_attn": "default (llama.cpp auto on Metal)",
763
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
764
+ },
765
+ "raw": "latency/runs/gguf-q4_k_m_24576_1q.json"
766
+ },
767
+ {
768
+ "backend": "gguf-q4_k_m",
769
+ "n_questions": 10,
770
+ "state_tokens": 24436,
771
+ "input_tokens_per_question": [
772
+ 24501,
773
+ 24476,
774
+ 24483,
775
+ 24469,
776
+ 24476,
777
+ 24479,
778
+ 24501,
779
+ 24475,
780
+ 24470,
781
+ 24477
782
+ ],
783
+ "load_s": 0.9967,
784
+ "cold_s": 19.4776,
785
+ "warm_median_s": 19.4309,
786
+ "warm_s": [
787
+ 19.4309,
788
+ 19.4251,
789
+ 19.5057
790
+ ],
791
+ "state_cached_median_s": 0.9616,
792
+ "peak_rss_bytes": {
793
+ "python": 354992128,
794
+ "jev_score_v2": 2226913280
795
+ },
796
+ "peak_phys_footprint_bytes": {
797
+ "python": 250791552,
798
+ "jev_score_v2": 870128320
799
+ },
800
+ "results_identical_cold_vs_state_cached": true,
801
+ "loadavg_at_end": [
802
+ 5.63,
803
+ 8.19,
804
+ 9.59
805
+ ],
806
+ "measured_unix": 1790443278.075001,
807
+ "settings": {
808
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
809
+ "n_gpu_layers": 999,
810
+ "flash_attn": "default (llama.cpp auto on Metal)",
811
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
812
+ },
813
+ "raw": "latency/runs/gguf-q4_k_m_24576_10q.json"
814
+ },
815
+ {
816
+ "backend": "mlx-bf16",
817
+ "n_questions": 1,
818
+ "state_tokens": 878,
819
+ "input_tokens_per_question": [
820
+ 943
821
+ ],
822
+ "load_s": 3.6602,
823
+ "cold_s": 0.64,
824
+ "warm_median_s": 0.5658,
825
+ "warm_s": [
826
+ 0.5615,
827
+ 0.5665,
828
+ 0.5658
829
+ ],
830
+ "state_cached_median_s": 0.0796,
831
+ "peak_rss_bytes": {
832
+ "python": 4401463296
833
+ },
834
+ "peak_phys_footprint_bytes": {
835
+ "python": 5205486080
836
+ },
837
+ "mlx_peak_memory_bytes": 4536436474,
838
+ "results_identical_cold_vs_state_cached": true,
839
+ "loadavg_at_end": [
840
+ 6.5,
841
+ 8.17,
842
+ 9.53
843
+ ],
844
+ "measured_unix": 1790443305.918901,
845
+ "settings": {
846
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
847
+ "precision": "bf16",
848
+ "compute_dtype": null
849
+ },
850
+ "raw": "latency/runs/mlx-bf16_1024_1q.json"
851
+ },
852
+ {
853
+ "backend": "mlx-bf16",
854
+ "n_questions": 10,
855
+ "state_tokens": 878,
856
+ "input_tokens_per_question": [
857
+ 943,
858
+ 918,
859
+ 925,
860
+ 911,
861
+ 918,
862
+ 921,
863
+ 943,
864
+ 917,
865
+ 912,
866
+ 919
867
+ ],
868
+ "load_s": 3.3114,
869
+ "cold_s": 1.1022,
870
+ "warm_median_s": 1.0651,
871
+ "warm_s": [
872
+ 1.0651,
873
+ 1.0647,
874
+ 1.0714
875
+ ],
876
+ "state_cached_median_s": 0.5726,
877
+ "peak_rss_bytes": {
878
+ "python": 4416684032
879
+ },
880
+ "peak_phys_footprint_bytes": {
881
+ "python": 5575518912
882
+ },
883
+ "mlx_peak_memory_bytes": 4536436474,
884
+ "results_identical_cold_vs_state_cached": true,
885
+ "loadavg_at_end": [
886
+ 6.03,
887
+ 8.01,
888
+ 9.46
889
+ ],
890
+ "measured_unix": 1790443316.3707972,
891
+ "settings": {
892
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
893
+ "precision": "bf16",
894
+ "compute_dtype": null
895
+ },
896
+ "raw": "latency/runs/mlx-bf16_1024_10q.json"
897
+ },
898
+ {
899
+ "backend": "mlx-bf16",
900
+ "n_questions": 1,
901
+ "state_tokens": 3950,
902
+ "input_tokens_per_question": [
903
+ 4015
904
+ ],
905
+ "load_s": 3.2848,
906
+ "cold_s": 2.265,
907
+ "warm_median_s": 2.2579,
908
+ "warm_s": [
909
+ 2.2462,
910
+ 2.2579,
911
+ 2.2701
912
+ ],
913
+ "state_cached_median_s": 0.0903,
914
+ "peak_rss_bytes": {
915
+ "python": 4408279040
916
+ },
917
+ "peak_phys_footprint_bytes": {
918
+ "python": 6899099712
919
+ },
920
+ "mlx_peak_memory_bytes": 5062198966,
921
+ "results_identical_cold_vs_state_cached": true,
922
+ "loadavg_at_end": [
923
+ 5.78,
924
+ 7.9,
925
+ 9.4
926
+ ],
927
+ "measured_unix": 1790443330.072621,
928
+ "settings": {
929
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
930
+ "precision": "bf16",
931
+ "compute_dtype": null
932
+ },
933
+ "raw": "latency/runs/mlx-bf16_4096_1q.json"
934
+ },
935
+ {
936
+ "backend": "mlx-bf16",
937
+ "n_questions": 10,
938
+ "state_tokens": 3950,
939
+ "input_tokens_per_question": [
940
+ 4015,
941
+ 3990,
942
+ 3997,
943
+ 3983,
944
+ 3990,
945
+ 3993,
946
+ 4015,
947
+ 3989,
948
+ 3984,
949
+ 3991
950
+ ],
951
+ "load_s": 3.3381,
952
+ "cold_s": 2.8286,
953
+ "warm_median_s": 2.7998,
954
+ "warm_s": [
955
+ 2.832,
956
+ 2.7968,
957
+ 2.7998
958
+ ],
959
+ "state_cached_median_s": 0.622,
960
+ "peak_rss_bytes": {
961
+ "python": 4406149120
962
+ },
963
+ "peak_phys_footprint_bytes": {
964
+ "python": 6943402624
965
+ },
966
+ "mlx_peak_memory_bytes": 5062395574,
967
+ "results_identical_cold_vs_state_cached": true,
968
+ "loadavg_at_end": [
969
+ 5.53,
970
+ 7.71,
971
+ 9.3
972
+ ],
973
+ "measured_unix": 1790443347.640775,
974
+ "settings": {
975
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
976
+ "precision": "bf16",
977
+ "compute_dtype": null
978
+ },
979
+ "raw": "latency/runs/mlx-bf16_4096_10q.json"
980
+ },
981
+ {
982
+ "backend": "mlx-bf16",
983
+ "n_questions": 1,
984
+ "state_tokens": 24436,
985
+ "input_tokens_per_question": [
986
+ 24501
987
+ ],
988
+ "load_s": 3.2657,
989
+ "cold_s": 15.472,
990
+ "warm_median_s": 15.5073,
991
+ "warm_s": [
992
+ 15.5202,
993
+ 15.5073,
994
+ 15.4734
995
+ ],
996
+ "state_cached_median_s": 0.1546,
997
+ "peak_rss_bytes": {
998
+ "python": 4403707904
999
+ },
1000
+ "peak_phys_footprint_bytes": {
1001
+ "python": 7889725824
1002
+ },
1003
+ "mlx_peak_memory_bytes": 5942822582,
1004
+ "results_identical_cold_vs_state_cached": true,
1005
+ "loadavg_at_end": [
1006
+ 6.15,
1007
+ 7.49,
1008
+ 9.1
1009
+ ],
1010
+ "measured_unix": 1790443414.47198,
1011
+ "settings": {
1012
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1013
+ "precision": "bf16",
1014
+ "compute_dtype": null
1015
+ },
1016
+ "raw": "latency/runs/mlx-bf16_24576_1q.json"
1017
+ },
1018
+ {
1019
+ "backend": "mlx-bf16",
1020
+ "n_questions": 10,
1021
+ "state_tokens": 24436,
1022
+ "input_tokens_per_question": [
1023
+ 24501,
1024
+ 24476,
1025
+ 24483,
1026
+ 24469,
1027
+ 24476,
1028
+ 24479,
1029
+ 24501,
1030
+ 24475,
1031
+ 24470,
1032
+ 24477
1033
+ ],
1034
+ "load_s": 3.278,
1035
+ "cold_s": 16.2813,
1036
+ "warm_median_s": 16.2351,
1037
+ "warm_s": [
1038
+ 16.2243,
1039
+ 16.2351,
1040
+ 16.2657
1041
+ ],
1042
+ "state_cached_median_s": 0.8797,
1043
+ "peak_rss_bytes": {
1044
+ "python": 4405805056
1045
+ },
1046
+ "peak_phys_footprint_bytes": {
1047
+ "python": 7863035904
1048
+ },
1049
+ "mlx_peak_memory_bytes": 5942806198,
1050
+ "results_identical_cold_vs_state_cached": true,
1051
+ "loadavg_at_end": [
1052
+ 8.07,
1053
+ 7.75,
1054
+ 9.06
1055
+ ],
1056
+ "measured_unix": 1790443486.522902,
1057
+ "settings": {
1058
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1059
+ "precision": "bf16",
1060
+ "compute_dtype": null
1061
+ },
1062
+ "raw": "latency/runs/mlx-bf16_24576_10q.json"
1063
+ },
1064
+ {
1065
+ "backend": "mlx-8bit",
1066
+ "n_questions": 1,
1067
+ "state_tokens": 878,
1068
+ "input_tokens_per_question": [
1069
+ 943
1070
+ ],
1071
+ "load_s": 3.3345,
1072
+ "cold_s": 0.7573,
1073
+ "warm_median_s": 0.72,
1074
+ "warm_s": [
1075
+ 0.7177,
1076
+ 0.72,
1077
+ 0.7228
1078
+ ],
1079
+ "state_cached_median_s": 0.0834,
1080
+ "peak_rss_bytes": {
1081
+ "python": 2640986112
1082
+ },
1083
+ "peak_phys_footprint_bytes": {
1084
+ "python": 3772211392
1085
+ },
1086
+ "mlx_peak_memory_bytes": 3074287404,
1087
+ "results_identical_cold_vs_state_cached": true,
1088
+ "loadavg_at_end": [
1089
+ 8.14,
1090
+ 7.77,
1091
+ 9.06
1092
+ ],
1093
+ "measured_unix": 1790443494.139926,
1094
+ "settings": {
1095
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1096
+ "precision": "8bit",
1097
+ "compute_dtype": null
1098
+ },
1099
+ "raw": "latency/runs/mlx-8bit_1024_1q.json"
1100
+ },
1101
+ {
1102
+ "backend": "mlx-8bit",
1103
+ "n_questions": 10,
1104
+ "state_tokens": 878,
1105
+ "input_tokens_per_question": [
1106
+ 943,
1107
+ 918,
1108
+ 925,
1109
+ 911,
1110
+ 918,
1111
+ 921,
1112
+ 943,
1113
+ 917,
1114
+ 912,
1115
+ 919
1116
+ ],
1117
+ "load_s": 3.1502,
1118
+ "cold_s": 1.3124,
1119
+ "warm_median_s": 1.2657,
1120
+ "warm_s": [
1121
+ 1.2719,
1122
+ 1.2657,
1123
+ 1.2621
1124
+ ],
1125
+ "state_cached_median_s": 0.6288,
1126
+ "peak_rss_bytes": {
1127
+ "python": 2635268096
1128
+ },
1129
+ "peak_phys_footprint_bytes": {
1130
+ "python": 4171391040
1131
+ },
1132
+ "mlx_peak_memory_bytes": 3074287404,
1133
+ "results_identical_cold_vs_state_cached": true,
1134
+ "loadavg_at_end": [
1135
+ 7.51,
1136
+ 7.65,
1137
+ 9.0
1138
+ ],
1139
+ "measured_unix": 1790443505.379929,
1140
+ "settings": {
1141
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1142
+ "precision": "8bit",
1143
+ "compute_dtype": null
1144
+ },
1145
+ "raw": "latency/runs/mlx-8bit_1024_10q.json"
1146
+ },
1147
+ {
1148
+ "backend": "mlx-8bit",
1149
+ "n_questions": 1,
1150
+ "state_tokens": 3950,
1151
+ "input_tokens_per_question": [
1152
+ 4015
1153
+ ],
1154
+ "load_s": 3.1569,
1155
+ "cold_s": 2.9904,
1156
+ "warm_median_s": 2.9581,
1157
+ "warm_s": [
1158
+ 2.9525,
1159
+ 2.9624,
1160
+ 2.9581
1161
+ ],
1162
+ "state_cached_median_s": 0.0922,
1163
+ "peak_rss_bytes": {
1164
+ "python": 2639839232
1165
+ },
1166
+ "peak_phys_footprint_bytes": {
1167
+ "python": 5536832896
1168
+ },
1169
+ "mlx_peak_memory_bytes": 3471550184,
1170
+ "results_identical_cold_vs_state_cached": true,
1171
+ "loadavg_at_end": [
1172
+ 8.57,
1173
+ 7.86,
1174
+ 9.04
1175
+ ],
1176
+ "measured_unix": 1790443521.78049,
1177
+ "settings": {
1178
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1179
+ "precision": "8bit",
1180
+ "compute_dtype": null
1181
+ },
1182
+ "raw": "latency/runs/mlx-8bit_4096_1q.json"
1183
+ },
1184
+ {
1185
+ "backend": "mlx-8bit",
1186
+ "n_questions": 10,
1187
+ "state_tokens": 3950,
1188
+ "input_tokens_per_question": [
1189
+ 4015,
1190
+ 3990,
1191
+ 3997,
1192
+ 3983,
1193
+ 3990,
1194
+ 3993,
1195
+ 4015,
1196
+ 3989,
1197
+ 3984,
1198
+ 3991
1199
+ ],
1200
+ "load_s": 3.1665,
1201
+ "cold_s": 3.572,
1202
+ "warm_median_s": 3.5432,
1203
+ "warm_s": [
1204
+ 3.5584,
1205
+ 3.5432,
1206
+ 3.5354
1207
+ ],
1208
+ "state_cached_median_s": 0.6768,
1209
+ "peak_rss_bytes": {
1210
+ "python": 2624536576
1211
+ },
1212
+ "peak_phys_footprint_bytes": {
1213
+ "python": 5698149760
1214
+ },
1215
+ "mlx_peak_memory_bytes": 3471615720,
1216
+ "results_identical_cold_vs_state_cached": true,
1217
+ "loadavg_at_end": [
1218
+ 7.82,
1219
+ 7.73,
1220
+ 8.97
1221
+ ],
1222
+ "measured_unix": 1790443542.268524,
1223
+ "settings": {
1224
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1225
+ "precision": "8bit",
1226
+ "compute_dtype": null
1227
+ },
1228
+ "raw": "latency/runs/mlx-8bit_4096_10q.json"
1229
+ },
1230
+ {
1231
+ "backend": "mlx-8bit",
1232
+ "n_questions": 1,
1233
+ "state_tokens": 24436,
1234
+ "input_tokens_per_question": [
1235
+ 24501
1236
+ ],
1237
+ "load_s": 3.1447,
1238
+ "cold_s": 19.75,
1239
+ "warm_median_s": 19.8614,
1240
+ "warm_s": [
1241
+ 19.8005,
1242
+ 19.8614,
1243
+ 19.9115
1244
+ ],
1245
+ "state_cached_median_s": 0.1562,
1246
+ "peak_rss_bytes": {
1247
+ "python": 2643197952
1248
+ },
1249
+ "peak_phys_footprint_bytes": {
1250
+ "python": 6518807424
1251
+ },
1252
+ "mlx_peak_memory_bytes": 4350994102,
1253
+ "results_identical_cold_vs_state_cached": true,
1254
+ "loadavg_at_end": [
1255
+ 6.7,
1256
+ 7.34,
1257
+ 8.69
1258
+ ],
1259
+ "measured_unix": 1790443626.3357399,
1260
+ "settings": {
1261
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1262
+ "precision": "8bit",
1263
+ "compute_dtype": null
1264
+ },
1265
+ "raw": "latency/runs/mlx-8bit_24576_1q.json"
1266
+ },
1267
+ {
1268
+ "backend": "mlx-8bit",
1269
+ "n_questions": 10,
1270
+ "state_tokens": 24436,
1271
+ "input_tokens_per_question": [
1272
+ 24501,
1273
+ 24476,
1274
+ 24483,
1275
+ 24469,
1276
+ 24476,
1277
+ 24479,
1278
+ 24501,
1279
+ 24475,
1280
+ 24470,
1281
+ 24477
1282
+ ],
1283
+ "load_s": 3.1668,
1284
+ "cold_s": 20.7629,
1285
+ "warm_median_s": 20.7793,
1286
+ "warm_s": [
1287
+ 20.6782,
1288
+ 20.7793,
1289
+ 21.8013
1290
+ ],
1291
+ "state_cached_median_s": 0.9543,
1292
+ "peak_rss_bytes": {
1293
+ "python": 2643705856
1294
+ },
1295
+ "peak_phys_footprint_bytes": {
1296
+ "python": 6522985664
1297
+ },
1298
+ "mlx_peak_memory_bytes": 4350944950,
1299
+ "results_identical_cold_vs_state_cached": true,
1300
+ "loadavg_at_end": [
1301
+ 8.26,
1302
+ 8.35,
1303
+ 8.99
1304
+ ],
1305
+ "measured_unix": 1790443717.49426,
1306
+ "settings": {
1307
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1308
+ "precision": "8bit",
1309
+ "compute_dtype": null
1310
+ },
1311
+ "raw": "latency/runs/mlx-8bit_24576_10q.json"
1312
+ },
1313
+ {
1314
+ "backend": "torch-mps-fp32",
1315
+ "n_questions": 1,
1316
+ "state_tokens": 878,
1317
+ "input_tokens_per_question": [
1318
+ 943
1319
+ ],
1320
+ "load_s": 7.2396,
1321
+ "cold_s": 2.0776,
1322
+ "warm_median_s": 1.82,
1323
+ "warm_s": [
1324
+ 1.82,
1325
+ 1.8185,
1326
+ 1.8657
1327
+ ],
1328
+ "state_cached_median_s": 0.2243,
1329
+ "peak_rss_bytes": {
1330
+ "python": 11777310720
1331
+ },
1332
+ "peak_phys_footprint_bytes": {
1333
+ "python": 9488876864
1334
+ },
1335
+ "results_identical_cold_vs_state_cached": true,
1336
+ "loadavg_at_end": [
1337
+ 11.08,
1338
+ 15.97,
1339
+ 14.21
1340
+ ],
1341
+ "measured_unix": 1790433935.3785548,
1342
+ "settings": {
1343
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1344
+ "device": "mps",
1345
+ "dtype": "float32",
1346
+ "attn_chunk": 1024
1347
+ },
1348
+ "raw": "latency/runs/torch-mps-fp32_1024_1q.json"
1349
+ },
1350
+ {
1351
+ "backend": "torch-mps-fp32",
1352
+ "n_questions": 10,
1353
+ "state_tokens": 878,
1354
+ "input_tokens_per_question": [
1355
+ 943,
1356
+ 918,
1357
+ 925,
1358
+ 911,
1359
+ 918,
1360
+ 921,
1361
+ 943,
1362
+ 917,
1363
+ 912,
1364
+ 919
1365
+ ],
1366
+ "load_s": 5.8844,
1367
+ "cold_s": 3.487,
1368
+ "warm_median_s": 3.126,
1369
+ "warm_s": [
1370
+ 3.126,
1371
+ 3.124,
1372
+ 3.1357
1373
+ ],
1374
+ "state_cached_median_s": 1.534,
1375
+ "peak_rss_bytes": {
1376
+ "python": 11766398976
1377
+ },
1378
+ "peak_phys_footprint_bytes": {
1379
+ "python": 9348826496
1380
+ },
1381
+ "results_identical_cold_vs_state_cached": true,
1382
+ "loadavg_at_end": [
1383
+ 10.92,
1384
+ 15.56,
1385
+ 14.11
1386
+ ],
1387
+ "measured_unix": 1790433960.3376808,
1388
+ "settings": {
1389
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1390
+ "device": "mps",
1391
+ "dtype": "float32",
1392
+ "attn_chunk": 1024
1393
+ },
1394
+ "raw": "latency/runs/torch-mps-fp32_1024_10q.json"
1395
+ },
1396
+ {
1397
+ "backend": "torch-mps-fp32",
1398
+ "n_questions": 1,
1399
+ "state_tokens": 3950,
1400
+ "input_tokens_per_question": [
1401
+ 4015
1402
+ ],
1403
+ "load_s": 6.6133,
1404
+ "cold_s": 7.9626,
1405
+ "warm_median_s": 7.7385,
1406
+ "warm_s": [
1407
+ 7.7135,
1408
+ 7.7385,
1409
+ 7.7569
1410
+ ],
1411
+ "state_cached_median_s": 0.2508,
1412
+ "peak_rss_bytes": {
1413
+ "python": 11774099456
1414
+ },
1415
+ "peak_phys_footprint_bytes": {
1416
+ "python": 9830892736
1417
+ },
1418
+ "results_identical_cold_vs_state_cached": true,
1419
+ "loadavg_at_end": [
1420
+ 8.41,
1421
+ 7.52,
1422
+ 7.06
1423
+ ],
1424
+ "measured_unix": 1790438685.325036,
1425
+ "settings": {
1426
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1427
+ "device": "mps",
1428
+ "dtype": "float32",
1429
+ "attn_chunk": 1024
1430
+ },
1431
+ "raw": "latency/runs/torch-mps-fp32_4096_1q.json"
1432
+ },
1433
+ {
1434
+ "backend": "torch-mps-fp32",
1435
+ "n_questions": 10,
1436
+ "state_tokens": 3950,
1437
+ "input_tokens_per_question": [
1438
+ 4015,
1439
+ 3990,
1440
+ 3997,
1441
+ 3983,
1442
+ 3990,
1443
+ 3993,
1444
+ 4015,
1445
+ 3989,
1446
+ 3984,
1447
+ 3991
1448
+ ],
1449
+ "load_s": 5.8354,
1450
+ "cold_s": 9.4943,
1451
+ "warm_median_s": 9.2321,
1452
+ "warm_s": [
1453
+ 9.1731,
1454
+ 9.2321,
1455
+ 9.2805
1456
+ ],
1457
+ "state_cached_median_s": 1.7373,
1458
+ "peak_rss_bytes": {
1459
+ "python": 11766579200
1460
+ },
1461
+ "peak_phys_footprint_bytes": {
1462
+ "python": 9740780928
1463
+ },
1464
+ "results_identical_cold_vs_state_cached": true,
1465
+ "loadavg_at_end": [
1466
+ 10.4,
1467
+ 8.12,
1468
+ 7.31
1469
+ ],
1470
+ "measured_unix": 1790438734.979529,
1471
+ "settings": {
1472
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1473
+ "device": "mps",
1474
+ "dtype": "float32",
1475
+ "attn_chunk": 1024
1476
+ },
1477
+ "raw": "latency/runs/torch-mps-fp32_4096_10q.json"
1478
+ },
1479
+ {
1480
+ "backend": "torch-mps-fp32",
1481
+ "n_questions": 1,
1482
+ "state_tokens": 24436,
1483
+ "input_tokens_per_question": [
1484
+ 24501
1485
+ ],
1486
+ "load_s": 7.5475,
1487
+ "cold_s": 56.4354,
1488
+ "warm_median_s": 61.9146,
1489
+ "warm_s": [
1490
+ 55.156,
1491
+ 61.9146,
1492
+ 80.3434
1493
+ ],
1494
+ "state_cached_median_s": 0.6,
1495
+ "peak_rss_bytes": {
1496
+ "python": 11774935040
1497
+ },
1498
+ "peak_phys_footprint_bytes": {
1499
+ "python": 13703111296
1500
+ },
1501
+ "results_identical_cold_vs_state_cached": true,
1502
+ "loadavg_at_end": [
1503
+ 7.27,
1504
+ 7.84,
1505
+ 7.44
1506
+ ],
1507
+ "measured_unix": 1790438999.6904268,
1508
+ "settings": {
1509
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1510
+ "device": "mps",
1511
+ "dtype": "float32",
1512
+ "attn_chunk": 1024
1513
+ },
1514
+ "raw": "latency/runs/torch-mps-fp32_24576_1q.json"
1515
+ },
1516
+ {
1517
+ "backend": "torch-mps-fp32",
1518
+ "n_questions": 10,
1519
+ "state_tokens": 24436,
1520
+ "input_tokens_per_question": [
1521
+ 24501,
1522
+ 24476,
1523
+ 24483,
1524
+ 24469,
1525
+ 24476,
1526
+ 24479,
1527
+ 24501,
1528
+ 24475,
1529
+ 24470,
1530
+ 24477
1531
+ ],
1532
+ "load_s": 7.9699,
1533
+ "cold_s": 72.6584,
1534
+ "warm_median_s": 72.3035,
1535
+ "warm_s": [
1536
+ 83.2192,
1537
+ 72.3035,
1538
+ 58.3757
1539
+ ],
1540
+ "state_cached_median_s": 2.7217,
1541
+ "peak_rss_bytes": {
1542
+ "python": 11775229952
1543
+ },
1544
+ "peak_phys_footprint_bytes": {
1545
+ "python": 13640000128
1546
+ },
1547
+ "results_identical_cold_vs_state_cached": true,
1548
+ "loadavg_at_end": [
1549
+ 9.6,
1550
+ 8.35,
1551
+ 7.72
1552
+ ],
1553
+ "measured_unix": 1790439304.235886,
1554
+ "settings": {
1555
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1556
+ "device": "mps",
1557
+ "dtype": "float32",
1558
+ "attn_chunk": 1024
1559
+ },
1560
+ "raw": "latency/runs/torch-mps-fp32_24576_10q.json"
1561
+ }
1562
+ ],
1563
+ "scripts": [
1564
+ "release_2b/latency/latency_run.py",
1565
+ "release_2b/latency/run_all.sh",
1566
+ "release_2b/latency/aggregate.py"
1567
+ ]
1568
+ }
validation/parity/PREDECLARED_RELEASE_GATES_2B.md ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Jev-Style-2B-Decision-v3 — release format gates (pre-declared 2026-09-26, BEFORE any trained-weight format was scored)
2
+
3
+ Reference: HF FP32 CPU, exact v2 block attention, on the released bf16 text-only checkpoint
4
+ runs/macjev/candidate_2b/hf-candidate (SHA256SUMS 279/279 verified). Temperature T = 0.8278650621 (macjev_readout.json).
5
+
6
+ Fixtures (frozen, sha256 in their .stats/.README):
7
+ - base fixture: runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl (35 req / 43 q, 214,700 tok, up to 25,600 tok, overflow, K<=151)
8
+ - gate fixture: runs/macjev/runtime_v2_dev/reference/gate_fixture_n1000.jsonl (1,000 real dev rows, <=4,096 tok)
9
+
10
+ Gates (training plan §11.2, unchanged): F16/bf16 top-1 >= .99; 8-bit top-1 >= .98; |dNLL| <= .02 (fp/16/8-bit);
11
+ accuracy drop Q8_0/MLX-8bit <= 0.3 pp, Q4_K_M <= 1.0 pp (measured on the gate fixture; n=1000 => 0.1 pp per flip).
12
+ Block-mask proof on the base fixture: every question closer to the block reference than to the causal and prefix-causal controls.
13
+
14
+ Release set (user choice): GGUF F16 / Q8_0 / Q4_K_M; MLX bf16 / 8bit (affine g64).
15
+ Recipe 1 (primary): stock llama-quantize defaults; mlx_lm.convert defaults + FP32 norm sidecar + MLXScorerV2.
16
+ Recipe 2 (declared fallback, used ONLY if recipe 1 fails a gate): keep the tied embedding / readout rows at f16
17
+ (llama-quantize --token-embedding-type f16; MLX: quant predicate excluding embed_tokens). Same gates, same fixtures.
18
+ If recipe 2 also fails, that format is NOT shipped. Gates are not relaxed after seeing results.
19
+
20
+ ## Public benchmarks (pre-declared 2026-09-26 22:01:49 AEST, before any 2B benchmark item was scored)
21
+ Timing (erratum added 2026-09-27 03:35 AEST): this section was appended by the same shell command that launched the
22
+ benchmark runs, immediately before the first item (file mtime 22:01:49; run log `runs/macjev/bench_2b/run_trained.log`
23
+ first line `[2026-09-26 22:01:49] START jevbench`; first result seen 22:06:28). The heading originally said
24
+ "22:10" by mistake. The format-gate section above was written when the file was created (21:46:03), before any
25
+ trained-weight format was scored (first format score started 21:53:00).
26
+ - Reported engine: GGUF F16 (runs/macjev/release_2b/gguf/model-f16.gguf, jev-score-v2, Metal), global T = 0.8278650621 only.
27
+ - JevBench v1.4.1 public (231) and elcronos zero-shot sets (tweet_topic / fin_topic / daily_dialog), each run ONCE,
28
+ same metric code as the 0.8B v3 card. No per-benchmark temperatures, no reruns, no prompt changes after seeing scores.
29
+ - Decision Index 0.2: not run locally; requested from the maintainer after release (user decision 2026-09-26).
validation/parity/cross_format_dp.json ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base:torch_fp32_cpu": {
3
+ "n": 43,
4
+ "top1_agree": 1.0,
5
+ "max_abs_dp": 4.782633115096857e-06,
6
+ "mean_max_abs_dp": 6.972375795818997e-07
7
+ },
8
+ "base:gguf_f16": {
9
+ "n": 43,
10
+ "top1_agree": 1.0,
11
+ "max_abs_dp": 0.0006297677378335198,
12
+ "mean_max_abs_dp": 0.00017737997866918594
13
+ },
14
+ "base:gguf_q8_0": {
15
+ "n": 43,
16
+ "top1_agree": 1.0,
17
+ "max_abs_dp": 0.021289327356281362,
18
+ "mean_max_abs_dp": 0.0044768677260933745
19
+ },
20
+ "base:gguf_q4_k_m": {
21
+ "n": 43,
22
+ "top1_agree": 0.9767441860465116,
23
+ "max_abs_dp": 0.15778925110927658,
24
+ "mean_max_abs_dp": 0.047022506153486646
25
+ },
26
+ "base:mlx_bf16": {
27
+ "n": 43,
28
+ "top1_agree": 1.0,
29
+ "max_abs_dp": 0.010765431767972844,
30
+ "mean_max_abs_dp": 0.003890125022245109
31
+ },
32
+ "base:mlx_8bit": {
33
+ "n": 43,
34
+ "top1_agree": 1.0,
35
+ "max_abs_dp": 0.04628699142360499,
36
+ "mean_max_abs_dp": 0.007026009064181663
37
+ },
38
+ "gate:gguf_f16": {
39
+ "n": 1000,
40
+ "top1_agree": 1.0,
41
+ "max_abs_dp": 0.0013728371250730786,
42
+ "mean_max_abs_dp": 0.00012138780447113503
43
+ },
44
+ "gate:gguf_q8_0": {
45
+ "n": 1000,
46
+ "top1_agree": 0.997,
47
+ "max_abs_dp": 0.033479594816581415,
48
+ "mean_max_abs_dp": 0.0020247272188215755
49
+ },
50
+ "gate:gguf_q4_k_m": {
51
+ "n": 1000,
52
+ "top1_agree": 0.957,
53
+ "max_abs_dp": 0.34599917206983777,
54
+ "mean_max_abs_dp": 0.02317308476687987
55
+ },
56
+ "gate:mlx_bf16": {
57
+ "n": 1000,
58
+ "top1_agree": 0.997,
59
+ "max_abs_dp": 0.03549758339084519,
60
+ "mean_max_abs_dp": 0.0025865935031305484
61
+ },
62
+ "gate:mlx_8bit": {
63
+ "n": 1000,
64
+ "top1_agree": 0.996,
65
+ "max_abs_dp": 0.1623697296878569,
66
+ "mean_max_abs_dp": 0.0039011049015487734
67
+ }
68
+ }
validation/runtime/parity_cpu_fp32_t4.log ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ load 4.1s device=cpu threads=4
3
+ [transformers] `causal_conv1d_fn` is falling back to its reference PyTorch implementation because `causal_conv1d` is not installed. This is correct but much slower; install `causal_conv1d` for the optimized kernel.
4
+ [transformers] `chunk_gated_delta_rule` is falling back to its reference PyTorch implementation because `flash-linear-attention` is not installed. This is correct but much slower; install `flash-linear-attention` for the optimized kernel.
5
+ multi_question-0 [450, 364, 356, 345, 319] d=7.391e-06 flips=0 meta_ok=True 28.5s
6
+ multi_question-1 [5416, 5471, 5425, 5517, 5428] d=4.530e-06 flips=0 meta_ok=True 55.1s
7
+ short_sb-themes-0 [105] d=2.980e-07 flips=0 meta_ok=True 8.2s
8
+ short_sb-public_ingests-0 [382] d=6.020e-06 flips=0 meta_ok=True 9.4s
9
+ short_sb-di_retrieval-0 [197] d=2.980e-07 flips=0 meta_ok=True 9.5s
10
+ short_sb-di_knowledge-0 [72] d=1.669e-06 flips=0 meta_ok=True 8.9s
11
+ short_sb-hard_short-0 [517] d=4.411e-06 flips=0 meta_ok=True 10.2s
12
+ short_sb-format-0 [166] d=3.576e-06 flips=0 meta_ok=True 9.3s
13
+ qtype_choice-0 [594] d=5.841e-06 flips=0 meta_ok=True 10.9s
14
+ qtype_noul-0 [246] d=1.609e-06 flips=0 meta_ok=True 10.0s
15
+ qtype_score-0 [873] d=3.934e-06 flips=0 meta_ok=True 12.3s
16
+ multi_block_3k-0 [2927] d=1.132e-05 flips=0 meta_ok=True 24.8s
17
+ multi_block_3k-1 [2932] d=2.146e-06 flips=0 meta_ok=True 23.2s
18
+ multi_block_5k-0 [5011] d=4.798e-06 flips=0 meta_ok=True 37.1s
19
+ multi_block_5k-1 [4767] d=5.364e-06 flips=0 meta_ok=True 36.3s
20
+ multi_block_9k-s0 [9083] d=2.384e-06 flips=0 meta_ok=True 69.7s
21
+ multi_block_9k-s1 [9568] d=2.310e-06 flips=0 meta_ok=True 77.4s
22
+ prefix_boundary-2047 [2178] d=2.623e-06 flips=0 meta_ok=True 19.7s
23
+ prefix_boundary-2048 [2203] d=3.457e-06 flips=0 meta_ok=True 20.8s
24
+ [transformers] `causal_conv1d_update` is falling back to its reference PyTorch implementation because `causal_conv1d` is not installed. This is correct but much slower; install `causal_conv1d` for the optimized kernel.
25
+ [transformers] `fused_recurrent_gated_delta_rule` is falling back to its reference PyTorch implementation because `flash-linear-attention` is not installed. This is correct but much slower; install `flash-linear-attention` for the optimized kernel.
26
+ prefix_boundary-2049 [2289] d=4.530e-06 flips=0 meta_ok=True 28.7s
27
+ prefix_boundary-4095 [4163] d=9.179e-06 flips=0 meta_ok=True 37.4s
28
+ prefix_boundary-4096 [4312] d=4.292e-06 flips=0 meta_ok=True 39.8s
29
+ prefix_boundary-4097 [4180] d=1.669e-06 flips=0 meta_ok=True 36.9s
30
+ prefix_boundary-6144 [6259] d=5.722e-06 flips=0 meta_ok=True 61.6s
31
+ catalogue_overflow-0 [4722] d=6.676e-06 flips=0 meta_ok=True 36.4s
32
+ catalogue_overflow-1 [4721] d=7.153e-06 flips=0 meta_ok=True 36.2s
33
+ catalogue_overflow-2 [4729] d=1.502e-05 flips=0 meta_ok=True 40.0s
34
+ catalogue_overflow_long_state-0 [8806] d=6.437e-06 flips=0 meta_ok=True 64.5s
35
+ many_options-0 [500] d=5.960e-06 flips=0 meta_ok=True 11.5s
36
+ many_options-1 [423] d=3.815e-06 flips=0 meta_ok=True 10.2s
37
+ long_16k-0 [15900] d=6.676e-06 flips=0 meta_ok=True 119.6s
38
+ long_16k-1 [16384] d=2.086e-06 flips=0 meta_ok=True 120.1s
39
+ long_16k-2 [16800] d=5.722e-06 flips=0 meta_ok=True 106.6s
40
+ long_24k-0 [24000] d=3.815e-06 flips=0 meta_ok=True 163.9s
41
+ long_24k-1 [25600] d=4.053e-06 flips=0 meta_ok=True 171.0s
42
+ catalogue_overflow n= 3 max|d|=1.502e-05 flips=0
43
+ catalogue_overflow_long_state n= 1 max|d|=6.437e-06 flips=0
44
+ long_16k n= 3 max|d|=6.676e-06 flips=0
45
+ long_24k n= 2 max|d|=4.053e-06 flips=0
46
+ many_options n= 2 max|d|=5.960e-06 flips=0
47
+ multi_block_3k n= 2 max|d|=1.132e-05 flips=0
48
+ multi_block_5k n= 2 max|d|=5.364e-06 flips=0
49
+ multi_block_9k n= 2 max|d|=2.384e-06 flips=0
50
+ multi_question n=10 max|d|=7.391e-06 flips=0
51
+ prefix_boundary n= 7 max|d|=9.179e-06 flips=0
52
+ qtype_choice n= 1 max|d|=5.841e-06 flips=0
53
+ qtype_noul n= 1 max|d|=1.609e-06 flips=0
54
+ qtype_score n= 1 max|d|=3.934e-06 flips=0
55
+ short_sb n= 6 max|d|=6.020e-06 flips=0
56
+ ALL requests=35 questions=43 max|d|=1.502e-05 flips=0 meta_ok_all=True
validation/runtime/parity_mps_fp32_long.log ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+
2
+ load 4.9s device=mps threads=2
3
+ [transformers] `causal_conv1d_fn` is falling back to its reference PyTorch implementation because `causal_conv1d` is not installed. This is correct but much slower; install `causal_conv1d` for the optimized kernel.
4
+ [transformers] `chunk_gated_delta_rule` is falling back to its reference PyTorch implementation because `flash-linear-attention` is not installed. This is correct but much slower; install `flash-linear-attention` for the optimized kernel.
5
+ long_16k-1 [16384] d=2.742e-06 flips=0 meta_ok=True 33.9s
6
+ long_24k-1 [25600] d=8.821e-06 flips=0 meta_ok=True 57.5s
7
+ long_16k n= 1 max|d|=2.742e-06 flips=0
8
+ long_24k n= 1 max|d|=8.821e-06 flips=0
9
+ ALL requests=2 questions=2 max|d|=8.821e-06 flips=0 meta_ok_all=True
validation/runtime/reverify_main.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "runtime_file": "jev_style_decision.py",
3
+ "runtime_sha256": "5ecba24cdfff4f0e043c2ea48c9c7f514907c6804f265906e21dec0f1b20dbd6",
4
+ "shared_core_sha256": "56deae095206fc59b9a30d8626ac5d27cacddcd94d7c7f08f6f4bc53d326a002",
5
+ "what": "PyTorch FP32 CPU, requests with <= 6000 tokens per question",
6
+ "loaded_from": "runs/macjev/hf_staging/Jev-Style-2B-Decision-v3",
7
+ "verify_manifest": true,
8
+ "compared_with": {
9
+ "file": "runs/macjev/runtime_v2_dev/reference_trained/ref_block_fp32.jsonl",
10
+ "sha256": "09ff7d4fdd97c52c28429139a4f8d53dd224a5e0bcc69b1a318978abad5e136c"
11
+ },
12
+ "fixture": {
13
+ "file": "runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl",
14
+ "sha256": "2b16d25eafa70c00a1106ba51b94890821b51ecb4fa393abb12403c8739ef314"
15
+ },
16
+ "questions": 34,
17
+ "skipped_questions": 9,
18
+ "max_abs_score_diff": 1.5020370483398438e-05,
19
+ "top1_same": "34/34",
20
+ "token_or_overflow_mismatches": [],
21
+ "seconds": 411.6,
22
+ "finished_unix": 1790449904.6816142
23
+ }
validation/runtime/reverify_main_mps_long.log ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+
2
+ load 4.6s device=mps threads=2
3
+ [transformers] `causal_conv1d_fn` is falling back to its reference PyTorch implementation because `causal_conv1d` is not installed. This is correct but much slower; install `causal_conv1d` for the optimized kernel.
4
+ [transformers] `chunk_gated_delta_rule` is falling back to its reference PyTorch implementation because `flash-linear-attention` is not installed. This is correct but much slower; install `flash-linear-attention` for the optimized kernel.
5
+ long_16k-1 [16384] d=2.742e-06 flips=0 meta_ok=True 37.8s
6
+ long_24k-1 [25600] d=8.821e-06 flips=0 meta_ok=True 61.0s
7
+ long_16k n= 1 max|d|=2.742e-06 flips=0
8
+ long_24k n= 1 max|d|=8.821e-06 flips=0
9
+ ALL requests=2 questions=2 max|d|=8.821e-06 flips=0 meta_ok_all=True
validation/runtime/v_jevstyle_e2e.json ADDED
@@ -0,0 +1,80 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "systemone_2b_release": {
3
+ "model": "jev-style-2b-decision-v3",
4
+ "answers": {
5
+ "team": {
6
+ "type": "choice",
7
+ "choice": "billing",
8
+ "confidence": 0.9802240272350504,
9
+ "probabilities": {
10
+ "billing": 0.9868160181567003,
11
+ "tech": 0.0073588517073361745,
12
+ "sales": 0.0058251301359634414
13
+ }
14
+ },
15
+ "urgent": {
16
+ "type": "noul",
17
+ "noul": 0.6980439916465264
18
+ },
19
+ "anger": {
20
+ "type": "score",
21
+ "score": 1.8349445223126968,
22
+ "confidence": 0.7639580848277534,
23
+ "legend": {
24
+ "0": "calm",
25
+ "1": "annoyed",
26
+ "2": "angry"
27
+ },
28
+ "probabilities": {
29
+ "0": 0.007694200905805288,
30
+ "1": 0.14966707587569247,
31
+ "2": 0.8426387232185022
32
+ }
33
+ }
34
+ },
35
+ "usage": {
36
+ "input_tokens": 168,
37
+ "state_tokens": 48,
38
+ "output_tokens": 0
39
+ },
40
+ "latency_ms": 17843.8,
41
+ "timing": {
42
+ "total_ms": 17843.8
43
+ },
44
+ "backend": "torch-cpu"
45
+ },
46
+ "models_2b_release": {
47
+ "object": "list",
48
+ "data": [
49
+ {
50
+ "id": "jev-style-2b-decision-v3",
51
+ "context_tokens": 25600,
52
+ "head_max_tokens": 25600,
53
+ "backend": "torch-cpu"
54
+ }
55
+ ],
56
+ "models": [
57
+ {
58
+ "name": "jev-style-2b-decision-v3",
59
+ "release_date": null,
60
+ "description": "Jev-Style 2B Decision v3: local typed decisions (noul / choice / score)"
61
+ }
62
+ ]
63
+ },
64
+ "over_budget": [
65
+ 422,
66
+ "input_budget_exceeded"
67
+ ],
68
+ "k60_answers": {
69
+ "q0": "opt1",
70
+ "q1": "opt1",
71
+ "q2": "opt2"
72
+ },
73
+ "adapter_vs_direct": [
74
+ 2.220446049250313e-16,
75
+ 0.0
76
+ ],
77
+ "default_release_model_id": "jev-style-0.8b-decision-v3",
78
+ "default_release_same_answers": true,
79
+ "text_path_vs_dev_ref_max_abs_d": 3.4570693969726562e-06
80
+ }
validation/runtime/v_tiny_and_render.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "render": {
3
+ "question_errors": 2,
4
+ "qerr_kinds": [
5
+ "question text ('ins') must be a non-empty string"
6
+ ],
7
+ "n_rendered_equal": 406,
8
+ "catalogue_overflow": 43,
9
+ "both_rejected": 2,
10
+ "edges": {
11
+ "short_2048": 2014,
12
+ "short_2049": [
13
+ 2015,
14
+ 2074
15
+ ]
16
+ },
17
+ "big_k_blocks": 14,
18
+ "big_k_tokens": 23102
19
+ },
20
+ "tiny": {
21
+ "cases": 24,
22
+ "max_abs_d_vs_dev_ref": 2.1457672119140625e-06
23
+ }
24
+ }