Text Classification
jev-style
Safetensors
Transformers
qwen3_5_text
text-generation
decision-model
system-one
calibration
classification
long-context
multilingual
qwen3.5
on-device
llm-routing
guardrails
Instructions to use chaoliangUNSW/Jev-Style-0.8B-Decision-v3 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- jev-style
How to use chaoliangUNSW/Jev-Style-0.8B-Decision-v3 with jev-style:
pip install "jev-style[torch]"
from jev_style import JevStyle, noul, choice js = JevStyle.from_pretrained("chaoliangUNSW/Jev-Style-0.8B-Decision-v3") out = js.decide("I was charged twice for one order.", { "billing": noul("This message is about billing."), "team": choice("Which team should handle it?", ["billing", "shipping", "tech"]), }) print(out["answers"]["team"]["choice"]) - Transformers
How to use chaoliangUNSW/Jev-Style-0.8B-Decision-v3 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="chaoliangUNSW/Jev-Style-0.8B-Decision-v3")# Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("chaoliangUNSW/Jev-Style-0.8B-Decision-v3") model = AutoModelForCausalLM.from_pretrained("chaoliangUNSW/Jev-Style-0.8B-Decision-v3", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Release Jev-Style-0.8B-Decision-v3
Browse files- .gitattributes +11 -0
- LICENSE +202 -0
- NOTICE +27 -0
- README.md +448 -0
- chat_template.jinja +154 -0
- config.json +75 -0
- figures/beyond_laya.json +94 -0
- figures/beyond_laya.png +3 -0
- figures/beyond_laya.svg +407 -0
- figures/calibration.data.json +99 -0
- figures/calibration.png +3 -0
- figures/calibration.svg +275 -0
- figures/design_table.data.json +175 -0
- figures/design_table.png +3 -0
- figures/design_table.svg +301 -0
- figures/headline_typed.data.json +90 -0
- figures/headline_typed.png +3 -0
- figures/headline_typed.svg +368 -0
- figures/jevbench.data.json +60 -0
- figures/jevbench.png +3 -0
- figures/jevbench.svg +204 -0
- figures/latency.data.json +51 -0
- figures/latency.png +3 -0
- figures/latency.svg +447 -0
- figures/long_context.json +216 -0
- figures/long_context.png +3 -0
- figures/long_context.svg +342 -0
- figures/multilingual.data.json +486 -0
- figures/multilingual.png +3 -0
- figures/multilingual.svg +1333 -0
- figures/quantization.data.json +151 -0
- figures/quantization.png +3 -0
- figures/quantization.svg +471 -0
- figures/zeroshot.json +58 -0
- figures/zeroshot.png +3 -0
- figures/zeroshot.svg +244 -0
- generation_config.json +6 -0
- jev_style_decision.py +517 -0
- manifest.json +181 -0
- model.safetensors +3 -0
- readout_config.json +247 -0
- release_config.json +186 -0
- requirements.txt +5 -0
- tokenizer.json +3 -0
- tokenizer_config.json +32 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,14 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
figures/beyond_laya.png filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
figures/calibration.png filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
figures/design_table.png filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
figures/headline_typed.png filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
figures/jevbench.png filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
figures/latency.png filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
figures/long_context.png filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
figures/multilingual.png filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
figures/quantization.png filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
figures/zeroshot.png filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
LICENSE
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Apache License
|
| 3 |
+
Version 2.0, January 2004
|
| 4 |
+
http://www.apache.org/licenses/
|
| 5 |
+
|
| 6 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 7 |
+
|
| 8 |
+
1. Definitions.
|
| 9 |
+
|
| 10 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 11 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 12 |
+
|
| 13 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 14 |
+
the copyright owner that is granting the License.
|
| 15 |
+
|
| 16 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 17 |
+
other entities that control, are controlled by, or are under common
|
| 18 |
+
control with that entity. For the purposes of this definition,
|
| 19 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 20 |
+
direction or management of such entity, whether by contract or
|
| 21 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 22 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 23 |
+
|
| 24 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 25 |
+
exercising permissions granted by this License.
|
| 26 |
+
|
| 27 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 28 |
+
including but not limited to software source code, documentation
|
| 29 |
+
source, and configuration files.
|
| 30 |
+
|
| 31 |
+
"Object" form shall mean any form resulting from mechanical
|
| 32 |
+
transformation or translation of a Source form, including but
|
| 33 |
+
not limited to compiled object code, generated documentation,
|
| 34 |
+
and conversions to other media types.
|
| 35 |
+
|
| 36 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 37 |
+
Object form, made available under the License, as indicated by a
|
| 38 |
+
copyright notice that is included in or attached to the work
|
| 39 |
+
(an example is provided in the Appendix below).
|
| 40 |
+
|
| 41 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 42 |
+
form, that is based on (or derived from) the Work and for which the
|
| 43 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 44 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 45 |
+
of this License, Derivative Works shall not include works that remain
|
| 46 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 47 |
+
the Work and Derivative Works thereof.
|
| 48 |
+
|
| 49 |
+
"Contribution" shall mean any work of authorship, including
|
| 50 |
+
the original version of the Work and any modifications or additions
|
| 51 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 52 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 53 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 54 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 55 |
+
means any form of electronic, verbal, or written communication sent
|
| 56 |
+
to the Licensor or its representatives, including but not limited to
|
| 57 |
+
communication on electronic mailing lists, source code control systems,
|
| 58 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 59 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 60 |
+
excluding communication that is conspicuously marked or otherwise
|
| 61 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 62 |
+
|
| 63 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 64 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 65 |
+
subsequently incorporated within the Work.
|
| 66 |
+
|
| 67 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 68 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 69 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 70 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 71 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 72 |
+
Work and such Derivative Works in Source or Object form.
|
| 73 |
+
|
| 74 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 75 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 76 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 77 |
+
(except as stated in this section) patent license to make, have made,
|
| 78 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 79 |
+
where such license applies only to those patent claims licensable
|
| 80 |
+
by such Contributor that are necessarily infringed by their
|
| 81 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 82 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 83 |
+
institute patent litigation against any entity (including a
|
| 84 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 85 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 86 |
+
or contributory patent infringement, then any patent licenses
|
| 87 |
+
granted to You under this License for that Work shall terminate
|
| 88 |
+
as of the date such litigation is filed.
|
| 89 |
+
|
| 90 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 91 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 92 |
+
modifications, and in Source or Object form, provided that You
|
| 93 |
+
meet the following conditions:
|
| 94 |
+
|
| 95 |
+
(a) You must give any other recipients of the Work or
|
| 96 |
+
Derivative Works a copy of this License; and
|
| 97 |
+
|
| 98 |
+
(b) You must cause any modified files to carry prominent notices
|
| 99 |
+
stating that You changed the files; and
|
| 100 |
+
|
| 101 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 102 |
+
that You distribute, all copyright, patent, trademark, and
|
| 103 |
+
attribution notices from the Source form of the Work,
|
| 104 |
+
excluding those notices that do not pertain to any part of
|
| 105 |
+
the Derivative Works; and
|
| 106 |
+
|
| 107 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 108 |
+
distribution, then any Derivative Works that You distribute must
|
| 109 |
+
include a readable copy of the attribution notices contained
|
| 110 |
+
within such NOTICE file, excluding those notices that do not
|
| 111 |
+
pertain to any part of the Derivative Works, in at least one
|
| 112 |
+
of the following places: within a NOTICE text file distributed
|
| 113 |
+
as part of the Derivative Works; within the Source form or
|
| 114 |
+
documentation, if provided along with the Derivative Works; or,
|
| 115 |
+
within a display generated by the Derivative Works, if and
|
| 116 |
+
wherever such third-party notices normally appear. The contents
|
| 117 |
+
of the NOTICE file are for informational purposes only and
|
| 118 |
+
do not modify the License. You may add Your own attribution
|
| 119 |
+
notices within Derivative Works that You distribute, alongside
|
| 120 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 121 |
+
that such additional attribution notices cannot be construed
|
| 122 |
+
as modifying the License.
|
| 123 |
+
|
| 124 |
+
You may add Your own copyright statement to Your modifications and
|
| 125 |
+
may provide additional or different license terms and conditions
|
| 126 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 127 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 128 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 129 |
+
the conditions stated in this License.
|
| 130 |
+
|
| 131 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 132 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 133 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 134 |
+
this License, without any additional terms or conditions.
|
| 135 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 136 |
+
the terms of any separate license agreement you may have executed
|
| 137 |
+
with Licensor regarding such Contributions.
|
| 138 |
+
|
| 139 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 140 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 141 |
+
except as required for reasonable and customary use in describing the
|
| 142 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 143 |
+
|
| 144 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 145 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 146 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 147 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 148 |
+
implied, including, without limitation, any warranties or conditions
|
| 149 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 150 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 151 |
+
appropriateness of using or redistributing the Work and assume any
|
| 152 |
+
risks associated with Your exercise of permissions under this License.
|
| 153 |
+
|
| 154 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 155 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 156 |
+
unless required by applicable law (such as deliberate and grossly
|
| 157 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 158 |
+
liable to You for damages, including any direct, indirect, special,
|
| 159 |
+
incidental, or consequential damages of any character arising as a
|
| 160 |
+
result of this License or out of the use or inability to use the
|
| 161 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 162 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 163 |
+
other commercial damages or losses), even if such Contributor
|
| 164 |
+
has been advised of the possibility of such damages.
|
| 165 |
+
|
| 166 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 167 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 168 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 169 |
+
or other liability obligations and/or rights consistent with this
|
| 170 |
+
License. However, in accepting such obligations, You may act only
|
| 171 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 172 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 173 |
+
defend, and hold each Contributor harmless for any liability
|
| 174 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 175 |
+
of your accepting any such warranty or additional liability.
|
| 176 |
+
|
| 177 |
+
END OF TERMS AND CONDITIONS
|
| 178 |
+
|
| 179 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 180 |
+
|
| 181 |
+
To apply the Apache License to your work, attach the following
|
| 182 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 183 |
+
replaced with your own identifying information. (Don't include
|
| 184 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 185 |
+
comment syntax for the file format. We also recommend that a
|
| 186 |
+
file or class name and description of purpose be included on the
|
| 187 |
+
same "printed page" as the copyright notice for easier
|
| 188 |
+
identification within third-party archives.
|
| 189 |
+
|
| 190 |
+
Copyright 2026 Alibaba Cloud
|
| 191 |
+
|
| 192 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 193 |
+
you may not use this file except in compliance with the License.
|
| 194 |
+
You may obtain a copy of the License at
|
| 195 |
+
|
| 196 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 197 |
+
|
| 198 |
+
Unless required by applicable law or agreed to in writing, software
|
| 199 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 200 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 201 |
+
See the License for the specific language governing permissions and
|
| 202 |
+
limitations under the License.
|
NOTICE
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
chaoliangUNSW/Jev-Style-0.8B-Decision-v3
|
| 2 |
+
Copyright 2026 chaoliangUNSW. Licensed under the Apache License, Version 2.0 (see LICENSE).
|
| 3 |
+
|
| 4 |
+
This model is a fine-tuned derivative of Qwen3.5-0.8B (https://huggingface.co/Qwen/Qwen3.5-0.8B,
|
| 5 |
+
revision 2fc06364715b967f1860aea9cf38778875588b17), Copyright 2026 Alibaba Cloud, licensed under the Apache
|
| 6 |
+
License, Version 2.0. The LICENSE file in this repository is the license file distributed with Qwen3.5-0.8B.
|
| 7 |
+
|
| 8 |
+
Modifications relative to Qwen3.5-0.8B:
|
| 9 |
+
- all text-model weights were fine-tuned (full fine-tuning, bf16 training) to score typed decision questions
|
| 10 |
+
(choice / score / true-false) with a verdict readout: logit(" yes") - logit(" no") at one " ->" slot per
|
| 11 |
+
option; calibration temperatures were fitted afterwards (readout_config.json);
|
| 12 |
+
- the vision tower (model.visual.*) and the multi-token-prediction head (mtp.*) were removed; the checkpoint
|
| 13 |
+
is a text-only Qwen3_5ForCausalLM with tied input/output embeddings;
|
| 14 |
+
- added the runtime script, readout/release configuration files, the integrity manifest and this NOTICE.
|
| 15 |
+
|
| 16 |
+
The question types (choice / score / noul) follow the typed-decision convention of Laya
|
| 17 |
+
(https://github.com/NandhaKishorM/laya, Apache-2.0) so both models can be evaluated on the same
|
| 18 |
+
inputs. No Laya code or weights are included.
|
| 19 |
+
|
| 20 |
+
Third generation (v3) of the Jev-Style decision series. Earlier generations: v1 =
|
| 21 |
+
chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision (public GGUF release: chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF) and
|
| 22 |
+
v2 = chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2, both 2B models built on Qwen3.5-2B-Base. v3 is fine-tuned from
|
| 23 |
+
Qwen3.5-0.8B; no weights of v1 or v2 were reused.
|
| 24 |
+
|
| 25 |
+
Not affiliated with, endorsed by or connected to TypeSafe or Jev. "Jev-Style" only describes the kind
|
| 26 |
+
of model (a small typed-decision model in a similar style); no Jev weights, code or outputs are included.
|
| 27 |
+
Not affiliated with or endorsed by Alibaba Cloud / the Qwen team or the Laya authors.
|
README.md
ADDED
|
@@ -0,0 +1,448 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
base_model: Qwen/Qwen3.5-0.8B
|
| 4 |
+
base_model_relation: finetune
|
| 5 |
+
library_name: transformers
|
| 6 |
+
pipeline_tag: text-classification
|
| 7 |
+
language:
|
| 8 |
+
- en
|
| 9 |
+
- zh
|
| 10 |
+
- ar
|
| 11 |
+
- bg
|
| 12 |
+
- de
|
| 13 |
+
- el
|
| 14 |
+
- es
|
| 15 |
+
- fr
|
| 16 |
+
- hi
|
| 17 |
+
- ja
|
| 18 |
+
- ko
|
| 19 |
+
- pt
|
| 20 |
+
- ru
|
| 21 |
+
- sw
|
| 22 |
+
- ta
|
| 23 |
+
- th
|
| 24 |
+
- tr
|
| 25 |
+
- ur
|
| 26 |
+
- vi
|
| 27 |
+
tags:
|
| 28 |
+
- decision-model
|
| 29 |
+
- jev-style
|
| 30 |
+
- system-one
|
| 31 |
+
- calibration
|
| 32 |
+
- classification
|
| 33 |
+
- long-context
|
| 34 |
+
- multilingual
|
| 35 |
+
- qwen3.5
|
| 36 |
+
---
|
| 37 |
+
|
| 38 |
+
# Jev-Style-0.8B-Decision-v3
|
| 39 |
+
|
| 40 |
+
**Jev-Style decision series:** [v1 · 2B](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF) → [v2 · 2B](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2) → **v3 · 0.8B (this model)** · **Website:** [jevstyle.com](https://jevstyle.com)
|
| 41 |
+
|
| 42 |
+
**One state. One pass. Every option scored.**
|
| 43 |
+
|
| 44 |
+
Jev-Style v3 is a 0.8B decision model. It takes inputs of up to 25,600 tokens, reads the state once and returns a calibrated
|
| 45 |
+
probability for every option of every question you ask about it. There is no letter cap on options, and it was
|
| 46 |
+
evaluated in 51 languages.
|
| 47 |
+
|
| 48 |
+
| **0.8B** | **25,600 tokens** | **77 options** | **51 languages** | **0.53 GB** |
|
| 49 |
+
|:---:|:---:|:---:|:---:|:---:|
|
| 50 |
+
| parameters, full fine-tune | input, preregistered 25K claim passed | scored in one pass (largest tested) | evaluated on MASSIVE intent | 4-bit Q4_K_M; same top-1 as FP32 on 240/240 parity rows |
|
| 51 |
+
|
| 52 |
+
| Build | Size | Runtime |
|
| 53 |
+
|---|---:|---|
|
| 54 |
+
| **Transformers safetensors (bf16) · this repository** | 1.50 GB | PyTorch on CUDA, Apple MPS or CPU (`jev_style_decision.py`) |
|
| 55 |
+
| [GGUF F16 / Q8_0 / Q4_K_M](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-GGUF) | 1.52 / 0.81 / 0.53 GB | llama.cpp + the bundled `jev-score` scorer |
|
| 56 |
+
| [MLX bf16](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-bf16) | 1.50 GB | Apple silicon, mlx-lm |
|
| 57 |
+
| [MLX 8-bit](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-8bit) | 0.80 GB | Apple silicon, mlx-lm |
|
| 58 |
+
|
| 59 |
+
## Highlights
|
| 60 |
+
|
| 61 |
+
- **79.2% on 2,000 typed decisions: +6.4 points over Jev, +5.7 over our 2B v2, +2.6 over Laya's typed
|
| 62 |
+
checkpoint.** Its Brier score is **3.2× lower than Jev's** (0.046 vs 0.148). v3 and Laya typed trained on
|
| 63 |
+
this dataset's train split; Jev's number is zero-shot, from the dataset card.
|
| 64 |
+
- **Up to +30.3 points over the best official Laya checkpoint on five decision tasks**, on identical rows: +30.3 on model routing,
|
| 65 |
+
+29.5 macro-F1 on toxicity, +29.4 on the 37 locales held out of MASSIVE training, +19.0 on 77-way Banking77
|
| 66 |
+
and +7.2 balanced accuracy on jailbreak detection. Every paired 95% CI excludes zero.
|
| 67 |
+
- **Ahead of Laya multilingual in 51 of 51 languages.** MASSIVE intent macro accuracy is 71.7% vs 40.1% (+31.7 points),
|
| 68 |
+
and v3 stays above 3× chance in every language.
|
| 69 |
+
- **4.5× lower NLL, 3.0× lower Brier and 5.5× lower ECE** than the best Laya checkpoint as deployed (with its shipped
|
| 70 |
+
temperatures), across 49 suites and 17,416 paired rows.
|
| 71 |
+
- **25,600-token inputs, with the preregistered 25K claim passed.** At 24K tokens v3 answers **98.3%** of 1,280
|
| 72 |
+
real items correctly. The question-only and state-swap controls stay at chance, and the preregistered controlled
|
| 73 |
+
accuracy is within 1.1 points of the 2K–4K reference.
|
| 74 |
+
- **64.1% zero-shot on JevBench v1.4.1**, ahead of Laya (58.4%) and every Qwen3.5-0.8B-based system on the board
|
| 75 |
+
(point estimates on 231 public items). On two topic sets it never trained on, v3 leads English Laya by +12.3 and
|
| 76 |
+
+12.5 points, and it comes within 4 points of Jev on tweet_topic.
|
| 77 |
+
- **Up to 4.6× faster than a Laya-architecture engine when 10 questions share one 4K-token state** (1,381 ms vs
|
| 78 |
+
6,364 ms with the GGUF runtime's `many_mode="batched"`; the engine is our round-1 MacLaya-4K, one call per question,
|
| 79 |
+
not an official Laya checkpoint). The
|
| 80 |
+
**0.53 GB Q4_K_M file matched full precision on 240 of 240 parity rows** (plus 6 of 6 at 16K and 25.6K tokens).
|
| 81 |
+
|
| 82 |
+
Each result below states its protocol and source under the figure.
|
| 83 |
+
|
| 84 |
+
## Results
|
| 85 |
+
|
| 86 |
+
### Typed decisions: 0.8B beats the 2B models and Jev
|
| 87 |
+
|
| 88 |
+

|
| 89 |
+
|
| 90 |
+
At 0.8B parameters, v3 scores **79.2%** on the 2,000 typed decisions. That is **+6.4 points over Jev**, +5.7 over
|
| 91 |
+
our 2B v2 and +2.6 over Laya's typed checkpoint, and the Brier score is **3.2× lower than Jev's** (0.046 vs 0.148).
|
| 92 |
+
|
| 93 |
+
<sub>Typed-decisions test set (LocalLLaMA/typed-decisions), 2,000 decisions from 400 states. In-domain for v3 and Laya typed (both trained on its train split); zero-shot for Jev (numbers from the dataset card, measured through the Jev API on all 2,000 decisions). Laya: official typed-decisions checkpoint re-run by us on identical rows with its shipped temperature. 2B v1/v2: teacher agreement as reported on the v2 card (same 2,000 decisions, scored by that card's harness; v1 was not trained on typed decisions, v2's training pool included typed workflow decisions). Jev's accuracy is published as 0.727, so the gap is 6.40–6.50 points. v3: 1,583 / 2,000 correct, 95% CI 77.3–80.9% (Wilson); v3 minus Laya typed, paired bootstrap 95% CI +1.0 to +4.2 points.</sub>
|
| 94 |
+
|
| 95 |
+
**Head-to-head against Laya's typed checkpoint.** Both models trained on this dataset's train split, and v3 wins on all four
|
| 96 |
+
metrics, each with a paired 95% CI that excludes zero:
|
| 97 |
+
|
| 98 |
+
| Metric (2,000 decisions) | Laya typed-decisions checkpoint | **Jev-Style 0.8B v3** | Difference, paired 95% CI |
|
| 99 |
+
|---|---:|---:|---|
|
| 100 |
+
| Accuracy ↑ | 76.6% | **79.2%** | +2.6 pts [+1.0, +4.2] |
|
| 101 |
+
| Soft accuracy ↑ | 47.1% | **52.4%** | +5.4 pts [+5.0, +5.7] |
|
| 102 |
+
| Brier vs soft labels ↓ | 0.061 | **0.046** | −0.016 [−0.019, −0.012] |
|
| 103 |
+
| Score-question MAE ↓ | 0.242 | **0.195** | −0.047 [−0.060, −0.035] |
|
| 104 |
+
|
| 105 |
+
<sub>Both in-domain; v3 also trained on 27,300 synthetic typed items from other workflows. Laya: official checkpoint re-run by us on identical rows with its shipped temperature. Paired case-cluster bootstrap within suites, 2,000 resamples.</sub>
|
| 106 |
+
|
| 107 |
+
### Beyond Laya: up to +30 points
|
| 108 |
+
|
| 109 |
+

|
| 110 |
+
|
| 111 |
+
**On five decision tasks scored on identical rows, the 0.8B v3 beats the best official Laya checkpoint on every
|
| 112 |
+
one:** +19.0 points on 77-way Banking77, +7.2 balanced accuracy on jailbreak detection, +29.5 macro-F1 on
|
| 113 |
+
toxicity, +30.3 on model routing and +29.4 across the 37 locales held out of MASSIVE training.
|
| 114 |
+
|
| 115 |
+
<sub>Laya numbers: official checkpoints (English, typed-decisions, multilingual) re-run by us on identical rows with their shipped temperatures and default token budgets; the best of the three is shown per task. v3 trained on tasks of the same kind from other datasets, never on these evaluation rows: intent (CLINC150/HWU64; Banking77 never trained), jailbreak (other permissive sets plus teacher data), toxicity (civil_comments plus teacher data; toxic-chat is evaluation-only), routing (teacher-written; the gsm8k/mbpp/AG rows are evaluation-only), MASSIVE in 14 other locales (no MASSIVE rows in these 37). n = 400 / 400 / 400 / 399 / 3,700 (37 × 100). Every gap's paired 95% bootstrap CI excludes zero.</sub>
|
| 116 |
+
|
| 117 |
+
### 51 languages, 51 wins
|
| 118 |
+
|
| 119 |
+

|
| 120 |
+
|
| 121 |
+
**One 0.8B model, 51 languages, 51 wins over Laya.** On MASSIVE intent (20 options per question) v3 averages
|
| 122 |
+
**71.7%** across 51 languages, against 40.1% for the official Laya multilingual checkpoint (+31.7 points). It
|
| 123 |
+
beats Laya multilingual in every one of the 51 languages, by at least 11 points, and stays above 3× chance in all
|
| 124 |
+
of them. That includes the 37 locales held out of MASSIVE training (65.5% vs 36.1%), 32 of them outside the 19
|
| 125 |
+
fine-tuning languages.
|
| 126 |
+
|
| 127 |
+
<sub>MASSIVE intent (mteb/amazon_massive_intent) test rows, 100 per language, 20 candidate intents per row (chance 5%, 3× chance 15%); accuracy = top-scored option. v3: in-domain for the 14 trained locales, held out for the other 37 (vi/th/el/ur had about 1.3K translated-NLI training rows each; zh-TW shares Chinese with zh-CN; a 69-row multilingual jailbreak set in training may include a few prompts in other held-out languages). Laya: official multilingual checkpoint re-run by us on identical rows with its shipped temperature and default token budget (held-out for Laya). v3 is also ahead of the best of the three official Laya checkpoints in all 51 languages (per-language point estimates on 100 rows each, smallest gap 10 points). Paired 95% CI of the 51-language macro difference: +30.1 to +33.1 points.</sub>
|
| 128 |
+
|
| 129 |
+
### Probabilities you can act on
|
| 130 |
+
|
| 131 |
+

|
| 132 |
+
|
| 133 |
+
**Across 49 suites and 17,416 identical rows, v3's probabilities beat the best official Laya checkpoint on all three
|
| 134 |
+
probability-quality metrics:** **4.5× lower NLL** (0.493 vs 2.213), **3.0× lower Brier** (0.239 vs 0.712) and
|
| 135 |
+
**5.5× lower ECE** (0.054 vs 0.299).
|
| 136 |
+
|
| 137 |
+
<sub>Macro average over 49 suites, 17,416 identical rows for both models. Mixed protocol for both v3 and Laya (in-domain on some suites, held-out on others). As deployed: v3 with its supplied temperatures; Laya with the shipped temperatures of its official checkpoints, re-run by us on identical rows. Best Laya = best of the three official checkpoints per metric (multilingual on all three). Paired case-cluster bootstrap 95% CIs (2,000 resamples) of every difference exclude zero.</sub>
|
| 138 |
+
|
| 139 |
+
**Stable under option shuffling.** When the options are presented in a different order, v3 changed its answer on
|
| 140 |
+
**1 of 200** option-order pairs (0.5%), against 22 of 200 (11.0%) for the best Laya checkpoint.
|
| 141 |
+
|
| 142 |
+
<sub>MASSIVE intent English, 200 option-permutation pairs, identical rows. In-domain for v3, held-out for Laya (typed-decisions checkpoint, the best of the three here). Paired 95% CI of the difference: −15.0 to −6.5 points.</sub>
|
| 143 |
+
|
| 144 |
+
### Long context: flat from 1K to 24K tokens
|
| 145 |
+
|
| 146 |
+

|
| 147 |
+
|
| 148 |
+
**v3 reads documents far past Laya's 512 / 1,024-token default budgets.** Controlled accuracy stays within 3.8
|
| 149 |
+
points across all seven length bins. At 24K it is 55.3%, 1.1 points from the 2K–4K reference (56.4%) and well
|
| 150 |
+
inside the preregistered ±5-point limit, so **the 25K claim passed**. In plain accuracy, v3 answers **98.3%**
|
| 151 |
+
(1,258 of 1,280) of the real 24K-token items correctly and at least 96.9% in every length bin. The same questions
|
| 152 |
+
with the state removed or swapped for another item's state fall to chance (28.9% and 28.5%, against 28.4% chance
|
| 153 |
+
at 24K), so the answers cannot be recovered from the question alone.
|
| 154 |
+
|
| 155 |
+
<sub>v3 only. Laya's default input budget is 512 tokens (English) / 1,024 (multilingual, typed) per the Laya README, so Laya is not plotted. Suite long_grid_plus, English and Chinese documents: preregistered 2026-09-24 and amended before any model was scored (+96 items per 24K depth decile, thresholds unchanged); 320 items per bin, 1,280 at 24K. Controlled accuracy = the real item is correct AND its question-only and state-swap controls pass; both controls are at chance in every length bin. 25K claim rule: |24K − 2K–4K reference| ≤ 5 points and every 24K evidence-depth decile within 10 points of it.</sub>
|
| 156 |
+
|
| 157 |
+
### JevBench: ahead of Laya and every Qwen3.5-0.8B-based system
|
| 158 |
+
|
| 159 |
+

|
| 160 |
+
|
| 161 |
+
**On the 231 public JevBench v1.4.1 items, v3 scores 64.1% zero-shot**: 5.6 points above Laya, and ahead of every
|
| 162 |
+
Qwen3.5-0.8B-based system on the board, including a dedicated 0.8B decision fine-tune (+4.8 points) and
|
| 163 |
+
SimpleJev on the same base (+9.5 points). Every answer is a valid option (231 of 231), because v3 can only score
|
| 164 |
+
the options it is given.
|
| 165 |
+
|
| 166 |
+
<sub>JevBench v1.4.1, public items only (231). v3: self-run zero-shot with the vendored official harness (commit 24b9b5c), 148 / 231 correct, 95% CI 57.7–70.0% (Wilson); training-pool contamination scan: 0 hits; not an official leaderboard entry. Other rows: public accuracy as published in the board's [v1.4.1 results file](https://github.com/fstandhartinger/jevbench). Shown: Laya plus every Qwen3.5-0.8B-based system on the board; other board systems are not shown. Laya's and M. Ghafiri's scores lie inside v3's 95% CI, so those two leads are point estimates, not significant at n = 231.</sub>
|
| 167 |
+
|
| 168 |
+
### Zero-shot topics: +12 points over English Laya
|
| 169 |
+
|
| 170 |
+

|
| 171 |
+
|
| 172 |
+
**On two topic sets it never trained on, v3 leads English Laya by +12.3 points on tweet_topic** (75.5% vs 63.2%)
|
| 173 |
+
**and +12.5 points on the 20-way fin_topic** (46.7% vs 34.2%). On tweet_topic it lands **within 4 points of Jev**
|
| 174 |
+
(75.5% vs 79.3%). Macro-F1 leads over English Laya are +13.8 points (59.9% vs 46.1%) and +8.9 points (45.2% vs 36.2%).
|
| 175 |
+
With its shipped temperature, its probabilities are also better calibrated than Jev's on both sets: ECE 0.027 vs
|
| 176 |
+
0.063 on tweet_topic and 0.046 vs 0.166 on fin_topic.
|
| 177 |
+
|
| 178 |
+
<sub>Zero-shot for every system: neither set is in v3's training pool; accuracy over every row of the pinned test files (n = 1,693 and 4,117). Jev (1.13, API) and English Laya: numbers published by the [elcronos jev-vs-open-decision-models study](https://github.com/elcronos/jev-vs-open-decision-models) with its own prompt (results/cross_dataset_summary.json @ a1901bc), not re-run by us. v3: scored by us on the identical rows, label sets and instruction, in v3's own input format; tweet_topic accuracy 95% CI 73.4–77.5%. ECE: 15 equal-width bins as in the study; v3 with its shipped global temperature (0.880, fitted on v3's own calibration split, never on these sets), Jev's ECE as published (raw API probabilities).</sub>
|
| 179 |
+
|
| 180 |
+
### Speed: many questions, one read
|
| 181 |
+
|
| 182 |
+

|
| 183 |
+
|
| 184 |
+
**Ask many questions about one long state and v3 pulls away.** v3 reads the state once and scores every question
|
| 185 |
+
in a single call (GGUF runtime, `many_mode="batched"`; the default exact mode shares whole 1,024-token chunks of the
|
| 186 |
+
state and gives results identical to one call per question). With 5 to 10 questions per state, that makes it **1.4× to 1.9× faster** than a Laya-architecture
|
| 187 |
+
engine (our round-1 MacLaya-4K) on 1K-token states and **2.6× to 4.6× faster** on 4K-token states (4K tokens with
|
| 188 |
+
10 questions: 1,381 ms vs 6,364 ms). It also answers
|
| 189 |
+
questions about 8K-token states in 2.3 to 2.6 s, which the 4K-budget engine cannot run at all.
|
| 190 |
+
|
| 191 |
+
<sub>Identical-architecture timing: untrained Qwen3.5-0.8B export (latency does not depend on the weights). v3 = llama.cpp GGUF F16, one call per state with all questions scored together (`many_mode="batched"` in the GGUF runtime). Comparison engine = round-1 MacLaya-4K, our own fine-tune of the Laya multilingual architecture (4,096-token budget, FP32 on Apple MPS), one call per question; it is not an official Laya checkpoint. Compared at 5 and 10 questions per state. Apple M1 Max 64 GB, warm end-to-end p50, idle run 2026-09-23, prefix reuse off.</sub>
|
| 192 |
+
|
| 193 |
+
### Quantization: 4-bit, 0.53 GB, same calls
|
| 194 |
+
|
| 195 |
+

|
| 196 |
+
|
| 197 |
+
**Quantize it to 4-bit and it still makes the same call.** Every shipped v3 format (GGUF F16, Q8_0 and Q4_K_M;
|
| 198 |
+
MLX bf16 and 8-bit) matches the PyTorch FP32 model on **240 of 240 parity rows**, plus 6 of 6 prompts of about 16K
|
| 199 |
+
and 25.6K tokens. The 0.53 GB Q4_K_M file is about **2.4× smaller** than the 2B v2's Q4_K_M (1.27 GB).
|
| 200 |
+
|
| 201 |
+
<sub>v3: top-1 agreement with the PyTorch FP32 reference on 240 parity rows (a mixed fixture drawn from the training pool, 22 categories, English and Chinese), plus 6 extra rows at about 16K and 25.6K tokens (3 each), where every format also agrees 6/6. v3 sizes are the exported weight files (GB = 10^9 bytes). 2B v1/v2 numbers and sizes are as reported on their public Hugging Face GGUF cards: 500 held-out decisions each, against bf16 for v1 and CUDA merged BF16 for v2. The fixtures (training-pool rows for v3, held-out rows for v1/v2) and references differ, so the rows are not a paired comparison and no agreement gap is claimed.</sub>
|
| 202 |
+
|
| 203 |
+
## Third generation: what changed
|
| 204 |
+
|
| 205 |
+

|
| 206 |
+
|
| 207 |
+
<sub>v1/v2: as reported on their public Hugging Face cards (v1 GGUF card; v2 and v2-GGUF cards; v1's typed-decisions number is reported on the v2 card). v3: release manifest, training config and evaluation files; Q4_K_M size = exported file (GB = 10^9 bytes), parity rows drawn from the training pool. Typed decisions: same 2,000 decisions from 400 states; v1/v2 scored by the v2 card's harness, v3 by ours. In-domain for v3; v1 was not trained on typed decisions; v2's pool included typed workflow decisions. "51 evaluated" = MASSIVE locales, 14 trained + 37 held out; the fine-tuning pool covers 19 languages. 25,600 tokens = the runtime's whole-input limit, 25× v2's 1,024-token prompt.</sub>
|
| 208 |
+
|
| 209 |
+
<details>
|
| 210 |
+
<summary><strong>The same comparison as a text table</strong></summary>
|
| 211 |
+
|
| 212 |
+
| | Jev-Style 2B v1 | Jev-Style 2B v2 | **Jev-Style 0.8B v3** |
|
| 213 |
+
|---|---|---|---|
|
| 214 |
+
| Parameters | 2B (Qwen3.5-2B-Base) | 2B (continued from v1) | **0.8B** (752M text-model parameters) |
|
| 215 |
+
| Training | LoRA rank 16 (all linear layers) | LoRA rank 32 (33.6M trainable parameters) | **Full fine-tune** (every weight trained) |
|
| 216 |
+
| Readout | Option-letter token (one letter per option) | Option-letter token (' A' ... ' Z') | **Verdict slot per option** (every option scored, one pass) |
|
| 217 |
+
| Options per decision | Up to 26 (20 via top_logprobs) | 2–26 (letter-capped) | **No letter cap** (tested with 77 options) |
|
| 218 |
+
| Context | Not stated (quickstart: server default) | 1,024-token prompt (quickstart runs -c 2048) | **25,600 tokens** (preregistered 25K claim passed) |
|
| 219 |
+
| Languages | English (five English task families) | English (English state required) | **51 evaluated** (MASSIVE locales; 19 languages in fine-tuning) |
|
| 220 |
+
| Questions per state read | 1 (one question per prompt) | 1 (one question per prompt) | **Many** (all questions in one call) |
|
| 221 |
+
| Q4_K_M file | 1.3 GB (as reported on the v1 GGUF card) | 1.27 GB (as reported on the v2 GGUF card) | **0.53 GB** (matches FP32 on 240 / 240 parity rows) |
|
| 222 |
+
| Typed decisions, teacher agreement | 53.35% (2,000 decisions / 400 states) | 73.45% (same 2,000 decisions) | **79.15%** (same 2,000; 1,583 correct) |
|
| 223 |
+
|
| 224 |
+
</details>
|
| 225 |
+
|
| 226 |
+
Three ceilings of v1 and v2 are gone in v3:
|
| 227 |
+
|
| 228 |
+
- **No letter cap.** v1 and v2 read one option-letter token, so a question could have at most 26 options. v3
|
| 229 |
+
scores a verdict slot per option, so the options are whatever you pass. Banking77 was run with all 77 intents
|
| 230 |
+
in one pass.
|
| 231 |
+
- **25× the prompt budget.** v2's interface is a 1,024-token prompt. v3 takes 25,600 tokens, and its long-context
|
| 232 |
+
claim was preregistered and passed.
|
| 233 |
+
- **Read once, ask many.** v1 and v2 put one question in each prompt. v3 renders the state once and answers any
|
| 234 |
+
number of questions about it. With 10 questions on a 4K-token state this is 4.6× faster (GGUF, `many_mode="batched"`) than a
|
| 235 |
+
Laya-architecture engine (our round-1 MacLaya-4K) that calls once per question.
|
| 236 |
+
|
| 237 |
+
## What "Jev-style" means
|
| 238 |
+
|
| 239 |
+
[Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev) (TypeSafe AI, 2026) introduced *System One*
|
| 240 |
+
decision models. Instead of generating text, the model takes a state and a typed question and returns a probability
|
| 241 |
+
for each allowed answer in a single pass. Jev-Style models follow that pattern with open weights:
|
| 242 |
+
|
| 243 |
+
- **choice**: pick one of N named options, with a probability for each;
|
| 244 |
+
- **noul** (yes/no): the probability that a statement about the state is true;
|
| 245 |
+
- **score**: a distribution over 2 to 10 ordered levels.
|
| 246 |
+
|
| 247 |
+
The model cannot answer outside the options it is given, and it never decodes text.
|
| 248 |
+
|
| 249 |
+
> **Independent project.** Jev-Style is not affiliated with, endorsed by or connected to TypeSafe AI or Jev, and no
|
| 250 |
+
> Jev weights, code or outputs are used. It is also not affiliated with the Laya authors or the Qwen team. Jev and
|
| 251 |
+
> Laya numbers on this card come from the sources named under each result.
|
| 252 |
+
|
| 253 |
+
## Quick start (transformers)
|
| 254 |
+
|
| 255 |
+
```bash
|
| 256 |
+
pip install -U huggingface_hub
|
| 257 |
+
hf download chaoliangUNSW/Jev-Style-0.8B-Decision-v3 --local-dir jev-v3
|
| 258 |
+
cd jev-v3
|
| 259 |
+
pip install -r requirements.txt # torch, transformers>=5.0 (Qwen3.5 support), tokenizers, numpy
|
| 260 |
+
```
|
| 261 |
+
|
| 262 |
+
The repository ships `jev_style_decision.py`, a self-contained runtime. It handles input rendering, the verdict
|
| 263 |
+
readout, the fitted temperatures and budget checks.
|
| 264 |
+
|
| 265 |
+
```python
|
| 266 |
+
from jev_style_decision import JevStyleDecision
|
| 267 |
+
|
| 268 |
+
m = JevStyleDecision(".") # float32 on CUDA, Apple MPS or CPU (device="cpu" to force)
|
| 269 |
+
r = m.decide(
|
| 270 |
+
{"ticket": "I was charged twice for my subscription this month.", "customer_tier": "pro"},
|
| 271 |
+
"Which team should handle this ticket?",
|
| 272 |
+
options={"billing": "payments, invoices, refunds",
|
| 273 |
+
"technical": "bugs and outages",
|
| 274 |
+
"sales": "new purchases"},
|
| 275 |
+
category="theme_routing",
|
| 276 |
+
)
|
| 277 |
+
print(r["answer"], r["probabilities"])
|
| 278 |
+
# billing (probabilities ≈ billing 0.978, sales 0.017, technical 0.004 on CPU, float32)
|
| 279 |
+
```
|
| 280 |
+
|
| 281 |
+
Other question types, and several questions about one state:
|
| 282 |
+
|
| 283 |
+
```python
|
| 284 |
+
state = "Order #1182: paid, packed, handed to the courier on Monday. Tracking shows 'delivered' on Wednesday."
|
| 285 |
+
m.decide(state, "Has the order been delivered?", qtype="noul") # {"false": p, "true": p}
|
| 286 |
+
m.decide(state, "How urgent is a follow-up?", qtype="score",
|
| 287 |
+
options=["not urgent", "somewhat urgent", "urgent", "critical"]) # levels "0".."3"
|
| 288 |
+
m.decide_many(state, [
|
| 289 |
+
{"t": "noul", "ins": "Was the order paid?", "crit": None},
|
| 290 |
+
{"t": "choice", "ins": "Which step is the order at?",
|
| 291 |
+
"crit": {"packing": None, "in transit": None, "delivered": None}},
|
| 292 |
+
])
|
| 293 |
+
```
|
| 294 |
+
|
| 295 |
+
From the command line:
|
| 296 |
+
|
| 297 |
+
```bash
|
| 298 |
+
python jev_style_decision.py --state "The film was excellent." \
|
| 299 |
+
--question "What is the sentiment of this review?" \
|
| 300 |
+
--options '["negative", "positive"]' --category general_sentiment
|
| 301 |
+
# -> "answer": "positive", probability 0.989
|
| 302 |
+
```
|
| 303 |
+
|
| 304 |
+
`decide` returns a dict with `answer`, `probabilities`, the raw `scores`, the `temperature` used,
|
| 305 |
+
`top_probability`, `entropy_concentration`, `input_tokens`, `head_tokens`, `model` and `backend`. Batch mode reads JSON lines (`--jsonl file|-`), and
|
| 306 |
+
`--verify` checks every file against the sha256 manifest before loading.
|
| 307 |
+
|
| 308 |
+
**GGUF (llama.cpp):** [Jev-Style-0.8B-Decision-v3-GGUF](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-GGUF).
|
| 309 |
+
It includes `jev_score.cpp`, a small libllama scorer that reads logits only at the verdict slots and scores many
|
| 310 |
+
questions on one decoded state, plus `jev_style_decision_gguf.py` with the same API.
|
| 311 |
+
|
| 312 |
+
**MLX (Apple silicon):** [bf16](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-bf16) and
|
| 313 |
+
[8-bit](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-8bit), each with
|
| 314 |
+
`jev_style_decision_mlx.py` and the same API.
|
| 315 |
+
|
| 316 |
+
## Input format and readout
|
| 317 |
+
|
| 318 |
+
Each segment is tokenised on its own and the pieces are concatenated. Text inside the state or the options is
|
| 319 |
+
tokenised with special tokens disabled, so a string such as `<|im_end|>` in user data stays plain text.
|
| 320 |
+
|
| 321 |
+
```text
|
| 322 |
+
State:
|
| 323 |
+
<state: plain text, or any JSON value serialised with ensure_ascii=False>
|
| 324 |
+
|
| 325 |
+
Question [<choice|noul|score>]: <question>
|
| 326 |
+
Options:
|
| 327 |
+
- <option 1>
|
| 328 |
+
- <option 2>
|
| 329 |
+
Judge each option:
|
| 330 |
+
<option 1> ->
|
| 331 |
+
<option 2> ->
|
| 332 |
+
```
|
| 333 |
+
|
| 334 |
+
- **Verdict slot.** The hidden state at the ` ->` token that ends option k's line is option k's verdict slot. Its
|
| 335 |
+
score is `logit(" yes") − logit(" no")` at that position, computed as `h_k · (w_yes − w_no)` from the final
|
| 336 |
+
normalised hidden state and the tied embedding rows, in float32. No parameters are added, so the weights stay a
|
| 337 |
+
standard Qwen3.5 text model.
|
| 338 |
+
- **Probabilities.** `softmax(scores / T)`. `T` is looked up by calibration family × question type × option-count
|
| 339 |
+
bucket in `readout_config.json` (20 fitted groups), and the global value 0.880 is the fallback. Pass
|
| 340 |
+
`category=` (for example `theme_routing`, `general_topic`, `intent`, `typed_official`) to pick the family. With
|
| 341 |
+
no `category`, the runtime uses the global temperature (0.880), and `temperature=1.0` gives the raw scores.
|
| 342 |
+
- **Option text.** Choice options render as `name` or `name: description`. Score levels render as
|
| 343 |
+
`level i: description`. Yes/no questions render as `false: …` / `true: …`, with default descriptions when none
|
| 344 |
+
are given.
|
| 345 |
+
|
| 346 |
+
## Usage notes
|
| 347 |
+
|
| 348 |
+
- **Input budget.** The whole rendered input, meaning state, question, options and readout, may be up to 25,600
|
| 349 |
+
tokens. The head (question, options and readout) may be up to 2,048 tokens. Over-budget inputs raise
|
| 350 |
+
`InputBudgetError`, and nothing is ever truncated silently.
|
| 351 |
+
- **Decisions only.** The model scores the options you give it and returns probabilities. It does not generate
|
| 352 |
+
text, and it takes no actions on its own.
|
| 353 |
+
- **Options.** `choice` takes any number of named options within the 2,048-token head (77 is the largest set we
|
| 354 |
+
evaluated; the GGUF runtime accepts up to 256 options per question). `score` takes 2 to 10 ordered levels, lowest first. `noul` needs no options.
|
| 355 |
+
- **Several questions about one state.** Use `decide_many`. In the GGUF runtime it sends all questions to the
|
| 356 |
+
bundled scorer in one request. By default the results are identical to one `decide` call per question; to keep
|
| 357 |
+
them identical, the state is shared only in whole 1,024-token blocks, so the time saved starts at 1,024-token
|
| 358 |
+
states and grows with the state length. `JevStyleDecisionGGUF(..., many_mode="batched")` reads the whole state
|
| 359 |
+
once and scores all questions together, as in the latency chart; its probabilities differed from `decide` by at
|
| 360 |
+
most 0.002 in our tests, and a near-tied top answer can change.
|
| 361 |
+
- **Precision.** The evaluation numbers on this card were computed with the PyTorch weights. The GGUF and MLX
|
| 362 |
+
builds were checked for top-1 parity with the PyTorch FP32 reference (see Quantization).
|
| 363 |
+
- **Weights.** This is a text-only `Qwen3_5ForCausalLM`: 752,393,024 parameters, 24 layers (18 Gated DeltaNet +
|
| 364 |
+
6 full attention), hidden size 1,024, tied embeddings. The vision tower and the multi-token-prediction head
|
| 365 |
+
were removed.
|
| 366 |
+
|
| 367 |
+
## Training
|
| 368 |
+
|
| 369 |
+
- **Base:** [Qwen/Qwen3.5-0.8B](https://huggingface.co/Qwen/Qwen3.5-0.8B) (revision `2fc06364`).
|
| 370 |
+
- **Run:** full fine-tune in bf16 on one NVIDIA H100 80GB. It took 994 optimizer steps over 131.4M tokens, with
|
| 371 |
+
a learning rate of 2e-5 and about 96 minutes of training steps (5,780 s). Checkpoint step 981 was selected by
|
| 372 |
+
the preregistered development score.
|
| 373 |
+
- **Mixture:** a 321,756-row training pool in 19 languages, with inputs up to 25,600 tokens for long documents and agent
|
| 374 |
+
histories:
|
| 375 |
+
- typed decisions: the LocalLLaMA/typed-decisions train split plus 27,300 synthetic typed items;
|
| 376 |
+
- general classification and QA: MNLI and translated NLI, AG News, GoEmotions, DAIR Emotion, SQuAD v2, SST-5
|
| 377 |
+
and BoolQ;
|
| 378 |
+
- intents: MASSIVE intent and scenario in 14 locales, CLINC150 and HWU64;
|
| 379 |
+
- application themes: spam, phishing, jailbreak and prompt injection, toxicity, ticket triage and model routing;
|
| 380 |
+
- Mac agent step-gate and goal-done checks;
|
| 381 |
+
- long-context retrieval, tables and QA.
|
| 382 |
+
- **Calibration:** 20 group temperatures plus a global one, fitted on 15,655 held-out calibration rows (never
|
| 383 |
+
test rows).
|
| 384 |
+
|
| 385 |
+
## Training data and licences
|
| 386 |
+
|
| 387 |
+
- **Base model:** Qwen/Qwen3.5-0.8B by the Qwen team (Alibaba Cloud), Apache-2.0. The Apache License 2.0 text is
|
| 388 |
+
in `LICENSE`, and `NOTICE` lists the modifications. This model is released under Apache-2.0.
|
| 389 |
+
- **Datasets with restrictive or unclear terms** (kept in training by the author's decision):
|
| 390 |
+
- DAIR Emotion (14,757 training rows): its dataset card says it should be used for educational and research
|
| 391 |
+
purposes only.
|
| 392 |
+
- AG News (30,000 training rows): its licence is listed as unknown, and its card describes it as provided by the
|
| 393 |
+
academic community for research and non-commercial use.
|
| 394 |
+
- SST-5, MNLI, an HWU64 mirror, QuALITY, Enron spam and a phishing-email dataset also carry their own terms.
|
| 395 |
+
Check each source before commercial use.
|
| 396 |
+
- **Outputs of other models** (kept by the author's decision): an OpenAI GPT model wrote the theme data for
|
| 397 |
+
routing, triage and jailbreak, the teacher-translated NLI data and the Chinese filler text for long documents.
|
| 398 |
+
OpenAI GPT and Anthropic Claude models designed the synthetic typed-decision workflows, and Anthropic Claude
|
| 399 |
+
models labelled them. The providers' terms of use may restrict how models trained on such outputs may be used,
|
| 400 |
+
so check them for your use case.
|
| 401 |
+
- **Evaluation-only data** (0 training rows): Banking77, toxic-chat, XNLI, ContractNLI, MS MARCO, tweet_topic,
|
| 402 |
+
fin_topic, the JevBench items and the support-ticket set.
|
| 403 |
+
|
| 404 |
+
<details>
|
| 405 |
+
<summary><strong>Evaluation records and vector charts</strong></summary>
|
| 406 |
+
|
| 407 |
+
- Every chart is also provided as SVG: [headline_typed](figures/headline_typed.svg),
|
| 408 |
+
[beyond_laya](figures/beyond_laya.svg), [multilingual](figures/multilingual.svg),
|
| 409 |
+
[calibration](figures/calibration.svg), [long_context](figures/long_context.svg),
|
| 410 |
+
[jevbench](figures/jevbench.svg), [zeroshot](figures/zeroshot.svg), [latency](figures/latency.svg),
|
| 411 |
+
[quantization](figures/quantization.svg), [design_table](figures/design_table.svg).
|
| 412 |
+
- Plotted values, sources and protocol labels for each chart:
|
| 413 |
+
[headline_typed](figures/headline_typed.data.json), [beyond_laya](figures/beyond_laya.json),
|
| 414 |
+
[multilingual](figures/multilingual.data.json), [calibration](figures/calibration.data.json),
|
| 415 |
+
[long_context](figures/long_context.json), [jevbench](figures/jevbench.data.json),
|
| 416 |
+
[zeroshot](figures/zeroshot.json), [latency](figures/latency.data.json),
|
| 417 |
+
[quantization](figures/quantization.data.json), [design_table](figures/design_table.data.json).
|
| 418 |
+
- Every v3 and re-run Laya number comes from prediction files that were each scored once. Paired differences use
|
| 419 |
+
a case-cluster bootstrap within suites (2,000 resamples). A win is only claimed when the 95% CI excludes zero,
|
| 420 |
+
except where a chart or note says otherwise (JevBench leads over Laya and M. Ghafiri, and per-language MASSIVE
|
| 421 |
+
gaps against the best of three Laya checkpoints, are point estimates).
|
| 422 |
+
- Public sources: [Laya](https://huggingface.co/convaiinnovations/laya) (official checkpoints and README;
|
| 423 |
+
[BENCHMARKS.md](https://github.com/NandhaKishorM/laya/blob/main/BENCHMARKS.md)),
|
| 424 |
+
[LocalLLaMA/typed-decisions](https://huggingface.co/datasets/LocalLLaMA/typed-decisions) (dataset card with Jev's
|
| 425 |
+
numbers), [JevBench](https://github.com/fstandhartinger/jevbench) (v1.4.1 results file),
|
| 426 |
+
[elcronos/jev-vs-open-decision-models](https://github.com/elcronos/jev-vs-open-decision-models) (zero-shot topic
|
| 427 |
+
study), and the [2B v1](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF) and
|
| 428 |
+
[2B v2](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2) cards.
|
| 429 |
+
|
| 430 |
+
</details>
|
| 431 |
+
|
| 432 |
+
## Citation
|
| 433 |
+
|
| 434 |
+
```bibtex
|
| 435 |
+
@misc{jevstyle2026v3,
|
| 436 |
+
title = {Jev-Style-0.8B-Decision-v3: a long-context, multilingual 0.8B decision model},
|
| 437 |
+
author = {chaoliangUNSW},
|
| 438 |
+
year = {2026},
|
| 439 |
+
howpublished = {\url{https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3}},
|
| 440 |
+
note = {Fine-tuned from Qwen/Qwen3.5-0.8B}
|
| 441 |
+
}
|
| 442 |
+
```
|
| 443 |
+
|
| 444 |
+
## Contact
|
| 445 |
+
|
| 446 |
+
I welcome internship, employment, and research collaboration opportunities. Please contact me at [**yanchaoliang369@gmail.com**](mailto:yanchaoliang369@gmail.com).
|
| 447 |
+
|
| 448 |
+
欢迎提供实习、工作及科研合作机会,请邮件联系:[yanchaoliang369@gmail.com](mailto:yanchaoliang369@gmail.com)。
|
chat_template.jinja
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{%- set image_count = namespace(value=0) %}
|
| 2 |
+
{%- set video_count = namespace(value=0) %}
|
| 3 |
+
{%- macro render_content(content, do_vision_count, is_system_content=false) %}
|
| 4 |
+
{%- if content is string %}
|
| 5 |
+
{{- content }}
|
| 6 |
+
{%- elif content is iterable and content is not mapping %}
|
| 7 |
+
{%- for item in content %}
|
| 8 |
+
{%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
|
| 9 |
+
{%- if is_system_content %}
|
| 10 |
+
{{- raise_exception('System message cannot contain images.') }}
|
| 11 |
+
{%- endif %}
|
| 12 |
+
{%- if do_vision_count %}
|
| 13 |
+
{%- set image_count.value = image_count.value + 1 %}
|
| 14 |
+
{%- endif %}
|
| 15 |
+
{%- if add_vision_id %}
|
| 16 |
+
{{- 'Picture ' ~ image_count.value ~ ': ' }}
|
| 17 |
+
{%- endif %}
|
| 18 |
+
{{- '<|vision_start|><|image_pad|><|vision_end|>' }}
|
| 19 |
+
{%- elif 'video' in item or item.type == 'video' %}
|
| 20 |
+
{%- if is_system_content %}
|
| 21 |
+
{{- raise_exception('System message cannot contain videos.') }}
|
| 22 |
+
{%- endif %}
|
| 23 |
+
{%- if do_vision_count %}
|
| 24 |
+
{%- set video_count.value = video_count.value + 1 %}
|
| 25 |
+
{%- endif %}
|
| 26 |
+
{%- if add_vision_id %}
|
| 27 |
+
{{- 'Video ' ~ video_count.value ~ ': ' }}
|
| 28 |
+
{%- endif %}
|
| 29 |
+
{{- '<|vision_start|><|video_pad|><|vision_end|>' }}
|
| 30 |
+
{%- elif 'text' in item %}
|
| 31 |
+
{{- item.text }}
|
| 32 |
+
{%- else %}
|
| 33 |
+
{{- raise_exception('Unexpected item type in content.') }}
|
| 34 |
+
{%- endif %}
|
| 35 |
+
{%- endfor %}
|
| 36 |
+
{%- elif content is none or content is undefined %}
|
| 37 |
+
{{- '' }}
|
| 38 |
+
{%- else %}
|
| 39 |
+
{{- raise_exception('Unexpected content type.') }}
|
| 40 |
+
{%- endif %}
|
| 41 |
+
{%- endmacro %}
|
| 42 |
+
{%- if not messages %}
|
| 43 |
+
{{- raise_exception('No messages provided.') }}
|
| 44 |
+
{%- endif %}
|
| 45 |
+
{%- if tools and tools is iterable and tools is not mapping %}
|
| 46 |
+
{{- '<|im_start|>system\n' }}
|
| 47 |
+
{{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
|
| 48 |
+
{%- for tool in tools %}
|
| 49 |
+
{{- "\n" }}
|
| 50 |
+
{{- tool | tojson }}
|
| 51 |
+
{%- endfor %}
|
| 52 |
+
{{- "\n</tools>" }}
|
| 53 |
+
{{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
|
| 54 |
+
{%- if messages[0].role == 'system' %}
|
| 55 |
+
{%- set content = render_content(messages[0].content, false, true)|trim %}
|
| 56 |
+
{%- if content %}
|
| 57 |
+
{{- '\n\n' + content }}
|
| 58 |
+
{%- endif %}
|
| 59 |
+
{%- endif %}
|
| 60 |
+
{{- '<|im_end|>\n' }}
|
| 61 |
+
{%- else %}
|
| 62 |
+
{%- if messages[0].role == 'system' %}
|
| 63 |
+
{%- set content = render_content(messages[0].content, false, true)|trim %}
|
| 64 |
+
{{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
|
| 65 |
+
{%- endif %}
|
| 66 |
+
{%- endif %}
|
| 67 |
+
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 68 |
+
{%- for message in messages[::-1] %}
|
| 69 |
+
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 70 |
+
{%- if ns.multi_step_tool and message.role == "user" %}
|
| 71 |
+
{%- set content = render_content(message.content, false)|trim %}
|
| 72 |
+
{%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
|
| 73 |
+
{%- set ns.multi_step_tool = false %}
|
| 74 |
+
{%- set ns.last_query_index = index %}
|
| 75 |
+
{%- endif %}
|
| 76 |
+
{%- endif %}
|
| 77 |
+
{%- endfor %}
|
| 78 |
+
{%- if ns.multi_step_tool %}
|
| 79 |
+
{{- raise_exception('No user query found in messages.') }}
|
| 80 |
+
{%- endif %}
|
| 81 |
+
{%- for message in messages %}
|
| 82 |
+
{%- set content = render_content(message.content, true)|trim %}
|
| 83 |
+
{%- if message.role == "system" %}
|
| 84 |
+
{%- if not loop.first %}
|
| 85 |
+
{{- raise_exception('System message must be at the beginning.') }}
|
| 86 |
+
{%- endif %}
|
| 87 |
+
{%- elif message.role == "user" %}
|
| 88 |
+
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 89 |
+
{%- elif message.role == "assistant" %}
|
| 90 |
+
{%- set reasoning_content = '' %}
|
| 91 |
+
{%- if message.reasoning_content is string %}
|
| 92 |
+
{%- set reasoning_content = message.reasoning_content %}
|
| 93 |
+
{%- else %}
|
| 94 |
+
{%- if '</think>' in content %}
|
| 95 |
+
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
| 96 |
+
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
| 97 |
+
{%- endif %}
|
| 98 |
+
{%- endif %}
|
| 99 |
+
{%- set reasoning_content = reasoning_content|trim %}
|
| 100 |
+
{%- if loop.index0 > ns.last_query_index %}
|
| 101 |
+
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
|
| 102 |
+
{%- else %}
|
| 103 |
+
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 104 |
+
{%- endif %}
|
| 105 |
+
{%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
|
| 106 |
+
{%- for tool_call in message.tool_calls %}
|
| 107 |
+
{%- if tool_call.function is defined %}
|
| 108 |
+
{%- set tool_call = tool_call.function %}
|
| 109 |
+
{%- endif %}
|
| 110 |
+
{%- if loop.first %}
|
| 111 |
+
{%- if content|trim %}
|
| 112 |
+
{{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 113 |
+
{%- else %}
|
| 114 |
+
{{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 115 |
+
{%- endif %}
|
| 116 |
+
{%- else %}
|
| 117 |
+
{{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 118 |
+
{%- endif %}
|
| 119 |
+
{%- if tool_call.arguments is defined %}
|
| 120 |
+
{%- for args_name, args_value in tool_call.arguments|items %}
|
| 121 |
+
{{- '<parameter=' + args_name + '>\n' }}
|
| 122 |
+
{%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
|
| 123 |
+
{{- args_value }}
|
| 124 |
+
{{- '\n</parameter>\n' }}
|
| 125 |
+
{%- endfor %}
|
| 126 |
+
{%- endif %}
|
| 127 |
+
{{- '</function>\n</tool_call>' }}
|
| 128 |
+
{%- endfor %}
|
| 129 |
+
{%- endif %}
|
| 130 |
+
{{- '<|im_end|>\n' }}
|
| 131 |
+
{%- elif message.role == "tool" %}
|
| 132 |
+
{%- if loop.previtem and loop.previtem.role != "tool" %}
|
| 133 |
+
{{- '<|im_start|>user' }}
|
| 134 |
+
{%- endif %}
|
| 135 |
+
{{- '\n<tool_response>\n' }}
|
| 136 |
+
{{- content }}
|
| 137 |
+
{{- '\n</tool_response>' }}
|
| 138 |
+
{%- if not loop.last and loop.nextitem.role != "tool" %}
|
| 139 |
+
{{- '<|im_end|>\n' }}
|
| 140 |
+
{%- elif loop.last %}
|
| 141 |
+
{{- '<|im_end|>\n' }}
|
| 142 |
+
{%- endif %}
|
| 143 |
+
{%- else %}
|
| 144 |
+
{{- raise_exception('Unexpected message role.') }}
|
| 145 |
+
{%- endif %}
|
| 146 |
+
{%- endfor %}
|
| 147 |
+
{%- if add_generation_prompt %}
|
| 148 |
+
{{- '<|im_start|>assistant\n' }}
|
| 149 |
+
{%- if enable_thinking is defined and enable_thinking is true %}
|
| 150 |
+
{{- '<think>\n' }}
|
| 151 |
+
{%- else %}
|
| 152 |
+
{{- '<think>\n\n</think>\n\n' }}
|
| 153 |
+
{%- endif %}
|
| 154 |
+
{%- endif %}
|
config.json
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"Qwen3_5ForCausalLM"
|
| 4 |
+
],
|
| 5 |
+
"attention_bias": false,
|
| 6 |
+
"attention_dropout": 0.0,
|
| 7 |
+
"attn_output_gate": true,
|
| 8 |
+
"bos_token_id": null,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eos_token_id": 248044,
|
| 11 |
+
"full_attention_interval": 4,
|
| 12 |
+
"head_dim": 256,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 1024,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 3584,
|
| 17 |
+
"layer_types": [
|
| 18 |
+
"linear_attention",
|
| 19 |
+
"linear_attention",
|
| 20 |
+
"linear_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"linear_attention",
|
| 23 |
+
"linear_attention",
|
| 24 |
+
"linear_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"linear_attention",
|
| 27 |
+
"linear_attention",
|
| 28 |
+
"linear_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"linear_attention",
|
| 31 |
+
"linear_attention",
|
| 32 |
+
"linear_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"linear_attention",
|
| 35 |
+
"linear_attention",
|
| 36 |
+
"linear_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"linear_attention",
|
| 39 |
+
"linear_attention",
|
| 40 |
+
"linear_attention",
|
| 41 |
+
"full_attention"
|
| 42 |
+
],
|
| 43 |
+
"linear_conv_kernel_dim": 4,
|
| 44 |
+
"linear_key_head_dim": 128,
|
| 45 |
+
"linear_num_key_heads": 16,
|
| 46 |
+
"linear_num_value_heads": 16,
|
| 47 |
+
"linear_value_head_dim": 128,
|
| 48 |
+
"mamba_ssm_dtype": "float32",
|
| 49 |
+
"max_position_embeddings": 262144,
|
| 50 |
+
"mlp_only_layers": [],
|
| 51 |
+
"model_type": "qwen3_5_text",
|
| 52 |
+
"mtp_num_hidden_layers": 1,
|
| 53 |
+
"mtp_use_dedicated_embeddings": false,
|
| 54 |
+
"num_attention_heads": 8,
|
| 55 |
+
"num_hidden_layers": 24,
|
| 56 |
+
"num_key_value_heads": 2,
|
| 57 |
+
"pad_token_id": null,
|
| 58 |
+
"partial_rotary_factor": 0.25,
|
| 59 |
+
"rms_norm_eps": 1e-06,
|
| 60 |
+
"rope_parameters": {
|
| 61 |
+
"mrope_interleaved": true,
|
| 62 |
+
"mrope_section": [
|
| 63 |
+
11,
|
| 64 |
+
11,
|
| 65 |
+
10
|
| 66 |
+
],
|
| 67 |
+
"partial_rotary_factor": 0.25,
|
| 68 |
+
"rope_theta": 10000000,
|
| 69 |
+
"rope_type": "default"
|
| 70 |
+
},
|
| 71 |
+
"tie_word_embeddings": true,
|
| 72 |
+
"transformers_version": "5.17.0",
|
| 73 |
+
"use_cache": true,
|
| 74 |
+
"vocab_size": 248320
|
| 75 |
+
}
|
figures/beyond_laya.json
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"chart": "beyond_laya",
|
| 3 |
+
"unit": "percent (value x 100)",
|
| 4 |
+
"source": "runs/macjev/report_r2/scoreboard/scoreboard.json via hf_staging/v3_card/chart_data.json",
|
| 5 |
+
"delta_ci_method": "case-cluster bootstrap within suites, 2000 resamples (scoreboard n_boot)",
|
| 6 |
+
"rows": [
|
| 7 |
+
{
|
| 8 |
+
"task": "Banking77",
|
| 9 |
+
"metric": "accuracy, 77 intents",
|
| 10 |
+
"v3": 0.6825,
|
| 11 |
+
"best_laya": 0.4925,
|
| 12 |
+
"best_laya_checkpoint": "typed",
|
| 13 |
+
"delta_pts": 19.0,
|
| 14 |
+
"delta_ci95_pts": [
|
| 15 |
+
13.99,
|
| 16 |
+
24.0
|
| 17 |
+
],
|
| 18 |
+
"n": 400,
|
| 19 |
+
"v3_entry": "banking77.v3",
|
| 20 |
+
"laya_entry": "banking77.laya_best",
|
| 21 |
+
"scoreboard_field": "metrics['banking77']",
|
| 22 |
+
"protocol_label": "v3: held-out (banking77 never trained; CLINC150/HWU64 intents are) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"task": "Jailbreak",
|
| 26 |
+
"metric": "balanced accuracy",
|
| 27 |
+
"v3": 0.9036238842064085,
|
| 28 |
+
"best_laya": 0.8311462000782389,
|
| 29 |
+
"best_laya_checkpoint": "multilingual",
|
| 30 |
+
"delta_pts": 7.25,
|
| 31 |
+
"delta_ci95_pts": [
|
| 32 |
+
2.99,
|
| 33 |
+
11.43
|
| 34 |
+
],
|
| 35 |
+
"n": 400,
|
| 36 |
+
"v3_entry": "theme.guardrails_jailbreak.balanced_accuracy.v3",
|
| 37 |
+
"laya_entry": "theme.guardrails_jailbreak.balanced_accuracy.laya_best",
|
| 38 |
+
"scoreboard_field": "metrics['theme.guardrails_jailbreak.balanced_accuracy']",
|
| 39 |
+
"protocol_label": "v3: held-out source (other permissive jailbreak sets + teacher) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
|
| 40 |
+
},
|
| 41 |
+
{
|
| 42 |
+
"task": "Toxicity",
|
| 43 |
+
"metric": "macro-F1",
|
| 44 |
+
"v3": 0.7089254055198327,
|
| 45 |
+
"best_laya": 0.41360068097985436,
|
| 46 |
+
"best_laya_checkpoint": "multilingual",
|
| 47 |
+
"delta_pts": 29.53,
|
| 48 |
+
"delta_ci95_pts": [
|
| 49 |
+
24.65,
|
| 50 |
+
34.37
|
| 51 |
+
],
|
| 52 |
+
"n": 400,
|
| 53 |
+
"v3_entry": "theme.moderation_toxicity.macro_f1.v3",
|
| 54 |
+
"laya_entry": "theme.moderation_toxicity.macro_f1.laya_best",
|
| 55 |
+
"scoreboard_field": "metrics['theme.moderation_toxicity.macro_f1']",
|
| 56 |
+
"protocol_label": "v3: held-out source (civil_comments + teacher; toxic-chat eval-only) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"task": "Model routing",
|
| 60 |
+
"metric": "accuracy",
|
| 61 |
+
"v3": 0.9624060150375939,
|
| 62 |
+
"best_laya": 0.6591478696741855,
|
| 63 |
+
"best_laya_checkpoint": "typed",
|
| 64 |
+
"delta_pts": 30.33,
|
| 65 |
+
"delta_ci95_pts": [
|
| 66 |
+
26.06,
|
| 67 |
+
35.09
|
| 68 |
+
],
|
| 69 |
+
"n": 399,
|
| 70 |
+
"v3_entry": "theme.model_routing_domain.v3",
|
| 71 |
+
"laya_entry": "theme.model_routing_domain.laya_best",
|
| 72 |
+
"scoreboard_field": "metrics['theme.model_routing_domain']",
|
| 73 |
+
"protocol_label": "v3: near-domain (teacher-written; gsm8k/mbpp/AG rows eval-only) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"task": "MASSIVE intent",
|
| 77 |
+
"metric": "37 held-out locales",
|
| 78 |
+
"v3": 0.6551351351351351,
|
| 79 |
+
"best_laya": 0.36108108108108106,
|
| 80 |
+
"best_laya_checkpoint": "multilingual",
|
| 81 |
+
"delta_pts": 29.41,
|
| 82 |
+
"delta_ci95_pts": [
|
| 83 |
+
27.54,
|
| 84 |
+
31.3
|
| 85 |
+
],
|
| 86 |
+
"n": 3700,
|
| 87 |
+
"v3_entry": "massive51.unseen37.macro_accuracy.v3",
|
| 88 |
+
"laya_entry": "massive51.unseen37.macro_accuracy.laya_best",
|
| 89 |
+
"scoreboard_field": "metrics['massive51.unseen37.macro_accuracy']",
|
| 90 |
+
"protocol_label": "v3: held-out (37 languages never trained) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
|
| 91 |
+
}
|
| 92 |
+
],
|
| 93 |
+
"footnote": "Laya numbers: official checkpoints (English, typed-decisions, multilingual) re-run by us on identical rows with their shipped temperatures and\ndefault token budgets; the best of the three is shown per task. v3 trained on same-kind tasks from other datasets, never on these eval rows:\nintent (CLINC150/HWU64; Banking77 never trained), jailbreak (other permissive sets + teacher), toxicity (civil_comments + teacher; toxic-chat\neval-only), routing (teacher-written; gsm8k/mbpp/AG rows eval-only), MASSIVE in 14 other locales (no MASSIVE rows in these 37).\nn = 400 / 400 / 400 / 399 / 3,700 (37 x 100). Every gap's paired 95% bootstrap CI excludes zero. Plotted values: figures/beyond_laya.json."
|
| 94 |
+
}
|
figures/beyond_laya.png
ADDED
|
Git LFS Details
|
figures/beyond_laya.svg
ADDED
|
|
figures/calibration.data.json
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"chart": "calibration",
|
| 3 |
+
"metric": "49-suite macro NLL / Brier / ECE, as deployed (lower is better)",
|
| 4 |
+
"protocol_label": "v3: mixed (49 suites) / Laya: mixed (49 suites). Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures.",
|
| 5 |
+
"rows": [
|
| 6 |
+
{
|
| 7 |
+
"panel": "NLL",
|
| 8 |
+
"metric": "49-suite macro NLL, as deployed",
|
| 9 |
+
"n_rows": 17416,
|
| 10 |
+
"v3": 0.4928839178581892,
|
| 11 |
+
"best_laya": 2.212565130608199,
|
| 12 |
+
"best_laya_checkpoint": "multilingual",
|
| 13 |
+
"all_laya_checkpoints": {
|
| 14 |
+
"english": 9.669232280968933,
|
| 15 |
+
"typed": 7.344998151307391,
|
| 16 |
+
"multilingual": 2.212565130608199
|
| 17 |
+
},
|
| 18 |
+
"ratio_best_laya_over_v3": 4.48901871301224,
|
| 19 |
+
"diff_v3_minus_best": -1.7196812127500098,
|
| 20 |
+
"diff_ci95": [
|
| 21 |
+
-1.7632826815264786,
|
| 22 |
+
-1.6751892907984907
|
| 23 |
+
],
|
| 24 |
+
"diff_ci_method": "case-cluster bootstrap within suites, 2000 resamples (scoreboard n_boot)",
|
| 25 |
+
"source_path": "runs/macjev/report_r2/scoreboard/scoreboard.json",
|
| 26 |
+
"fields": [
|
| 27 |
+
"metrics['t4.macro_nll'].ours_value",
|
| 28 |
+
"metrics['t4.macro_nll'].laya_best"
|
| 29 |
+
],
|
| 30 |
+
"entries": [
|
| 31 |
+
"t4.macro_nll.v3",
|
| 32 |
+
"t4.macro_nll.laya_best"
|
| 33 |
+
],
|
| 34 |
+
"claim": "t4.macro_nll_vs_best_laya"
|
| 35 |
+
},
|
| 36 |
+
{
|
| 37 |
+
"panel": "Brier score",
|
| 38 |
+
"metric": "49-suite macro Brier, as deployed",
|
| 39 |
+
"n_rows": 17416,
|
| 40 |
+
"v3": 0.23930092306086748,
|
| 41 |
+
"best_laya": 0.7120664110846202,
|
| 42 |
+
"best_laya_checkpoint": "multilingual",
|
| 43 |
+
"all_laya_checkpoints": {
|
| 44 |
+
"english": 1.0124090901595464,
|
| 45 |
+
"typed": 0.9474950671291875,
|
| 46 |
+
"multilingual": 0.7120664110846202
|
| 47 |
+
},
|
| 48 |
+
"ratio_best_laya_over_v3": 2.975610799894417,
|
| 49 |
+
"diff_v3_minus_best": -0.47276548802375273,
|
| 50 |
+
"diff_ci95": [
|
| 51 |
+
-0.4843787573639837,
|
| 52 |
+
-0.46074418926883626
|
| 53 |
+
],
|
| 54 |
+
"diff_ci_method": "case-cluster bootstrap within suites, 2000 resamples (scoreboard n_boot)",
|
| 55 |
+
"source_path": "runs/macjev/report_r2/scoreboard/scoreboard.json",
|
| 56 |
+
"fields": [
|
| 57 |
+
"metrics['t4.macro_brier'].ours_value",
|
| 58 |
+
"metrics['t4.macro_brier'].laya_best"
|
| 59 |
+
],
|
| 60 |
+
"entries": [
|
| 61 |
+
"t4.macro_brier.v3",
|
| 62 |
+
"t4.macro_brier.laya_best"
|
| 63 |
+
],
|
| 64 |
+
"claim": "t4.macro_brier_vs_best_laya"
|
| 65 |
+
},
|
| 66 |
+
{
|
| 67 |
+
"panel": "ECE",
|
| 68 |
+
"metric": "49-suite macro ECE, as deployed",
|
| 69 |
+
"n_rows": 17416,
|
| 70 |
+
"v3": 0.05416919804670487,
|
| 71 |
+
"best_laya": 0.2994167384382702,
|
| 72 |
+
"best_laya_checkpoint": "multilingual",
|
| 73 |
+
"all_laya_checkpoints": {
|
| 74 |
+
"english": 0.4647782759946152,
|
| 75 |
+
"typed": 0.40828649732151606,
|
| 76 |
+
"multilingual": 0.2994167384382702
|
| 77 |
+
},
|
| 78 |
+
"ratio_best_laya_over_v3": 5.527435318132494,
|
| 79 |
+
"diff_v3_minus_best": -0.24524754039156535,
|
| 80 |
+
"diff_ci95": [
|
| 81 |
+
-0.24448168599481393,
|
| 82 |
+
-0.22893121774401398
|
| 83 |
+
],
|
| 84 |
+
"diff_ci_method": "case-cluster bootstrap within suites, 2000 resamples (scoreboard n_boot)",
|
| 85 |
+
"source_path": "runs/macjev/report_r2/scoreboard/scoreboard.json",
|
| 86 |
+
"fields": [
|
| 87 |
+
"metrics['t4.macro_ece_as_deployed'].ours_value",
|
| 88 |
+
"metrics['t4.macro_ece_as_deployed'].laya_best"
|
| 89 |
+
],
|
| 90 |
+
"entries": [
|
| 91 |
+
"t4.macro_ece_as_deployed.v3",
|
| 92 |
+
"t4.macro_ece_as_deployed.laya_best"
|
| 93 |
+
],
|
| 94 |
+
"claim": "t4.macro_ece_as_deployed_vs_best_laya"
|
| 95 |
+
}
|
| 96 |
+
],
|
| 97 |
+
"footnote": "Macro average over 49 suites, 17,416 identical rows for both models. Protocol: mixed for both v3 and Laya (in-domain on some suites, held-out on others). As deployed: v3 with its supplied deployment temperatures; Laya numbers: official checkpoints re-run by us on identical rows with their shipped temperatures. Best Laya = best of the three official checkpoints (English, typed-decisions, multilingual) per metric; multilingual is best on all three. Paired case-cluster bootstrap 95% CIs (2,000 resamples) of every v3 minus Laya difference exclude zero. Plotted values: figures/calibration.data.json.",
|
| 98 |
+
"not_plotted": "optional typed-decisions reliability diagram omitted (see agent report)"
|
| 99 |
+
}
|
figures/calibration.png
ADDED
|
Git LFS Details
|
figures/calibration.svg
ADDED
|
|
figures/design_table.data.json
ADDED
|
@@ -0,0 +1,175 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"chart": "design_table",
|
| 3 |
+
"columns": [
|
| 4 |
+
"Jev-Style 2B v1",
|
| 5 |
+
"Jev-Style 2B v2",
|
| 6 |
+
"Jev-Style 0.8B v3"
|
| 7 |
+
],
|
| 8 |
+
"rows": [
|
| 9 |
+
{
|
| 10 |
+
"row": "Parameters",
|
| 11 |
+
"Jev-Style 2B v1": {
|
| 12 |
+
"main": "2B",
|
| 13 |
+
"sub": "Qwen3.5-2B-Base"
|
| 14 |
+
},
|
| 15 |
+
"Jev-Style 2B v2": {
|
| 16 |
+
"main": "2B",
|
| 17 |
+
"sub": "continued from v1"
|
| 18 |
+
},
|
| 19 |
+
"Jev-Style 0.8B v3": {
|
| 20 |
+
"main": "0.8B",
|
| 21 |
+
"sub": "752M text-model params"
|
| 22 |
+
}
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"row": "Training",
|
| 26 |
+
"Jev-Style 2B v1": {
|
| 27 |
+
"main": "LoRA rank 16",
|
| 28 |
+
"sub": "all linear layers"
|
| 29 |
+
},
|
| 30 |
+
"Jev-Style 2B v2": {
|
| 31 |
+
"main": "LoRA rank 32",
|
| 32 |
+
"sub": "33.6M trainable params"
|
| 33 |
+
},
|
| 34 |
+
"Jev-Style 0.8B v3": {
|
| 35 |
+
"main": "Full fine-tune",
|
| 36 |
+
"sub": "every weight trained"
|
| 37 |
+
}
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"row": "Readout",
|
| 41 |
+
"Jev-Style 2B v1": {
|
| 42 |
+
"main": "Option-letter token",
|
| 43 |
+
"sub": "one letter per option"
|
| 44 |
+
},
|
| 45 |
+
"Jev-Style 2B v2": {
|
| 46 |
+
"main": "Option-letter token",
|
| 47 |
+
"sub": "' A' ... ' Z'"
|
| 48 |
+
},
|
| 49 |
+
"Jev-Style 0.8B v3": {
|
| 50 |
+
"main": "Verdict slot per option",
|
| 51 |
+
"sub": "every option scored, one pass"
|
| 52 |
+
}
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"row": "Options per decision",
|
| 56 |
+
"Jev-Style 2B v1": {
|
| 57 |
+
"main": "Up to 26",
|
| 58 |
+
"sub": "20 via top_logprobs"
|
| 59 |
+
},
|
| 60 |
+
"Jev-Style 2B v2": {
|
| 61 |
+
"main": "2-26",
|
| 62 |
+
"sub": "letter-capped"
|
| 63 |
+
},
|
| 64 |
+
"Jev-Style 0.8B v3": {
|
| 65 |
+
"main": "No letter cap",
|
| 66 |
+
"sub": "tested with 77 options"
|
| 67 |
+
}
|
| 68 |
+
},
|
| 69 |
+
{
|
| 70 |
+
"row": "Context",
|
| 71 |
+
"Jev-Style 2B v1": {
|
| 72 |
+
"main": "Not stated",
|
| 73 |
+
"sub": "quickstart: server default"
|
| 74 |
+
},
|
| 75 |
+
"Jev-Style 2B v2": {
|
| 76 |
+
"main": "1,024-token prompt",
|
| 77 |
+
"sub": "quickstart runs -c 2048"
|
| 78 |
+
},
|
| 79 |
+
"Jev-Style 0.8B v3": {
|
| 80 |
+
"main": "25,600 tokens",
|
| 81 |
+
"sub": "preregistered 25K claim passed"
|
| 82 |
+
}
|
| 83 |
+
},
|
| 84 |
+
{
|
| 85 |
+
"row": "Languages",
|
| 86 |
+
"Jev-Style 2B v1": {
|
| 87 |
+
"main": "English",
|
| 88 |
+
"sub": "five English task families"
|
| 89 |
+
},
|
| 90 |
+
"Jev-Style 2B v2": {
|
| 91 |
+
"main": "English",
|
| 92 |
+
"sub": "English state required"
|
| 93 |
+
},
|
| 94 |
+
"Jev-Style 0.8B v3": {
|
| 95 |
+
"main": "51 evaluated",
|
| 96 |
+
"sub": "MASSIVE locales; 19 in fine-tuning"
|
| 97 |
+
}
|
| 98 |
+
},
|
| 99 |
+
{
|
| 100 |
+
"row": "Questions per state read",
|
| 101 |
+
"Jev-Style 2B v1": {
|
| 102 |
+
"main": "1",
|
| 103 |
+
"sub": "one question per prompt"
|
| 104 |
+
},
|
| 105 |
+
"Jev-Style 2B v2": {
|
| 106 |
+
"main": "1",
|
| 107 |
+
"sub": "one question per prompt"
|
| 108 |
+
},
|
| 109 |
+
"Jev-Style 0.8B v3": {
|
| 110 |
+
"main": "Many",
|
| 111 |
+
"sub": "all questions in one call"
|
| 112 |
+
}
|
| 113 |
+
},
|
| 114 |
+
{
|
| 115 |
+
"row": "Q4_K_M file",
|
| 116 |
+
"Jev-Style 2B v1": {
|
| 117 |
+
"main": "1.3 GB",
|
| 118 |
+
"sub": "as reported on the v1 GGUF card"
|
| 119 |
+
},
|
| 120 |
+
"Jev-Style 2B v2": {
|
| 121 |
+
"main": "1.27 GB",
|
| 122 |
+
"sub": "as reported on the v2 GGUF card"
|
| 123 |
+
},
|
| 124 |
+
"Jev-Style 0.8B v3": {
|
| 125 |
+
"main": "0.53 GB",
|
| 126 |
+
"sub": "matches FP32 on 240/240 parity rows"
|
| 127 |
+
}
|
| 128 |
+
},
|
| 129 |
+
{
|
| 130 |
+
"row": "Typed decisions, teacher agreement",
|
| 131 |
+
"Jev-Style 2B v1": {
|
| 132 |
+
"main": "53.35%",
|
| 133 |
+
"sub": "2,000 decisions / 400 states"
|
| 134 |
+
},
|
| 135 |
+
"Jev-Style 2B v2": {
|
| 136 |
+
"main": "73.45%",
|
| 137 |
+
"sub": "same 2,000 decisions"
|
| 138 |
+
},
|
| 139 |
+
"Jev-Style 0.8B v3": {
|
| 140 |
+
"main": "79.15%",
|
| 141 |
+
"sub": "same 2,000 \u00b7 1,583 correct"
|
| 142 |
+
}
|
| 143 |
+
}
|
| 144 |
+
],
|
| 145 |
+
"numbers": {
|
| 146 |
+
"v1_q4_k_m_agreement": 0.944,
|
| 147 |
+
"v2_q4_k_m_agreement": 0.914,
|
| 148 |
+
"v3_q4_k_m_agreement": {
|
| 149 |
+
"agree": 240,
|
| 150 |
+
"n": 240,
|
| 151 |
+
"value": 1.0
|
| 152 |
+
},
|
| 153 |
+
"typed_teacher_agreement": {
|
| 154 |
+
"v1": 0.5335,
|
| 155 |
+
"v2": 0.7345,
|
| 156 |
+
"v3": 0.7915,
|
| 157 |
+
"v3_correct": 1583,
|
| 158 |
+
"n": 2000,
|
| 159 |
+
"states": 400
|
| 160 |
+
},
|
| 161 |
+
"params": {
|
| 162 |
+
"v1": "2B",
|
| 163 |
+
"v2": "2B",
|
| 164 |
+
"v3_text_model": 752393024
|
| 165 |
+
},
|
| 166 |
+
"size_ratio_v3_over_2b": 0.4,
|
| 167 |
+
"context_ratio_v3_over_v2_prompt": 25.0
|
| 168 |
+
},
|
| 169 |
+
"sources": {
|
| 170 |
+
"v1": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF/raw/main/README.md (file table 'Same decision as bf16' Q4_K_M 94.4%; 'LoRA rank 16 on all linear layers'; 'Up to 26 options (20 when ... top_logprobs)'; 'Trained on five English task families'; quickstart llama-server without -c; prompt has one [Question])",
|
| 171 |
+
"v2": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/raw/main/README.md (Q4_K_M 91.4% choice agreement vs CUDA BF16 on frozen 500-decision subset; '2-26 unique options, within a 1,024-token prompt'; 'English state'; quickstart -c 2048; rank-32 LoRA, 33,638,400 trainable; continued from v1; ' A' through ' Z' readout; typed-decisions 53.35% v1 / 73.45% v2, 2,000 decisions from 400 states)",
|
| 172 |
+
"v3": "chart_data.json: design.generations (export_manifest.json parameters_text_model=752393024; config.resolved.json readout=verdict, no LoRA keys; jevbench/results.json scorer.max_len=25600; apps.jsonl banking77_full 400 rows x 77 options; scoreboard long_grid_plus claim_25k_ok=true), quant.v3 gguf-q4_k_m 240/240 (export_r2/main/parity/report.json), typed.accuracy.v3 1583/2000 (typed_test.jsonl, 400 group_ids)"
|
| 173 |
+
},
|
| 174 |
+
"footnote": "v1/v2: as reported on their public Hugging Face cards (v1 GGUF card; v2 and v2-GGUF cards; v1's typed-decisions number is reported on the v2 card). v3: release manifest, training config and eval files; Q4_K_M size = exported file (GB = 10^9 bytes), parity rows drawn from the training pool. Typed decisions: same 2,000 decisions from 400 states; v1/v2 scored by the v2 card's harness, v3 by ours. In-domain for v3; v1 not trained on typed decisions; v2's pool included typed workflow decisions. '51 evaluated' = MASSIVE locales (14 trained, 37 held out); fine-tuning covers 19 languages. 25,600 tokens = the runtime's whole-input limit; 25x = vs v2's 1,024-token prompt."
|
| 175 |
+
}
|
figures/design_table.png
ADDED
|
Git LFS Details
|
figures/design_table.svg
ADDED
|
|
figures/headline_typed.data.json
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"figure": "headline_typed",
|
| 3 |
+
"panels": {
|
| 4 |
+
"accuracy_pct": [
|
| 5 |
+
{
|
| 6 |
+
"label": "Jev-Style 2B v1",
|
| 7 |
+
"entry": "typed.teacher_agreement.v1",
|
| 8 |
+
"raw": 0.5335,
|
| 9 |
+
"plotted": 53.4,
|
| 10 |
+
"source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2/raw/main/README.md",
|
| 11 |
+
"field": "card text: 'The separate typed-decisions group contains 2,000 teacher-reference decisions from 400 states. Teacher agreement is 53.35% for v1, 37.55% for English Laya and 73.45% for v2'"
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"label": "Jev",
|
| 15 |
+
"entry": "typed.accuracy.jev",
|
| 16 |
+
"raw": 0.727,
|
| 17 |
+
"plotted": 72.7,
|
| 18 |
+
"source": "docs/round2_audit/track_jev_results.json",
|
| 19 |
+
"field": "string 'Jev typed-decisions (zero-shot)': '0.727 acc; KL 1.442; Brier 0.148; ECE 0.144; soft-acc 0.580' (from LocalLLaMA/typed-decisions dataset card / Laya HF card)"
|
| 20 |
+
},
|
| 21 |
+
{
|
| 22 |
+
"label": "Jev-Style 2B v2",
|
| 23 |
+
"entry": "typed.teacher_agreement.v2",
|
| 24 |
+
"raw": 0.7345,
|
| 25 |
+
"plotted": 73.5,
|
| 26 |
+
"source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2/raw/main/README.md",
|
| 27 |
+
"field": "card text: 'The separate typed-decisions group contains 2,000 teacher-reference decisions from 400 states. Teacher agreement is 53.35% for v1, 37.55% for English Laya and 73.45% for v2'"
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
"label": "Laya (typed ckpt)",
|
| 31 |
+
"entry": "typed.accuracy.laya_typed",
|
| 32 |
+
"raw": 0.766,
|
| 33 |
+
"plotted": 76.6,
|
| 34 |
+
"source": "runs/macjev/report_r2/scoreboard/scoreboard.json",
|
| 35 |
+
"field": "metrics['typed.accuracy'].laya.typed.value"
|
| 36 |
+
},
|
| 37 |
+
{
|
| 38 |
+
"label": "Jev-Style 0.8B v3",
|
| 39 |
+
"entry": "typed.accuracy.v3",
|
| 40 |
+
"raw": 0.7915,
|
| 41 |
+
"plotted": 79.2,
|
| 42 |
+
"source": "runs/macjev/report_r2/scoreboard/scoreboard.json",
|
| 43 |
+
"field": "metrics['typed.accuracy'].ours_value"
|
| 44 |
+
}
|
| 45 |
+
],
|
| 46 |
+
"brier_vs_soft": [
|
| 47 |
+
{
|
| 48 |
+
"label": "Jev",
|
| 49 |
+
"entry": "typed.Brier_vs_soft_labels.jev",
|
| 50 |
+
"raw": 0.148,
|
| 51 |
+
"plotted": 0.148,
|
| 52 |
+
"source": "docs/round2_audit/track_jev_results.json",
|
| 53 |
+
"field": "string 'Jev typed-decisions (zero-shot)': '0.727 acc; KL 1.442; Brier 0.148; ECE 0.144; soft-acc 0.580' (from LocalLLaMA/typed-decisions dataset card / Laya HF card)"
|
| 54 |
+
},
|
| 55 |
+
{
|
| 56 |
+
"label": "Laya (typed ckpt)",
|
| 57 |
+
"entry": "typed.brier_vs_soft.laya_typed",
|
| 58 |
+
"raw": 0.06146485330135357,
|
| 59 |
+
"plotted": 0.061,
|
| 60 |
+
"source": "runs/macjev/report_r2/scoreboard/scoreboard.json",
|
| 61 |
+
"field": "metrics['typed.brier_vs_soft'].laya.typed.value"
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"label": "Jev-Style 0.8B v3",
|
| 65 |
+
"entry": "typed.brier_vs_soft.v3",
|
| 66 |
+
"raw": 0.04583493309263009,
|
| 67 |
+
"plotted": 0.046,
|
| 68 |
+
"source": "runs/macjev/report_r2/scoreboard/scoreboard.json",
|
| 69 |
+
"field": "metrics['typed.brier_vs_soft'].ours_value"
|
| 70 |
+
}
|
| 71 |
+
]
|
| 72 |
+
},
|
| 73 |
+
"annotations": {
|
| 74 |
+
"delta_vs_jev_pts": 6.4,
|
| 75 |
+
"delta_vs_v2_pts": 5.7,
|
| 76 |
+
"delta_vs_laya_typed_pts": 2.6,
|
| 77 |
+
"brier_ratio_jev_over_v3": 3.229,
|
| 78 |
+
"brier_pct_lower_than_laya_typed": 25.4,
|
| 79 |
+
"v3_acc_ci95_wilson": [
|
| 80 |
+
0.7731,
|
| 81 |
+
0.8087
|
| 82 |
+
],
|
| 83 |
+
"v3_minus_laya_typed_paired_ci95": [
|
| 84 |
+
0.010499999999999954,
|
| 85 |
+
0.04249999999999998
|
| 86 |
+
]
|
| 87 |
+
},
|
| 88 |
+
"protocol": "in-domain for v3 and Laya typed; zero-shot for Jev (dataset card); 2B v1/v2 as reported on the v2 card",
|
| 89 |
+
"footnote": "Typed-decisions test set (LocalLLaMA/typed-decisions), 2,000 decisions from 400 states. In-domain for v3 and Laya typed (both trained on its\ntrain split); zero-shot for Jev (dataset-card numbers, Jev API, all 2,000 decisions). Laya: official typed-decisions checkpoint re-run by us on\nidentical rows with its shipped temperature. 2B v1/v2: teacher agreement as reported on the v2 card (same 2,000 decisions, that card's harness;\nv1 was not trained on typed decisions, v2's pool included typed workflow decisions). v3 95% CI 77.3\u201380.9% (Wilson);\nv3 minus Laya typed, paired bootstrap 95% CI +1.0 to +4.2 pts. Plotted values: figures/headline_typed.data.json."
|
| 90 |
+
}
|
figures/headline_typed.png
ADDED
|
Git LFS Details
|
figures/headline_typed.svg
ADDED
|
|
figures/jevbench.data.json
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"chart": "jevbench",
|
| 3 |
+
"metric": "JevBench v1.4.1 public accuracy (231 items)",
|
| 4 |
+
"rows": [
|
| 5 |
+
{
|
| 6 |
+
"label": "Jev-Style 0.8B v3",
|
| 7 |
+
"key": "v3",
|
| 8 |
+
"accuracy": 0.6406926406926406,
|
| 9 |
+
"correct": 148,
|
| 10 |
+
"n": 231,
|
| 11 |
+
"colour": "#3456F0",
|
| 12 |
+
"source": "runs/macjev/received/ext_evals/main/jevbench/results.json :: public_accuracy"
|
| 13 |
+
},
|
| 14 |
+
{
|
| 15 |
+
"label": "Qwen3.5-0.8B Decision Model (M. Ghafiri)",
|
| 16 |
+
"key": "mghafiri-qwen3.5-0.8b-decision-model",
|
| 17 |
+
"accuracy": 0.5930735930735931,
|
| 18 |
+
"correct": 137,
|
| 19 |
+
"n": 231,
|
| 20 |
+
"colour": "#C4C4CB",
|
| 21 |
+
"source": "data/external/jevbench_v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[key=mghafiri-qwen3.5-0.8b-decision-model].public_accuracy",
|
| 22 |
+
"v3_lead_pts": 4.76
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"label": "Laya (ModernBERT-large, 421M)",
|
| 26 |
+
"key": "laya",
|
| 27 |
+
"accuracy": 0.5844155844155844,
|
| 28 |
+
"correct": 135,
|
| 29 |
+
"n": 231,
|
| 30 |
+
"colour": "#D0D0D6",
|
| 31 |
+
"source": "data/external/jevbench_v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[key=laya].public_accuracy",
|
| 32 |
+
"v3_lead_pts": 5.63
|
| 33 |
+
},
|
| 34 |
+
{
|
| 35 |
+
"label": "SimpleJev (Qwen3.5-0.8B)",
|
| 36 |
+
"key": "simplejev-qwen3.5-0.8b",
|
| 37 |
+
"accuracy": 0.5454545454545454,
|
| 38 |
+
"correct": 126,
|
| 39 |
+
"n": 231,
|
| 40 |
+
"colour": "#C4C4CB",
|
| 41 |
+
"source": "data/external/jevbench_v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[key=simplejev-qwen3.5-0.8b].public_accuracy",
|
| 42 |
+
"v3_lead_pts": 9.52
|
| 43 |
+
}
|
| 44 |
+
],
|
| 45 |
+
"footnote": "JevBench v1.4.1, public items only (231). v3: self-run zero-shot with the vendored official harness (commit 24b9b5c); training-pool contamination scan: 0 hits; not an official leaderboard entry. Other rows: public accuracy as published in the board's v1.4.1 results file (github.com/fstandhartinger/jevbench). Shown: Laya plus every Qwen3.5-0.8B-based system on the board. Laya and M. Ghafiri lie inside v3's 95% CI (57.7\u201370.0%, Wilson): those two leads are point estimates, not significant at n = 231.",
|
| 46 |
+
"v3_run": {
|
| 47 |
+
"source_path": "runs/macjev/received/ext_evals/main/jevbench/results.json",
|
| 48 |
+
"field": "public_accuracy (148/231; independent check equal)",
|
| 49 |
+
"protocol_label": "v3 self-run with the vendored official harness on the 231 public items, zero-shot (contamination scan of the training pool: 0 hits, runs/macjev/external/jevbench/contamination.json); other rows as published in jevbench v1.4.1 results.json (full runs by the board); not an official leaderboard entry",
|
| 50 |
+
"ci95": [
|
| 51 |
+
0.577,
|
| 52 |
+
0.6998
|
| 53 |
+
],
|
| 54 |
+
"ci_method": "Wilson 95%"
|
| 55 |
+
},
|
| 56 |
+
"board": {
|
| 57 |
+
"source_path": "data/external/jevbench_v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json",
|
| 58 |
+
"protocol_label": "as published by the board (commit 24b9b5c1609a, tag v1.4.1)"
|
| 59 |
+
}
|
| 60 |
+
}
|
figures/jevbench.png
ADDED
|
Git LFS Details
|
figures/jevbench.svg
ADDED
|
|
figures/latency.data.json
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"chart": "latency",
|
| 3 |
+
"from_chart_data_entry": "latency.idle",
|
| 4 |
+
"source_path": "runs/macjev/latency/m1max_untrained_base_2026-09-23_review_idle/latency.json",
|
| 5 |
+
"field": "results['llamacpp-f16'|'laya-multilingual-mps-fp32'][cell].p50_ms",
|
| 6 |
+
"protocol_label": "M1 Max 64 GB, warm end-to-end p50 ms, idle run 2026-09-23 (load avg 3-8); untrained identical-architecture Qwen3.5-0.8B export (latency does not depend on weights); v3 = llama.cpp GGUF F16, one call per state; comparison engine = round-1 MacLaya-4K (our Laya-multilingual fine-tune, FP32/MPS, 4,096-token budget), one decide() per question; prefix reuse off",
|
| 7 |
+
"plotted": [
|
| 8 |
+
{
|
| 9 |
+
"cell": "1024x5",
|
| 10 |
+
"v3_p50_ms": 394,
|
| 11 |
+
"maclaya4k_p50_ms": 543,
|
| 12 |
+
"speedup_label": "1.4x"
|
| 13 |
+
},
|
| 14 |
+
{
|
| 15 |
+
"cell": "1024x10",
|
| 16 |
+
"v3_p50_ms": 525,
|
| 17 |
+
"maclaya4k_p50_ms": 1019,
|
| 18 |
+
"speedup_label": "1.9x"
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"cell": "4096x5",
|
| 22 |
+
"v3_p50_ms": 1212,
|
| 23 |
+
"maclaya4k_p50_ms": 3204,
|
| 24 |
+
"speedup_label": "2.6x"
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"cell": "4096x10",
|
| 28 |
+
"v3_p50_ms": 1381,
|
| 29 |
+
"maclaya4k_p50_ms": 6364,
|
| 30 |
+
"speedup_label": "4.6x"
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"cell": "8192x1",
|
| 34 |
+
"v3_p50_ms": 2295,
|
| 35 |
+
"maclaya4k_p50_ms": null,
|
| 36 |
+
"speedup_label": "v3 only"
|
| 37 |
+
},
|
| 38 |
+
{
|
| 39 |
+
"cell": "8192x5",
|
| 40 |
+
"v3_p50_ms": 2424,
|
| 41 |
+
"maclaya4k_p50_ms": null,
|
| 42 |
+
"speedup_label": "v3 only"
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"cell": "8192x10",
|
| 46 |
+
"v3_p50_ms": 2613,
|
| 47 |
+
"maclaya4k_p50_ms": null,
|
| 48 |
+
"speedup_label": "v3 only"
|
| 49 |
+
}
|
| 50 |
+
]
|
| 51 |
+
}
|
figures/latency.png
ADDED
|
Git LFS Details
|
figures/latency.svg
ADDED
|
|
figures/long_context.json
ADDED
|
@@ -0,0 +1,216 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"figure": "long_context",
|
| 3 |
+
"plotted": [
|
| 4 |
+
{
|
| 5 |
+
"length_bin": 1024,
|
| 6 |
+
"items": 320,
|
| 7 |
+
"controlled_pct": 57.81,
|
| 8 |
+
"ci95_pct": [
|
| 9 |
+
52.34,
|
| 10 |
+
63.1
|
| 11 |
+
]
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"length_bin": 2048,
|
| 15 |
+
"items": 320,
|
| 16 |
+
"controlled_pct": 54.69,
|
| 17 |
+
"ci95_pct": [
|
| 18 |
+
49.21,
|
| 19 |
+
60.05
|
| 20 |
+
]
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"length_bin": 4096,
|
| 24 |
+
"items": 320,
|
| 25 |
+
"controlled_pct": 58.13,
|
| 26 |
+
"ci95_pct": [
|
| 27 |
+
52.65,
|
| 28 |
+
63.4
|
| 29 |
+
]
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"length_bin": 8192,
|
| 33 |
+
"items": 320,
|
| 34 |
+
"controlled_pct": 54.37,
|
| 35 |
+
"ci95_pct": [
|
| 36 |
+
48.9,
|
| 37 |
+
59.75
|
| 38 |
+
]
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"length_bin": 12288,
|
| 42 |
+
"items": 320,
|
| 43 |
+
"controlled_pct": 54.37,
|
| 44 |
+
"ci95_pct": [
|
| 45 |
+
48.9,
|
| 46 |
+
59.75
|
| 47 |
+
]
|
| 48 |
+
},
|
| 49 |
+
{
|
| 50 |
+
"length_bin": 16384,
|
| 51 |
+
"items": 320,
|
| 52 |
+
"controlled_pct": 54.69,
|
| 53 |
+
"ci95_pct": [
|
| 54 |
+
49.21,
|
| 55 |
+
60.05
|
| 56 |
+
]
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"length_bin": 24576,
|
| 60 |
+
"items": 1280,
|
| 61 |
+
"controlled_pct": 55.31,
|
| 62 |
+
"ci95_pct": [
|
| 63 |
+
52.58,
|
| 64 |
+
58.02
|
| 65 |
+
]
|
| 66 |
+
}
|
| 67 |
+
],
|
| 68 |
+
"reference_controlled_2k_4k_pct": 56.41,
|
| 69 |
+
"gap_24k_vs_ref_pts": -1.09,
|
| 70 |
+
"spread_all_bins_pts": 3.75,
|
| 71 |
+
"claim_25k_ok": true,
|
| 72 |
+
"a_gap_ok": true,
|
| 73 |
+
"b_deciles_ok": true,
|
| 74 |
+
"c_controls_ok": true,
|
| 75 |
+
"verified_length_bin": 4096,
|
| 76 |
+
"caution": "verified_length_bin = 4096: the stricter every-bin (8K-16K) decile criterion fails in some deciles; the card may say 'preregistered 25K claim passed' but not 'verified at every length'",
|
| 77 |
+
"controlled_24k_by_decile": {
|
| 78 |
+
"0": 0.5234375,
|
| 79 |
+
"1": 0.515625,
|
| 80 |
+
"2": 0.546875,
|
| 81 |
+
"3": 0.5390625,
|
| 82 |
+
"4": 0.5546875,
|
| 83 |
+
"5": 0.5546875,
|
| 84 |
+
"6": 0.5703125,
|
| 85 |
+
"7": 0.5625,
|
| 86 |
+
"8": 0.5390625,
|
| 87 |
+
"9": 0.625
|
| 88 |
+
},
|
| 89 |
+
"controls_pooled_by_length": {
|
| 90 |
+
"1024": {
|
| 91 |
+
"question_only": {
|
| 92 |
+
"n": 320,
|
| 93 |
+
"accuracy": 0.290625,
|
| 94 |
+
"chance": 0.28003348214285717,
|
| 95 |
+
"p_above_chance": 0.3511374281963401,
|
| 96 |
+
"at_chance": true
|
| 97 |
+
},
|
| 98 |
+
"state_swap": {
|
| 99 |
+
"n": 320,
|
| 100 |
+
"accuracy": 0.315625,
|
| 101 |
+
"chance": 0.28003348214285717,
|
| 102 |
+
"p_above_chance": 0.07484199516389486,
|
| 103 |
+
"at_chance": true
|
| 104 |
+
}
|
| 105 |
+
},
|
| 106 |
+
"2048": {
|
| 107 |
+
"question_only": {
|
| 108 |
+
"n": 320,
|
| 109 |
+
"accuracy": 0.2875,
|
| 110 |
+
"chance": 0.28029017857142857,
|
| 111 |
+
"p_above_chance": 0.40558180434986946,
|
| 112 |
+
"at_chance": true
|
| 113 |
+
},
|
| 114 |
+
"state_swap": {
|
| 115 |
+
"n": 320,
|
| 116 |
+
"accuracy": 0.303125,
|
| 117 |
+
"chance": 0.28029017857142857,
|
| 118 |
+
"p_above_chance": 0.18406466252955953,
|
| 119 |
+
"at_chance": true
|
| 120 |
+
}
|
| 121 |
+
},
|
| 122 |
+
"4096": {
|
| 123 |
+
"question_only": {
|
| 124 |
+
"n": 320,
|
| 125 |
+
"accuracy": 0.253125,
|
| 126 |
+
"chance": 0.28010044642857146,
|
| 127 |
+
"p_above_chance": 0.8864307901965104,
|
| 128 |
+
"at_chance": true
|
| 129 |
+
},
|
| 130 |
+
"state_swap": {
|
| 131 |
+
"n": 320,
|
| 132 |
+
"accuracy": 0.29375,
|
| 133 |
+
"chance": 0.28010044642857146,
|
| 134 |
+
"p_above_chance": 0.304486548332512,
|
| 135 |
+
"at_chance": true
|
| 136 |
+
}
|
| 137 |
+
},
|
| 138 |
+
"8192": {
|
| 139 |
+
"question_only": {
|
| 140 |
+
"n": 320,
|
| 141 |
+
"accuracy": 0.28125,
|
| 142 |
+
"chance": 0.28131324404761904,
|
| 143 |
+
"p_above_chance": 0.5273741085290076,
|
| 144 |
+
"at_chance": true
|
| 145 |
+
},
|
| 146 |
+
"state_swap": {
|
| 147 |
+
"n": 320,
|
| 148 |
+
"accuracy": 0.284375,
|
| 149 |
+
"chance": 0.28131324404761904,
|
| 150 |
+
"p_above_chance": 0.4747527191420883,
|
| 151 |
+
"at_chance": true
|
| 152 |
+
}
|
| 153 |
+
},
|
| 154 |
+
"12288": {
|
| 155 |
+
"question_only": {
|
| 156 |
+
"n": 320,
|
| 157 |
+
"accuracy": 0.278125,
|
| 158 |
+
"chance": 0.2804017857142857,
|
| 159 |
+
"p_above_chance": 0.5644923218252806,
|
| 160 |
+
"at_chance": true
|
| 161 |
+
},
|
| 162 |
+
"state_swap": {
|
| 163 |
+
"n": 320,
|
| 164 |
+
"accuracy": 0.284375,
|
| 165 |
+
"chance": 0.2804017857142857,
|
| 166 |
+
"p_above_chance": 0.4593971620159274,
|
| 167 |
+
"at_chance": true
|
| 168 |
+
}
|
| 169 |
+
},
|
| 170 |
+
"16384": {
|
| 171 |
+
"question_only": {
|
| 172 |
+
"n": 320,
|
| 173 |
+
"accuracy": 0.265625,
|
| 174 |
+
"chance": 0.28533854166666667,
|
| 175 |
+
"p_above_chance": 0.8139517721332229,
|
| 176 |
+
"at_chance": true
|
| 177 |
+
},
|
| 178 |
+
"state_swap": {
|
| 179 |
+
"n": 320,
|
| 180 |
+
"accuracy": 0.28125,
|
| 181 |
+
"chance": 0.28533854166666667,
|
| 182 |
+
"p_above_chance": 0.5936977453471026,
|
| 183 |
+
"at_chance": true
|
| 184 |
+
}
|
| 185 |
+
},
|
| 186 |
+
"24576": {
|
| 187 |
+
"question_only": {
|
| 188 |
+
"n": 1280,
|
| 189 |
+
"accuracy": 0.2890625,
|
| 190 |
+
"chance": 0.2837481398809524,
|
| 191 |
+
"p_above_chance": 0.33937799270227736,
|
| 192 |
+
"at_chance": true
|
| 193 |
+
},
|
| 194 |
+
"state_swap": {
|
| 195 |
+
"n": 1280,
|
| 196 |
+
"accuracy": 0.28515625,
|
| 197 |
+
"chance": 0.2837481398809524,
|
| 198 |
+
"p_above_chance": 0.4658977510199284,
|
| 199 |
+
"at_chance": true
|
| 200 |
+
}
|
| 201 |
+
}
|
| 202 |
+
},
|
| 203 |
+
"laya_budgets": {
|
| 204 |
+
"laya_english": 512,
|
| 205 |
+
"laya_multilingual": 1024,
|
| 206 |
+
"laya_typed_decisions": 1024,
|
| 207 |
+
"source": "third_party/laya/README.md lines 366-368 (defaults); models/laya-official/.../multilingual/rl_agent_config.json max_len 1024"
|
| 208 |
+
},
|
| 209 |
+
"v3_context_tokens": 25600,
|
| 210 |
+
"sources": [
|
| 211 |
+
"runs/macjev/received/own_evals/main/test.own/long_grid_plus.jsonl + runs/macjev/received/own_evals/main/temperatures.json via macjev.eval.aggregate.suite_metrics('long_grid_plus') (recomputed); matches runs/macjev/report_r2/scoreboard/scoreboard.json long_grid_plus",
|
| 212 |
+
"chart_data.json#long_grid_plus.bins"
|
| 213 |
+
],
|
| 214 |
+
"protocol_label": "v3 only (Laya cannot run >4K; not applicable). Controlled = real row correct AND question_only + state_swap controls pass; Wilson CI; 2K-4K reference; preregistered 2026-09-24, amended pre-data (+96 items per 24K decile, thresholds unchanged)",
|
| 215 |
+
"footnote": "v3 only. Laya's default input budget is 512 tokens (English) / 1,024 (multilingual, typed) per the Laya README;\nlonger rows exceed Laya's default budgets, so Laya is not plotted. Suite long_grid_plus: preregistered 2026-09-24, amended pre-data\n(+96 items per 24K depth decile, thresholds unchanged); 320 items per bin, 1,280 at 24K. Controlled accuracy = real-state\nanswer correct AND its question-only and state-swap controls pass; both controls are at chance in every length bin.\n25K claim rule: |24K \u2212 2K\u20134K reference| \u2264 5 pts and every 24K evidence-depth decile within 10 pts of it.\nPlotted values: figures/long_context.json (bin-level controlled accuracy and CIs)."
|
| 216 |
+
}
|
figures/long_context.png
ADDED
|
Git LFS Details
|
figures/long_context.svg
ADDED
|
|
figures/multilingual.data.json
ADDED
|
@@ -0,0 +1,486 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"chart": "multilingual",
|
| 3 |
+
"metric": "per-language MASSIVE intent accuracy, argmax over 20 scored options, 100 rows/language",
|
| 4 |
+
"sources": {
|
| 5 |
+
"v3": "runs/macjev/received/20260924-0313/evals/main/test/massive51.jsonl",
|
| 6 |
+
"laya_multilingual": "runs/macjev/laya_baselines/multilingual/massive51.jsonl",
|
| 7 |
+
"chart_data_entry": "chart_data.json#massive51.per_language (re-verified against raw files)"
|
| 8 |
+
},
|
| 9 |
+
"protocol": "14 trained + 37 locales held out of MASSIVE training for v3; held-out for all Laya checkpoints. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures. Wilson 95% CI per language (n=100)",
|
| 10 |
+
"chance": 0.05,
|
| 11 |
+
"three_x_chance": 0.15000000000000002,
|
| 12 |
+
"summary": {
|
| 13 |
+
"languages": 51,
|
| 14 |
+
"v3_beats_laya_multilingual": 51,
|
| 15 |
+
"v3_beats_best_laya_checkpoint": 51,
|
| 16 |
+
"v3_beats_laya_multilingual_never_trained": 37,
|
| 17 |
+
"v3_above_3x_chance": 51,
|
| 18 |
+
"laya_multilingual_above_3x_chance": 48,
|
| 19 |
+
"macro_v3": 0.7174509803921569,
|
| 20 |
+
"macro_laya_multilingual": 0.4007843137254902,
|
| 21 |
+
"macro_v3_never_trained37": 0.6551351351351351,
|
| 22 |
+
"macro_laya_multilingual_never_trained37": 0.36108108108108106,
|
| 23 |
+
"min_margin_vs_laya_multilingual": 0.11
|
| 24 |
+
},
|
| 25 |
+
"rows_plotted_sorted": [
|
| 26 |
+
{
|
| 27 |
+
"lang": "zh-CN",
|
| 28 |
+
"name": "Chinese (CN)",
|
| 29 |
+
"trained": true,
|
| 30 |
+
"v3": 0.95,
|
| 31 |
+
"laya_multilingual": 0.65,
|
| 32 |
+
"best_laya": 0.65,
|
| 33 |
+
"best_laya_ckpt": "multilingual"
|
| 34 |
+
},
|
| 35 |
+
{
|
| 36 |
+
"lang": "ja",
|
| 37 |
+
"name": "Japanese",
|
| 38 |
+
"trained": true,
|
| 39 |
+
"v3": 0.95,
|
| 40 |
+
"laya_multilingual": 0.64,
|
| 41 |
+
"best_laya": 0.64,
|
| 42 |
+
"best_laya_ckpt": "multilingual"
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"lang": "es",
|
| 46 |
+
"name": "Spanish",
|
| 47 |
+
"trained": true,
|
| 48 |
+
"v3": 0.95,
|
| 49 |
+
"laya_multilingual": 0.58,
|
| 50 |
+
"best_laya": 0.58,
|
| 51 |
+
"best_laya_ckpt": "multilingual"
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"lang": "ru",
|
| 55 |
+
"name": "Russian",
|
| 56 |
+
"trained": true,
|
| 57 |
+
"v3": 0.94,
|
| 58 |
+
"laya_multilingual": 0.57,
|
| 59 |
+
"best_laya": 0.57,
|
| 60 |
+
"best_laya_ckpt": "multilingual"
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"lang": "en",
|
| 64 |
+
"name": "English",
|
| 65 |
+
"trained": true,
|
| 66 |
+
"v3": 0.93,
|
| 67 |
+
"laya_multilingual": 0.71,
|
| 68 |
+
"best_laya": 0.82,
|
| 69 |
+
"best_laya_ckpt": "english"
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"lang": "zh-TW",
|
| 73 |
+
"name": "Chinese (TW)",
|
| 74 |
+
"trained": false,
|
| 75 |
+
"v3": 0.93,
|
| 76 |
+
"laya_multilingual": 0.61,
|
| 77 |
+
"best_laya": 0.61,
|
| 78 |
+
"best_laya_ckpt": "multilingual"
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
"lang": "de",
|
| 82 |
+
"name": "German",
|
| 83 |
+
"trained": true,
|
| 84 |
+
"v3": 0.93,
|
| 85 |
+
"laya_multilingual": 0.5,
|
| 86 |
+
"best_laya": 0.5,
|
| 87 |
+
"best_laya_ckpt": "multilingual"
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"lang": "fr",
|
| 91 |
+
"name": "French",
|
| 92 |
+
"trained": true,
|
| 93 |
+
"v3": 0.9,
|
| 94 |
+
"laya_multilingual": 0.6,
|
| 95 |
+
"best_laya": 0.6,
|
| 96 |
+
"best_laya_ckpt": "typed"
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"lang": "it",
|
| 100 |
+
"name": "Italian",
|
| 101 |
+
"trained": false,
|
| 102 |
+
"v3": 0.89,
|
| 103 |
+
"laya_multilingual": 0.52,
|
| 104 |
+
"best_laya": 0.52,
|
| 105 |
+
"best_laya_ckpt": "multilingual"
|
| 106 |
+
},
|
| 107 |
+
{
|
| 108 |
+
"lang": "pl",
|
| 109 |
+
"name": "Polish",
|
| 110 |
+
"trained": false,
|
| 111 |
+
"v3": 0.89,
|
| 112 |
+
"laya_multilingual": 0.5,
|
| 113 |
+
"best_laya": 0.5,
|
| 114 |
+
"best_laya_ckpt": "multilingual"
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"lang": "ko",
|
| 118 |
+
"name": "Korean",
|
| 119 |
+
"trained": true,
|
| 120 |
+
"v3": 0.89,
|
| 121 |
+
"laya_multilingual": 0.47,
|
| 122 |
+
"best_laya": 0.47,
|
| 123 |
+
"best_laya_ckpt": "multilingual"
|
| 124 |
+
},
|
| 125 |
+
{
|
| 126 |
+
"lang": "pt",
|
| 127 |
+
"name": "Portuguese",
|
| 128 |
+
"trained": true,
|
| 129 |
+
"v3": 0.88,
|
| 130 |
+
"laya_multilingual": 0.5,
|
| 131 |
+
"best_laya": 0.5,
|
| 132 |
+
"best_laya_ckpt": "typed"
|
| 133 |
+
},
|
| 134 |
+
{
|
| 135 |
+
"lang": "hi",
|
| 136 |
+
"name": "Hindi",
|
| 137 |
+
"trained": true,
|
| 138 |
+
"v3": 0.87,
|
| 139 |
+
"laya_multilingual": 0.46,
|
| 140 |
+
"best_laya": 0.46,
|
| 141 |
+
"best_laya_ckpt": "multilingual"
|
| 142 |
+
},
|
| 143 |
+
{
|
| 144 |
+
"lang": "sv",
|
| 145 |
+
"name": "Swedish",
|
| 146 |
+
"trained": false,
|
| 147 |
+
"v3": 0.85,
|
| 148 |
+
"laya_multilingual": 0.49,
|
| 149 |
+
"best_laya": 0.49,
|
| 150 |
+
"best_laya_ckpt": "multilingual"
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"lang": "ar",
|
| 154 |
+
"name": "Arabic",
|
| 155 |
+
"trained": true,
|
| 156 |
+
"v3": 0.85,
|
| 157 |
+
"laya_multilingual": 0.46,
|
| 158 |
+
"best_laya": 0.46,
|
| 159 |
+
"best_laya_ckpt": "multilingual"
|
| 160 |
+
},
|
| 161 |
+
{
|
| 162 |
+
"lang": "vi",
|
| 163 |
+
"name": "Vietnamese",
|
| 164 |
+
"trained": false,
|
| 165 |
+
"v3": 0.85,
|
| 166 |
+
"laya_multilingual": 0.34,
|
| 167 |
+
"best_laya": 0.34,
|
| 168 |
+
"best_laya_ckpt": "multilingual"
|
| 169 |
+
},
|
| 170 |
+
{
|
| 171 |
+
"lang": "fa",
|
| 172 |
+
"name": "Persian",
|
| 173 |
+
"trained": false,
|
| 174 |
+
"v3": 0.84,
|
| 175 |
+
"laya_multilingual": 0.51,
|
| 176 |
+
"best_laya": 0.51,
|
| 177 |
+
"best_laya_ckpt": "multilingual"
|
| 178 |
+
},
|
| 179 |
+
{
|
| 180 |
+
"lang": "id",
|
| 181 |
+
"name": "Indonesian",
|
| 182 |
+
"trained": false,
|
| 183 |
+
"v3": 0.84,
|
| 184 |
+
"laya_multilingual": 0.51,
|
| 185 |
+
"best_laya": 0.51,
|
| 186 |
+
"best_laya_ckpt": "multilingual"
|
| 187 |
+
},
|
| 188 |
+
{
|
| 189 |
+
"lang": "tr",
|
| 190 |
+
"name": "Turkish",
|
| 191 |
+
"trained": true,
|
| 192 |
+
"v3": 0.83,
|
| 193 |
+
"laya_multilingual": 0.4,
|
| 194 |
+
"best_laya": 0.4,
|
| 195 |
+
"best_laya_ckpt": "multilingual"
|
| 196 |
+
},
|
| 197 |
+
{
|
| 198 |
+
"lang": "nl",
|
| 199 |
+
"name": "Dutch",
|
| 200 |
+
"trained": false,
|
| 201 |
+
"v3": 0.82,
|
| 202 |
+
"laya_multilingual": 0.47,
|
| 203 |
+
"best_laya": 0.47,
|
| 204 |
+
"best_laya_ckpt": "multilingual"
|
| 205 |
+
},
|
| 206 |
+
{
|
| 207 |
+
"lang": "nb",
|
| 208 |
+
"name": "Norwegian",
|
| 209 |
+
"trained": false,
|
| 210 |
+
"v3": 0.81,
|
| 211 |
+
"laya_multilingual": 0.56,
|
| 212 |
+
"best_laya": 0.56,
|
| 213 |
+
"best_laya_ckpt": "multilingual"
|
| 214 |
+
},
|
| 215 |
+
{
|
| 216 |
+
"lang": "da",
|
| 217 |
+
"name": "Danish",
|
| 218 |
+
"trained": false,
|
| 219 |
+
"v3": 0.81,
|
| 220 |
+
"laya_multilingual": 0.52,
|
| 221 |
+
"best_laya": 0.52,
|
| 222 |
+
"best_laya_ckpt": "multilingual"
|
| 223 |
+
},
|
| 224 |
+
{
|
| 225 |
+
"lang": "ro",
|
| 226 |
+
"name": "Romanian",
|
| 227 |
+
"trained": false,
|
| 228 |
+
"v3": 0.77,
|
| 229 |
+
"laya_multilingual": 0.37,
|
| 230 |
+
"best_laya": 0.37,
|
| 231 |
+
"best_laya_ckpt": "multilingual"
|
| 232 |
+
},
|
| 233 |
+
{
|
| 234 |
+
"lang": "af",
|
| 235 |
+
"name": "Afrikaans",
|
| 236 |
+
"trained": false,
|
| 237 |
+
"v3": 0.76,
|
| 238 |
+
"laya_multilingual": 0.35,
|
| 239 |
+
"best_laya": 0.35,
|
| 240 |
+
"best_laya_ckpt": "multilingual"
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"lang": "ta",
|
| 244 |
+
"name": "Tamil",
|
| 245 |
+
"trained": true,
|
| 246 |
+
"v3": 0.74,
|
| 247 |
+
"laya_multilingual": 0.31,
|
| 248 |
+
"best_laya": 0.31,
|
| 249 |
+
"best_laya_ckpt": "multilingual"
|
| 250 |
+
},
|
| 251 |
+
{
|
| 252 |
+
"lang": "sw",
|
| 253 |
+
"name": "Swahili",
|
| 254 |
+
"trained": true,
|
| 255 |
+
"v3": 0.74,
|
| 256 |
+
"laya_multilingual": 0.23,
|
| 257 |
+
"best_laya": 0.23,
|
| 258 |
+
"best_laya_ckpt": "multilingual"
|
| 259 |
+
},
|
| 260 |
+
{
|
| 261 |
+
"lang": "ms",
|
| 262 |
+
"name": "Malay",
|
| 263 |
+
"trained": false,
|
| 264 |
+
"v3": 0.73,
|
| 265 |
+
"laya_multilingual": 0.41,
|
| 266 |
+
"best_laya": 0.41,
|
| 267 |
+
"best_laya_ckpt": "multilingual"
|
| 268 |
+
},
|
| 269 |
+
{
|
| 270 |
+
"lang": "el",
|
| 271 |
+
"name": "Greek",
|
| 272 |
+
"trained": false,
|
| 273 |
+
"v3": 0.72,
|
| 274 |
+
"laya_multilingual": 0.44,
|
| 275 |
+
"best_laya": 0.44,
|
| 276 |
+
"best_laya_ckpt": "multilingual"
|
| 277 |
+
},
|
| 278 |
+
{
|
| 279 |
+
"lang": "he",
|
| 280 |
+
"name": "Hebrew",
|
| 281 |
+
"trained": false,
|
| 282 |
+
"v3": 0.72,
|
| 283 |
+
"laya_multilingual": 0.37,
|
| 284 |
+
"best_laya": 0.37,
|
| 285 |
+
"best_laya_ckpt": "multilingual"
|
| 286 |
+
},
|
| 287 |
+
{
|
| 288 |
+
"lang": "ur",
|
| 289 |
+
"name": "Urdu",
|
| 290 |
+
"trained": false,
|
| 291 |
+
"v3": 0.7,
|
| 292 |
+
"laya_multilingual": 0.42,
|
| 293 |
+
"best_laya": 0.42,
|
| 294 |
+
"best_laya_ckpt": "multilingual"
|
| 295 |
+
},
|
| 296 |
+
{
|
| 297 |
+
"lang": "az",
|
| 298 |
+
"name": "Azerbaijani",
|
| 299 |
+
"trained": false,
|
| 300 |
+
"v3": 0.66,
|
| 301 |
+
"laya_multilingual": 0.36,
|
| 302 |
+
"best_laya": 0.36,
|
| 303 |
+
"best_laya_ckpt": "multilingual"
|
| 304 |
+
},
|
| 305 |
+
{
|
| 306 |
+
"lang": "bn",
|
| 307 |
+
"name": "Bengali",
|
| 308 |
+
"trained": false,
|
| 309 |
+
"v3": 0.65,
|
| 310 |
+
"laya_multilingual": 0.45,
|
| 311 |
+
"best_laya": 0.45,
|
| 312 |
+
"best_laya_ckpt": "multilingual"
|
| 313 |
+
},
|
| 314 |
+
{
|
| 315 |
+
"lang": "sl",
|
| 316 |
+
"name": "Slovenian",
|
| 317 |
+
"trained": false,
|
| 318 |
+
"v3": 0.65,
|
| 319 |
+
"laya_multilingual": 0.37,
|
| 320 |
+
"best_laya": 0.37,
|
| 321 |
+
"best_laya_ckpt": "multilingual"
|
| 322 |
+
},
|
| 323 |
+
{
|
| 324 |
+
"lang": "th",
|
| 325 |
+
"name": "Thai",
|
| 326 |
+
"trained": false,
|
| 327 |
+
"v3": 0.64,
|
| 328 |
+
"laya_multilingual": 0.48,
|
| 329 |
+
"best_laya": 0.48,
|
| 330 |
+
"best_laya_ckpt": "multilingual"
|
| 331 |
+
},
|
| 332 |
+
{
|
| 333 |
+
"lang": "hu",
|
| 334 |
+
"name": "Hungarian",
|
| 335 |
+
"trained": false,
|
| 336 |
+
"v3": 0.63,
|
| 337 |
+
"laya_multilingual": 0.36,
|
| 338 |
+
"best_laya": 0.36,
|
| 339 |
+
"best_laya_ckpt": "multilingual"
|
| 340 |
+
},
|
| 341 |
+
{
|
| 342 |
+
"lang": "te",
|
| 343 |
+
"name": "Telugu",
|
| 344 |
+
"trained": false,
|
| 345 |
+
"v3": 0.62,
|
| 346 |
+
"laya_multilingual": 0.22,
|
| 347 |
+
"best_laya": 0.22,
|
| 348 |
+
"best_laya_ckpt": "multilingual"
|
| 349 |
+
},
|
| 350 |
+
{
|
| 351 |
+
"lang": "kn",
|
| 352 |
+
"name": "Kannada",
|
| 353 |
+
"trained": false,
|
| 354 |
+
"v3": 0.61,
|
| 355 |
+
"laya_multilingual": 0.3,
|
| 356 |
+
"best_laya": 0.3,
|
| 357 |
+
"best_laya_ckpt": "multilingual"
|
| 358 |
+
},
|
| 359 |
+
{
|
| 360 |
+
"lang": "fi",
|
| 361 |
+
"name": "Finnish",
|
| 362 |
+
"trained": false,
|
| 363 |
+
"v3": 0.6,
|
| 364 |
+
"laya_multilingual": 0.34,
|
| 365 |
+
"best_laya": 0.34,
|
| 366 |
+
"best_laya_ckpt": "multilingual"
|
| 367 |
+
},
|
| 368 |
+
{
|
| 369 |
+
"lang": "jv",
|
| 370 |
+
"name": "Javanese",
|
| 371 |
+
"trained": false,
|
| 372 |
+
"v3": 0.6,
|
| 373 |
+
"laya_multilingual": 0.3,
|
| 374 |
+
"best_laya": 0.3,
|
| 375 |
+
"best_laya_ckpt": "multilingual"
|
| 376 |
+
},
|
| 377 |
+
{
|
| 378 |
+
"lang": "hy",
|
| 379 |
+
"name": "Armenian",
|
| 380 |
+
"trained": false,
|
| 381 |
+
"v3": 0.59,
|
| 382 |
+
"laya_multilingual": 0.25,
|
| 383 |
+
"best_laya": 0.25,
|
| 384 |
+
"best_laya_ckpt": "multilingual"
|
| 385 |
+
},
|
| 386 |
+
{
|
| 387 |
+
"lang": "ka",
|
| 388 |
+
"name": "Georgian",
|
| 389 |
+
"trained": false,
|
| 390 |
+
"v3": 0.56,
|
| 391 |
+
"laya_multilingual": 0.15,
|
| 392 |
+
"best_laya": 0.15,
|
| 393 |
+
"best_laya_ckpt": "multilingual"
|
| 394 |
+
},
|
| 395 |
+
{
|
| 396 |
+
"lang": "tl",
|
| 397 |
+
"name": "Tagalog",
|
| 398 |
+
"trained": false,
|
| 399 |
+
"v3": 0.54,
|
| 400 |
+
"laya_multilingual": 0.35,
|
| 401 |
+
"best_laya": 0.35,
|
| 402 |
+
"best_laya_ckpt": "multilingual"
|
| 403 |
+
},
|
| 404 |
+
{
|
| 405 |
+
"lang": "km",
|
| 406 |
+
"name": "Khmer",
|
| 407 |
+
"trained": false,
|
| 408 |
+
"v3": 0.54,
|
| 409 |
+
"laya_multilingual": 0.2,
|
| 410 |
+
"best_laya": 0.2,
|
| 411 |
+
"best_laya_ckpt": "multilingual"
|
| 412 |
+
},
|
| 413 |
+
{
|
| 414 |
+
"lang": "sq",
|
| 415 |
+
"name": "Albanian",
|
| 416 |
+
"trained": false,
|
| 417 |
+
"v3": 0.52,
|
| 418 |
+
"laya_multilingual": 0.3,
|
| 419 |
+
"best_laya": 0.3,
|
| 420 |
+
"best_laya_ckpt": "multilingual"
|
| 421 |
+
},
|
| 422 |
+
{
|
| 423 |
+
"lang": "ml",
|
| 424 |
+
"name": "Malayalam",
|
| 425 |
+
"trained": false,
|
| 426 |
+
"v3": 0.52,
|
| 427 |
+
"laya_multilingual": 0.28,
|
| 428 |
+
"best_laya": 0.28,
|
| 429 |
+
"best_laya_ckpt": "multilingual"
|
| 430 |
+
},
|
| 431 |
+
{
|
| 432 |
+
"lang": "lv",
|
| 433 |
+
"name": "Latvian",
|
| 434 |
+
"trained": false,
|
| 435 |
+
"v3": 0.51,
|
| 436 |
+
"laya_multilingual": 0.31,
|
| 437 |
+
"best_laya": 0.31,
|
| 438 |
+
"best_laya_ckpt": "multilingual"
|
| 439 |
+
},
|
| 440 |
+
{
|
| 441 |
+
"lang": "is",
|
| 442 |
+
"name": "Icelandic",
|
| 443 |
+
"trained": false,
|
| 444 |
+
"v3": 0.5,
|
| 445 |
+
"laya_multilingual": 0.35,
|
| 446 |
+
"best_laya": 0.35,
|
| 447 |
+
"best_laya_ckpt": "multilingual"
|
| 448 |
+
},
|
| 449 |
+
{
|
| 450 |
+
"lang": "my",
|
| 451 |
+
"name": "Burmese",
|
| 452 |
+
"trained": false,
|
| 453 |
+
"v3": 0.46,
|
| 454 |
+
"laya_multilingual": 0.16,
|
| 455 |
+
"best_laya": 0.16,
|
| 456 |
+
"best_laya_ckpt": "multilingual"
|
| 457 |
+
},
|
| 458 |
+
{
|
| 459 |
+
"lang": "cy",
|
| 460 |
+
"name": "Welsh",
|
| 461 |
+
"trained": false,
|
| 462 |
+
"v3": 0.37,
|
| 463 |
+
"laya_multilingual": 0.13,
|
| 464 |
+
"best_laya": 0.17,
|
| 465 |
+
"best_laya_ckpt": "typed"
|
| 466 |
+
},
|
| 467 |
+
{
|
| 468 |
+
"lang": "mn",
|
| 469 |
+
"name": "Mongolian",
|
| 470 |
+
"trained": false,
|
| 471 |
+
"v3": 0.28,
|
| 472 |
+
"laya_multilingual": 0.16,
|
| 473 |
+
"best_laya": 0.16,
|
| 474 |
+
"best_laya_ckpt": "multilingual"
|
| 475 |
+
},
|
| 476 |
+
{
|
| 477 |
+
"lang": "am",
|
| 478 |
+
"name": "Amharic",
|
| 479 |
+
"trained": false,
|
| 480 |
+
"v3": 0.26,
|
| 481 |
+
"laya_multilingual": 0.15,
|
| 482 |
+
"best_laya": 0.16,
|
| 483 |
+
"best_laya_ckpt": "typed"
|
| 484 |
+
}
|
| 485 |
+
]
|
| 486 |
+
}
|
figures/multilingual.png
ADDED
|
Git LFS Details
|
figures/multilingual.svg
ADDED
|
|
figures/quantization.data.json
ADDED
|
@@ -0,0 +1,151 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"chart": "quantization",
|
| 3 |
+
"metric": "top-1 agreement with the full-precision reference (%), plus file size (GB = bytes/1e9)",
|
| 4 |
+
"rows": [
|
| 5 |
+
{
|
| 6 |
+
"tier": "16-bit",
|
| 7 |
+
"model": "v3",
|
| 8 |
+
"label": "Jev-Style 0.8B v3 \u00b7 GGUF F16",
|
| 9 |
+
"agree": 100.0,
|
| 10 |
+
"n": 240,
|
| 11 |
+
"agree_count": 240,
|
| 12 |
+
"size_gb": 1.516744128,
|
| 13 |
+
"size_note": "local export file (bytes / 1e9)",
|
| 14 |
+
"reference": "torch FP32",
|
| 15 |
+
"source": "runs/macjev/export_r2/main/parity/report.json; sizes: chart_data.json export.sizes"
|
| 16 |
+
},
|
| 17 |
+
{
|
| 18 |
+
"tier": "16-bit",
|
| 19 |
+
"model": "v3",
|
| 20 |
+
"label": "Jev-Style 0.8B v3 \u00b7 MLX bf16",
|
| 21 |
+
"agree": 100.0,
|
| 22 |
+
"n": 240,
|
| 23 |
+
"agree_count": 240,
|
| 24 |
+
"size_gb": 1.504827355,
|
| 25 |
+
"size_note": "local export file (bytes / 1e9)",
|
| 26 |
+
"reference": "torch FP32",
|
| 27 |
+
"source": "runs/macjev/export_r2/main/parity/report.json; sizes: chart_data.json export.sizes"
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
"tier": "16-bit",
|
| 31 |
+
"model": "v2",
|
| 32 |
+
"label": "Jev-Style 2B v2 \u00b7 GGUF BF16",
|
| 33 |
+
"agree": 99.6,
|
| 34 |
+
"n": 500,
|
| 35 |
+
"size_gb": 3.78,
|
| 36 |
+
"size_note": "as reported on card",
|
| 37 |
+
"reference": "CUDA merged BF16",
|
| 38 |
+
"source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/raw/main/README.md"
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"tier": "16-bit",
|
| 42 |
+
"model": "v2",
|
| 43 |
+
"label": "Jev-Style 2B v2 \u00b7 MLX BF16",
|
| 44 |
+
"agree": 99.6,
|
| 45 |
+
"n": 500,
|
| 46 |
+
"size_gb": 3.76,
|
| 47 |
+
"size_note": "as reported on card",
|
| 48 |
+
"reference": "CUDA merged BF16",
|
| 49 |
+
"source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/raw/main/README.md (table: 'Native MLX BF16 | 3.76 GB | 99.6% choice agreement')"
|
| 50 |
+
},
|
| 51 |
+
{
|
| 52 |
+
"tier": "16-bit",
|
| 53 |
+
"model": "v1",
|
| 54 |
+
"label": "Jev-Style 2B v1 \u00b7 GGUF BF16",
|
| 55 |
+
"agree": 99.8,
|
| 56 |
+
"n": 500,
|
| 57 |
+
"size_gb": 3.9,
|
| 58 |
+
"size_note": "as reported on card",
|
| 59 |
+
"reference": "bf16",
|
| 60 |
+
"source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF/raw/main/README.md"
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"tier": "8-bit",
|
| 64 |
+
"model": "v3",
|
| 65 |
+
"label": "Jev-Style 0.8B v3 \u00b7 GGUF Q8_0",
|
| 66 |
+
"agree": 100.0,
|
| 67 |
+
"n": 240,
|
| 68 |
+
"agree_count": 240,
|
| 69 |
+
"size_gb": 0.811843008,
|
| 70 |
+
"size_note": "local export file (bytes / 1e9)",
|
| 71 |
+
"reference": "torch FP32",
|
| 72 |
+
"source": "runs/macjev/export_r2/main/parity/report.json; sizes: chart_data.json export.sizes"
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"tier": "8-bit",
|
| 76 |
+
"model": "v3",
|
| 77 |
+
"label": "Jev-Style 0.8B v3 \u00b7 MLX 8-bit",
|
| 78 |
+
"agree": 100.0,
|
| 79 |
+
"n": 240,
|
| 80 |
+
"agree_count": 240,
|
| 81 |
+
"size_gb": 0.799973748,
|
| 82 |
+
"size_note": "local export file (bytes / 1e9)",
|
| 83 |
+
"reference": "torch FP32",
|
| 84 |
+
"source": "runs/macjev/export_r2/main/parity/report.json; sizes: chart_data.json export.sizes"
|
| 85 |
+
},
|
| 86 |
+
{
|
| 87 |
+
"tier": "8-bit",
|
| 88 |
+
"model": "v2",
|
| 89 |
+
"label": "Jev-Style 2B v2 \u00b7 GGUF Q8_0",
|
| 90 |
+
"agree": 99.2,
|
| 91 |
+
"n": 500,
|
| 92 |
+
"size_gb": 2.01,
|
| 93 |
+
"size_note": "as reported on card",
|
| 94 |
+
"reference": "CUDA merged BF16",
|
| 95 |
+
"source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/raw/main/README.md"
|
| 96 |
+
},
|
| 97 |
+
{
|
| 98 |
+
"tier": "8-bit",
|
| 99 |
+
"model": "v1",
|
| 100 |
+
"label": "Jev-Style 2B v1 \u00b7 GGUF Q8_0",
|
| 101 |
+
"agree": 99.4,
|
| 102 |
+
"n": 500,
|
| 103 |
+
"size_gb": 2.1,
|
| 104 |
+
"size_note": "as reported on card",
|
| 105 |
+
"reference": "bf16",
|
| 106 |
+
"source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF/raw/main/README.md"
|
| 107 |
+
},
|
| 108 |
+
{
|
| 109 |
+
"tier": "4-bit",
|
| 110 |
+
"model": "v3",
|
| 111 |
+
"label": "Jev-Style 0.8B v3 \u00b7 GGUF Q4_K_M",
|
| 112 |
+
"agree": 100.0,
|
| 113 |
+
"n": 240,
|
| 114 |
+
"agree_count": 240,
|
| 115 |
+
"size_gb": 0.529296832,
|
| 116 |
+
"size_note": "local export file (bytes / 1e9)",
|
| 117 |
+
"reference": "torch FP32",
|
| 118 |
+
"source": "runs/macjev/export_r2/main/parity/report.json; sizes: chart_data.json export.sizes"
|
| 119 |
+
},
|
| 120 |
+
{
|
| 121 |
+
"tier": "4-bit",
|
| 122 |
+
"model": "v2",
|
| 123 |
+
"label": "Jev-Style 2B v2 \u00b7 GGUF Q4_K_M",
|
| 124 |
+
"agree": 91.4,
|
| 125 |
+
"n": 500,
|
| 126 |
+
"size_gb": 1.27,
|
| 127 |
+
"size_note": "as reported on card",
|
| 128 |
+
"reference": "CUDA merged BF16",
|
| 129 |
+
"source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-GGUF/raw/main/README.md"
|
| 130 |
+
},
|
| 131 |
+
{
|
| 132 |
+
"tier": "4-bit",
|
| 133 |
+
"model": "v1",
|
| 134 |
+
"label": "Jev-Style 2B v1 \u00b7 GGUF Q4_K_M",
|
| 135 |
+
"agree": 94.39999999999999,
|
| 136 |
+
"n": 500,
|
| 137 |
+
"size_gb": 1.3,
|
| 138 |
+
"size_note": "as reported on card",
|
| 139 |
+
"reference": "bf16",
|
| 140 |
+
"source": "https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF/raw/main/README.md"
|
| 141 |
+
}
|
| 142 |
+
],
|
| 143 |
+
"deltas": {
|
| 144 |
+
"v2 Q4_K_M size / v3 Q4_K_M size": 2.4
|
| 145 |
+
},
|
| 146 |
+
"no_agreement_delta_claimed": "v3 parity rows are training-pool rows; v1/v2 used 500 held-out decisions, so no agreement gap is claimed",
|
| 147 |
+
"long_points": "v3 formats also 100% top-1 at 16,384 and 25,600 tokens (3 rows each), parity report long_points",
|
| 148 |
+
"not_plotted": "mlx-4bit (98.75%, 237/240) is not a published format and is omitted",
|
| 149 |
+
"footnote": "v3: top-1 agreement with the PyTorch FP32 reference on a 240-row parity fixture drawn from the training pool (22 categories, en+zh), plus 6 extra rows at about 16K and 25.6K tokens (6/6 agree). v3 sizes = exported weight files (GB = 10^9 bytes). 2B v1/v2: as reported on their public HF GGUF cards (500 held-out decisions each; vs bf16 for v1, vs CUDA merged BF16 for v2; card sizes). Different fixtures (training-pool rows for v3, held-out rows for v1/v2) and references: rows are not a paired comparison. x-axis starts at 88%.",
|
| 150 |
+
"protocol_label": "v3 G5 parity: 240-row mixed fixture (drawn from training-pool rows, 22 categories, en+zh) vs torch FP32; plus 16K and 25.6K long points (3 rows each). 2B numbers are from their cards on different fixtures (500 decisions vs bf16)"
|
| 151 |
+
}
|
figures/quantization.png
ADDED
|
Git LFS Details
|
figures/quantization.svg
ADDED
|
|
figures/zeroshot.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"chart": "zeroshot",
|
| 3 |
+
"metric": "accuracy (argmax over the set's labels; all rows, none unsupported)",
|
| 4 |
+
"plotted": {
|
| 5 |
+
"tweet_topic": {
|
| 6 |
+
"v3": 75.49,
|
| 7 |
+
"laya_en": 63.2,
|
| 8 |
+
"delta_pts_vs_laya_en": 12.29
|
| 9 |
+
},
|
| 10 |
+
"fin_topic": {
|
| 11 |
+
"v3": 46.71,
|
| 12 |
+
"laya_en": 34.2,
|
| 13 |
+
"delta_pts_vs_laya_en": 12.51
|
| 14 |
+
}
|
| 15 |
+
},
|
| 16 |
+
"jev_marker": {
|
| 17 |
+
"set": "tweet_topic",
|
| 18 |
+
"jev": 79.33,
|
| 19 |
+
"gap_pts": 3.84
|
| 20 |
+
},
|
| 21 |
+
"v3_recomputed": {
|
| 22 |
+
"tweet_topic": {
|
| 23 |
+
"correct": 1278,
|
| 24 |
+
"n": 1693,
|
| 25 |
+
"ci95": [
|
| 26 |
+
0.7341996455995274,
|
| 27 |
+
0.7749556999409333
|
| 28 |
+
]
|
| 29 |
+
},
|
| 30 |
+
"fin_topic": {
|
| 31 |
+
"correct": 1923,
|
| 32 |
+
"n": 4117,
|
| 33 |
+
"ci95": [
|
| 34 |
+
0.45202817585620597,
|
| 35 |
+
0.48239008987126547
|
| 36 |
+
]
|
| 37 |
+
}
|
| 38 |
+
},
|
| 39 |
+
"entries": [
|
| 40 |
+
"zeroshot.tweet_topic.accuracy.v3",
|
| 41 |
+
"zeroshot.tweet_topic.accuracy.laya_en",
|
| 42 |
+
"zeroshot.tweet_topic.accuracy.jev",
|
| 43 |
+
"zeroshot.fin_topic.accuracy.v3",
|
| 44 |
+
"zeroshot.fin_topic.accuracy.laya_en",
|
| 45 |
+
"zeroshot.fin_topic.accuracy.jev"
|
| 46 |
+
],
|
| 47 |
+
"claims": [
|
| 48 |
+
"tweet_topic_accuracy_vs_laya_en",
|
| 49 |
+
"fin_topic_accuracy_vs_laya_en",
|
| 50 |
+
"tweet_topic_accuracy_vs_jev"
|
| 51 |
+
],
|
| 52 |
+
"sources": {
|
| 53 |
+
"v3": "runs/macjev/received/ext_evals/main/zeroshot_topics/metrics.json comparison.clean[*].ours.accuracy; recomputed from ext.zeroshot.<set>.jsonl",
|
| 54 |
+
"jev_laya_en": "src/macjev/eval/external/zeroshot_topics.py PUBLISHED = elcronos results/cross_dataset_summary.json @ a1901bc3d520e73936de8d4326545c0cdcf742fb"
|
| 55 |
+
},
|
| 56 |
+
"not_plotted_on_purpose": "Jev on fin_topic (not a win, not within 5 pts); macro-F1 vs Jev",
|
| 57 |
+
"footnote": "Zero-shot for every system: neither set is in v3's training pool; accuracy over every row of the pinned test files. Jev (1.13, API) and English Laya: numbers published by the elcronos jev-vs-open-decision-models study with its own prompt (results/cross_dataset_summary.json @ a1901bc), not re-run by us. v3: scored by us on the identical rows, label sets and instruction, in v3's own input format."
|
| 58 |
+
}
|
figures/zeroshot.png
ADDED
|
Git LFS Details
|
figures/zeroshot.svg
ADDED
|
|
generation_config.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_from_model_config": true,
|
| 3 |
+
"eos_token_id": 248044,
|
| 4 |
+
"transformers_version": "5.17.0",
|
| 5 |
+
"use_cache": true
|
| 6 |
+
}
|
jev_style_decision.py
ADDED
|
@@ -0,0 +1,517 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Jev-Style-0.8B-Decision-v3: typed decisions with transformers / PyTorch (CUDA, MPS, CPU).
|
| 2 |
+
|
| 3 |
+
Self-contained runtime for chaoliangUNSW/Jev-Style-0.8B-Decision-v3 (Apache-2.0). No dependency on any training code:
|
| 4 |
+
rendering, verdict readout and calibration are implemented below and reproduce the reference
|
| 5 |
+
implementation used for evaluation (see release_config.json -> "runtime_parity").
|
| 6 |
+
|
| 7 |
+
Calibration temperature: with no category (the default) probabilities use the global temperature of
|
| 8 |
+
readout_config.json (temperatures.global = 0.880); pass category=... (CLI --category, JSONL "category")
|
| 9 |
+
for the fitted group temperature of that category's family x question type x option-count bucket, or
|
| 10 |
+
temperature=... to override (1.0 = uncalibrated scores).
|
| 11 |
+
"""
|
| 12 |
+
# ----------------------------------------------------------------------------------------------
|
| 13 |
+
# Shared core (identical in jev_style_decision.py, jev_style_decision_gguf.py and
|
| 14 |
+
# jev_style_decision_mlx.py): input rendering, verdict readout, calibrated probabilities.
|
| 15 |
+
#
|
| 16 |
+
# Input layout ("macjev-render-v1"; token segments are encoded separately and concatenated):
|
| 17 |
+
#
|
| 18 |
+
# State:\n<state>\n\n
|
| 19 |
+
# Question [<type>]: <question>\nOptions:\n
|
| 20 |
+
# - <option 1>\n ... - <option K>\n
|
| 21 |
+
# Judge each option:\n
|
| 22 |
+
# <option 1> ->\n ... <option K> ->\n
|
| 23 |
+
#
|
| 24 |
+
# Score of option k = logit(" yes") - logit(" no") at the k-th " ->" token (computed from the final
|
| 25 |
+
# hidden state and the tied embedding rows, float32). Probabilities = softmax(scores / T), where T is
|
| 26 |
+
# the calibration temperature shipped in readout_config.json:
|
| 27 |
+
# * no category given (the default): T = temperatures.global (the file's global temperature);
|
| 28 |
+
# * category="..." given: T = the fitted group temperature of (family of that category x question
|
| 29 |
+
# type x option-count bucket), or temperatures.global when that group was not fitted;
|
| 30 |
+
# * temperature=... given: that value (1.0 = uncalibrated scores).
|
| 31 |
+
# T is clamped to temperatures.clamp. Text inside the state or options is tokenised with special
|
| 32 |
+
# tokens disabled, so e.g. "<|im_end|>" in user text can never act as a control token.
|
| 33 |
+
#
|
| 34 |
+
# Budgets: whole input <= 25,600 tokens; question + options + readout ("head") <= 2,048 tokens.
|
| 35 |
+
# Larger inputs raise InputBudgetError. Nothing is ever truncated.
|
| 36 |
+
# ----------------------------------------------------------------------------------------------
|
| 37 |
+
import argparse
|
| 38 |
+
import hashlib
|
| 39 |
+
import json
|
| 40 |
+
import math
|
| 41 |
+
import sys
|
| 42 |
+
from pathlib import Path
|
| 43 |
+
|
| 44 |
+
import numpy as np
|
| 45 |
+
|
| 46 |
+
MODEL_NAME = "Jev-Style-0.8B-Decision-v3"
|
| 47 |
+
TEMPLATE_VERSION = "macjev-render-v1"
|
| 48 |
+
READOUT_FORMAT = "macjev-readout-v1"
|
| 49 |
+
CONTEXT_LIMIT = 25_600 # state + question + options + readout
|
| 50 |
+
HARD_HEAD_MAX = 2048 # question + options + readout
|
| 51 |
+
QTYPES = ("choice", "score", "noul")
|
| 52 |
+
HERE = Path(__file__).resolve().parent
|
| 53 |
+
|
| 54 |
+
# calibration families (category prefix -> family), same table the temperatures were fitted with
|
| 55 |
+
FAMILY_BY_CATEGORY_PREFIX = (("typed_official", "typed"), ("typed_synthetic", "typed_synth"), ("general_", "general"),
|
| 56 |
+
("intent", "intent"), ("nli", "nli"), ("theme_", "theme"), ("mac_", "mac"),
|
| 57 |
+
("long_", "long"))
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
class InputBudgetError(ValueError):
|
| 61 |
+
"""The rendered input exceeds a token budget. Nothing was truncated."""
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
class QuestionError(ValueError):
|
| 65 |
+
"""The question/options are malformed."""
|
| 66 |
+
|
| 67 |
+
|
| 68 |
+
# -- questions ------------------------------------------------------------------------------------
|
| 69 |
+
def option_names(question):
|
| 70 |
+
"""Canonical option identifiers, in the order the probabilities are returned."""
|
| 71 |
+
if not isinstance(question, dict):
|
| 72 |
+
raise QuestionError("question must be a dict {'t', 'ins', 'crit'}")
|
| 73 |
+
t, crit = question.get("t"), question.get("crit")
|
| 74 |
+
if not isinstance(question.get("ins"), str) or not question["ins"].strip():
|
| 75 |
+
raise QuestionError("question text ('ins') must be a non-empty string")
|
| 76 |
+
if t == "choice":
|
| 77 |
+
if not isinstance(crit, dict) or not crit:
|
| 78 |
+
raise QuestionError("choice needs a non-empty dict {option name: description or None}")
|
| 79 |
+
return [str(k) for k in crit]
|
| 80 |
+
if t == "score":
|
| 81 |
+
if not isinstance(crit, list) or not 2 <= len(crit) <= 10:
|
| 82 |
+
raise QuestionError("score needs a list of 2..10 level descriptions")
|
| 83 |
+
return [str(i) for i in range(len(crit))]
|
| 84 |
+
if t == "noul":
|
| 85 |
+
if crit is not None and not isinstance(crit, dict):
|
| 86 |
+
raise QuestionError("noul criteria must be None or {'false': ..., 'true': ...}")
|
| 87 |
+
return ["false", "true"]
|
| 88 |
+
raise QuestionError(f"unknown question type {t!r} (expected one of {QTYPES})")
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
def make_question(question, options=None, qtype=None):
|
| 92 |
+
"""Build a typed question.
|
| 93 |
+
|
| 94 |
+
* ``question`` already a dict {"t", "ins", "crit"}: validated and returned.
|
| 95 |
+
* ``qtype="choice"`` (default when ``options`` is given): ``options`` = {name: description or None}
|
| 96 |
+
or a list of names.
|
| 97 |
+
* ``qtype="score"``: ``options`` = list of 2..10 level descriptions (level 0 first).
|
| 98 |
+
* ``qtype="noul"`` (default when no options): a true/false statement; ``options`` may be
|
| 99 |
+
{"false": "...", "true": "..."} to describe the two outcomes.
|
| 100 |
+
"""
|
| 101 |
+
if isinstance(question, dict):
|
| 102 |
+
q = dict(question)
|
| 103 |
+
else:
|
| 104 |
+
if qtype is None:
|
| 105 |
+
qtype = "choice" if options is not None else "noul"
|
| 106 |
+
if qtype == "choice":
|
| 107 |
+
if isinstance(options, (list, tuple)):
|
| 108 |
+
if len(set(map(str, options))) != len(options):
|
| 109 |
+
raise QuestionError("duplicate option names")
|
| 110 |
+
crit = {str(o): None for o in options}
|
| 111 |
+
else:
|
| 112 |
+
crit = options
|
| 113 |
+
elif qtype == "score":
|
| 114 |
+
crit = list(options) if options is not None else None
|
| 115 |
+
else:
|
| 116 |
+
crit = options
|
| 117 |
+
q = {"t": qtype, "ins": question, "crit": crit}
|
| 118 |
+
option_names(q)
|
| 119 |
+
return q
|
| 120 |
+
|
| 121 |
+
|
| 122 |
+
def serialize_state(state):
|
| 123 |
+
"""Strings pass through unchanged; any other JSON value is serialised (ensure_ascii=False)."""
|
| 124 |
+
if isinstance(state, str):
|
| 125 |
+
return state
|
| 126 |
+
return json.dumps(state, ensure_ascii=False)
|
| 127 |
+
|
| 128 |
+
|
| 129 |
+
def _criterion(value):
|
| 130 |
+
if isinstance(value, str):
|
| 131 |
+
return value
|
| 132 |
+
return json.dumps(value, ensure_ascii=False, separators=(", ", ": "), default=str)
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
def render_options(question):
|
| 136 |
+
t, crit = question["t"], question.get("crit")
|
| 137 |
+
if t == "choice":
|
| 138 |
+
return [k if v is None or v == "" else f"{k}: {_criterion(v)}" for k, v in crit.items()]
|
| 139 |
+
if t == "score":
|
| 140 |
+
return [f"level {i}: {_criterion(c)}" for i, c in enumerate(crit)]
|
| 141 |
+
crit = crit or {}
|
| 142 |
+
false_c, true_c = crit.get("false"), crit.get("true")
|
| 143 |
+
return ["false: " + (_criterion(false_c) if false_c not in (None, "") else "no, the statement does not hold"),
|
| 144 |
+
"true: " + (_criterion(true_c) if true_c not in (None, "") else "yes, the statement holds")]
|
| 145 |
+
|
| 146 |
+
|
| 147 |
+
# -- tokenizer + renderer -----------------------------------------------------------------------
|
| 148 |
+
class TextEncoder:
|
| 149 |
+
"""HF ``tokenizers`` tokenizer.json; no BOS/EOS, special tokens in text are split (never control tokens)."""
|
| 150 |
+
|
| 151 |
+
def __init__(self, tokenizer_json):
|
| 152 |
+
from tokenizers import Tokenizer
|
| 153 |
+
self.tk = Tokenizer.from_file(str(tokenizer_json))
|
| 154 |
+
self.tk.encode_special_tokens = True
|
| 155 |
+
|
| 156 |
+
def __call__(self, text):
|
| 157 |
+
return self.tk.encode(text, add_special_tokens=False).ids
|
| 158 |
+
|
| 159 |
+
|
| 160 |
+
class Rendered:
|
| 161 |
+
__slots__ = ("ids", "prefix_len", "slots", "names", "head_tokens")
|
| 162 |
+
|
| 163 |
+
def __init__(self, ids, prefix_len, slots, names, head_tokens):
|
| 164 |
+
self.ids, self.prefix_len, self.slots, self.names, self.head_tokens = ids, prefix_len, slots, names, head_tokens
|
| 165 |
+
|
| 166 |
+
|
| 167 |
+
class Renderer:
|
| 168 |
+
def __init__(self, encode, readout_cfg, max_len=CONTEXT_LIMIT, head_max=HARD_HEAD_MAX):
|
| 169 |
+
if readout_cfg.get("format") != READOUT_FORMAT or readout_cfg.get("template") != TEMPLATE_VERSION:
|
| 170 |
+
raise ValueError("readout_config.json is not a macjev-readout-v1 / macjev-render-v1 config")
|
| 171 |
+
if readout_cfg.get("readout") != "verdict":
|
| 172 |
+
raise ValueError("this runtime implements the verdict readout only")
|
| 173 |
+
if not 0 < int(max_len) <= CONTEXT_LIMIT:
|
| 174 |
+
raise ValueError(f"max_len must be in 1..{CONTEXT_LIMIT}")
|
| 175 |
+
if not 0 < int(head_max) <= HARD_HEAD_MAX:
|
| 176 |
+
raise ValueError(f"head_max must be in 1..{HARD_HEAD_MAX}")
|
| 177 |
+
self.enc, self.max_len, self.head_max = encode, int(max_len), int(head_max)
|
| 178 |
+
st = readout_cfg["slot_tokens"]
|
| 179 |
+
self.yes, self.no, arrow = int(st["yes"]["id"]), int(st["no"]["id"]), int(st["verdict_slot"]["id"])
|
| 180 |
+
for text, want in ((" yes", self.yes), (" no", self.no), (" ->", arrow)):
|
| 181 |
+
got = self.enc(text)
|
| 182 |
+
if got != [want]:
|
| 183 |
+
raise ValueError(f"tokenizer mismatch: {text!r} -> {got}, readout_config expects [{want}]")
|
| 184 |
+
self.arrow = [arrow]
|
| 185 |
+
self.newline = self.enc("\n")
|
| 186 |
+
self.dash = self.enc("- ")
|
| 187 |
+
self.judge = self.enc("Judge each option:\n")
|
| 188 |
+
|
| 189 |
+
def prefix_ids(self, state):
|
| 190 |
+
return self.enc("State:\n") + self.enc(serialize_state(state)) + self.enc("\n\n")
|
| 191 |
+
|
| 192 |
+
def render(self, state, question, head_max=None, max_len=None):
|
| 193 |
+
head_max = self.head_max if head_max is None else int(head_max)
|
| 194 |
+
max_len = self.max_len if max_len is None else min(int(max_len), self.max_len)
|
| 195 |
+
if head_max > HARD_HEAD_MAX:
|
| 196 |
+
raise InputBudgetError(f"head_max may not exceed {HARD_HEAD_MAX}")
|
| 197 |
+
names = option_names(question)
|
| 198 |
+
opts = [self.enc(o) for o in render_options(question)]
|
| 199 |
+
suffix = self.enc(f"Question [{question['t']}]: {question['ins']}\nOptions:\n")
|
| 200 |
+
for o in opts:
|
| 201 |
+
suffix += self.dash + o + self.newline
|
| 202 |
+
suffix += self.judge
|
| 203 |
+
rel = []
|
| 204 |
+
for o in opts:
|
| 205 |
+
suffix += o + self.arrow
|
| 206 |
+
rel.append(len(suffix) - 1)
|
| 207 |
+
suffix += self.newline
|
| 208 |
+
if len(suffix) > head_max:
|
| 209 |
+
raise InputBudgetError(f"question + options + readout need {len(suffix)} tokens; the head budget is "
|
| 210 |
+
f"{head_max} (hard cap {HARD_HEAD_MAX}). Nothing was truncated: shorten the "
|
| 211 |
+
f"question/options or split the options over several questions.")
|
| 212 |
+
prefix = self.prefix_ids(state)
|
| 213 |
+
ids = prefix + suffix
|
| 214 |
+
if len(ids) > max_len:
|
| 215 |
+
raise InputBudgetError(f"input needs {len(ids)} tokens (state {len(prefix)} + head {len(suffix)}); the "
|
| 216 |
+
f"limit is {max_len} (model maximum {CONTEXT_LIMIT}). Nothing was truncated: "
|
| 217 |
+
f"shorten the state.")
|
| 218 |
+
return Rendered(ids, len(prefix), [len(prefix) + s for s in rel], names, len(suffix))
|
| 219 |
+
|
| 220 |
+
|
| 221 |
+
# -- calibration ----------------------------------------------------------------------------------
|
| 222 |
+
def family(category):
|
| 223 |
+
for prefix, fam in FAMILY_BY_CATEGORY_PREFIX:
|
| 224 |
+
if category.startswith(prefix):
|
| 225 |
+
return fam
|
| 226 |
+
return "other"
|
| 227 |
+
|
| 228 |
+
|
| 229 |
+
def option_bucket(k):
|
| 230 |
+
return "2" if k <= 2 else "3-5" if k <= 5 else "6-10" if k <= 10 else "11-20" if k <= 20 else "21+"
|
| 231 |
+
|
| 232 |
+
|
| 233 |
+
def lookup_temperature(temps, category, qtype, n_options):
|
| 234 |
+
"""Calibration temperature. ``category`` None/"" -> the global temperature; otherwise the fitted
|
| 235 |
+
group (family(category) x qtype x option bucket), falling back to the global temperature."""
|
| 236 |
+
g = None
|
| 237 |
+
if category:
|
| 238 |
+
g = (temps.get("groups") or {}).get(f"{family(category)}|{qtype}|{option_bucket(n_options)}")
|
| 239 |
+
t = g["T"] if g else temps.get("global", 1.0)
|
| 240 |
+
lo, hi = temps.get("clamp", [0.3, 5.0])
|
| 241 |
+
return float(min(hi, max(lo, t)))
|
| 242 |
+
|
| 243 |
+
|
| 244 |
+
def concentration(p):
|
| 245 |
+
k = len(p)
|
| 246 |
+
if k < 2:
|
| 247 |
+
return 1.0
|
| 248 |
+
ent = -(p * np.log(np.clip(p, 1e-12, 1.0))).sum()
|
| 249 |
+
return float(np.clip(1.0 - ent / math.log(k), 0.0, 1.0))
|
| 250 |
+
|
| 251 |
+
|
| 252 |
+
def _sha256(path):
|
| 253 |
+
h = hashlib.sha256()
|
| 254 |
+
with open(path, "rb") as f:
|
| 255 |
+
for b in iter(lambda: f.read(1 << 22), b""):
|
| 256 |
+
h.update(b)
|
| 257 |
+
return h.hexdigest()
|
| 258 |
+
|
| 259 |
+
|
| 260 |
+
def verify_manifest(model_dir, only=None):
|
| 261 |
+
"""Re-hash the files listed in manifest.json (all, or those whose path starts with one of ``only``).
|
| 262 |
+
Documentation (README.md, figures/, assets/) is recorded in the manifest but not checked here, so
|
| 263 |
+
a card edit never makes the runtime refuse to load."""
|
| 264 |
+
model_dir = Path(model_dir)
|
| 265 |
+
man = json.loads((model_dir / "manifest.json").read_text())
|
| 266 |
+
bad, missing, checked = [], [], 0
|
| 267 |
+
for name, rec in man["files"].items():
|
| 268 |
+
if name == "README.md" or name.startswith(("assets/", "figures/")):
|
| 269 |
+
continue
|
| 270 |
+
if only and not any(name == o or name.startswith(o.rstrip("/") + "/") for o in only):
|
| 271 |
+
continue
|
| 272 |
+
p = model_dir / name
|
| 273 |
+
if not p.exists():
|
| 274 |
+
missing.append(name)
|
| 275 |
+
elif _sha256(p) != rec["sha256"]:
|
| 276 |
+
bad.append(name)
|
| 277 |
+
checked += 1
|
| 278 |
+
return {"ok": not bad and not missing, "checked": checked, "bad": bad, "missing": missing}
|
| 279 |
+
|
| 280 |
+
|
| 281 |
+
class DecisionBase:
|
| 282 |
+
"""Backend-independent part. Subclasses implement ``_scores(rendered) -> list[float]`` and may
|
| 283 |
+
override ``_scores_many(list of rendered) -> list of list[float]`` (several questions, one state).
|
| 284 |
+
|
| 285 |
+
Calibration: ``category`` (constructor default or per call) selects the fitted group temperature
|
| 286 |
+
of that category's family; with no category anywhere, the global temperature of
|
| 287 |
+
readout_config.json (temperatures.global) is used."""
|
| 288 |
+
backend = "base"
|
| 289 |
+
|
| 290 |
+
def _setup(self, model_dir, tokenizer_json, category=None, head_max=HARD_HEAD_MAX, max_len=CONTEXT_LIMIT):
|
| 291 |
+
self.model_dir = Path(model_dir)
|
| 292 |
+
self.readout_config = json.loads((self.model_dir / "readout_config.json").read_text())
|
| 293 |
+
self.temperatures = self.readout_config["temperatures"]
|
| 294 |
+
self.default_category = category or None # None -> temperatures.global
|
| 295 |
+
self.encode = TextEncoder(tokenizer_json)
|
| 296 |
+
self.renderer = Renderer(self.encode, self.readout_config, max_len=max_len, head_max=head_max)
|
| 297 |
+
|
| 298 |
+
def temperature(self, question, category=None):
|
| 299 |
+
"""T for ``question``: group temperature of ``category`` (or the constructor's default category);
|
| 300 |
+
the global temperature when neither is given."""
|
| 301 |
+
return lookup_temperature(self.temperatures, category or self.default_category, question["t"],
|
| 302 |
+
len(option_names(question)))
|
| 303 |
+
|
| 304 |
+
def _scores_many(self, rendered):
|
| 305 |
+
return [self._scores(r) for r in rendered]
|
| 306 |
+
|
| 307 |
+
def _result(self, r, q, scores, category=None, temperature=None):
|
| 308 |
+
scores = [float(x) for x in scores]
|
| 309 |
+
t = float(temperature) if temperature is not None else self.temperature(q, category)
|
| 310 |
+
z = np.asarray(scores, float) / t
|
| 311 |
+
if not np.all(np.isfinite(z)):
|
| 312 |
+
raise FloatingPointError("non-finite decision scores")
|
| 313 |
+
p = np.exp(z - z.max())
|
| 314 |
+
p /= p.sum()
|
| 315 |
+
i = int(p.argmax())
|
| 316 |
+
return {"answer": r.names[i], "probabilities": dict(zip(r.names, p.tolist())),
|
| 317 |
+
"scores": dict(zip(r.names, scores)), "temperature": t, "top_probability": float(p[i]),
|
| 318 |
+
"entropy_concentration": concentration(p), "input_tokens": len(r.ids), "head_tokens": r.head_tokens,
|
| 319 |
+
"model": MODEL_NAME, "backend": self.backend}
|
| 320 |
+
|
| 321 |
+
def decide(self, state, question, options=None, qtype=None, category=None, temperature=None, head_max=None):
|
| 322 |
+
"""Score one question about ``state``.
|
| 323 |
+
|
| 324 |
+
Returns {"answer", "probabilities" {option: p}, "scores" {option: logit(yes)-logit(no)},
|
| 325 |
+
"temperature", "top_probability", "entropy_concentration", "input_tokens", "head_tokens"}.
|
| 326 |
+
Temperature: with no ``category`` (here or in the constructor) the global temperature of
|
| 327 |
+
readout_config.json is used; ``category`` picks the fitted group temperature of its family
|
| 328 |
+
(e.g. "mac_gate", "general_topic", "theme_routing", "intent", "typed_official");
|
| 329 |
+
``temperature`` overrides both (1.0 = uncalibrated scores).
|
| 330 |
+
Raises InputBudgetError (never truncates) or QuestionError.
|
| 331 |
+
"""
|
| 332 |
+
q = make_question(question, options, qtype)
|
| 333 |
+
r = self.renderer.render(state, q, head_max=head_max)
|
| 334 |
+
return self._result(r, q, self._scores(r), category, temperature)
|
| 335 |
+
|
| 336 |
+
def decide_many(self, state, questions, category=None, temperature=None, head_max=None):
|
| 337 |
+
"""Several questions about the same state (each a dict {"t","ins","crit"}); results in order.
|
| 338 |
+
Same outputs as calling decide() per question. The llama.cpp runtime sends all questions in one
|
| 339 |
+
request and shares the state in whole 1,024-token ubatches (see JevStyleDecisionGGUF, also for
|
| 340 |
+
its faster, not bit-identical many_mode="batched"). All questions are rendered and
|
| 341 |
+
budget-checked before any scoring."""
|
| 342 |
+
qs = [make_question(q) for q in questions]
|
| 343 |
+
rs = [self.renderer.render(state, q, head_max=head_max) for q in qs]
|
| 344 |
+
if not rs:
|
| 345 |
+
return []
|
| 346 |
+
return [self._result(r, q, sc, category, temperature) for r, q, sc in zip(rs, qs, self._scores_many(rs))]
|
| 347 |
+
|
| 348 |
+
|
| 349 |
+
def base_arg_parser(description):
|
| 350 |
+
ap = argparse.ArgumentParser(description=description)
|
| 351 |
+
ap.add_argument("--model-dir", default=str(HERE), help="folder with the weights and readout_config.json")
|
| 352 |
+
ap.add_argument("--state", help="state as plain text")
|
| 353 |
+
ap.add_argument("--state-json", help="state as a JSON value")
|
| 354 |
+
ap.add_argument("--question", help="question text (or a JSON question {'t','ins','crit'})")
|
| 355 |
+
ap.add_argument("--options", help="JSON: {name: description} or [names] (choice); [levels] (score)")
|
| 356 |
+
ap.add_argument("--qtype", choices=QTYPES)
|
| 357 |
+
ap.add_argument("--category", help="calibration family key, e.g. mac_gate, general_topic, theme_routing, intent "
|
| 358 |
+
"(default: none -> the global temperature of readout_config.json)")
|
| 359 |
+
ap.add_argument("--temperature", type=float, help="override the calibrated temperature")
|
| 360 |
+
ap.add_argument("--head-max", type=int, default=HARD_HEAD_MAX)
|
| 361 |
+
ap.add_argument("--max-len", type=int, default=CONTEXT_LIMIT)
|
| 362 |
+
ap.add_argument("--jsonl", help="batch mode: input JSON lines {id?, state, question, options?, qtype?, "
|
| 363 |
+
"category?}; one JSON result per line on stdout")
|
| 364 |
+
ap.add_argument("--verify", action="store_true", help="check sha256 of the files in manifest.json first")
|
| 365 |
+
return ap
|
| 366 |
+
|
| 367 |
+
|
| 368 |
+
def run_cli(args, engine):
|
| 369 |
+
def one(rec):
|
| 370 |
+
q = rec["question"]
|
| 371 |
+
return engine.decide(rec.get("state", ""), q, options=rec.get("options"), qtype=rec.get("qtype"),
|
| 372 |
+
category=rec.get("category"), temperature=rec.get("temperature", args.temperature))
|
| 373 |
+
if args.jsonl:
|
| 374 |
+
src = sys.stdin if args.jsonl == "-" else open(args.jsonl, encoding="utf-8")
|
| 375 |
+
for n, line in enumerate(src):
|
| 376 |
+
if not line.strip():
|
| 377 |
+
continue
|
| 378 |
+
rec = json.loads(line)
|
| 379 |
+
rid = rec.get("id", n)
|
| 380 |
+
try:
|
| 381 |
+
out = {"id": rid, **one(rec)}
|
| 382 |
+
except (InputBudgetError, QuestionError) as e:
|
| 383 |
+
out = {"id": rid, "error": f"{type(e).__name__}: {e}"}
|
| 384 |
+
print(json.dumps(out, ensure_ascii=False), flush=True)
|
| 385 |
+
return 0
|
| 386 |
+
if args.question is None:
|
| 387 |
+
raise SystemExit("--question (or --jsonl) is required")
|
| 388 |
+
state = json.loads(args.state_json) if args.state_json is not None else (args.state or "")
|
| 389 |
+
question = args.question
|
| 390 |
+
if question.lstrip().startswith("{"):
|
| 391 |
+
question = json.loads(question)
|
| 392 |
+
rec = {"state": state, "question": question, "options": json.loads(args.options) if args.options else None,
|
| 393 |
+
"qtype": args.qtype, "category": args.category}
|
| 394 |
+
print(json.dumps(one(rec), ensure_ascii=False, indent=2))
|
| 395 |
+
return 0
|
| 396 |
+
# ---------------------------------------------------------------------------- end of shared core
|
| 397 |
+
|
| 398 |
+
|
| 399 |
+
# ------------------------------------------------------------------------------ PyTorch backend
|
| 400 |
+
ATTN_CHUNK = 1024
|
| 401 |
+
_CHUNKED_NAME = "jev_chunked_sdpa"
|
| 402 |
+
_CHUNKED_REGISTERED = False
|
| 403 |
+
|
| 404 |
+
|
| 405 |
+
def _chunked_sdpa_forward(module, query, key, value, attention_mask, dropout=0.0, scaling=None, is_causal=None,
|
| 406 |
+
**kwargs):
|
| 407 |
+
"""Query-chunked SDPA (MPS / CPU): the same computation as transformers' sdpa, but the
|
| 408 |
+
[heads x queries x keys] score matrix is built ATTN_CHUNK queries at a time, so a 25,600-token
|
| 409 |
+
input does not need ~21 GB for one attention call. Inputs of <= ATTN_CHUNK tokens take the
|
| 410 |
+
unchanged sdpa path."""
|
| 411 |
+
import torch
|
| 412 |
+
from transformers.integrations.sdpa_attention import sdpa_attention_forward
|
| 413 |
+
q_len, kv_len = query.shape[2], key.shape[2]
|
| 414 |
+
chunk = int(getattr(module, "jev_attn_chunk", ATTN_CHUNK) or ATTN_CHUNK)
|
| 415 |
+
if q_len <= chunk:
|
| 416 |
+
return sdpa_attention_forward(module, query, key, value, attention_mask, dropout=dropout, scaling=scaling,
|
| 417 |
+
is_causal=is_causal, **kwargs)
|
| 418 |
+
causal = is_causal if is_causal is not None else getattr(module, "is_causal", True)
|
| 419 |
+
past = kv_len - q_len
|
| 420 |
+
outs = []
|
| 421 |
+
for s in range(0, q_len, chunk):
|
| 422 |
+
e = min(q_len, s + chunk)
|
| 423 |
+
if attention_mask is not None:
|
| 424 |
+
m = attention_mask if attention_mask.shape[-2] == 1 else attention_mask[:, :, s:e, :]
|
| 425 |
+
elif causal:
|
| 426 |
+
qpos = torch.arange(s + past, e + past, device=query.device)
|
| 427 |
+
kpos = torch.arange(kv_len, device=query.device)
|
| 428 |
+
m = (kpos[None, :] <= qpos[:, None])[None, None]
|
| 429 |
+
else:
|
| 430 |
+
m = None
|
| 431 |
+
out, _ = sdpa_attention_forward(module, query[:, :, s:e], key, value, m, dropout=dropout, scaling=scaling,
|
| 432 |
+
is_causal=False, **kwargs)
|
| 433 |
+
outs.append(out)
|
| 434 |
+
return torch.cat(outs, dim=1), None
|
| 435 |
+
|
| 436 |
+
|
| 437 |
+
def _enable_chunked_attention(model, chunk=ATTN_CHUNK):
|
| 438 |
+
global _CHUNKED_REGISTERED
|
| 439 |
+
if getattr(model.config, "_attn_implementation", None) not in ("sdpa", _CHUNKED_NAME):
|
| 440 |
+
return False
|
| 441 |
+
try:
|
| 442 |
+
from transformers import AttentionInterface
|
| 443 |
+
from transformers.masking_utils import ALL_MASK_ATTENTION_FUNCTIONS, AttentionMaskInterface
|
| 444 |
+
except ImportError:
|
| 445 |
+
return False
|
| 446 |
+
if not _CHUNKED_REGISTERED:
|
| 447 |
+
AttentionInterface.register(_CHUNKED_NAME, _chunked_sdpa_forward)
|
| 448 |
+
AttentionMaskInterface.register(_CHUNKED_NAME, ALL_MASK_ATTENTION_FUNCTIONS["sdpa"])
|
| 449 |
+
_CHUNKED_REGISTERED = True
|
| 450 |
+
model.set_attn_implementation(_CHUNKED_NAME)
|
| 451 |
+
for m in model.modules():
|
| 452 |
+
if hasattr(m, "is_causal"):
|
| 453 |
+
m.jev_attn_chunk = int(chunk)
|
| 454 |
+
return True
|
| 455 |
+
|
| 456 |
+
|
| 457 |
+
class JevStyleDecision(DecisionBase):
|
| 458 |
+
"""Transformers / PyTorch runtime (CUDA, Apple MPS or CPU).
|
| 459 |
+
|
| 460 |
+
>>> m = JevStyleDecision(".") # float32 on the best available device
|
| 461 |
+
>>> m.decide({"messages": ["Refund still missing after 3 weeks"]},
|
| 462 |
+
... "Which team should handle this ticket?",
|
| 463 |
+
... options={"billing": "payments, refunds", "tech": "bugs, crashes", "sales": "pricing, plans"},
|
| 464 |
+
... category="theme_routing")["probabilities"]
|
| 465 |
+
"""
|
| 466 |
+
backend = "torch"
|
| 467 |
+
|
| 468 |
+
def __init__(self, model_dir=HERE, device=None, dtype="float32", category=None, head_max=HARD_HEAD_MAX,
|
| 469 |
+
max_len=CONTEXT_LIMIT, attn_chunk=ATTN_CHUNK, verify=False):
|
| 470 |
+
import torch
|
| 471 |
+
self.torch = torch
|
| 472 |
+
model_dir = Path(model_dir)
|
| 473 |
+
if verify:
|
| 474 |
+
res = verify_manifest(model_dir)
|
| 475 |
+
if not res["ok"]:
|
| 476 |
+
raise RuntimeError(f"integrity check failed: {res}")
|
| 477 |
+
self._setup(model_dir, model_dir / "tokenizer.json", category, head_max, max_len)
|
| 478 |
+
if device is None:
|
| 479 |
+
device = ("cuda" if torch.cuda.is_available() else
|
| 480 |
+
"mps" if getattr(torch.backends, "mps", None) and torch.backends.mps.is_available() else "cpu")
|
| 481 |
+
dt = getattr(torch, dtype) if isinstance(dtype, str) else dtype
|
| 482 |
+
try:
|
| 483 |
+
from transformers.models.qwen3_5.modeling_qwen3_5 import Qwen3_5ForCausalLM as cls
|
| 484 |
+
except ImportError as e:
|
| 485 |
+
raise ImportError("this model needs a transformers version with Qwen3.5 support "
|
| 486 |
+
"(transformers.models.qwen3_5)") from e
|
| 487 |
+
try:
|
| 488 |
+
model = cls.from_pretrained(str(model_dir), dtype=dt)
|
| 489 |
+
except TypeError: # transformers 4.x keyword
|
| 490 |
+
model = cls.from_pretrained(str(model_dir), torch_dtype=dt)
|
| 491 |
+
self.model = model.to(device).eval()
|
| 492 |
+
self.device, self.dtype = device, dt
|
| 493 |
+
self.chunked_attention = _enable_chunked_attention(self.model, attn_chunk) if device != "cuda" else False
|
| 494 |
+
w = self.model.get_output_embeddings().weight
|
| 495 |
+
self.direction = (w[self.renderer.yes].float() - w[self.renderer.no].float()).detach()
|
| 496 |
+
|
| 497 |
+
def _scores(self, r):
|
| 498 |
+
torch = self.torch
|
| 499 |
+
with torch.no_grad():
|
| 500 |
+
ids = torch.tensor([r.ids], device=self.device)
|
| 501 |
+
h = self.model.model(input_ids=ids, use_cache=False).last_hidden_state # final normed hidden states
|
| 502 |
+
hs = h[0, torch.tensor(r.slots, device=self.device)].float()
|
| 503 |
+
return (hs @ self.direction).cpu().tolist()
|
| 504 |
+
|
| 505 |
+
|
| 506 |
+
def main(argv=None):
|
| 507 |
+
ap = base_arg_parser(f"{MODEL_NAME}: typed decisions with transformers / PyTorch")
|
| 508 |
+
ap.add_argument("--device", choices=["cuda", "mps", "cpu"])
|
| 509 |
+
ap.add_argument("--dtype", default="float32", choices=["float32", "bfloat16", "float16"])
|
| 510 |
+
args = ap.parse_args(argv)
|
| 511 |
+
engine = JevStyleDecision(args.model_dir, device=args.device, dtype=args.dtype, category=args.category,
|
| 512 |
+
head_max=args.head_max, max_len=args.max_len, verify=args.verify)
|
| 513 |
+
return run_cli(args, engine)
|
| 514 |
+
|
| 515 |
+
|
| 516 |
+
if __name__ == "__main__":
|
| 517 |
+
raise SystemExit(main())
|
manifest.json
ADDED
|
@@ -0,0 +1,181 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"format": "jev-style-manifest-v1",
|
| 3 |
+
"repo": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3",
|
| 4 |
+
"created_unix": 1790261965.8405728,
|
| 5 |
+
"files": {
|
| 6 |
+
"LICENSE": {
|
| 7 |
+
"sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a",
|
| 8 |
+
"bytes": 11544
|
| 9 |
+
},
|
| 10 |
+
"NOTICE": {
|
| 11 |
+
"sha256": "3c1bee42827901d754cf64e437cf8837ea94241348f5f4be9e6c4e14fbfc5716",
|
| 12 |
+
"bytes": 1966
|
| 13 |
+
},
|
| 14 |
+
"README.md": {
|
| 15 |
+
"sha256": "ccf12d32e307f97aa6818c098fc874aec0f3666c79af3834fee799f98eb6871b",
|
| 16 |
+
"bytes": 31905
|
| 17 |
+
},
|
| 18 |
+
"chat_template.jinja": {
|
| 19 |
+
"sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80",
|
| 20 |
+
"bytes": 7755
|
| 21 |
+
},
|
| 22 |
+
"config.json": {
|
| 23 |
+
"sha256": "3f56b6210db3c52e0bafb50da005eb52b6d340ab4ad980091879b397743675fc",
|
| 24 |
+
"bytes": 1790
|
| 25 |
+
},
|
| 26 |
+
"figures/beyond_laya.json": {
|
| 27 |
+
"sha256": "a6dfd582fe3aca32bcd67cf01b85190d92084a0790f7744dd03180df5923f946",
|
| 28 |
+
"bytes": 4357
|
| 29 |
+
},
|
| 30 |
+
"figures/beyond_laya.png": {
|
| 31 |
+
"sha256": "0dbd24837759acf05ed3a668016ecaea75a41a093b64588417d2fd750b95779b",
|
| 32 |
+
"bytes": 185526
|
| 33 |
+
},
|
| 34 |
+
"figures/beyond_laya.svg": {
|
| 35 |
+
"sha256": "4b64a4e71d3c9c05c9a785003bbfbd7269c5341bcc8abd560a2b156cfb771d6b",
|
| 36 |
+
"bytes": 22000
|
| 37 |
+
},
|
| 38 |
+
"figures/calibration.data.json": {
|
| 39 |
+
"sha256": "0396f5d48ba839c65718a7f1b449e7fa074503cbd1b9db48a2d40096c2d6142a",
|
| 40 |
+
"bytes": 3765
|
| 41 |
+
},
|
| 42 |
+
"figures/calibration.png": {
|
| 43 |
+
"sha256": "3176339f6e4f58c9efa583b101f5862141a326468cf177f2c50c3ce2d1adf06f",
|
| 44 |
+
"bytes": 148880
|
| 45 |
+
},
|
| 46 |
+
"figures/calibration.svg": {
|
| 47 |
+
"sha256": "7cdb177072406c736a03bcec696adacf037cf810f156dd8f94ab6cf24e1d76ec",
|
| 48 |
+
"bytes": 15047
|
| 49 |
+
},
|
| 50 |
+
"figures/design_table.data.json": {
|
| 51 |
+
"sha256": "84fd672a2c69d6881c199dc09ccbe7b1313b73bf7f56f54f574cf4eefe052003",
|
| 52 |
+
"bytes": 5313
|
| 53 |
+
},
|
| 54 |
+
"figures/design_table.png": {
|
| 55 |
+
"sha256": "5b907319a5a4f6ee8eea5800fbc732b95d15e941b1d0cd8d774d3705a1341744",
|
| 56 |
+
"bytes": 331936
|
| 57 |
+
},
|
| 58 |
+
"figures/design_table.svg": {
|
| 59 |
+
"sha256": "a0a23216766b235c950a071a614691b010dcea9f3a0c5bb8c6f8cc2a7ecae0bc",
|
| 60 |
+
"bytes": 26655
|
| 61 |
+
},
|
| 62 |
+
"figures/headline_typed.data.json": {
|
| 63 |
+
"sha256": "04380aa3823f80489ae37f893de0bda67ae2dae38fd07e37f88700f7c173f0ec",
|
| 64 |
+
"bytes": 3819
|
| 65 |
+
},
|
| 66 |
+
"figures/headline_typed.png": {
|
| 67 |
+
"sha256": "9e763e4cbd86a25299785b94a89945f2c3a7cab4b1c7d5000f9456c4d220a1f4",
|
| 68 |
+
"bytes": 175884
|
| 69 |
+
},
|
| 70 |
+
"figures/headline_typed.svg": {
|
| 71 |
+
"sha256": "a49ec30b28319d682d9465145bf6435ac4ec4ad11c46a28c552c9b851fb27f37",
|
| 72 |
+
"bytes": 20359
|
| 73 |
+
},
|
| 74 |
+
"figures/jevbench.data.json": {
|
| 75 |
+
"sha256": "f9173e65f5cd1fe9fadad8c93a8d00dbe5181e7314d49f18c6f547a0ed168543",
|
| 76 |
+
"bytes": 2645
|
| 77 |
+
},
|
| 78 |
+
"figures/jevbench.png": {
|
| 79 |
+
"sha256": "b75ab6864b4487d0a323c1019f7ce405fedcdb5a4d8dc910c86a29b3141b5388",
|
| 80 |
+
"bytes": 142720
|
| 81 |
+
},
|
| 82 |
+
"figures/jevbench.svg": {
|
| 83 |
+
"sha256": "2014d7664f268c2621e68226d4ec92eec6dd8dc7a04029024b5110a9341b812f",
|
| 84 |
+
"bytes": 12181
|
| 85 |
+
},
|
| 86 |
+
"figures/latency.data.json": {
|
| 87 |
+
"sha256": "c0eb510a69dfe1b1cd85beca6ba3c354be4984d588b4aae2672331606a119cbc",
|
| 88 |
+
"bytes": 1402
|
| 89 |
+
},
|
| 90 |
+
"figures/latency.png": {
|
| 91 |
+
"sha256": "d6d0d34b875997b8a733d4776dc34a1b564ddc64b48cea9d0def7f72a34af79e",
|
| 92 |
+
"bytes": 203081
|
| 93 |
+
},
|
| 94 |
+
"figures/latency.svg": {
|
| 95 |
+
"sha256": "c7bdf011eec488aae7197d653cb8486dfd8ccbd1a58f5ba222873de7410aca30",
|
| 96 |
+
"bytes": 24079
|
| 97 |
+
},
|
| 98 |
+
"figures/long_context.json": {
|
| 99 |
+
"sha256": "e6289e61da3bf4117ed4ed05820b99620e541c25cdeb5262e3a580fa9febff62",
|
| 100 |
+
"bytes": 5553
|
| 101 |
+
},
|
| 102 |
+
"figures/long_context.png": {
|
| 103 |
+
"sha256": "957fb443f67d1ed91ef61fa83b0496347299823799bc97692d25893060b1e26d",
|
| 104 |
+
"bytes": 224537
|
| 105 |
+
},
|
| 106 |
+
"figures/long_context.svg": {
|
| 107 |
+
"sha256": "ad7be559f77b11b56df84b4f286f9cea9c076d4caf47a2f8fb0f67ab46f7968b",
|
| 108 |
+
"bytes": 21403
|
| 109 |
+
},
|
| 110 |
+
"figures/multilingual.data.json": {
|
| 111 |
+
"sha256": "21018e6ed6ec35f30b2739a5fd0adfae10eee6aa6d24e0e877651317d7cc76e7",
|
| 112 |
+
"bytes": 9921
|
| 113 |
+
},
|
| 114 |
+
"figures/multilingual.png": {
|
| 115 |
+
"sha256": "544a3982019f7d0d8f788f4931ba88bde317956c87f7a7138fcd5e4ec22013d4",
|
| 116 |
+
"bytes": 300658
|
| 117 |
+
},
|
| 118 |
+
"figures/multilingual.svg": {
|
| 119 |
+
"sha256": "c4442d53ba20f1a3bfa3b5764db211a7a18db95c48ec834f78d454ca13efd7e4",
|
| 120 |
+
"bytes": 63997
|
| 121 |
+
},
|
| 122 |
+
"figures/quantization.data.json": {
|
| 123 |
+
"sha256": "a140816ef145a6dd096ef471e4afbff1cecd33adb663c6cb856e19845349fc2d",
|
| 124 |
+
"bytes": 5473
|
| 125 |
+
},
|
| 126 |
+
"figures/quantization.png": {
|
| 127 |
+
"sha256": "bc75c212caa11008ca929a685d3085040c2b4698afb58d04a0ad692b815be230",
|
| 128 |
+
"bytes": 253992
|
| 129 |
+
},
|
| 130 |
+
"figures/quantization.svg": {
|
| 131 |
+
"sha256": "70720bb46ec1e718dcbfac9a83d3b5d11d50b3a5cc3356915465d70d95f3d43b",
|
| 132 |
+
"bytes": 28756
|
| 133 |
+
},
|
| 134 |
+
"figures/zeroshot.json": {
|
| 135 |
+
"sha256": "003c087074b876efab789fadf94f51d134c604d5f2efce417557732de2272e9d",
|
| 136 |
+
"bytes": 1867
|
| 137 |
+
},
|
| 138 |
+
"figures/zeroshot.png": {
|
| 139 |
+
"sha256": "255e196d449c086130d0f06c493c64a7cc21730a38d21c1c2188641e4c08c002",
|
| 140 |
+
"bytes": 128831
|
| 141 |
+
},
|
| 142 |
+
"figures/zeroshot.svg": {
|
| 143 |
+
"sha256": "a148fa7cb8f4e0b6e529996441fadfa883595bfeb05dc93c4048609071655790",
|
| 144 |
+
"bytes": 14474
|
| 145 |
+
},
|
| 146 |
+
"generation_config.json": {
|
| 147 |
+
"sha256": "62153eb6c69f2e1f426beaa8002b7186437e949c7588167085df14e10e9c0a73",
|
| 148 |
+
"bytes": 116
|
| 149 |
+
},
|
| 150 |
+
"jev_style_decision.py": {
|
| 151 |
+
"sha256": "be38df723d83ef5fdffb8f7dfaee63b3c31e632b233fc3c7d5f6eb8c83fd994f",
|
| 152 |
+
"bytes": 25917
|
| 153 |
+
},
|
| 154 |
+
"model.safetensors": {
|
| 155 |
+
"sha256": "0f8c861605dcdfb356a63e056baa2106e81042e8981e8d0d26fe50b34599541e",
|
| 156 |
+
"bytes": 1504827608
|
| 157 |
+
},
|
| 158 |
+
"readout_config.json": {
|
| 159 |
+
"sha256": "01de9bcce7effbfd1ae0a3fa13e52d7fa9aa4dd1cd1e47977d6bbdcd5e63054a",
|
| 160 |
+
"bytes": 6236
|
| 161 |
+
},
|
| 162 |
+
"release_config.json": {
|
| 163 |
+
"sha256": "7ee881089c1ca5b019f98a26e60da4b0a454100d8cd557e5e352df3813aa31c0",
|
| 164 |
+
"bytes": 7088
|
| 165 |
+
},
|
| 166 |
+
"requirements.txt": {
|
| 167 |
+
"sha256": "855306c7cd8db9fcea2f865d61a0c53330c530c9dcf07e955a9250aaae2993f1",
|
| 168 |
+
"bytes": 201
|
| 169 |
+
},
|
| 170 |
+
"tokenizer.json": {
|
| 171 |
+
"sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
|
| 172 |
+
"bytes": 19989325
|
| 173 |
+
},
|
| 174 |
+
"tokenizer_config.json": {
|
| 175 |
+
"sha256": "66e427c470fe580fe8c7b5725d857af23d8417e37fae62667ec698306a19987b",
|
| 176 |
+
"bytes": 1124
|
| 177 |
+
}
|
| 178 |
+
},
|
| 179 |
+
"readme_hashed": true,
|
| 180 |
+
"note": "manifest.json hashes every file of the repo except itself, README.md and figures/ included. The runtime --verify check skips the documentation (README.md, figures/, assets/) and checks every other file."
|
| 181 |
+
}
|
model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0f8c861605dcdfb356a63e056baa2106e81042e8981e8d0d26fe50b34599541e
|
| 3 |
+
size 1504827608
|
readout_config.json
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"format": "macjev-readout-v1",
|
| 3 |
+
"model_name": "Jev-Style-0.8B-Decision-v3",
|
| 4 |
+
"readout": "verdict",
|
| 5 |
+
"template": "macjev-render-v1",
|
| 6 |
+
"slot_tokens": {
|
| 7 |
+
"yes": {
|
| 8 |
+
"text": " yes",
|
| 9 |
+
"id": 9542
|
| 10 |
+
},
|
| 11 |
+
"no": {
|
| 12 |
+
"text": " no",
|
| 13 |
+
"id": 874
|
| 14 |
+
},
|
| 15 |
+
"verdict_slot": {
|
| 16 |
+
"text": " ->",
|
| 17 |
+
"id": 1411
|
| 18 |
+
},
|
| 19 |
+
"letters": {
|
| 20 |
+
"A": 357,
|
| 21 |
+
"B": 417,
|
| 22 |
+
"C": 351,
|
| 23 |
+
"D": 414,
|
| 24 |
+
"E": 458,
|
| 25 |
+
"F": 426,
|
| 26 |
+
"G": 469,
|
| 27 |
+
"H": 462,
|
| 28 |
+
"I": 353,
|
| 29 |
+
"J": 604,
|
| 30 |
+
"K": 710,
|
| 31 |
+
"L": 436,
|
| 32 |
+
"M": 380,
|
| 33 |
+
"N": 443,
|
| 34 |
+
"O": 496,
|
| 35 |
+
"P": 387,
|
| 36 |
+
"Q": 1167,
|
| 37 |
+
"R": 423,
|
| 38 |
+
"S": 326,
|
| 39 |
+
"T": 345,
|
| 40 |
+
"U": 533,
|
| 41 |
+
"V": 629,
|
| 42 |
+
"W": 457,
|
| 43 |
+
"X": 1543,
|
| 44 |
+
"Y": 783,
|
| 45 |
+
"Z": 1799,
|
| 46 |
+
"a": 264,
|
| 47 |
+
"b": 292,
|
| 48 |
+
"c": 272,
|
| 49 |
+
"d": 293,
|
| 50 |
+
"e": 378,
|
| 51 |
+
"f": 281,
|
| 52 |
+
"g": 338,
|
| 53 |
+
"h": 304,
|
| 54 |
+
"i": 585,
|
| 55 |
+
"j": 492,
|
| 56 |
+
"k": 580,
|
| 57 |
+
"l": 324,
|
| 58 |
+
"m": 295,
|
| 59 |
+
"n": 307,
|
| 60 |
+
"o": 296,
|
| 61 |
+
"p": 280,
|
| 62 |
+
"q": 2715,
|
| 63 |
+
"r": 427,
|
| 64 |
+
"s": 274,
|
| 65 |
+
"t": 259,
|
| 66 |
+
"u": 560,
|
| 67 |
+
"v": 343,
|
| 68 |
+
"w": 288,
|
| 69 |
+
"x": 830,
|
| 70 |
+
"y": 374,
|
| 71 |
+
"z": 1110
|
| 72 |
+
}
|
| 73 |
+
},
|
| 74 |
+
"score": "per option k: logit[' yes'] - logit[' no'] at the k-th ' ->' slot, computed as h_slot . (w_yes - w_no) from the final normed hidden state and the tied embedding rows (float32)",
|
| 75 |
+
"probabilities": "softmax(scores / T); T = temperatures.groups['<family>|<qtype>|<option bucket>'].T (else temperatures.global), clamped to temperatures.clamp",
|
| 76 |
+
"families": {
|
| 77 |
+
"typed_official*": "typed",
|
| 78 |
+
"typed_synthetic*": "typed_synth",
|
| 79 |
+
"general_*": "general",
|
| 80 |
+
"intent*": "intent",
|
| 81 |
+
"nli*": "nli",
|
| 82 |
+
"theme_*": "theme",
|
| 83 |
+
"mac_*": "mac",
|
| 84 |
+
"long_*": "long",
|
| 85 |
+
"anything else": "other (global T)"
|
| 86 |
+
},
|
| 87 |
+
"option_buckets": [
|
| 88 |
+
"2",
|
| 89 |
+
"3-5",
|
| 90 |
+
"6-10",
|
| 91 |
+
"11-20",
|
| 92 |
+
"21+"
|
| 93 |
+
],
|
| 94 |
+
"default_category": null,
|
| 95 |
+
"default_category_note": "no category given -> temperatures.global (the global temperature); pass category=... for the fitted group temperature of that category's family",
|
| 96 |
+
"tokenizer_sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
|
| 97 |
+
"budgets": {
|
| 98 |
+
"max_len": 25600,
|
| 99 |
+
"head_max": 2048,
|
| 100 |
+
"hard_head_max": 2048,
|
| 101 |
+
"note": "whole input <= max_len tokens, question+options+readout <= head_max; larger inputs raise InputBudgetError, nothing is truncated"
|
| 102 |
+
},
|
| 103 |
+
"temperatures": {
|
| 104 |
+
"version": "macjev-temperatures-v1",
|
| 105 |
+
"global": 0.8800546821789332,
|
| 106 |
+
"groups": {
|
| 107 |
+
"general|choice|3-5": {
|
| 108 |
+
"T": 0.8563796906101172,
|
| 109 |
+
"T_raw": 0.8524962478467872,
|
| 110 |
+
"n": 600,
|
| 111 |
+
"weight": 0.8571428571428571
|
| 112 |
+
},
|
| 113 |
+
"general|choice|6-10": {
|
| 114 |
+
"T": 0.7883347591845132,
|
| 115 |
+
"T_raw": 0.7797058230459033,
|
| 116 |
+
"n": 1000,
|
| 117 |
+
"weight": 0.9090909090909091
|
| 118 |
+
},
|
| 119 |
+
"general|noul|2": {
|
| 120 |
+
"T": 0.9702474579038656,
|
| 121 |
+
"T_raw": 1.0023209290239254,
|
| 122 |
+
"n": 300,
|
| 123 |
+
"weight": 0.75
|
| 124 |
+
},
|
| 125 |
+
"general|score|3-5": {
|
| 126 |
+
"T": 0.9583320444900012,
|
| 127 |
+
"T_raw": 0.9748039508642105,
|
| 128 |
+
"n": 500,
|
| 129 |
+
"weight": 0.8333333333333334
|
| 130 |
+
},
|
| 131 |
+
"intent|choice|11-20": {
|
| 132 |
+
"T": 0.8536586990515279,
|
| 133 |
+
"T_raw": 0.8530447902066461,
|
| 134 |
+
"n": 4233,
|
| 135 |
+
"weight": 0.9769213016385876
|
| 136 |
+
},
|
| 137 |
+
"intent|choice|21+": {
|
| 138 |
+
"T": 0.7508751181814466,
|
| 139 |
+
"T_raw": 0.737974406022733,
|
| 140 |
+
"n": 916,
|
| 141 |
+
"weight": 0.9015748031496063
|
| 142 |
+
},
|
| 143 |
+
"long|choice|2": {
|
| 144 |
+
"T": 1.0101601686032944,
|
| 145 |
+
"T_raw": 2.5327604766936602,
|
| 146 |
+
"n": 15,
|
| 147 |
+
"weight": 0.13043478260869565
|
| 148 |
+
},
|
| 149 |
+
"long|choice|3-5": {
|
| 150 |
+
"T": 0.9789458925825579,
|
| 151 |
+
"T_raw": 0.9984431087771811,
|
| 152 |
+
"n": 540,
|
| 153 |
+
"weight": 0.84375
|
| 154 |
+
},
|
| 155 |
+
"long|choice|6-10": {
|
| 156 |
+
"T": 0.6918332634850016,
|
| 157 |
+
"T_raw": 0.6260903832047418,
|
| 158 |
+
"n": 241,
|
| 159 |
+
"weight": 0.7067448680351907
|
| 160 |
+
},
|
| 161 |
+
"long|noul|2": {
|
| 162 |
+
"T": 0.8183260460233314,
|
| 163 |
+
"T_raw": 0.7857458244621078,
|
| 164 |
+
"n": 179,
|
| 165 |
+
"weight": 0.6415770609318996
|
| 166 |
+
},
|
| 167 |
+
"long|score|3-5": {
|
| 168 |
+
"T": 0.8412851094005789,
|
| 169 |
+
"T_raw": 0.7952160414763221,
|
| 170 |
+
"n": 80,
|
| 171 |
+
"weight": 0.4444444444444444
|
| 172 |
+
},
|
| 173 |
+
"mac|choice|3-5": {
|
| 174 |
+
"T": 0.7640449061346866,
|
| 175 |
+
"T_raw": 0.753267027600698,
|
| 176 |
+
"n": 995,
|
| 177 |
+
"weight": 0.908675799086758
|
| 178 |
+
},
|
| 179 |
+
"mac|noul|2": {
|
| 180 |
+
"T": 0.922463080251143,
|
| 181 |
+
"T_raw": 0.9300179680535603,
|
| 182 |
+
"n": 577,
|
| 183 |
+
"weight": 0.8522895125553914
|
| 184 |
+
},
|
| 185 |
+
"mac|score|3-5": {
|
| 186 |
+
"T": 2.729450780811877,
|
| 187 |
+
"T_raw": 4.999707266277221,
|
| 188 |
+
"n": 187,
|
| 189 |
+
"weight": 0.6515679442508711
|
| 190 |
+
},
|
| 191 |
+
"nli|choice|3-5": {
|
| 192 |
+
"T": 1.003611674961589,
|
| 193 |
+
"T_raw": 1.0108425661039413,
|
| 194 |
+
"n": 1830,
|
| 195 |
+
"weight": 0.9481865284974094
|
| 196 |
+
},
|
| 197 |
+
"theme|choice|6-10": {
|
| 198 |
+
"T": 0.9848729122522978,
|
| 199 |
+
"T_raw": 0.9981552921230235,
|
| 200 |
+
"n": 840,
|
| 201 |
+
"weight": 0.8936170212765957
|
| 202 |
+
},
|
| 203 |
+
"theme|noul|2": {
|
| 204 |
+
"T": 0.896799495386032,
|
| 205 |
+
"T_raw": 0.89763584528285,
|
| 206 |
+
"n": 2022,
|
| 207 |
+
"weight": 0.9528746465598492
|
| 208 |
+
},
|
| 209 |
+
"typed|choice|3-5": {
|
| 210 |
+
"T": 0.9751541851571508,
|
| 211 |
+
"T_raw": 1.0336976619219436,
|
| 212 |
+
"n": 176,
|
| 213 |
+
"weight": 0.6376811594202898
|
| 214 |
+
},
|
| 215 |
+
"typed|noul|2": {
|
| 216 |
+
"T": 0.9892589014450378,
|
| 217 |
+
"T_raw": 1.054189849787923,
|
| 218 |
+
"n": 184,
|
| 219 |
+
"weight": 0.647887323943662
|
| 220 |
+
},
|
| 221 |
+
"typed|score|3-5": {
|
| 222 |
+
"T": 1.0056628114581752,
|
| 223 |
+
"T_raw": 1.0631515969968885,
|
| 224 |
+
"n": 240,
|
| 225 |
+
"weight": 0.7058823529411765
|
| 226 |
+
}
|
| 227 |
+
},
|
| 228 |
+
"clamp": [
|
| 229 |
+
0.3,
|
| 230 |
+
5.0
|
| 231 |
+
],
|
| 232 |
+
"shrinkage_k": 100.0,
|
| 233 |
+
"key": "family|qtype|option_bucket",
|
| 234 |
+
"n_rows": 15655,
|
| 235 |
+
"fitted_on": [
|
| 236 |
+
"pool_cal.jsonl:dfb7e9beff96c6c5"
|
| 237 |
+
],
|
| 238 |
+
"fit_quality": {
|
| 239 |
+
"nll_before": 0.37752055301970966,
|
| 240 |
+
"nll_after": 0.36671188108641545,
|
| 241 |
+
"ece_before": 0.03289307373502533,
|
| 242 |
+
"ece_after": 0.011376234101433989,
|
| 243 |
+
"n": 15655
|
| 244 |
+
}
|
| 245 |
+
},
|
| 246 |
+
"temperatures_sha256": "39ad8f6633934ff67770725993d7354ebebd6e7a08f73e50cab04df24715f2ce"
|
| 247 |
+
}
|
release_config.json
ADDED
|
@@ -0,0 +1,186 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"format": "jev-style-release-v1",
|
| 3 |
+
"model_name": "Jev-Style-0.8B-Decision-v3",
|
| 4 |
+
"repo": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3",
|
| 5 |
+
"generation": "v3 (third generation of the Jev-Style decision series)",
|
| 6 |
+
"lineage": {
|
| 7 |
+
"v1": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision",
|
| 8 |
+
"v1_public_gguf": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF",
|
| 9 |
+
"v2": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2",
|
| 10 |
+
"v3": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3"
|
| 11 |
+
},
|
| 12 |
+
"base_model": "Qwen/Qwen3.5-0.8B",
|
| 13 |
+
"base_model_revision": "2fc06364715b967f1860aea9cf38778875588b17",
|
| 14 |
+
"base_model_relation": "finetune",
|
| 15 |
+
"architecture": "Qwen3_5ForCausalLM (text only, 24 layers: 18 Gated DeltaNet + 6 full attention, hidden 1024, tied embeddings, 752,393,024 parameters)",
|
| 16 |
+
"readout": "verdict",
|
| 17 |
+
"template": "macjev-render-v1",
|
| 18 |
+
"readout_config": "readout_config.json",
|
| 19 |
+
"budgets": {
|
| 20 |
+
"max_len": 25600,
|
| 21 |
+
"head_max": 2048
|
| 22 |
+
},
|
| 23 |
+
"source": {
|
| 24 |
+
"checkpoint_sha256": {
|
| 25 |
+
"best-0.safetensors": "b10d249adfa3e1475f0f633ab961130c06f16c898db23c846b035d24bdc5c945"
|
| 26 |
+
},
|
| 27 |
+
"text_only_model_safetensors_sha256": "0f8c861605dcdfb356a63e056baa2106e81042e8981e8d0d26fe50b34599541e",
|
| 28 |
+
"release_manifest_sha256": "4bc9f89795ccf2d749008de7739bc5294f3ec543404ca2db9cac6a3914cfb84a",
|
| 29 |
+
"llama_cpp_commit": "441df11f65ea0b6d0c72965aaf70c8241070ddcb",
|
| 30 |
+
"mlx": "0.32.2",
|
| 31 |
+
"mlx_lm": "0.31.3"
|
| 32 |
+
},
|
| 33 |
+
"calibration": {
|
| 34 |
+
"version": "macjev-temperatures-v1",
|
| 35 |
+
"global_T": 0.8800546821789332,
|
| 36 |
+
"groups": 20,
|
| 37 |
+
"n_rows": 15655,
|
| 38 |
+
"fit_quality": {
|
| 39 |
+
"nll_before": 0.37752055301970966,
|
| 40 |
+
"nll_after": 0.36671188108641545,
|
| 41 |
+
"ece_before": 0.03289307373502533,
|
| 42 |
+
"ece_after": 0.011376234101433989,
|
| 43 |
+
"n": 15655
|
| 44 |
+
},
|
| 45 |
+
"fitted_on": "calibration pool (dev/cal rows, never test rows)"
|
| 46 |
+
},
|
| 47 |
+
"g5_parity": {
|
| 48 |
+
"reference": "PyTorch float32 (same weights, same rendered inputs)",
|
| 49 |
+
"rows": 240,
|
| 50 |
+
"rows_note": "non-sealed training-distribution rows, 22 categories, en 215 / zh 25",
|
| 51 |
+
"report_sha256": "056dc3a3d1de5616f97dab3f8b5ec1f7e2c6c85258aa03f2a57b9fd65a71022b",
|
| 52 |
+
"g5_pass": true,
|
| 53 |
+
"verdict_readout": {
|
| 54 |
+
"gguf-f16": {
|
| 55 |
+
"top1_agreement": 1.0,
|
| 56 |
+
"dnll": 2.436279701012456e-05,
|
| 57 |
+
"n": 240,
|
| 58 |
+
"nan_rows": 0,
|
| 59 |
+
"gate": "top1>=0.99, |dNLL|<=0.02, no NaN, every reference row scored",
|
| 60 |
+
"pass": true
|
| 61 |
+
},
|
| 62 |
+
"gguf-q8_0": {
|
| 63 |
+
"top1_agreement": 1.0,
|
| 64 |
+
"dnll": -3.093173930673876e-05,
|
| 65 |
+
"n": 240,
|
| 66 |
+
"nan_rows": 0,
|
| 67 |
+
"gate": "top1>=0.98, |dNLL|<=0.02, no NaN, every reference row scored",
|
| 68 |
+
"pass": true
|
| 69 |
+
},
|
| 70 |
+
"gguf-q4_k_m": {
|
| 71 |
+
"top1_agreement": 1.0,
|
| 72 |
+
"dnll": 0.006187511546346225,
|
| 73 |
+
"n": 240,
|
| 74 |
+
"nan_rows": 0,
|
| 75 |
+
"gate": "report_only",
|
| 76 |
+
"pass": true
|
| 77 |
+
},
|
| 78 |
+
"mlx-bf16": {
|
| 79 |
+
"top1_agreement": 1.0,
|
| 80 |
+
"dnll": 0.00031018251516545803,
|
| 81 |
+
"n": 240,
|
| 82 |
+
"nan_rows": 0,
|
| 83 |
+
"gate": "top1>=0.99, |dNLL|<=0.02, no NaN, every reference row scored",
|
| 84 |
+
"pass": true
|
| 85 |
+
},
|
| 86 |
+
"mlx-bf16-f32act": {
|
| 87 |
+
"top1_agreement": 1.0,
|
| 88 |
+
"dnll": 0.00015659911501769708,
|
| 89 |
+
"n": 240,
|
| 90 |
+
"nan_rows": 0,
|
| 91 |
+
"gate": "top1>=0.99, |dNLL|<=0.02, no NaN, every reference row scored",
|
| 92 |
+
"pass": true
|
| 93 |
+
},
|
| 94 |
+
"mlx-8bit": {
|
| 95 |
+
"top1_agreement": 1.0,
|
| 96 |
+
"dnll": 0.00023178691602621093,
|
| 97 |
+
"n": 240,
|
| 98 |
+
"nan_rows": 0,
|
| 99 |
+
"gate": "top1>=0.98, |dNLL|<=0.02, no NaN, every reference row scored",
|
| 100 |
+
"pass": true
|
| 101 |
+
},
|
| 102 |
+
"mlx-4bit": {
|
| 103 |
+
"top1_agreement": 0.9875,
|
| 104 |
+
"dnll": 0.010326217052955111,
|
| 105 |
+
"n": 240,
|
| 106 |
+
"nan_rows": 0,
|
| 107 |
+
"gate": "report_only",
|
| 108 |
+
"pass": true
|
| 109 |
+
}
|
| 110 |
+
},
|
| 111 |
+
"gates": "top-1 >= 0.99 (16-bit) / >= 0.98 (8-bit), |dNLL| <= 0.02, no NaN; 4-bit report-only"
|
| 112 |
+
},
|
| 113 |
+
"decision_index": "not run for this release (optional follow-up)",
|
| 114 |
+
"related_repos": {
|
| 115 |
+
"main": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3",
|
| 116 |
+
"gguf": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3-GGUF",
|
| 117 |
+
"mlx-bf16": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-bf16",
|
| 118 |
+
"mlx-8bit": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX-8bit"
|
| 119 |
+
},
|
| 120 |
+
"not_released": {
|
| 121 |
+
"mlx-4bit": "affine4-g64 reported only (G5 top-1 0.9875, report-only gate); not released"
|
| 122 |
+
},
|
| 123 |
+
"tested_with": {
|
| 124 |
+
"python": "3.12",
|
| 125 |
+
"torch": "2.14.0",
|
| 126 |
+
"transformers": "5.17.0",
|
| 127 |
+
"tokenizers": "0.23.2",
|
| 128 |
+
"numpy": "2.5.3",
|
| 129 |
+
"mlx": "0.32.2",
|
| 130 |
+
"mlx-lm": "0.31.3",
|
| 131 |
+
"llama.cpp": "441df11f65ea0b6d0c72965aaf70c8241070ddcb"
|
| 132 |
+
},
|
| 133 |
+
"weights": {
|
| 134 |
+
"model.safetensors": "bfloat16 (text-only export of the trained checkpoint)"
|
| 135 |
+
},
|
| 136 |
+
"runtime": {
|
| 137 |
+
"script": "jev_style_decision.py",
|
| 138 |
+
"class": "JevStyleDecision",
|
| 139 |
+
"default_dtype": "float32",
|
| 140 |
+
"devices": [
|
| 141 |
+
"cuda",
|
| 142 |
+
"mps",
|
| 143 |
+
"cpu"
|
| 144 |
+
],
|
| 145 |
+
"long_inputs": "query-chunked SDPA on MPS/CPU (1024 queries per chunk)"
|
| 146 |
+
},
|
| 147 |
+
"runtime_parity": {
|
| 148 |
+
"protocol": "2026-09-24: each runtime script run in a clean subprocess (cwd = repo folder, empty PYTHONPATH, HF_HUB_OFFLINE=1, --verify) on 24 fixed parity-fixture rows (every 10th of the 240 G5 rows; 12 categories families, choice/score/noul, 92-8156 tokens) and compared with the reference (training-code) scorer of the same format on the same inputs, probabilities with the same calibration temperature; plus 2 synthetic long states (16,381 and 25,582 tokens). Rendered token ids identical to the reference renderer on 244/244 rows (240 fixture rows + 4 adversarial special-token/unicode rows). Re-run 2026-09-24 after the rename to Jev-Style-0.8B-Decision-v3, the no-category = global-temperature default and the GGUF general.name metadata edit: all six formats again identical to the reference scorer run on the original (pre-edit) files (max |prob diff| 0.0 same backend, 0 top-1 changes). Re-run 2026-09-25 after the GGUF decide_many fix (runtime code change in decide_many only, docstrings in all scripts): all six formats gave outputs identical (0.0) to the 2026-09-24 run on the 24 rows.",
|
| 149 |
+
"results": {
|
| 150 |
+
"torch-fp32-cpu": {
|
| 151 |
+
"rows": 24,
|
| 152 |
+
"errors": 0,
|
| 153 |
+
"vs_reference_same_backend": {
|
| 154 |
+
"max_abs_prob_diff": 0.0,
|
| 155 |
+
"max_abs_score_diff": 0.0,
|
| 156 |
+
"top1_changes": 0,
|
| 157 |
+
"top1_agreement": 1.0
|
| 158 |
+
},
|
| 159 |
+
"vs_reference_torch_fp32": {
|
| 160 |
+
"max_abs_prob_diff": 0.0,
|
| 161 |
+
"max_abs_score_diff": 0.0,
|
| 162 |
+
"top1_changes": 0,
|
| 163 |
+
"top1_agreement": 1.0
|
| 164 |
+
}
|
| 165 |
+
},
|
| 166 |
+
"torch-fp32-mps-long": {
|
| 167 |
+
"long16384:torch-mps-fp32": {
|
| 168 |
+
"tokens": 16381,
|
| 169 |
+
"answer_matches_gguf_mlx": true,
|
| 170 |
+
"max_abs_prob_diff_vs_gguf_f16": 0.0004853327842702093
|
| 171 |
+
},
|
| 172 |
+
"long25600:torch-mps-fp32": {
|
| 173 |
+
"tokens": 25582,
|
| 174 |
+
"answer_matches_gguf_mlx": true,
|
| 175 |
+
"max_abs_prob_diff_vs_gguf_f16": 0.0002457350394169111
|
| 176 |
+
}
|
| 177 |
+
},
|
| 178 |
+
"render_token_identity": {
|
| 179 |
+
"rows": 244,
|
| 180 |
+
"identical": 244,
|
| 181 |
+
"different": 0,
|
| 182 |
+
"bad": []
|
| 183 |
+
}
|
| 184 |
+
}
|
| 185 |
+
}
|
| 186 |
+
}
|
requirements.txt
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# tested with: python 3.12, torch 2.14.0, transformers 5.17.0, tokenizers 0.23.2, numpy 2.5.3
|
| 2 |
+
torch>=2.4
|
| 3 |
+
transformers>=5.0 # needs Qwen3.5 support (transformers.models.qwen3_5)
|
| 4 |
+
tokenizers>=0.21
|
| 5 |
+
numpy
|
tokenizer.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523
|
| 3 |
+
size 19989325
|
tokenizer_config.json
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"audio_bos_token": "<|audio_start|>",
|
| 4 |
+
"audio_eos_token": "<|audio_end|>",
|
| 5 |
+
"audio_token": "<|audio_pad|>",
|
| 6 |
+
"backend": "tokenizers",
|
| 7 |
+
"bos_token": null,
|
| 8 |
+
"clean_up_tokenization_spaces": false,
|
| 9 |
+
"eos_token": "<|im_end|>",
|
| 10 |
+
"errors": "replace",
|
| 11 |
+
"image_token": "<|image_pad|>",
|
| 12 |
+
"is_local": true,
|
| 13 |
+
"local_files_only": false,
|
| 14 |
+
"model_max_length": 262144,
|
| 15 |
+
"model_specific_special_tokens": {
|
| 16 |
+
"audio_bos_token": "<|audio_start|>",
|
| 17 |
+
"audio_eos_token": "<|audio_end|>",
|
| 18 |
+
"audio_token": "<|audio_pad|>",
|
| 19 |
+
"image_token": "<|image_pad|>",
|
| 20 |
+
"video_token": "<|video_pad|>",
|
| 21 |
+
"vision_bos_token": "<|vision_start|>",
|
| 22 |
+
"vision_eos_token": "<|vision_end|>"
|
| 23 |
+
},
|
| 24 |
+
"pad_token": "<|endoftext|>",
|
| 25 |
+
"pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
| 26 |
+
"split_special_tokens": false,
|
| 27 |
+
"tokenizer_class": "Qwen2Tokenizer",
|
| 28 |
+
"unk_token": null,
|
| 29 |
+
"video_token": "<|video_pad|>",
|
| 30 |
+
"vision_bos_token": "<|vision_start|>",
|
| 31 |
+
"vision_eos_token": "<|vision_end|>"
|
| 32 |
+
}
|