developerjeremylive areneau commited on
Commit
144957b
·
0 Parent(s):

Duplicate from Cloudflare/clef

Browse files

Co-authored-by: Alex Reneau <areneau@users.noreply.huggingface.co>

.gitattributes ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
README.md ADDED
@@ -0,0 +1,225 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ library_name: transformers
4
+ pipeline_tag: image-text-to-text
5
+ base_model: Qwen/Qwen3.8-27B
6
+ base_model_relation: finetune
7
+ tags:
8
+ - clef
9
+ - cloudflare
10
+ - systemone
11
+ - qwen3.8
12
+ - post-train
13
+ - image-text-to-typed-output
14
+ - multimodal
15
+ - structured-output
16
+ - classification
17
+ - custom-code
18
+ ---
19
+
20
+ # Clef
21
+
22
+ - **Announcement:** [Clef decision models on the Cloudflare blog](https://blog.cloudflare.com/clef-decision-models)
23
+ - **Decision Index leaderboard:** [clef-evals.workers-ai-mle.workers.dev](https://clef-evals.workers-ai-mle.workers.dev)
24
+
25
+ Clef is a 27B multimodal model that turns a state and a schema of typed
26
+ questions into decisions. It reads the state as text, JSON, images, or video, and returns a
27
+ probability for every allowed option of every question in a single forward pass. There is no
28
+ free-form text generation and no output parsing.
29
+
30
+ The Clef API is fully compatible with Jev and SystemOne.
31
+
32
+ Clef is post-trained from [Qwen/Qwen3.8-27B](https://huggingface.co/Qwen/Qwen3.8-27B). See
33
+ [Clef-Flash](https://huggingface.co/Cloudflare/clef-flash) for the
34
+ smaller, faster variant.
35
+
36
+ ## Model
37
+
38
+ - **Backbone:** Qwen/Qwen3.8-27B with its vision encoder, stored as standard sharded safetensors.
39
+ - **Joint schema head:** a small transformer head that reads the backbone's final hidden states,
40
+ routes evidence from the state to each question, and scores all options of all questions jointly.
41
+ - **Output:** one logit per allowed option for each question. Apply a softmax per question to get
42
+ probabilities.
43
+
44
+ ## Files
45
+
46
+ | File | Purpose |
47
+ |---|---|
48
+ | `model-*.safetensors`, `model.safetensors.index.json`, `config.json`, `generation_config.json` | Backbone, including the vision encoder |
49
+ | `joint_head.safetensors`, `joint_head_config.json` | Joint schema head |
50
+ | `joint_schema_model.py` | Record encoding, batching, the model, `load_release_model`, and `systemone` |
51
+ | `tokenizer.json`, `tokenizer_config.json`, `chat_template.jinja`, `processor_config.json` | Tokenizer and image/video processor |
52
+ | `LICENSE` | Apache-2.0 license |
53
+
54
+ ## Usage
55
+
56
+ Tested with `torch` 2.11 and `transformers` 5.10.2 on a single H200. Image and video inputs also
57
+ need `pillow`.
58
+
59
+ ```python
60
+ import sys
61
+
62
+ import torch
63
+ from huggingface_hub import snapshot_download
64
+
65
+ path = snapshot_download("Cloudflare/clef")
66
+ sys.path.insert(0, path)
67
+ from joint_schema_model import collate_records, encode_record, load_release_model
68
+
69
+ model, processor = load_release_model(path, device="cuda")
70
+
71
+ record = {
72
+ "state": {"invoice": {"vendor": "Acme", "total": 1250.0, "currency": "USD", "status": "overdue"}},
73
+ "questions": {
74
+ "status": {
75
+ "type": "choice",
76
+ "instructions": "What is the invoice status?",
77
+ "criteria": {"paid": "Invoice is paid.", "overdue": "Invoice is past due.", "draft": "Not sent."},
78
+ },
79
+ "large": {"type": "noul", "instructions": "Is the total above 1000 USD?"},
80
+ },
81
+ }
82
+
83
+ encoded = encode_record(processor.tokenizer, record, processor=processor)
84
+ batch = collate_records([encoded], processor.tokenizer.pad_token_id, torch.device("cuda"))
85
+ with torch.inference_mode():
86
+ logits = model(batch)[0]
87
+
88
+ for question, question_logits in zip(encoded.questions, logits):
89
+ probabilities = question_logits.float().softmax(-1).tolist()
90
+ print(question.question_id, dict(zip(question.option_ids, probabilities)))
91
+ ```
92
+
93
+ ### Jev / SystemOne API
94
+
95
+ `systemone` takes a Jev/SystemOne `POST /v1/systemone` request body and returns the same response
96
+ body: `model`, `answers` keyed by question ID, and `usage`. A `choice` answer has `choice`,
97
+ `confidence`, and `probabilities`; a `score` answer has the expected `score`, `confidence`, `legend`,
98
+ and `probabilities`; a `noul` answer has the probability of true. `instructions` is optional, and
99
+ `images` and `videos` may be added to the request.
100
+
101
+ ```python
102
+ from joint_schema_model import systemone
103
+
104
+ response = systemone(model, processor, {
105
+ "model": "clef",
106
+ "state": "Our checkout started returning errors and orders are blocked.",
107
+ "questions": {
108
+ "department": {
109
+ "type": "choice",
110
+ "instructions": "Which team should handle the message?",
111
+ "criteria": {"billing": "Payments or invoices", "technical": "Bugs or outages"},
112
+ },
113
+ "urgency": {"type": "score", "criteria": ["Can wait", "This week", "Today"]},
114
+ "outage": {"type": "noul", "instructions": "Is a service down?"},
115
+ },
116
+ })
117
+ print(response["answers"])
118
+ ```
119
+
120
+ ### Images and video
121
+
122
+ Add `images` (PIL images) or `videos` (frame arrays) to the record and pass the processor to
123
+ `encode_record`. Optional processor arguments go in `media_kwargs`.
124
+
125
+ ```python
126
+ from PIL import Image
127
+
128
+ record = {
129
+ "state": {"task": "Review the attached receipt."},
130
+ "images": [Image.open("receipt.jpg")],
131
+ "questions": {
132
+ "legible": {"type": "noul", "instructions": "Is the receipt total legible?"},
133
+ },
134
+ }
135
+ encoded = encode_record(processor.tokenizer, record, processor=processor)
136
+ ```
137
+
138
+ Text-only and multimodal records can be mixed in the same batch.
139
+
140
+ ## Input format
141
+
142
+ | Field | Description |
143
+ |---|---|
144
+ | `state` | Any string or JSON value describing the situation to decide on |
145
+ | `images`, `videos` | Optional lists of images or video frame arrays |
146
+ | `media_kwargs` | Optional keyword arguments for the image/video processor |
147
+ | `questions` | Mapping of question ID to question |
148
+
149
+ Each question has:
150
+
151
+ - `type`: `noul` (true/false), `choice` (named options), or `score` (ordered options)
152
+ - `instructions`: what to decide; optional, and the question ID is used when it is omitted
153
+ - `criteria`: for `choice`, a mapping of option ID to description; for `score`, a list of option
154
+ descriptions indexed from 0; for `noul`, optional descriptions for `true` and `false`
155
+
156
+ `encode_record` accepts `max_length` (default 16,384 tokens) and `max_state_tokens` to bound the input.
157
+
158
+ ## Results
159
+
160
+ ### Decision Index
161
+
162
+ Per-benchmark results from our internal run of the [Decision Index](https://clef-evals.workers-ai-mle.workers.dev) 0.2.1 suite. Scores are percentages; ForecastBench is a Brier score, where lower is better. The last two rows are request latency in milliseconds, where lower is better. The best value in each row is in bold.
163
+
164
+ | Benchmark | Clef | Clef-flash | Jev | DiffusionGemma Jev | Kev 9B | Laya |
165
+ |---|---|---|---|---|---|---|
166
+ | BFCL (case exact accuracy) | 98.5 | **98.8** | 95.8 | 96.5 | 94.5 | 38.1 |
167
+ | ToolRet (nDCG@10) | **69.2** | 66.4 | 65.3 | 61.2 | 64.3 | 12.8 |
168
+ | API-Bank (accuracy) | 91.9 | **93.1** | 88.2 | 83.7 | 56.3 | 11.5 |
169
+ | BANKING77 (macro-F1) | **94.2** | 90.9 | 79.7 | 74.3 | 84.8 | 14.3 |
170
+ | CLINC150+OOS (macro-F1) | **97.4** | 66.8 | 89.3 | 83.5 | 79.0 | 3.2 |
171
+ | RouterBench (selected quality) | 79.7 | 79.9 | 79.9 | 79.0 | **80.0** | 57.1 |
172
+ | Home appliance simulator (case exact accuracy) | 83.0 | **97.7** | 52.3 | 42.0 | 25.0 | 0.0 |
173
+ | SGD/SGD-X (macro-F1) | 43.8 | 34.2 | 43.0 | 40.6 | **64.0** | 42.4 |
174
+ | ContractNLI (macro-F1) | 81.4 | **84.3** | 71.7 | 76.0 | 57.8 | 29.0 |
175
+ | ANLI (macro-F1) | 69.8 | 59.1 | **74.8** | 66.4 | 56.3 | 48.7 |
176
+ | BPoMP (accuracy) | **96.9** | 95.4 | 90.6 | 86.9 | 67.0 | 51.6 |
177
+ | Humicroedit (accuracy) | 66.7 | **75.1** | 61.9 | 63.0 | 55.8 | 47.2 |
178
+ | POP909-CL (accuracy) | 15.8 | 1.6 | **18.1** | 2.5 | 10.8 | 5.1 |
179
+ | cfcolor (accuracy) | **66.0** | 65.8 | 64.7 | 58.2 | 56.3 | 52.3 |
180
+ | MMLU (accuracy) | 90.3 | **91.8** | 91.7 | 79.3 | 75.3 | 30.7 |
181
+ | GPQA Diamond (accuracy) | 48.0 | 51.0 | **78.3** | 44.9 | 38.8 | 27.6 |
182
+ | ARC-Easy (accuracy) | 99.0 | **99.5** | 99.3 | 98.2 | 97.7 | 47.0 |
183
+ | ARC-Challenge (accuracy) | 97.7 | **98.3** | 97.8 | 94.5 | 93.7 | 28.6 |
184
+ | WinoGrande (accuracy) | 93.5 | **97.5** | 92.0 | 73.6 | 73.2 | 50.5 |
185
+ | HellaSwag (accuracy) | 98.2 | **98.6** | 94.5 | 83.3 | 81.9 | 33.1 |
186
+ | GSM8K (accuracy) | **80.8** | 67.3 | 79.9 | 50.3 | 48.7 | 21.6 |
187
+ | ChessBench (accuracy) | **24.7** | 23.0 | 17.2 | 14.2 | 11.2 | 7.7 |
188
+ | MuSR (accuracy) | 83.5 | **86.0** | 66.1 | 61.2 | 57.9 | 43.2 |
189
+ | SATA-Bench (case exact accuracy) | 33.8 | **36.7** | 26.4 | 27.5 | 26.7 | 0.3 |
190
+ | BRIGHT (nDCG@10) | 45.9 | 39.3 | **47.5** | 42.9 | 38.5 | 19.9 |
191
+ | Amazon ESCI (macro-F1) | **57.5** | 57.4 | 55.2 | 53.4 | 49.2 | 24.4 |
192
+ | ACOS (per-review F1) | **33.3** | 25.9 | 29.5 | 24.5 | 18.3 | 3.5 |
193
+ | FinEntity (macro-F1) | 96.2 | **97.1** | 87.0 | 89.0 | 88.4 | 61.0 |
194
+ | VAST (macro-F1) | 59.5 | 49.6 | **64.6** | 55.7 | 55.4 | 40.5 |
195
+ | NLI4CT (macro-F1) | 82.9 | 78.6 | **84.1** | 78.4 | 74.9 | 47.7 |
196
+ | CRUXEval (accuracy) | **86.7** | 86.1 | 73.0 | 64.7 | 51.2 | 40.2 |
197
+ | CLadder (accuracy) | 94.0 | **97.7** | 72.6 | 67.8 | 62.0 | 52.9 |
198
+ | ForecastBench (Brier, lower is better) | 13.9 | **10.6** | 17.4 | 29.6 | 17.6 | 41.1 |
199
+ | Habermas Machine (accuracy) | 68.7 | **71.8** | 45.9 | 45.0 | 39.4 | 33.4 |
200
+ | PhishNChips (accuracy) | 79.6 | 75.0 | 62.5 | **85.4** | 50.7 | 50.1 |
201
+ | MMLU-Pro (accuracy) | 65.9 | 65.3 | **82.7** | 56.9 | 51.1 | 13.6 |
202
+ | BBH (accuracy) | 73.7 | 68.9 | **92.9** | 70.7 | 65.2 | 34.1 |
203
+ | RAGTruth (hallucination F1) | **79.4** | 35.6 | 76.5 | 70.4 | 46.2 | 48.8 |
204
+ | HoVer (accuracy) | 65.2 | 61.2 | **72.9** | 70.9 | 58.8 | 55.8 |
205
+ | When2Call MCQ (accuracy) | 72.4 | 65.6 | **81.0** | 75.4 | 49.6 | 11.9 |
206
+ | New Yorker (accuracy) | 69.5 | 66.1 | **70.1** | 63.6 | 58.1 | 27.1 |
207
+ | Median latency (ms) | 209.3 | 38.8 | 524.1 | 84.4 | 51.4 | **5.8** |
208
+ | p95 latency (ms) | 238.6 | **122.4** | 536.0 | 211.2 | 187.9 | 222.5 |
209
+
210
+ ### Workflow evals
211
+
212
+ Decision accuracy on four end-to-end business workflows from [Typesafe Evals](https://evals.typesafe.ai/), scored against consensus reference labels. All models are scored on the same dataset revision and case cohort.
213
+
214
+ | Workflow | Metric | Clef | Clef-flash | Jev |
215
+ |---|---|---:|---:|---:|
216
+ | Invoice processing | Exact actions | **64.7** | 57.1 | 61.8 |
217
+ | Invoice processing | Primary action | **86.2** | 73.3 | 83.1 |
218
+ | Customer service | Exact actions | 76.3 | **77.0** | 76.0 |
219
+ | Security incidents | Exact actions | **62.9** | 61.7 | 61.7 |
220
+ | Agent trace observability | Primary action | 68.5 | 69.8 | **71.6** |
221
+
222
+ ## License
223
+
224
+ Released under the Apache-2.0 license, following the base model
225
+ [Qwen/Qwen3.8-27B](https://huggingface.co/Qwen/Qwen3.8-27B).
chat_template.jinja ADDED
@@ -0,0 +1,170 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- set reasoning_instructions = '' %}
46
+ {%- if enable_thinking is undefined or enable_thinking is true %}
47
+ {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}
48
+ {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}
49
+ {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}
50
+ {%- endif %}
51
+ {%- if resolved_reasoning_effort == 'xhigh' %}
52
+ {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}
53
+ {%- elif resolved_reasoning_effort == 'low' %}
54
+ {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}
55
+ {%- endif %}
56
+ {%- endif %}
57
+ {%- if tools and tools is iterable and tools is not mapping %}
58
+ {{- '<|im_start|>system\n' }}
59
+ {%- if reasoning_instructions %}
60
+ {{- reasoning_instructions + '\n\n' }}
61
+ {%- endif %}
62
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
63
+ {%- for tool in tools %}
64
+ {{- "\n" }}
65
+ {{- tool | tojson }}
66
+ {%- endfor %}
67
+ {{- "\n</tools>" }}
68
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
69
+ {%- if messages[0].role == 'system' %}
70
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
71
+ {%- if content %}
72
+ {{- '\n\n' + content }}
73
+ {%- endif %}
74
+ {%- endif %}
75
+ {{- '<|im_end|>\n' }}
76
+ {%- else %}
77
+ {%- if messages[0].role == 'system' %}
78
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
79
+ {%- if content %}
80
+ {{- '<|im_start|>system\n' + (reasoning_instructions + '\n\n' if reasoning_instructions else '') + content + '<|im_end|>\n' }}
81
+ {%- elif reasoning_instructions %}
82
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
83
+ {%- endif %}
84
+ {%- elif reasoning_instructions %}
85
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
86
+ {%- endif %}
87
+ {%- endif %}
88
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
89
+ {%- for message in messages[::-1] %}
90
+ {%- set index = (messages|length - 1) - loop.index0 %}
91
+ {%- if ns.multi_step_tool and message.role == "user" %}
92
+ {%- set content = render_content(message.content, false)|trim %}
93
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
94
+ {%- set ns.multi_step_tool = false %}
95
+ {%- set ns.last_query_index = index %}
96
+ {%- endif %}
97
+ {%- endif %}
98
+ {%- endfor %}
99
+ {%- if ns.multi_step_tool %}
100
+ {{- raise_exception('No user query found in messages.') }}
101
+ {%- endif %}
102
+ {%- for message in messages %}
103
+ {%- set content = render_content(message.content, true)|trim %}
104
+ {%- if message.role == "system" %}
105
+ {%- if not loop.first %}
106
+ {{- raise_exception('System message must be at the beginning.') }}
107
+ {%- endif %}
108
+ {%- elif message.role == "user" %}
109
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
110
+ {%- elif message.role == "assistant" %}
111
+ {%- set reasoning_content = '' %}
112
+ {%- if message.reasoning_content is string %}
113
+ {%- set reasoning_content = message.reasoning_content %}
114
+ {%- endif %}
115
+ {%- set reasoning_content = reasoning_content|trim %}
116
+ {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}
117
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
118
+ {%- else %}
119
+ {{- '<|im_start|>' + message.role + '\n' + content }}
120
+ {%- endif %}
121
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
122
+ {%- for tool_call in message.tool_calls %}
123
+ {%- if tool_call.function is defined %}
124
+ {%- set tool_call = tool_call.function %}
125
+ {%- endif %}
126
+ {%- if loop.first %}
127
+ {%- if content|trim %}
128
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
129
+ {%- else %}
130
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
131
+ {%- endif %}
132
+ {%- else %}
133
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
134
+ {%- endif %}
135
+ {%- if tool_call.arguments is defined and tool_call.arguments != '' %}
136
+ {%- for args_name, args_value in tool_call.arguments|items %}
137
+ {{- '<parameter=' + args_name + '>\n' }}
138
+ {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}
139
+ {{- args_value }}
140
+ {{- '\n</parameter>\n' }}
141
+ {%- endfor %}
142
+ {%- endif %}
143
+ {{- '</function>\n</tool_call>' }}
144
+ {%- endfor %}
145
+ {%- endif %}
146
+ {{- '<|im_end|>\n' }}
147
+ {%- elif message.role == "tool" %}
148
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
149
+ {{- '<|im_start|>user' }}
150
+ {%- endif %}
151
+ {{- '\n<tool_response>\n' }}
152
+ {{- content }}
153
+ {{- '\n</tool_response>' }}
154
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
155
+ {{- '<|im_end|>\n' }}
156
+ {%- elif loop.last %}
157
+ {{- '<|im_end|>\n' }}
158
+ {%- endif %}
159
+ {%- else %}
160
+ {{- raise_exception('Unexpected message role.') }}
161
+ {%- endif %}
162
+ {%- endfor %}
163
+ {%- if add_generation_prompt %}
164
+ {{- '<|im_start|>assistant\n' }}
165
+ {%- if enable_thinking is defined and enable_thinking is false %}
166
+ {{- '<think>\n\n</think>\n\n' }}
167
+ {%- else %}
168
+ {{- '<think>\n' }}
169
+ {%- endif %}
170
+ {%- endif %}
config.json ADDED
@@ -0,0 +1,142 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForConditionalGeneration"
4
+ ],
5
+ "dtype": "bfloat16",
6
+ "image_token_id": 248056,
7
+ "language_model_only": false,
8
+ "model_type": "qwen3_5",
9
+ "text_config": {
10
+ "attention_bias": false,
11
+ "attention_dropout": 0.0,
12
+ "attn_output_gate": true,
13
+ "bos_token_id": 248044,
14
+ "dtype": "bfloat16",
15
+ "eos_token_id": 248044,
16
+ "full_attention_interval": 4,
17
+ "head_dim": 256,
18
+ "hidden_act": "silu",
19
+ "hidden_size": 5120,
20
+ "initializer_range": 0.02,
21
+ "intermediate_size": 17408,
22
+ "layer_types": [
23
+ "linear_attention",
24
+ "linear_attention",
25
+ "linear_attention",
26
+ "full_attention",
27
+ "linear_attention",
28
+ "linear_attention",
29
+ "linear_attention",
30
+ "full_attention",
31
+ "linear_attention",
32
+ "linear_attention",
33
+ "linear_attention",
34
+ "full_attention",
35
+ "linear_attention",
36
+ "linear_attention",
37
+ "linear_attention",
38
+ "full_attention",
39
+ "linear_attention",
40
+ "linear_attention",
41
+ "linear_attention",
42
+ "full_attention",
43
+ "linear_attention",
44
+ "linear_attention",
45
+ "linear_attention",
46
+ "full_attention",
47
+ "linear_attention",
48
+ "linear_attention",
49
+ "linear_attention",
50
+ "full_attention",
51
+ "linear_attention",
52
+ "linear_attention",
53
+ "linear_attention",
54
+ "full_attention",
55
+ "linear_attention",
56
+ "linear_attention",
57
+ "linear_attention",
58
+ "full_attention",
59
+ "linear_attention",
60
+ "linear_attention",
61
+ "linear_attention",
62
+ "full_attention",
63
+ "linear_attention",
64
+ "linear_attention",
65
+ "linear_attention",
66
+ "full_attention",
67
+ "linear_attention",
68
+ "linear_attention",
69
+ "linear_attention",
70
+ "full_attention",
71
+ "linear_attention",
72
+ "linear_attention",
73
+ "linear_attention",
74
+ "full_attention",
75
+ "linear_attention",
76
+ "linear_attention",
77
+ "linear_attention",
78
+ "full_attention",
79
+ "linear_attention",
80
+ "linear_attention",
81
+ "linear_attention",
82
+ "full_attention",
83
+ "linear_attention",
84
+ "linear_attention",
85
+ "linear_attention",
86
+ "full_attention"
87
+ ],
88
+ "linear_conv_kernel_dim": 4,
89
+ "linear_key_head_dim": 128,
90
+ "linear_num_key_heads": 16,
91
+ "linear_num_value_heads": 48,
92
+ "linear_value_head_dim": 128,
93
+ "mamba_ssm_dtype": "float32",
94
+ "max_position_embeddings": 262144,
95
+ "model_type": "qwen3_5_text",
96
+ "mtp_num_hidden_layers": 0,
97
+ "mtp_use_dedicated_embeddings": false,
98
+ "num_attention_heads": 24,
99
+ "num_hidden_layers": 64,
100
+ "num_key_value_heads": 4,
101
+ "output_gate_type": "swish",
102
+ "pad_token_id": null,
103
+ "partial_rotary_factor": 0.25,
104
+ "rms_norm_eps": 1e-06,
105
+ "rope_parameters": {
106
+ "mrope_interleaved": true,
107
+ "mrope_section": [
108
+ 11,
109
+ 11,
110
+ 10
111
+ ],
112
+ "partial_rotary_factor": 0.25,
113
+ "rope_theta": 10000000,
114
+ "rope_type": "default"
115
+ },
116
+ "tie_word_embeddings": false,
117
+ "use_cache": true,
118
+ "vocab_size": 248320
119
+ },
120
+ "tie_word_embeddings": false,
121
+ "transformers_version": "5.10.2",
122
+ "video_token_id": 248057,
123
+ "vision_config": {
124
+ "deepstack_visual_indexes": [],
125
+ "depth": 27,
126
+ "dtype": "bfloat16",
127
+ "hidden_act": "gelu_pytorch_tanh",
128
+ "hidden_size": 1152,
129
+ "in_channels": 3,
130
+ "initializer_range": 0.02,
131
+ "intermediate_size": 4304,
132
+ "model_type": "qwen3_5_vision",
133
+ "num_heads": 16,
134
+ "num_position_embeddings": 2304,
135
+ "out_hidden_size": 5120,
136
+ "patch_size": 16,
137
+ "spatial_merge_size": 2,
138
+ "temporal_patch_size": 2
139
+ },
140
+ "vision_end_token_id": 248054,
141
+ "vision_start_token_id": 248053
142
+ }
generation_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 248044,
3
+ "do_sample": true,
4
+ "eos_token_id": [
5
+ 248046,
6
+ 248044
7
+ ],
8
+ "pad_token_id": 248044,
9
+ "temperature": 1.0,
10
+ "top_k": 20,
11
+ "top_p": 0.95,
12
+ "transformers_version": "5.10.2"
13
+ }
joint_head.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a010ac04f078e699988e4049cbea5e62c962393f59fec366640b64e8d69a4953
3
+ size 256125024
joint_head_config.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "hidden_size": 5120,
3
+ "width": 1024,
4
+ "routing_layers": 2,
5
+ "layers": 4,
6
+ "heads": 16,
7
+ "feedforward": 4096
8
+ }
joint_schema_model.py ADDED
@@ -0,0 +1,576 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Clef: a multimodal Qwen backbone with a joint schema head for typed decisions.
2
+
3
+ A record provides a ``state`` (any JSON value), optional ``images`` and ``videos``,
4
+ and ``questions``. Each question has a ``type`` (``noul``, ``choice``, or ``score``),
5
+ ``instructions``, and, for ``choice`` and ``score``, ``criteria`` describing the
6
+ allowed options. The model returns one logit per allowed option for every question.
7
+ ``systemone`` answers a Jev/SystemOne ``/v1/systemone`` request body with the same
8
+ response body.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import math
15
+ from dataclasses import dataclass
16
+ from dataclasses import field as dataclass_field
17
+ from pathlib import Path
18
+ from typing import Any
19
+
20
+ import torch
21
+ import torch.nn.functional as functional
22
+
23
+
24
+ SYSTEM_PROMPT = (
25
+ "Read the complete state and schema. Decide every field jointly. Each answer "
26
+ "must be exactly one of that field's allowed options."
27
+ )
28
+ IMAGE_PLACEHOLDER = "<|vision_start|><|image_pad|><|vision_end|>"
29
+ VIDEO_PLACEHOLDER = "<|vision_start|><|video_pad|><|vision_end|>"
30
+ MEDIA_BATCH_KEYS = ("pixel_values", "image_grid_thw", "pixel_values_videos", "video_grid_thw")
31
+ MEDIA_TOKEN_KEYS = ("mm_token_type_ids",)
32
+ QUESTION_TYPES = {"noul": 0, "choice": 1, "score": 2}
33
+
34
+
35
+ def render(value: Any) -> str:
36
+ if isinstance(value, str):
37
+ return value
38
+ return json.dumps(
39
+ value,
40
+ ensure_ascii=False,
41
+ separators=(",", ":"),
42
+ sort_keys=True,
43
+ )
44
+
45
+
46
+ def question_options(question: dict[str, Any]) -> list[tuple[str, Any]]:
47
+ question_type = str(question["type"])
48
+ if question_type == "noul":
49
+ criteria = {
50
+ "true": "The proposition is true or the answer is yes.",
51
+ "false": "The proposition is false or the answer is no.",
52
+ }
53
+ criteria.update(question.get("criteria") or {})
54
+ return [(key, criteria[key]) for key in ("true", "false")]
55
+ if question_type == "choice":
56
+ return sorted((str(key), value) for key, value in question["criteria"].items())
57
+ return [(str(index), value) for index, value in enumerate(question["criteria"])]
58
+
59
+
60
+ @dataclass(frozen=True)
61
+ class EncodedQuestion:
62
+ question_id: str
63
+ question_type: int
64
+ question_span: tuple[int, int]
65
+ option_spans: tuple[tuple[int, int], ...]
66
+ option_ids: tuple[str, ...]
67
+
68
+
69
+ @dataclass(frozen=True)
70
+ class EncodedRecord:
71
+ input_ids: tuple[int, ...]
72
+ questions: tuple[EncodedQuestion, ...]
73
+ record_id: str
74
+ media: dict[str, Any] | None = dataclass_field(default=None, compare=False, repr=False)
75
+
76
+
77
+ def _tokens(tokenizer: Any, text: str) -> list[int]:
78
+ return tokenizer(text, add_special_tokens=False).input_ids
79
+
80
+
81
+ def _encode_media(processor: Any, record: dict[str, Any]) -> tuple[list[int], dict[str, Any] | None]:
82
+ images = list(record.get("images") or [])
83
+ videos = list(record.get("videos") or [])
84
+ if not images and not videos:
85
+ return [], None
86
+ if processor is None:
87
+ raise ValueError("records with images or videos require a processor")
88
+ text = IMAGE_PLACEHOLDER * len(images) + VIDEO_PLACEHOLDER * len(videos) + "\n"
89
+ encoded = processor(
90
+ text=[text],
91
+ images=images or None,
92
+ videos=videos or None,
93
+ return_tensors="pt",
94
+ **(record.get("media_kwargs") or {}),
95
+ )
96
+ media = {key: encoded[key] for key in MEDIA_BATCH_KEYS if key in encoded}
97
+ for key in MEDIA_TOKEN_KEYS:
98
+ if key in encoded:
99
+ media[key] = encoded[key][0].tolist()
100
+ return encoded["input_ids"][0].tolist(), media
101
+
102
+
103
+ def encode_record(
104
+ tokenizer: Any,
105
+ record: dict[str, Any],
106
+ max_length: int = 16384,
107
+ max_state_tokens: int | None = None,
108
+ processor: Any | None = None,
109
+ ) -> EncodedRecord:
110
+ schema_ids = _tokens(tokenizer, "\n\nSCHEMA FIELDS:\n")
111
+ questions: list[EncodedQuestion] = []
112
+ for question_index, (question_id, question) in enumerate(record["questions"].items()):
113
+ schema_ids.extend(
114
+ _tokens(
115
+ tokenizer,
116
+ f"\nFIELD {question_index + 1}\nID: {question_id}\nTYPE: {question['type']}\nINSTRUCTION: ",
117
+ )
118
+ )
119
+ question_start = len(schema_ids)
120
+ instructions = question.get("instructions")
121
+ if instructions is None or instructions == "":
122
+ instructions = str(question_id)
123
+ schema_ids.extend(_tokens(tokenizer, render(instructions)))
124
+ question_end = len(schema_ids)
125
+ schema_ids.extend(_tokens(tokenizer, "\nALLOWED OPTIONS:\n"))
126
+
127
+ option_spans: list[tuple[int, int]] = []
128
+ option_ids: list[str] = []
129
+ for option_index, (option_id, description) in enumerate(question_options(question)):
130
+ schema_ids.extend(_tokens(tokenizer, f"OPTION {option_index + 1}: "))
131
+ option_start = len(schema_ids)
132
+ semantics = {"option_id": option_id}
133
+ if description is not None:
134
+ semantics["description"] = description
135
+ schema_ids.extend(_tokens(tokenizer, render(semantics)))
136
+ option_spans.append((option_start, len(schema_ids)))
137
+ option_ids.append(option_id)
138
+ schema_ids.extend(_tokens(tokenizer, "\n"))
139
+ schema_ids.extend(_tokens(tokenizer, "END FIELD\n"))
140
+ questions.append(
141
+ EncodedQuestion(
142
+ question_id=str(question_id),
143
+ question_type=QUESTION_TYPES[str(question["type"])],
144
+ question_span=(question_start, question_end),
145
+ option_spans=tuple(option_spans),
146
+ option_ids=tuple(option_ids),
147
+ )
148
+ )
149
+
150
+ prefix_ids = _tokens(
151
+ tokenizer,
152
+ f"<|im_start|>system\n{SYSTEM_PROMPT}<|im_end|>\n<|im_start|>user\nSTATE:\n",
153
+ )
154
+ suffix_ids = _tokens(
155
+ tokenizer,
156
+ "\n<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\nJOINT SCHEMA DECISIONS:",
157
+ )
158
+ media_ids, media = _encode_media(processor, record)
159
+ if media is not None:
160
+ media["token_offset"] = len(prefix_ids)
161
+ prefix_ids = prefix_ids + media_ids
162
+ state_ids = _tokens(tokenizer, render(record["state"]))
163
+ if max_state_tokens is not None:
164
+ state_ids = state_ids[:max_state_tokens]
165
+ fixed_length = len(prefix_ids) + len(schema_ids) + len(suffix_ids)
166
+ if fixed_length > max_length:
167
+ raise ValueError(
168
+ f"schema requires {fixed_length} tokens before state; maximum is {max_length}"
169
+ )
170
+ state_ids = state_ids[: max_length - fixed_length]
171
+ schema_offset = len(prefix_ids) + len(state_ids)
172
+ shifted_questions = tuple(
173
+ EncodedQuestion(
174
+ question_id=question.question_id,
175
+ question_type=question.question_type,
176
+ question_span=(
177
+ question.question_span[0] + schema_offset,
178
+ question.question_span[1] + schema_offset,
179
+ ),
180
+ option_spans=tuple(
181
+ (start + schema_offset, end + schema_offset)
182
+ for start, end in question.option_spans
183
+ ),
184
+ option_ids=question.option_ids,
185
+ )
186
+ for question in questions
187
+ )
188
+ input_ids = tuple(prefix_ids + state_ids + schema_ids + suffix_ids)
189
+ if not input_ids or not shifted_questions:
190
+ raise ValueError("record produced no model input or questions")
191
+ return EncodedRecord(
192
+ input_ids=input_ids,
193
+ questions=shifted_questions,
194
+ record_id=str(record.get("id", "unknown")),
195
+ media=media,
196
+ )
197
+
198
+
199
+ def collate_records(
200
+ records: list[EncodedRecord],
201
+ pad_token_id: int,
202
+ device: torch.device,
203
+ ) -> dict[str, Any]:
204
+ maximum_length = max(len(record.input_ids) for record in records)
205
+ input_ids = torch.full(
206
+ (len(records), maximum_length),
207
+ pad_token_id,
208
+ dtype=torch.long,
209
+ device=device,
210
+ )
211
+ attention_mask = torch.zeros(
212
+ (len(records), maximum_length),
213
+ dtype=torch.long,
214
+ device=device,
215
+ )
216
+ for index, record in enumerate(records):
217
+ length = len(record.input_ids)
218
+ input_ids[index, :length] = torch.tensor(record.input_ids, device=device)
219
+ attention_mask[index, :length] = 1
220
+ media: dict[str, torch.Tensor] = {}
221
+ for key in MEDIA_BATCH_KEYS:
222
+ values = [record.media[key] for record in records if record.media and key in record.media]
223
+ if values:
224
+ media[key] = torch.cat(values, dim=0).to(device)
225
+ for key in MEDIA_TOKEN_KEYS:
226
+ if any(record.media and key in record.media for record in records):
227
+ token_values = torch.zeros((len(records), maximum_length), dtype=torch.long, device=device)
228
+ for index, record in enumerate(records):
229
+ if record.media and key in record.media:
230
+ offset = record.media["token_offset"]
231
+ values = torch.tensor(record.media[key], dtype=torch.long, device=device)
232
+ token_values[index, offset : offset + len(values)] = values
233
+ media[key] = token_values
234
+ return {
235
+ "input_ids": input_ids,
236
+ "attention_mask": attention_mask,
237
+ "records": records,
238
+ "media": media,
239
+ }
240
+
241
+
242
+ class EvidenceRoutingLayer(torch.nn.Module):
243
+ def __init__(
244
+ self,
245
+ width: int,
246
+ heads: int,
247
+ feedforward: int,
248
+ dropout: float = 0.0,
249
+ ) -> None:
250
+ super().__init__()
251
+ self.query_norm = torch.nn.LayerNorm(width)
252
+ self.memory_norm = torch.nn.LayerNorm(width)
253
+ self.attention = torch.nn.MultiheadAttention(
254
+ width,
255
+ heads,
256
+ dropout=dropout,
257
+ batch_first=True,
258
+ )
259
+ self.attention_dropout = torch.nn.Dropout(dropout)
260
+ self.feedforward_norm = torch.nn.LayerNorm(width)
261
+ self.feedforward = torch.nn.Sequential(
262
+ torch.nn.Linear(width, feedforward),
263
+ torch.nn.GELU(),
264
+ torch.nn.Dropout(dropout),
265
+ torch.nn.Linear(feedforward, width),
266
+ torch.nn.Dropout(dropout),
267
+ )
268
+
269
+ def forward(self, queries: torch.Tensor, memory: torch.Tensor) -> torch.Tensor:
270
+ normalized_queries = self.query_norm(queries)
271
+ routed, _ = self.attention(
272
+ normalized_queries,
273
+ self.memory_norm(memory),
274
+ self.memory_norm(memory),
275
+ need_weights=False,
276
+ )
277
+ queries = queries + self.attention_dropout(routed)
278
+ return queries + self.feedforward(self.feedforward_norm(queries))
279
+
280
+
281
+ class JointSchemaHead(torch.nn.Module):
282
+ def __init__(
283
+ self,
284
+ hidden_size: int,
285
+ width: int,
286
+ routing_layers: int,
287
+ layers: int,
288
+ heads: int,
289
+ feedforward: int,
290
+ dropout: float = 0.0,
291
+ ) -> None:
292
+ super().__init__()
293
+ self.hidden_norm = torch.nn.LayerNorm(hidden_size)
294
+ self.memory_projection = torch.nn.Linear(hidden_size, width, bias=False)
295
+ self.question_projection = torch.nn.Linear(hidden_size, width, bias=False)
296
+ self.option_question_projection = torch.nn.Linear(hidden_size, width, bias=False)
297
+ self.global_projection = torch.nn.Linear(hidden_size, width, bias=False)
298
+ self.option_context_projection = torch.nn.Linear(hidden_size, width, bias=False)
299
+ self.option_lexical_projection = torch.nn.Linear(hidden_size, width, bias=False)
300
+ self.type_embedding = torch.nn.Embedding(3, width)
301
+ self.evidence_layers = torch.nn.ModuleList(
302
+ [
303
+ EvidenceRoutingLayer(
304
+ width=width,
305
+ heads=heads,
306
+ feedforward=feedforward,
307
+ dropout=dropout,
308
+ )
309
+ for _ in range(routing_layers)
310
+ ]
311
+ )
312
+ self.option_summary_norm = torch.nn.LayerNorm(width)
313
+ self.layers = torch.nn.ModuleList(
314
+ [
315
+ torch.nn.TransformerDecoderLayer(
316
+ d_model=width,
317
+ nhead=heads,
318
+ dim_feedforward=feedforward,
319
+ dropout=dropout,
320
+ activation="gelu",
321
+ batch_first=True,
322
+ norm_first=True,
323
+ )
324
+ for _ in range(layers)
325
+ ]
326
+ )
327
+ self.field_norm = torch.nn.LayerNorm(width)
328
+ self.option_norm = torch.nn.LayerNorm(width)
329
+ self.residual_scorer = torch.nn.Sequential(
330
+ torch.nn.Linear(width * 4, width),
331
+ torch.nn.GELU(),
332
+ torch.nn.Dropout(dropout),
333
+ torch.nn.Linear(width, 1),
334
+ )
335
+ self.prior_logit_scale = torch.nn.Parameter(torch.zeros(()))
336
+ self.joint_logit_scale = torch.nn.Parameter(torch.zeros(()))
337
+ self.residual_gate = torch.nn.Parameter(torch.zeros(()))
338
+
339
+ @staticmethod
340
+ def _mean_span(values: torch.Tensor, span: tuple[int, int]) -> torch.Tensor:
341
+ start, end = span
342
+ return values[start:end].mean(dim=0)
343
+
344
+ def forward(
345
+ self,
346
+ hidden_states: torch.Tensor,
347
+ input_ids: torch.Tensor,
348
+ attention_mask: torch.Tensor,
349
+ records: list[EncodedRecord],
350
+ output_embedding_weight: torch.Tensor,
351
+ ) -> list[list[torch.Tensor]]:
352
+ results: list[list[torch.Tensor]] = []
353
+ normalized_hidden = self.hidden_norm(hidden_states)
354
+ for batch_index, record in enumerate(records):
355
+ sequence_length = int(attention_mask[batch_index].sum().item())
356
+ sequence_hidden = normalized_hidden[batch_index, :sequence_length]
357
+ memory = self.memory_projection(sequence_hidden).unsqueeze(0)
358
+ global_vector = sequence_hidden[-1]
359
+ question_vectors = torch.stack(
360
+ [
361
+ self._mean_span(sequence_hidden, question.question_span)
362
+ for question in record.questions
363
+ ]
364
+ )
365
+ type_ids = torch.tensor(
366
+ [question.question_type for question in record.questions],
367
+ device=hidden_states.device,
368
+ )
369
+ option_contexts: list[torch.Tensor] = []
370
+ lexical_options: list[torch.Tensor] = []
371
+ option_counts = []
372
+ for question in record.questions:
373
+ context_vectors = torch.stack(
374
+ [
375
+ self._mean_span(sequence_hidden, span)
376
+ for span in question.option_spans
377
+ ]
378
+ )
379
+ lexical_vectors = []
380
+ for start, end in question.option_spans:
381
+ token_ids = input_ids[batch_index, start:end]
382
+ lexical_vectors.append(output_embedding_weight[token_ids].mean(dim=0))
383
+ lexical = torch.stack(lexical_vectors)
384
+ option_contexts.append(context_vectors)
385
+ lexical_options.append(lexical)
386
+ option_counts.append(len(question.option_spans))
387
+
388
+ option_queries = []
389
+ for question_index, (context_vectors, lexical) in enumerate(
390
+ zip(option_contexts, lexical_options)
391
+ ):
392
+ option_queries.append(
393
+ self.option_context_projection(context_vectors)
394
+ + self.option_lexical_projection(lexical)
395
+ + self.option_question_projection(
396
+ question_vectors[question_index]
397
+ ).unsqueeze(0)
398
+ )
399
+ routed_options = torch.cat(option_queries, dim=0).unsqueeze(0)
400
+ for layer in self.evidence_layers:
401
+ routed_options = layer(routed_options, memory)
402
+ routed_options = routed_options[0]
403
+ split_options = list(torch.split(routed_options, option_counts, dim=0))
404
+
405
+ base_fields = self.question_projection(question_vectors)
406
+ option_summaries = []
407
+ for field, options in zip(base_fields, split_options):
408
+ routing_weights = torch.softmax(
409
+ torch.matmul(options, field) / math.sqrt(options.shape[-1]),
410
+ dim=0,
411
+ )
412
+ option_summaries.append(
413
+ torch.sum(routing_weights.unsqueeze(-1) * options, dim=0)
414
+ )
415
+ fields = (
416
+ base_fields
417
+ + self.option_summary_norm(torch.stack(option_summaries))
418
+ + self.global_projection(global_vector).unsqueeze(0)
419
+ + self.type_embedding(type_ids)
420
+ )
421
+ fields = fields.unsqueeze(0)
422
+ for layer in self.layers:
423
+ fields = layer(fields, memory)
424
+ fields = self.field_norm(fields[0])
425
+
426
+ record_logits: list[torch.Tensor] = []
427
+ for field, question, lexical, routed in zip(
428
+ fields,
429
+ record.questions,
430
+ lexical_options,
431
+ split_options,
432
+ ):
433
+ anchor = functional.normalize(
434
+ question_vectors[len(record_logits)] + global_vector,
435
+ dim=-1,
436
+ )
437
+ lexical_anchor = functional.normalize(lexical, dim=-1)
438
+ prior_scale = self.prior_logit_scale.clamp(max=math.log(100.0)).exp()
439
+ prior = prior_scale * torch.matmul(lexical_anchor, anchor)
440
+ options = self.option_norm(routed)
441
+ repeated_field = field.unsqueeze(0).expand_as(options)
442
+ cosine = functional.cosine_similarity(repeated_field, options, dim=-1)
443
+ features = torch.cat(
444
+ [
445
+ repeated_field,
446
+ options,
447
+ repeated_field * options,
448
+ torch.abs(repeated_field - options),
449
+ ],
450
+ dim=-1,
451
+ )
452
+ residual = self.residual_scorer(features).squeeze(-1)
453
+ joint_scale = self.joint_logit_scale.clamp(max=math.log(100.0)).exp()
454
+ joint = joint_scale * cosine + residual
455
+ record_logits.append(
456
+ prior + torch.sigmoid(self.residual_gate) * joint
457
+ )
458
+ results.append(record_logits)
459
+ return results
460
+
461
+
462
+ class ClefModel(torch.nn.Module):
463
+ def __init__(self, language_model: Any, head: JointSchemaHead) -> None:
464
+ super().__init__()
465
+ self.language_model = language_model
466
+ self.head = head
467
+
468
+ def forward(self, batch: dict[str, Any]) -> list[list[torch.Tensor]]:
469
+ base_model = (
470
+ self.language_model.get_base_model()
471
+ if hasattr(self.language_model, "get_base_model")
472
+ else self.language_model
473
+ )
474
+ media = batch.get("media") or {}
475
+ text_model = base_model.model
476
+ if not media and hasattr(text_model, "language_model"):
477
+ text_model = text_model.language_model
478
+ outputs = text_model(
479
+ input_ids=batch["input_ids"],
480
+ attention_mask=batch["attention_mask"],
481
+ use_cache=False,
482
+ return_dict=True,
483
+ **media,
484
+ )
485
+ return self.head(
486
+ outputs.last_hidden_state,
487
+ batch["input_ids"],
488
+ batch["attention_mask"],
489
+ batch["records"],
490
+ base_model.get_output_embeddings().weight,
491
+ )
492
+
493
+
494
+ def load_release_model(
495
+ model_path: str | Path,
496
+ device: str | torch.device = "cuda",
497
+ dtype: torch.dtype = torch.bfloat16,
498
+ **from_pretrained_kwargs: Any,
499
+ ) -> tuple[ClefModel, Any]:
500
+ """Load a Clef release (merged backbone, joint schema head, and processor)."""
501
+ from huggingface_hub import snapshot_download
502
+ from safetensors.torch import load_file
503
+ from transformers import AutoProcessor, Qwen3_5ForConditionalGeneration
504
+
505
+ path = Path(model_path)
506
+ if not path.is_dir():
507
+ path = Path(snapshot_download(str(model_path)))
508
+ backbone = Qwen3_5ForConditionalGeneration.from_pretrained(
509
+ path,
510
+ dtype=dtype,
511
+ device_map={"": str(device)},
512
+ **from_pretrained_kwargs,
513
+ )
514
+ backbone.config.use_cache = False
515
+ head_config = json.loads((path / "joint_head_config.json").read_text())
516
+ head = JointSchemaHead(**head_config)
517
+ head.load_state_dict(load_file(path / "joint_head.safetensors"), strict=True)
518
+ head = head.to(device=device, dtype=dtype)
519
+ processor = AutoProcessor.from_pretrained(path)
520
+ return ClefModel(backbone, head).eval(), processor
521
+
522
+
523
+ def systemone_answer(question: dict[str, Any], probabilities: dict[str, float]) -> dict[str, Any]:
524
+ """Convert per-option probabilities for one question into a SystemOne answer."""
525
+ if question["type"] == "noul":
526
+ return {"type": "noul", "noul": round(probabilities["true"], 4)}
527
+ if question["type"] == "choice":
528
+ options = [str(option) for option in question["criteria"]]
529
+ choice = max(options, key=probabilities.__getitem__)
530
+ return {
531
+ "type": "choice",
532
+ "choice": choice,
533
+ "confidence": round(probabilities[choice], 4),
534
+ "probabilities": {option: round(probabilities[option], 4) for option in options},
535
+ }
536
+ levels = [str(index) for index in range(len(question["criteria"]))]
537
+ return {
538
+ "type": "score",
539
+ "score": round(sum(index * probabilities[level] for index, level in enumerate(levels)), 4),
540
+ "confidence": round(max(probabilities[level] for level in levels), 4),
541
+ "legend": dict(zip(levels, question["criteria"])),
542
+ "probabilities": {level: round(probabilities[level], 4) for level in levels},
543
+ }
544
+
545
+
546
+ @torch.inference_mode()
547
+ def systemone(model: ClefModel, processor: Any, request: dict[str, Any], max_length: int = 16384) -> dict[str, Any]:
548
+ """Answer a Jev/SystemOne ``/v1/systemone`` request body with a SystemOne response body.
549
+
550
+ The request has ``model``, ``state``, and ``questions``, plus optional ``images`` and ``videos``.
551
+ """
552
+ questions = request.get("questions")
553
+ if not isinstance(request.get("model"), str) or "state" not in request:
554
+ raise ValueError("model and state are required")
555
+ if not isinstance(questions, dict) or not questions:
556
+ raise ValueError("at least one question is required")
557
+ for question_id, question in questions.items():
558
+ if question.get("type") not in QUESTION_TYPES:
559
+ raise ValueError(f"{question_id}: type must be noul, choice, or score")
560
+ if question["type"] != "noul" and not question.get("criteria"):
561
+ raise ValueError(f"{question_id}: criteria must not be empty")
562
+ encoded = encode_record(processor.tokenizer, request, max_length=max_length, processor=processor)
563
+ device = next(model.parameters()).device
564
+ logits = model(collate_records([encoded], processor.tokenizer.pad_token_id, device))[0]
565
+ answers = {
566
+ question.question_id: systemone_answer(
567
+ questions[question.question_id],
568
+ dict(zip(question.option_ids, question_logits.float().softmax(-1).tolist())),
569
+ )
570
+ for question, question_logits in zip(encoded.questions, logits)
571
+ }
572
+ return {
573
+ "model": request["model"],
574
+ "answers": answers,
575
+ "usage": {"input_tokens": len(encoded.input_ids), "output_tokens": 0},
576
+ }
model-00001-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:54d83c1d36631de231876217a8e0c2483eccee8746369a482b79442bdfc5d958
3
+ size 2542796928
model-00002-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:464086af08be8e2ec14960a4dcff083ebc39974ade00d79d35497385f960ab3a
3
+ size 4842451920
model-00003-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:092212d3a02fafacd6424723eda59d60e5282d2068d68f0e37cb891f63bbb658
3
+ size 4965227944
model-00004-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d06ff197668c782145fafa74bba61bbc296fb27e39afb15fd522918ce3514dc5
3
+ size 4912819264
model-00005-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cc693b8829614a72e0c2be303fb290cf05dbb4d7ded872a817b97bb222a78427
3
+ size 4986198544
model-00006-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e7cce15da2443cb8b84aaed66a9a71f0c87dc9d043f83b5f58f4a89f64ba60ad
3
+ size 4912819320
model-00007-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:75fc7e76b57d5d17a5d85fff3e879d07dd33edc885a8ee04ad437a899bcd5307
3
+ size 4932703272
model-00008-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:189b15cb6b1af48d5f118951446e15639bfeaf76081d5f20aed1f1b4253afe1d
3
+ size 4966314576
model-00009-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8101e2664bb14684fc7051f2e1f84903dbdf5489b17cae3212ac08a0af744a60
3
+ size 4964162248
model-00010-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a8e69016a1a8dab1c9ce8dd151c0d3224412ab864cf06d3a5e2b8a5775cfdfb8
3
+ size 4933789824
model-00011-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b328d21c36ab384696e40a30ac86a95aaf6dd82da89438ba009cb97d87c1d6b9
3
+ size 4965228032
model-00012-of-00012.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7505eed910a84ea18e66953572476f8a8a6a6ef54192a6643d6e9cd21a3bb978
3
+ size 2789094896
model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
processor_config.json ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "image_processor": {
3
+ "do_convert_rgb": true,
4
+ "do_normalize": true,
5
+ "do_rescale": true,
6
+ "do_resize": true,
7
+ "image_mean": [
8
+ 0.5,
9
+ 0.5,
10
+ 0.5
11
+ ],
12
+ "image_processor_type": "Qwen2VLImageProcessor",
13
+ "image_std": [
14
+ 0.5,
15
+ 0.5,
16
+ 0.5
17
+ ],
18
+ "merge_size": 2,
19
+ "patch_size": 16,
20
+ "resample": 3,
21
+ "rescale_factor": 0.00392156862745098,
22
+ "size": {
23
+ "longest_edge": 16777216,
24
+ "shortest_edge": 65536
25
+ },
26
+ "temporal_patch_size": 2
27
+ },
28
+ "processor_class": "Qwen3VLProcessor",
29
+ "video_processor": {
30
+ "do_convert_rgb": true,
31
+ "do_normalize": true,
32
+ "do_rescale": true,
33
+ "do_resize": true,
34
+ "do_sample_frames": true,
35
+ "fps": 2,
36
+ "image_mean": [
37
+ 0.5,
38
+ 0.5,
39
+ 0.5
40
+ ],
41
+ "image_std": [
42
+ 0.5,
43
+ 0.5,
44
+ 0.5
45
+ ],
46
+ "max_frames": 768,
47
+ "merge_size": 2,
48
+ "min_frames": 4,
49
+ "patch_size": 16,
50
+ "resample": 3,
51
+ "rescale_factor": 0.00392156862745098,
52
+ "return_metadata": false,
53
+ "size": {
54
+ "longest_edge": 25165824,
55
+ "shortest_edge": 4096
56
+ },
57
+ "temporal_patch_size": 2,
58
+ "video_processor_type": "Qwen3VLVideoProcessor"
59
+ }
60
+ }
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523
3
+ size 19989325
tokenizer_config.json ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|im_end|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "model_max_length": 262144,
13
+ "model_specific_special_tokens": {
14
+ "audio_bos_token": "<|audio_start|>",
15
+ "audio_eos_token": "<|audio_end|>",
16
+ "audio_token": "<|audio_pad|>",
17
+ "image_token": "<|image_pad|>",
18
+ "video_token": "<|video_pad|>",
19
+ "vision_bos_token": "<|vision_start|>",
20
+ "vision_eos_token": "<|vision_end|>"
21
+ },
22
+ "pad_token": "<|endoftext|>",
23
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
24
+ "split_special_tokens": false,
25
+ "tokenizer_class": "Qwen2Tokenizer",
26
+ "unk_token": null,
27
+ "video_token": "<|video_pad|>",
28
+ "vision_bos_token": "<|vision_start|>",
29
+ "vision_eos_token": "<|vision_end|>"
30
+ }