ukisai commited on
Commit
37e7e5e
·
0 Parent(s):

initial release

Browse files
.gitattributes ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ ukisai-banner.png filter=lfs diff=lfs merge=lfs -text
38
+ swift-speed-demo.mp4 filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,328 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: ukisai/Swift-Qwen3.8-27b
3
+ library_name: transformers
4
+ license: other
5
+ license_name: swift-open-license-1.0
6
+ pipeline_tag: image-text-to-text
7
+ tags:
8
+ - qwen3_8
9
+ - efficient-thinking
10
+ - reasoning
11
+ - token-efficient
12
+ - amd
13
+ - rocm
14
+ - int4
15
+ - awq
16
+ - quark
17
+ - w4a16
18
+ base_model_relation: quantized
19
+ ---
20
+
21
+ <div align="center">
22
+ <a href="https://ukisai.com"><img src="ukisai-banner.png" alt="UkisAI" style="width:100%;max-width:100%;height:auto;display:block;margin-bottom:0.6em;" /></a>
23
+ <div style="display:flex;justify-content:center;gap:0.6em;margin-bottom:1em;">
24
+ <a href="https://ukisai.com"><strong>Website</strong></a> &nbsp;&bull;&nbsp;
25
+ <a href="https://ukisai.com/products/swift"><strong>Learn more</strong></a> &nbsp;&bull;&nbsp;
26
+ <a href="https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF"><strong>GGUF</strong></a> &nbsp;&bull;&nbsp;
27
+ <a href="#license-and-access"><strong>Enterprise licensing</strong></a>
28
+ </div>
29
+ </div>
30
+
31
+ # Swift-Qwen3.8-27b-int4-AMD
32
+
33
+ AMD Quark AWQ INT4 (W4A16) edition of Swift. The following introduction describes the base Swift results; release-specific details are below.
34
+
35
+ Swift-Qwen3.8-27B is UkisAI's reasoning-efficient derivative of Qwen3.8-27B,
36
+ using **58.3% fewer thinking tokens** while maintaining near-identical performance
37
+ (**&lt;1% loss**) and as a result getting a **x1.95 speed-up** on several tasks.
38
+
39
+ <video controls autoplay muted loop playsinline style="width:100%;max-width:100%;height:auto;display:block;border-radius:12px;margin:0.8em 0 1.4em;" src="https://huggingface.co/ukisai/Swift-Qwen3.8-27b-int4-AMD/resolve/main/swift-speed-demo.mp4"></video>
40
+ <p align="center" style="font-size:13px;color:#8C94A8;margin:-0.6em 0 1.4em;">The prompt is a sample from LiveCodeBench v6</p>
41
+
42
+ ## AMD Quark INT4 release
43
+
44
+ This is the **INT4 W4A16 quantization of Swift for AMD hardware workflows**, produced
45
+ with [AMD Quark](https://github.com/amd/Quark). It uses Quark's **PyTorch** workflow
46
+ and native Hugging Face safetensors export: signed symmetric INT4 weights,
47
+ groups of 128, and BF16 activations.
48
+
49
+ The full-precision companion is
50
+ [Swift-Qwen3.8-27b-BF16-AMD](https://huggingface.co/ukisai/Swift-Qwen3.8-27b-BF16-AMD).
51
+
52
+ | Property | This checkpoint |
53
+ | --- | --- |
54
+ | Source | [Swift-Qwen3.8-27B](https://huggingface.co/ukisai/Swift-Qwen3.8-27b) |
55
+ | Quantizer | AMD Quark AWQ |
56
+ | Weight / activation precision | INT4 / BF16 (W4A16) |
57
+ | Weight grouping | Symmetric, group size 128 |
58
+ | Format | Native Quark safetensors, `real_quantized`, `reorder` packing |
59
+ | Weight files | 19.513 GB; BF16 source: 55.563 GB |
60
+ | Calibration | 128 Pile validation samples, 512 tokens each |
61
+ | Quantized layers | 496 eligible language-model linear layers |
62
+ | Preserved components | BF16 vision tower, output head, embeddings, and all 15 MTP tensors |
63
+
64
+ Quark supports preparing models for AMD deployment. **This checkpoint was quantized
65
+ and validated on an NVIDIA H100; AMD/ROCm serving and throughput have not yet been
66
+ validated.** Serving needs a runtime that supports this native Quark INT4 format.
67
+ The [Quark project](https://github.com/amd/Quark) and
68
+ [installation guide](https://quark.docs.amd.com/latest/install.html) describe its
69
+ supported CUDA and ROCm environments.
70
+
71
+ ### Checkpoint validation
72
+
73
+ | Sanity check | BF16 | This INT4 export |
74
+ | --- | ---: | ---: |
75
+ | Wikitext perplexity | 9.16197 | 9.54254 |
76
+ | Arithmetic generation | Pass | Pass |
77
+ | JSON generation | Pass | Pass |
78
+
79
+ Perplexity uses the same eight non-overlapping 512-token Wikitext-2 test windows.
80
+ The 4.15% perplexity increase is a small sanity result, not a full accuracy benchmark.
81
+ The packed checkpoint was independently reloaded, including its final configuration
82
+ and index, and reproduced the evaluation NLLs exactly. All floating tensors are
83
+ finite; 349 preserved vision/output-head/MTP tensors match the source exactly.
84
+ Vision inference and MTP decoding were not exercised in this validation.
85
+ See [quantization_report.json](quantization_report.json).
86
+
87
+ **The Swift benchmarks and speed demonstration below are reproduced from the base
88
+ Swift model card. They do not measure this Quark export or AMD hardware.**
89
+
90
+ ## Training approach
91
+
92
+ We built Swift by identifying reasoning-marker tokens that, in our analysis, trigger overthinking in Qwen’s
93
+ reasoning rollouts. We then fine-tuned Qwen by penalizing usage of those tokens while it reasons.
94
+
95
+ Swift produces shorter reasoning traces. In our testing, we also observe fewer overthinking errors.
96
+
97
+ For maximum gains, Swift also includes a transfer component derived from
98
+ [BottleCap AI's ThinkingCap-Qwen3.6-27B](https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.6-27B).
99
+
100
+ ## Evaluation scope
101
+
102
+ > All results below compare the Qwen3.8-27B BF16 base with the same base plus the
103
+ > Swift adapter.
104
+
105
+ ## Benchmarks
106
+
107
+ <style>
108
+ .swift-table { width:100%; table-layout:fixed; border-collapse:separate; border-spacing:0; overflow:hidden; border:1px solid #27344A; border-radius:20px; background:#0D111B; font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,sans-serif; font-size:14px; color:#BFBDBD; }
109
+ .swift-table th { padding:13px 8px; text-align:center; font-weight:700; color:#AEB5C7; background:#0D111B; border-right:1px solid #27344A; border-bottom:1px solid #27344A; }
110
+ .swift-table td { padding:14px 8px; text-align:center; color:#BFBDBD; background:#0D111B; border-right:1px solid #27344A; border-bottom:1px solid #27344A; vertical-align:middle; overflow-wrap:break-word; }
111
+ .swift-table tr > :last-child { border-right:0; }
112
+ .swift-table tbody tr:last-child td { border-bottom:0; }
113
+ .swift-table .benchmark-heading { color:#B7BDCD; background:#0D111B; border-bottom:3px solid #7D45B5; }
114
+ .swift-table .score-heading { color:#F0C5FF; background:#52239E; border-bottom:3px solid #7D45B5; }
115
+ .swift-table .tokens-heading, .swift-table .median-heading { color:#D4E8FF; background:#304FC2; border-bottom:3px solid #5687E6; }
116
+ .swift-table .benchmark { padding-left:18px; text-align:left; color:#FFFFFF; font-weight:600; }
117
+ .swift-table strong { color:#FFFFFF; }
118
+ .swift-table .section { padding:12px 18px; text-align:left; color:#B489FF; background:#2A2541; font-weight:700; letter-spacing:.08em; text-transform:uppercase; border-top:1px solid #3A3159; border-bottom:1px solid #3A3159; }
119
+ .swift-table .swift { background:#171127; }
120
+ .swift-table thead tr:nth-child(2) .swift { color:#D3A0FF; }
121
+ .swift-table .reduction { color:#69BFFF; background:#101B2C; font-weight:700; }
122
+ .swift-table .detail { color:#8C94A8; font-size:12px; font-weight:500; }
123
+
124
+ @media (max-width: 640px) {
125
+ .swift-table { display:block !important; width:100% !important; max-width:100%; overflow-x:auto !important; -webkit-overflow-scrolling:touch; table-layout:auto !important; }
126
+ .swift-table th, .swift-table td { min-width:100px; }
127
+ .swift-table th:first-child, .swift-table td:first-child { min-width:160px; }
128
+ }
129
+ </style>
130
+
131
+ <table class="swift-table">
132
+ <thead>
133
+ <tr>
134
+ <th rowspan="2" class="benchmark-heading" style="width:32%;text-align:left;padding-left:18px;vertical-align:bottom;">Benchmark</th>
135
+ <th colspan="2" class="score-heading">Score</th>
136
+ <th colspan="3" class="tokens-heading">Mean tokens</th>
137
+ <th class="median-heading" style="width:14%;">Median tokens</th>
138
+ </tr>
139
+ <tr>
140
+ <th>Base</th>
141
+ <th class="swift">Swift</th>
142
+ <th>Base</th>
143
+ <th class="swift">Swift</th>
144
+ <th class="reduction">Reduction</th>
145
+ <th class="reduction">Reduction</th>
146
+ </tr>
147
+ </thead>
148
+ <tbody>
149
+ <tr><td class="section" colspan="7">General reasoning</td></tr>
150
+ <tr><td class="benchmark">GPQA-Diamond</td><td>88.38%</td><td class="swift">88.28%</td><td>15,014</td><td class="swift"><strong>8,855</strong></td><td class="reduction">&darr; 41.0%</td><td class="reduction">&darr; 58.3%</td></tr>
151
+ <tr><td class="benchmark">MMLU-Pro</td><td>85.47%</td><td class="swift">84.95%</td><td>2,980</td><td class="swift"><strong>1,603</strong></td><td class="reduction">&darr; 46.2%</td><td class="reduction">&darr; 28.3%</td></tr>
152
+ <tr><td class="benchmark">C-Eval</td><td>90.00%</td><td class="swift">90.62%</td><td>1,492</td><td class="swift"><strong>804</strong></td><td class="reduction">&darr; 46.1%</td><td class="reduction">&darr; 19.3%</td></tr>
153
+ <tr><td class="benchmark">IFBench</td><td>73.53%</td><td class="swift">71.80%</td><td>8,052</td><td class="swift"><strong>4,657</strong></td><td class="reduction">&darr; 42.2%</td><td class="reduction">&darr; 50.5%</td></tr>
154
+ <tr><td class="section" colspan="7">Mathematics</td></tr>
155
+ <tr><td class="benchmark">AIME 2026</td><td>98.67%</td><td class="swift">94.00%</td><td>22,014</td><td class="swift"><strong>16,143</strong></td><td class="reduction">&darr; 26.7%</td><td class="reduction">&darr; 50.2%</td></tr>
156
+ <tr><td class="benchmark">HMMT (Nov 2025)</td><td>99.33%</td><td class="swift">96.00%</td><td>22,032</td><td class="swift"><strong>15,189</strong></td><td class="reduction">&darr; 31.1%</td><td class="reduction">&darr; 45.9%</td></tr>
157
+ <tr><td class="section" colspan="7">Multimodal</td></tr>
158
+ <tr><td class="benchmark">ERQA</td><td>67.45%</td><td class="swift">66.30%</td><td>4,137</td><td class="swift"><strong>2,045</strong></td><td class="reduction">&darr; 50.6%</td><td class="reduction">&darr; 54.6%</td></tr>
159
+ <tr><td class="section" colspan="7">Agentic coding</td></tr>
160
+ <tr><td class="benchmark">Terminal-Bench 2.1</td><td>66.74%</td><td class="swift">65.84%</td><td>37,086</td><td class="swift"><strong>27,272</strong></td><td class="reduction">&darr; 26.5%</td><td class="reduction">&darr; 38.7%</td></tr>
161
+ <tr><td class="benchmark">LiveCodeBench v6</td><td>76.76%</td><td class="swift">81.55%</td><td>11,374</td><td class="swift"><strong>8,615</strong></td><td class="reduction">&darr; 24.3%</td><td class="reduction">&darr; 45.8%</td></tr>
162
+ </tbody>
163
+ </table>
164
+
165
+ <details>
166
+ <summary><strong>How to reproduce</strong></summary>
167
+
168
+ <p style="font-size:13px;line-height:1.5;margin:8px 0;"><strong>Serving:</strong> BF16 · vLLM 0.27.1 · Qwen3 parser · context 262,144 · thinking xhigh.<br>
169
+ <strong>Sampling:</strong> temperature 1.0 · top_p 0.95 · top_k 20 · min_p 0 · presence_penalty 0 · repetition_penalty 1.<br>
170
+ <strong>Benchmarks:</strong> averages over five seeds (0–4) per model; five trials per task for Terminal-Bench.</p>
171
+
172
+ <table style="display:table;width:100%;border-collapse:collapse;font-size:13px;line-height:1.3;margin:8px 0;">
173
+ <thead><tr><th style="padding:4px 8px;text-align:left;">Benchmark</th><th style="padding:4px 8px;text-align:right;">Output cap</th></tr></thead>
174
+ <tbody>
175
+ <tr><td style="padding:3px 8px;">GPQA-Diamond</td><td style="padding:3px 8px;text-align:right;">100,000</td></tr>
176
+ <tr><td style="padding:3px 8px;">MMLU-Pro</td><td style="padding:3px 8px;text-align:right;">100,000</td></tr>
177
+ <tr><td style="padding:3px 8px;">C-Eval</td><td style="padding:3px 8px;text-align:right;">16,384</td></tr>
178
+ <tr><td style="padding:3px 8px;">IFBench</td><td style="padding:3px 8px;text-align:right;">81,920</td></tr>
179
+ <tr><td style="padding:3px 8px;">AIME 2026</td><td style="padding:3px 8px;text-align:right;">250,000</td></tr>
180
+ <tr><td style="padding:3px 8px;">HMMT Nov 2025</td><td style="padding:3px 8px;text-align:right;">250,000</td></tr>
181
+ <tr><td style="padding:3px 8px;">ERQA</td><td style="padding:3px 8px;text-align:right;">100,000</td></tr>
182
+ <tr><td style="padding:3px 8px;">Terminal-Bench 2.1</td><td style="padding:3px 8px;text-align:right;">Agent/task limits</td></tr>
183
+ <tr><td style="padding:3px 8px;">LiveCodeBench v6</td><td style="padding:3px 8px;text-align:right;">32,768</td></tr>
184
+ </tbody>
185
+ </table>
186
+
187
+ </details>
188
+
189
+ ## Efficiency across and versus reasoning efforts
190
+
191
+ Qwen3.8's `reasoning_effort` setting lets users choose how much the model thinks.
192
+ For Swift to be useful across these settings, it needs to reduce thinking while
193
+ keeping accuracy close to the base. We therefore tested `xhigh`, `medium`, and `low`:
194
+ thinking-token savings persist at every level.
195
+
196
+ <table class="swift-table" style="display:table;width:100%;table-layout:fixed;">
197
+ <thead>
198
+ <tr>
199
+ <th class="benchmark-heading" style="width:50%;text-align:left;padding-left:18px;white-space:normal;">Reasoning effort</th>
200
+ <th class="tokens-heading" style="width:50%;white-space:normal;">Mean thinking reduction</th>
201
+ </tr>
202
+ </thead>
203
+ <tbody>
204
+ <tr><td class="benchmark">Xhigh</td><td class="reduction">&darr; 41.0%</td></tr>
205
+ <tr><td class="benchmark">Medium</td><td class="reduction">&darr; 22.7%</td></tr>
206
+ <tr><td class="benchmark">Low</td><td class="reduction">&darr; 25.8%</td></tr>
207
+ </tbody>
208
+ </table>
209
+
210
+ The efficiency also holds up against the base's own lower effort settings. On
211
+ GPQA-Diamond (198 questions, 5 seeds, 990 paired calls), Swift at `xhigh` is
212
+ compared with the base at `xhigh` and at `medium`:
213
+
214
+ <table class="swift-table" style="display:table;width:100%;table-layout:fixed;">
215
+ <thead>
216
+ <tr>
217
+ <th class="benchmark-heading" style="width:34%;text-align:left;padding-left:18px;white-space:normal;">GPQA-Diamond</th>
218
+ <th class="score-heading" style="width:22%;white-space:normal;">Score</th>
219
+ <th class="tokens-heading" style="width:22%;white-space:normal;">Mean tokens</th>
220
+ <th class="median-heading" style="width:22%;white-space:normal;">Median tokens</th>
221
+ </tr>
222
+ </thead>
223
+ <tbody>
224
+ <tr><td class="benchmark">Base &middot; xhigh</td><td>88.38%</td><td>15,014</td><td>6,642</td></tr>
225
+ <tr class="swift"><td class="benchmark swift">Swift &middot; xhigh</td><td class="swift"><strong>88.28%</strong></td><td class="swift"><strong>8,855</strong></td><td class="swift"><strong>2,771</strong></td></tr>
226
+ <tr><td class="benchmark">Base &middot; medium</td><td>84.14%</td><td>4,451</td><td>1,753</td></tr>
227
+ </tbody>
228
+ </table>
229
+
230
+ Swift retains the accuracy of `xhigh` while using about half the tokens, although
231
+ it uses about double the tokens of `medium`.
232
+
233
+ ## Quantized models
234
+
235
+ Quantized deployment is the intended use for Swift: lower-memory weights paired with
236
+ shorter reasoning. The INT4 evaluations below retain token savings across GPQA,
237
+ IFBench, and AIME. On AIME, Swift matches or improves accuracy and reduces output-cap
238
+ failures by **31–33%**.
239
+
240
+ <table class="swift-table" style="display:table;width:100%;table-layout:fixed;">
241
+ <thead><tr>
242
+ <th class="benchmark-heading" style="width:32%;text-align:left;padding-left:18px;white-space:normal;">Benchmark / quantization</th>
243
+ <th class="score-heading" style="width:16%;white-space:normal;">Base accuracy</th>
244
+ <th class="score-heading" style="width:16%;white-space:normal;">Swift accuracy</th>
245
+ <th class="tokens-heading" style="width:18%;white-space:normal;">Mean token reduction</th>
246
+ <th class="median-heading" style="width:18%;white-space:normal;">Median token reduction</th>
247
+ </tr></thead>
248
+ <tbody>
249
+ <tr><td class="benchmark">GPQA-Diamond<br><span class="detail">Mixed-precision quant W4A16 · thinking tokens</span></td><td>88.69%</td><td class="swift">88.38%</td><td class="reduction">&darr; 32.1%</td><td class="reduction">&darr; 50.2%</td></tr>
250
+ <tr><td class="benchmark">IFBench<br><span class="detail">Mixed-precision quant W4A16 · completion tokens</span></td><td>72.58%</td><td class="swift">71.25%</td><td class="reduction">&darr; 30.1%</td><td class="reduction">&darr; 38.0%</td></tr>
251
+ <tr><td class="benchmark">AIME 2026<br><span class="detail">Mixed-precision quant W4A16 · completion tokens</span></td><td>84.00%</td><td class="swift">84.00%</td><td class="reduction">&darr; 19.0%</td><td class="reduction">&darr; 37.5%</td></tr>
252
+ <tr><td class="benchmark">AIME 2026<br><span class="detail">AWQ INT4 · completion tokens</span></td><td>82.67%</td><td class="swift">84.00%</td><td class="reduction">&darr; 22.8%</td><td class="reduction">&darr; 34.8%</td></tr>
253
+ </tbody>
254
+ </table>
255
+
256
+ <details>
257
+ <summary><strong>Quantized evaluation settings</strong></summary>
258
+
259
+ Each row compares the same quantized base with and without the Swift adapter.
260
+ GPQA and AIME use five seeds; IFBench uses four samples per prompt and strict scoring.
261
+ Output caps: GPQA 100,000; IFBench 81,920; AIME 32,768. GPQA and IFBench use saved
262
+ historical base runs. AIME uses template-default effort and counts truncated answers
263
+ as incorrect. Its shorter cap makes it a separate comparison from the BF16 table.
264
+
265
+ </details>
266
+
267
+ ## How to use
268
+
269
+ ### PyTorch with AMD Quark
270
+
271
+ Install the GPU-specific PyTorch and Quark packages from the
272
+ [official installation guide](https://quark.docs.amd.com/latest/install.html).
273
+ Validation used Python 3.12, PyTorch 2.11.0+cu128, Transformers 5.2.0,
274
+ AMD Quark 0.12.post1+cu128.torch2.11, Accelerate 1.15.0, and Safetensors 0.8.0.
275
+ For AMD, select the corresponding supported ROCm environment.
276
+
277
+ Download this repository and run the included loader:
278
+
279
+ ```bash
280
+ hf download ukisai/Swift-Qwen3.8-27b-int4-AMD --local-dir Swift-Qwen3.8-27b-int4-AMD
281
+ python Swift-Qwen3.8-27b-int4-AMD/load_quark.py \
282
+ --model Swift-Qwen3.8-27b-int4-AMD \
283
+ --prompt "What is 17 multiplied by 23? Answer with only the number."
284
+ ```
285
+
286
+ [`load_quark.py`](load_quark.py) imports the packed weights through Quark's PyTorch
287
+ API. The included [`quark_compat.py`](quark_compat.py) handles the public Quark
288
+ 0.12 dense-Qwen reload path. The model uses the `qwen3_5` Transformers architecture
289
+ identifier. [`recipe.py`](recipe.py) records the AWQ configuration.
290
+ The example disables thinking for a short deterministic smoke check.
291
+
292
+ ### Serving
293
+
294
+ A serving engine must support native Quark W4A16 signed INT4 with `reorder` packing
295
+ and this Qwen architecture. As of September 14, 2026,
296
+ [vLLM's native Quark INT4 support PR](https://github.com/vllm-project/vllm/pull/48606)
297
+ remains open. Stock vLLM compatibility and AMD performance are not established by
298
+ the PyTorch validation above. The preserved MTP head also needs compatible runtime
299
+ support before speculative decoding can be used.
300
+
301
+ For standard BF16 serving instructions, see the
302
+ [BF16 companion](https://huggingface.co/ukisai/Swift-Qwen3.8-27b-BF16-AMD).
303
+ The [base Swift card](https://huggingface.co/ukisai/Swift-Qwen3.8-27b#how-to-use) also documents the
304
+ UkisAI API and other Swift formats; that API is separate from this downloadable
305
+ Quark checkpoint.
306
+
307
+ ## License and access
308
+
309
+ Swift weights are distributed under the **Swift Open License v1.0**.
310
+ Personal, research, educational, evaluation, and commercial use are free for individuals
311
+ and organizations with annual recurring revenue, including affiliates, of up to
312
+ US$1,000,000. Above that threshold, commercial use requires a separate **Swift Enterprise
313
+ License**. Contact [UkisAI](https://ukisai.com/contact) for terms.
314
+
315
+ ## Citation
316
+
317
+ ```bibtex
318
+ @misc{swift-qwen3.8-27b,
319
+ title = {Swift-Qwen3.8-27B},
320
+ author = {UkisAI},
321
+ year = {2026},
322
+ url = {https://huggingface.co/ukisai/Swift-Qwen3.8-27b}
323
+ }
324
+ ```
325
+
326
+ ## Acknowledgements
327
+
328
+ We acknowledge the [NVIDIA Innovation Lab](https://www.nvidia.com/en-us/data-center/innovation-lab/) for providing access to **8× NVIDIA H100 GPUs** to train Swift.
SHA256SUMS ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ee88be55447e3bfb8aef842fd7eb4819c75578189d3ecc86c3f0032dbac97422 README.md
2
+ c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041 chat_template.jinja
3
+ fbc21d89ddaa82af0b79d3ccd328a76b863f36d4190b8c4a0100a29faf853d89 config.json
4
+ e70c136c1b78ddc1fb0905bac8e733a4dc448d4f852a5dd75143fffc70be550e generation_config.json
5
+ 86a714fcfc05b610e2d664b0d8385ad5a38421e9370712096194efd9e72dd8b1 load_quark.py
6
+ a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d merges.txt
7
+ 0f4e5cb71a806ff2ced530144b243d3ab28ca7068d41d89f5e42aa609d6f99db model.safetensors.index.json
8
+ 27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516 preprocessor_config.json
9
+ 73f2751bc256b5f4ffc7df65b1a1cf201b27dbcb3024677e9d0f470b20e7fa3a quantization_report.json
10
+ 28333acab30e77ba4801a32682251830a7a26dfe490f67999e26de3b6987463a quark_compat.py
11
+ 5d8f0744f0b4aab61ca58c03fc83698072cdf5f2d149880871021b0babb94dbc recipe.py
12
+ 0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3 tokenizer.json
13
+ b11349aafa7cdc6a320767cf7ceb29ed82f7eda5d65e8e0819e76f0ce947bf27 tokenizer_config.json
14
+ 3e9e39451b4586f86a8c7c12dfbba21a5adc16778543d93e39b139ca8627f0c7 ukisai-banner.png
15
+ 7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13 video_preprocessor_config.json
16
+ ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003 vocab.json
17
+ a792a58dcc5b8f5b663f81c3f5024e0bc7cccc5290e2254e006f2845fdd5c247 model.safetensors
chat_template.jinja ADDED
@@ -0,0 +1,170 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- set reasoning_instructions = '' %}
46
+ {%- if enable_thinking is undefined or enable_thinking is true %}
47
+ {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}
48
+ {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}
49
+ {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}
50
+ {%- endif %}
51
+ {%- if resolved_reasoning_effort == 'xhigh' %}
52
+ {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}
53
+ {%- elif resolved_reasoning_effort == 'low' %}
54
+ {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}
55
+ {%- endif %}
56
+ {%- endif %}
57
+ {%- if tools and tools is iterable and tools is not mapping %}
58
+ {{- '<|im_start|>system\n' }}
59
+ {%- if reasoning_instructions %}
60
+ {{- reasoning_instructions + '\n\n' }}
61
+ {%- endif %}
62
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
63
+ {%- for tool in tools %}
64
+ {{- "\n" }}
65
+ {{- tool | tojson }}
66
+ {%- endfor %}
67
+ {{- "\n</tools>" }}
68
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
69
+ {%- if messages[0].role == 'system' %}
70
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
71
+ {%- if content %}
72
+ {{- '\n\n' + content }}
73
+ {%- endif %}
74
+ {%- endif %}
75
+ {{- '<|im_end|>\n' }}
76
+ {%- else %}
77
+ {%- if messages[0].role == 'system' %}
78
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
79
+ {%- if content %}
80
+ {{- '<|im_start|>system\n' + (reasoning_instructions + '\n\n' if reasoning_instructions else '') + content + '<|im_end|>\n' }}
81
+ {%- elif reasoning_instructions %}
82
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
83
+ {%- endif %}
84
+ {%- elif reasoning_instructions %}
85
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
86
+ {%- endif %}
87
+ {%- endif %}
88
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
89
+ {%- for message in messages[::-1] %}
90
+ {%- set index = (messages|length - 1) - loop.index0 %}
91
+ {%- if ns.multi_step_tool and message.role == "user" %}
92
+ {%- set content = render_content(message.content, false)|trim %}
93
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
94
+ {%- set ns.multi_step_tool = false %}
95
+ {%- set ns.last_query_index = index %}
96
+ {%- endif %}
97
+ {%- endif %}
98
+ {%- endfor %}
99
+ {%- if ns.multi_step_tool %}
100
+ {{- raise_exception('No user query found in messages.') }}
101
+ {%- endif %}
102
+ {%- for message in messages %}
103
+ {%- set content = render_content(message.content, true)|trim %}
104
+ {%- if message.role == "system" %}
105
+ {%- if not loop.first %}
106
+ {{- raise_exception('System message must be at the beginning.') }}
107
+ {%- endif %}
108
+ {%- elif message.role == "user" %}
109
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
110
+ {%- elif message.role == "assistant" %}
111
+ {%- set reasoning_content = '' %}
112
+ {%- if message.reasoning_content is string %}
113
+ {%- set reasoning_content = message.reasoning_content %}
114
+ {%- endif %}
115
+ {%- set reasoning_content = reasoning_content|trim %}
116
+ {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}
117
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
118
+ {%- else %}
119
+ {{- '<|im_start|>' + message.role + '\n' + content }}
120
+ {%- endif %}
121
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
122
+ {%- for tool_call in message.tool_calls %}
123
+ {%- if tool_call.function is defined %}
124
+ {%- set tool_call = tool_call.function %}
125
+ {%- endif %}
126
+ {%- if loop.first %}
127
+ {%- if content|trim %}
128
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
129
+ {%- else %}
130
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
131
+ {%- endif %}
132
+ {%- else %}
133
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
134
+ {%- endif %}
135
+ {%- if tool_call.arguments is defined and tool_call.arguments != '' %}
136
+ {%- for args_name, args_value in tool_call.arguments|items %}
137
+ {{- '<parameter=' + args_name + '>\n' }}
138
+ {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}
139
+ {{- args_value }}
140
+ {{- '\n</parameter>\n' }}
141
+ {%- endfor %}
142
+ {%- endif %}
143
+ {{- '</function>\n</tool_call>' }}
144
+ {%- endfor %}
145
+ {%- endif %}
146
+ {{- '<|im_end|>\n' }}
147
+ {%- elif message.role == "tool" %}
148
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
149
+ {{- '<|im_start|>user' }}
150
+ {%- endif %}
151
+ {{- '\n<tool_response>\n' }}
152
+ {{- content }}
153
+ {{- '\n</tool_response>' }}
154
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
155
+ {{- '<|im_end|>\n' }}
156
+ {%- elif loop.last %}
157
+ {{- '<|im_end|>\n' }}
158
+ {%- endif %}
159
+ {%- else %}
160
+ {{- raise_exception('Unexpected message role.') }}
161
+ {%- endif %}
162
+ {%- endfor %}
163
+ {%- if add_generation_prompt %}
164
+ {{- '<|im_start|>assistant\n' }}
165
+ {%- if enable_thinking is defined and enable_thinking is false %}
166
+ {{- '<think>\n\n</think>\n\n' }}
167
+ {%- else %}
168
+ {{- '<think>\n' }}
169
+ {%- endif %}
170
+ {%- endif %}
config.json ADDED
@@ -0,0 +1,335 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForConditionalGeneration"
4
+ ],
5
+ "image_token_id": 248056,
6
+ "language_model_only": false,
7
+ "model_type": "qwen3_5",
8
+ "text_config": {
9
+ "attention_bias": false,
10
+ "attention_dropout": 0.0,
11
+ "attn_output_gate": true,
12
+ "bos_token_id": 248044,
13
+ "dtype": "bfloat16",
14
+ "eos_token_id": 248044,
15
+ "full_attention_interval": 4,
16
+ "head_dim": 256,
17
+ "hidden_act": "silu",
18
+ "hidden_size": 5120,
19
+ "initializer_range": 0.02,
20
+ "intermediate_size": 17408,
21
+ "layer_types": [
22
+ "linear_attention",
23
+ "linear_attention",
24
+ "linear_attention",
25
+ "full_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "linear_attention",
29
+ "full_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "linear_attention",
33
+ "full_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "linear_attention",
37
+ "full_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "linear_attention",
41
+ "full_attention",
42
+ "linear_attention",
43
+ "linear_attention",
44
+ "linear_attention",
45
+ "full_attention",
46
+ "linear_attention",
47
+ "linear_attention",
48
+ "linear_attention",
49
+ "full_attention",
50
+ "linear_attention",
51
+ "linear_attention",
52
+ "linear_attention",
53
+ "full_attention",
54
+ "linear_attention",
55
+ "linear_attention",
56
+ "linear_attention",
57
+ "full_attention",
58
+ "linear_attention",
59
+ "linear_attention",
60
+ "linear_attention",
61
+ "full_attention",
62
+ "linear_attention",
63
+ "linear_attention",
64
+ "linear_attention",
65
+ "full_attention",
66
+ "linear_attention",
67
+ "linear_attention",
68
+ "linear_attention",
69
+ "full_attention",
70
+ "linear_attention",
71
+ "linear_attention",
72
+ "linear_attention",
73
+ "full_attention",
74
+ "linear_attention",
75
+ "linear_attention",
76
+ "linear_attention",
77
+ "full_attention",
78
+ "linear_attention",
79
+ "linear_attention",
80
+ "linear_attention",
81
+ "full_attention",
82
+ "linear_attention",
83
+ "linear_attention",
84
+ "linear_attention",
85
+ "full_attention"
86
+ ],
87
+ "linear_conv_kernel_dim": 4,
88
+ "linear_key_head_dim": 128,
89
+ "linear_num_key_heads": 16,
90
+ "linear_num_value_heads": 48,
91
+ "linear_value_head_dim": 128,
92
+ "mamba_ssm_dtype": "float32",
93
+ "max_position_embeddings": 262144,
94
+ "model_type": "qwen3_5_text",
95
+ "mtp_num_hidden_layers": 1,
96
+ "mtp_use_dedicated_embeddings": false,
97
+ "num_attention_heads": 24,
98
+ "num_hidden_layers": 64,
99
+ "num_key_value_heads": 4,
100
+ "output_gate_type": "swish",
101
+ "pad_token_id": null,
102
+ "partial_rotary_factor": 0.25,
103
+ "rms_norm_eps": 1e-06,
104
+ "rope_parameters": {
105
+ "mrope_interleaved": true,
106
+ "mrope_section": [
107
+ 11,
108
+ 11,
109
+ 10
110
+ ],
111
+ "partial_rotary_factor": 0.25,
112
+ "rope_theta": 10000000,
113
+ "rope_type": "default"
114
+ },
115
+ "tie_word_embeddings": false,
116
+ "use_cache": true,
117
+ "vocab_size": 248320
118
+ },
119
+ "tie_word_embeddings": false,
120
+ "transformers_version": "5.8.0.dev0",
121
+ "video_token_id": 248057,
122
+ "vision_config": {
123
+ "deepstack_visual_indexes": [],
124
+ "depth": 27,
125
+ "hidden_act": "gelu_pytorch_tanh",
126
+ "hidden_size": 1152,
127
+ "in_channels": 3,
128
+ "initializer_range": 0.02,
129
+ "intermediate_size": 4304,
130
+ "model_type": "qwen3_5",
131
+ "num_heads": 16,
132
+ "num_position_embeddings": 2304,
133
+ "out_hidden_size": 5120,
134
+ "patch_size": 16,
135
+ "spatial_merge_size": 2,
136
+ "temporal_patch_size": 2
137
+ },
138
+ "vision_end_token_id": 248054,
139
+ "vision_start_token_id": 248053,
140
+ "quantization_config": {
141
+ "algo_config": [
142
+ {
143
+ "model_decoder_layers": "model.language_model.layers",
144
+ "name": "awq",
145
+ "scaling_layers": [
146
+ {
147
+ "inp": "mlp.gate_proj",
148
+ "layers": [
149
+ "mlp.gate_proj",
150
+ "mlp.up_proj"
151
+ ],
152
+ "module2inspect": "mlp",
153
+ "prev_op": "post_attention_layernorm"
154
+ },
155
+ {
156
+ "inp": "mlp.down_proj",
157
+ "layers": [
158
+ "mlp.down_proj"
159
+ ],
160
+ "prev_op": "mlp.up_proj"
161
+ }
162
+ ]
163
+ }
164
+ ],
165
+ "exclude": [
166
+ "lm_head",
167
+ "model.visual.blocks.0.attn.proj",
168
+ "model.visual.blocks.0.attn.qkv",
169
+ "model.visual.blocks.0.mlp.linear_fc1",
170
+ "model.visual.blocks.0.mlp.linear_fc2",
171
+ "model.visual.blocks.1.attn.proj",
172
+ "model.visual.blocks.1.attn.qkv",
173
+ "model.visual.blocks.1.mlp.linear_fc1",
174
+ "model.visual.blocks.1.mlp.linear_fc2",
175
+ "model.visual.blocks.10.attn.proj",
176
+ "model.visual.blocks.10.attn.qkv",
177
+ "model.visual.blocks.10.mlp.linear_fc1",
178
+ "model.visual.blocks.10.mlp.linear_fc2",
179
+ "model.visual.blocks.11.attn.proj",
180
+ "model.visual.blocks.11.attn.qkv",
181
+ "model.visual.blocks.11.mlp.linear_fc1",
182
+ "model.visual.blocks.11.mlp.linear_fc2",
183
+ "model.visual.blocks.12.attn.proj",
184
+ "model.visual.blocks.12.attn.qkv",
185
+ "model.visual.blocks.12.mlp.linear_fc1",
186
+ "model.visual.blocks.12.mlp.linear_fc2",
187
+ "model.visual.blocks.13.attn.proj",
188
+ "model.visual.blocks.13.attn.qkv",
189
+ "model.visual.blocks.13.mlp.linear_fc1",
190
+ "model.visual.blocks.13.mlp.linear_fc2",
191
+ "model.visual.blocks.14.attn.proj",
192
+ "model.visual.blocks.14.attn.qkv",
193
+ "model.visual.blocks.14.mlp.linear_fc1",
194
+ "model.visual.blocks.14.mlp.linear_fc2",
195
+ "model.visual.blocks.15.attn.proj",
196
+ "model.visual.blocks.15.attn.qkv",
197
+ "model.visual.blocks.15.mlp.linear_fc1",
198
+ "model.visual.blocks.15.mlp.linear_fc2",
199
+ "model.visual.blocks.16.attn.proj",
200
+ "model.visual.blocks.16.attn.qkv",
201
+ "model.visual.blocks.16.mlp.linear_fc1",
202
+ "model.visual.blocks.16.mlp.linear_fc2",
203
+ "model.visual.blocks.17.attn.proj",
204
+ "model.visual.blocks.17.attn.qkv",
205
+ "model.visual.blocks.17.mlp.linear_fc1",
206
+ "model.visual.blocks.17.mlp.linear_fc2",
207
+ "model.visual.blocks.18.attn.proj",
208
+ "model.visual.blocks.18.attn.qkv",
209
+ "model.visual.blocks.18.mlp.linear_fc1",
210
+ "model.visual.blocks.18.mlp.linear_fc2",
211
+ "model.visual.blocks.19.attn.proj",
212
+ "model.visual.blocks.19.attn.qkv",
213
+ "model.visual.blocks.19.mlp.linear_fc1",
214
+ "model.visual.blocks.19.mlp.linear_fc2",
215
+ "model.visual.blocks.2.attn.proj",
216
+ "model.visual.blocks.2.attn.qkv",
217
+ "model.visual.blocks.2.mlp.linear_fc1",
218
+ "model.visual.blocks.2.mlp.linear_fc2",
219
+ "model.visual.blocks.20.attn.proj",
220
+ "model.visual.blocks.20.attn.qkv",
221
+ "model.visual.blocks.20.mlp.linear_fc1",
222
+ "model.visual.blocks.20.mlp.linear_fc2",
223
+ "model.visual.blocks.21.attn.proj",
224
+ "model.visual.blocks.21.attn.qkv",
225
+ "model.visual.blocks.21.mlp.linear_fc1",
226
+ "model.visual.blocks.21.mlp.linear_fc2",
227
+ "model.visual.blocks.22.attn.proj",
228
+ "model.visual.blocks.22.attn.qkv",
229
+ "model.visual.blocks.22.mlp.linear_fc1",
230
+ "model.visual.blocks.22.mlp.linear_fc2",
231
+ "model.visual.blocks.23.attn.proj",
232
+ "model.visual.blocks.23.attn.qkv",
233
+ "model.visual.blocks.23.mlp.linear_fc1",
234
+ "model.visual.blocks.23.mlp.linear_fc2",
235
+ "model.visual.blocks.24.attn.proj",
236
+ "model.visual.blocks.24.attn.qkv",
237
+ "model.visual.blocks.24.mlp.linear_fc1",
238
+ "model.visual.blocks.24.mlp.linear_fc2",
239
+ "model.visual.blocks.25.attn.proj",
240
+ "model.visual.blocks.25.attn.qkv",
241
+ "model.visual.blocks.25.mlp.linear_fc1",
242
+ "model.visual.blocks.25.mlp.linear_fc2",
243
+ "model.visual.blocks.26.attn.proj",
244
+ "model.visual.blocks.26.attn.qkv",
245
+ "model.visual.blocks.26.mlp.linear_fc1",
246
+ "model.visual.blocks.26.mlp.linear_fc2",
247
+ "model.visual.blocks.3.attn.proj",
248
+ "model.visual.blocks.3.attn.qkv",
249
+ "model.visual.blocks.3.mlp.linear_fc1",
250
+ "model.visual.blocks.3.mlp.linear_fc2",
251
+ "model.visual.blocks.4.attn.proj",
252
+ "model.visual.blocks.4.attn.qkv",
253
+ "model.visual.blocks.4.mlp.linear_fc1",
254
+ "model.visual.blocks.4.mlp.linear_fc2",
255
+ "model.visual.blocks.5.attn.proj",
256
+ "model.visual.blocks.5.attn.qkv",
257
+ "model.visual.blocks.5.mlp.linear_fc1",
258
+ "model.visual.blocks.5.mlp.linear_fc2",
259
+ "model.visual.blocks.6.attn.proj",
260
+ "model.visual.blocks.6.attn.qkv",
261
+ "model.visual.blocks.6.mlp.linear_fc1",
262
+ "model.visual.blocks.6.mlp.linear_fc2",
263
+ "model.visual.blocks.7.attn.proj",
264
+ "model.visual.blocks.7.attn.qkv",
265
+ "model.visual.blocks.7.mlp.linear_fc1",
266
+ "model.visual.blocks.7.mlp.linear_fc2",
267
+ "model.visual.blocks.8.attn.proj",
268
+ "model.visual.blocks.8.attn.qkv",
269
+ "model.visual.blocks.8.mlp.linear_fc1",
270
+ "model.visual.blocks.8.mlp.linear_fc2",
271
+ "model.visual.blocks.9.attn.proj",
272
+ "model.visual.blocks.9.attn.qkv",
273
+ "model.visual.blocks.9.mlp.linear_fc1",
274
+ "model.visual.blocks.9.mlp.linear_fc2",
275
+ "model.visual.merger.linear_fc1",
276
+ "model.visual.merger.linear_fc2",
277
+ "model.visual.pos_embed",
278
+ "mtp.*",
279
+ "mtp.fc.weight",
280
+ "mtp.layers.0.input_layernorm.weight",
281
+ "mtp.layers.0.mlp.down_proj.weight",
282
+ "mtp.layers.0.mlp.gate_proj.weight",
283
+ "mtp.layers.0.mlp.up_proj.weight",
284
+ "mtp.layers.0.post_attention_layernorm.weight",
285
+ "mtp.layers.0.self_attn.k_norm.weight",
286
+ "mtp.layers.0.self_attn.k_proj.weight",
287
+ "mtp.layers.0.self_attn.o_proj.weight",
288
+ "mtp.layers.0.self_attn.q_norm.weight",
289
+ "mtp.layers.0.self_attn.q_proj.weight",
290
+ "mtp.layers.0.self_attn.v_proj.weight",
291
+ "mtp.norm.weight",
292
+ "mtp.pre_fc_norm_embedding.weight",
293
+ "mtp.pre_fc_norm_hidden.weight"
294
+ ],
295
+ "export": {
296
+ "kv_cache_group": [],
297
+ "min_kv_scale": 0.0,
298
+ "pack_method": "reorder",
299
+ "weight_format": "real_quantized",
300
+ "weight_merge_groups": null
301
+ },
302
+ "global_quant_config": {
303
+ "bias": null,
304
+ "input_tensors": null,
305
+ "output_tensors": null,
306
+ "target_device": null,
307
+ "weight": {
308
+ "block_size": null,
309
+ "ch_axis": -1,
310
+ "dtype": "int4",
311
+ "enable_buffer_reuse": false,
312
+ "group_size": 128,
313
+ "is_dynamic": false,
314
+ "is_scale_quant": false,
315
+ "max_input_numel": 4194304,
316
+ "mx_element_dtype": null,
317
+ "observer_cls": "PerGroupMinMaxObserver",
318
+ "qscheme": "per_group",
319
+ "round_method": "half_even",
320
+ "scale_calculation_mode": null,
321
+ "scale_format": null,
322
+ "scale_type": "float",
323
+ "symmetric": true
324
+ }
325
+ },
326
+ "kv_cache_post_rope": false,
327
+ "kv_cache_quant_config": {},
328
+ "layer_quant_config": {},
329
+ "layer_type_quant_config": {},
330
+ "quant_method": "quark",
331
+ "quant_mode": "eager_mode",
332
+ "softmax_quant_spec": null,
333
+ "version": "0.12.post1+cu128.torch2.11"
334
+ }
335
+ }
generation_config.json ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 248044,
3
+ "do_sample": true,
4
+ "eos_token_id": [
5
+ 248046,
6
+ 248044
7
+ ],
8
+ "pad_token_id": 248044,
9
+ "temperature": 1.0,
10
+ "top_k": 20,
11
+ "top_p": 0.95
12
+ }
load_quark.py ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Load Swift's native Quark INT4 checkpoint for PyTorch inference."""
2
+ import argparse
3
+ from pathlib import Path
4
+
5
+
6
+ def main():
7
+ parser = argparse.ArgumentParser(description=__doc__)
8
+ parser.add_argument("--model", default="ukisai/Swift-Qwen3.8-27b-int4-AMD")
9
+ parser.add_argument("--prompt", default="What is 17 multiplied by 23? Answer with only the number.")
10
+ parser.add_argument("--max-new-tokens", type=int, default=128)
11
+ parser.add_argument("--device", default="cuda:0")
12
+ args = parser.parse_args()
13
+
14
+ import torch
15
+ from accelerate import init_empty_weights
16
+ from huggingface_hub import snapshot_download
17
+ from transformers import AutoConfig, AutoTokenizer, GenerationConfig, Qwen3_5ForConditionalGeneration
18
+ from transformers.initialization import no_init_weights
19
+ from quark.torch import import_model_from_safetensors
20
+ from quark_compat import dense_qwen_reload_support
21
+
22
+ checkpoint = args.model if Path(args.model).is_dir() else snapshot_download(
23
+ args.model, allow_patterns=["*.safetensors", "*.json", "*.jinja", "*.txt"]
24
+ )
25
+ config = AutoConfig.from_pretrained(checkpoint)
26
+ config._attn_implementation = "eager"
27
+ with no_init_weights(), init_empty_weights():
28
+ model = Qwen3_5ForConditionalGeneration(config)
29
+ with dense_qwen_reload_support():
30
+ model = import_model_from_safetensors(model, checkpoint, device="cpu")
31
+ model.generation_config = GenerationConfig.from_pretrained(checkpoint)
32
+ model.to(args.device).eval()
33
+ tokenizer = AutoTokenizer.from_pretrained(checkpoint)
34
+ inputs = tokenizer.apply_chat_template(
35
+ [{"role": "user", "content": args.prompt}],
36
+ tokenize=True, add_generation_prompt=True, enable_thinking=False,
37
+ return_tensors="pt", return_dict=True,
38
+ ).to(args.device)
39
+ with torch.inference_mode():
40
+ outputs = model.generate(
41
+ **inputs, max_new_tokens=args.max_new_tokens, do_sample=False,
42
+ use_cache=True, pad_token_id=tokenizer.eos_token_id,
43
+ )
44
+ print(tokenizer.decode(outputs[0, inputs["input_ids"].shape[1]:], skip_special_tokens=True))
45
+
46
+
47
+ if __name__ == "__main__":
48
+ main()
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a792a58dcc5b8f5b663f81c3f5024e0bc7cccc5290e2254e006f2845fdd5c247
3
+ size 19512909752
model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
preprocessor_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "size": {
3
+ "longest_edge": 16777216,
4
+ "shortest_edge": 65536
5
+ },
6
+ "patch_size": 16,
7
+ "temporal_patch_size": 2,
8
+ "merge_size": 2,
9
+ "image_mean": [
10
+ 0.5,
11
+ 0.5,
12
+ 0.5
13
+ ],
14
+ "image_std": [
15
+ 0.5,
16
+ 0.5,
17
+ 0.5
18
+ ],
19
+ "processor_class": "Qwen3VLProcessor",
20
+ "image_processor_type": "Qwen2VLImageProcessorFast"
21
+ }
quantization_report.json ADDED
@@ -0,0 +1,114 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "source_model": "ukisai/Swift-Qwen3.8-27b",
3
+ "source_revision": "1b30aaaf753fe5c1cb51ada2ea0367a53445359c",
4
+ "checkpoint": "ukisai/Swift-Qwen3.8-27b-int4-AMD",
5
+ "framework": "PyTorch",
6
+ "validation_hardware": "NVIDIA H100",
7
+ "amd_hardware_tested": false,
8
+ "versions": {
9
+ "torch": "2.11.0+cu128",
10
+ "transformers": "5.2.0",
11
+ "amd-quark": "0.12.post1+cu128.torch2.11",
12
+ "accelerate": "1.15.0",
13
+ "safetensors": "0.8.0"
14
+ },
15
+ "evaluation_data": {
16
+ "dataset": "Salesforce/wikitext",
17
+ "config": "wikitext-2-raw-v1",
18
+ "split": "test",
19
+ "fingerprint": "a46124b21ac53738",
20
+ "windows": 8,
21
+ "window_length": 512,
22
+ "selection": "first 4096 tokens; small sanity check, not full benchmark",
23
+ "input_sha256": "2ebdb31fd4394419aabae16a635375c8e70d306c14462051d849056893cd2b27"
24
+ },
25
+ "evaluation": {
26
+ "label": "int4-final-export",
27
+ "nlls": [
28
+ 1.820142388343811,
29
+ 2.3994803428649902,
30
+ 2.3420217037200928,
31
+ 2.2320547103881836,
32
+ 2.1242759227752686,
33
+ 2.34279727935791,
34
+ 2.3792335987091064,
35
+ 2.4060728549957275
36
+ ],
37
+ "mean_nll": 2.2557598501443863,
38
+ "perplexity": 9.542541457549426,
39
+ "samples": [
40
+ {
41
+ "prompt": "What is 17 multiplied by 23? Answer with only the number.",
42
+ "answer": "391",
43
+ "generated_tokens": 4
44
+ },
45
+ {
46
+ "prompt": "Return only valid JSON with keys \"status\" set to \"ok\" and \"count\" set to 3.",
47
+ "answer": "{\"status\": \"ok\", \"count\": 3}",
48
+ "generated_tokens": 13
49
+ }
50
+ ],
51
+ "evaluation_seconds": 49.92316593322903,
52
+ "scope": "8 non-overlapping 512-token Wikitext test windows; sanity check only"
53
+ },
54
+ "quantization": "Quark AWQ signed INT4, symmetric groups of 128, BF16 activations",
55
+ "calibration": {
56
+ "dataset": "mit-han-lab/pile-val-backup",
57
+ "split": "validation",
58
+ "fingerprint": "fa7b9ce01fad1b22",
59
+ "samples": 128,
60
+ "sequence_length": 512,
61
+ "selection": "first 128 rows, matching AMD's example default",
62
+ "padding": "left, EOS token, matching Quark get_tokenizer",
63
+ "input_sha256": "9bbae467c3c633d882fdfad4f398aea0515c211c26eafea482585ddb73995b15"
64
+ },
65
+ "audit": {
66
+ "tensor_bytes": 19512618464,
67
+ "weight_scale_tensors": 496,
68
+ "preserved_mtp_tensors": 15,
69
+ "exact_preserved_tensors": 349,
70
+ "mtp_tensors_already_exported": 15,
71
+ "mtp_tensors_added": 0,
72
+ "copied_file_sha256": {
73
+ "chat_template.jinja": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
74
+ "generation_config.json": "e70c136c1b78ddc1fb0905bac8e733a4dc448d4f852a5dd75143fffc70be550e",
75
+ "merges.txt": "a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d",
76
+ "preprocessor_config.json": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516",
77
+ "tokenizer_config.json": "b11349aafa7cdc6a320767cf7ceb29ed82f7eda5d65e8e0819e76f0ce947bf27",
78
+ "video_preprocessor_config.json": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13",
79
+ "vocab.json": "ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003",
80
+ "tokenizer.json": "0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3"
81
+ }
82
+ },
83
+ "round_trip_max_nll_difference": 0.0,
84
+ "bf16_reference": {
85
+ "label": "bf16",
86
+ "nlls": [
87
+ 1.7963014841079712,
88
+ 2.3732309341430664,
89
+ 2.2346465587615967,
90
+ 2.221674919128418,
91
+ 2.0747108459472656,
92
+ 2.3078770637512207,
93
+ 2.3398988246917725,
94
+ 2.372152090072632
95
+ ],
96
+ "mean_nll": 2.215061590075493,
97
+ "perplexity": 9.161973380864124,
98
+ "samples": [
99
+ {
100
+ "prompt": "What is 17 multiplied by 23? Answer with only the number.",
101
+ "answer": "391",
102
+ "generated_tokens": 4
103
+ },
104
+ {
105
+ "prompt": "Return only valid JSON with keys \"status\" set to \"ok\" and \"count\" set to 3.",
106
+ "answer": "{\"status\": \"ok\", \"count\": 3}",
107
+ "generated_tokens": 13
108
+ }
109
+ ],
110
+ "evaluation_seconds": 6.864971877075732,
111
+ "scope": "8 non-overlapping 512-token Wikitext test windows; sanity check only"
112
+ },
113
+ "perplexity_ratio_int4_over_bf16": 1.041537784586906
114
+ }
quark_compat.py ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Process-local compatibility for dense Qwen3.5/3.8 in Quark 0.12.
2
+
3
+ Quark's reload path invokes its MoE preparation helper even for dense models.
4
+ That helper explicitly rejects qwen3_5. Dense Qwen needs no expert conversion;
5
+ all quantized modules are ordinary Linear layers. Keep every other architecture
6
+ on the original path and restore the helper after each import.
7
+ """
8
+ from contextlib import contextmanager
9
+
10
+
11
+ @contextmanager
12
+ def dense_qwen_reload_support():
13
+ from quark.torch.utils.llm import model_preparation
14
+
15
+ original = model_preparation._prepare_for_moe_quant
16
+
17
+ def prepare(model, reload=False):
18
+ if getattr(model.config, "model_type", None) == "qwen3_5":
19
+ text_config = model.config.text_config
20
+ if getattr(text_config, "num_experts", 0):
21
+ raise ValueError("This compatibility path only supports dense Qwen")
22
+ if any("expert" in type(module).__name__.lower() for module in model.modules()):
23
+ raise ValueError("Unexpected expert module in dense Qwen")
24
+ return
25
+ return original(model, reload)
26
+
27
+ model_preparation._prepare_for_moe_quant = prepare
28
+ try:
29
+ yield
30
+ finally:
31
+ model_preparation._prepare_for_moe_quant = original
recipe.py ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Quark recipe matching AMD's published Qwen3.8-27B AWQ configuration."""
2
+ from quark.torch.quantization.config.config import (
3
+ AWQConfig,
4
+ Int4PerGroupSpec,
5
+ QConfig,
6
+ QLayerConfig,
7
+ )
8
+
9
+
10
+ def make_config():
11
+ return QConfig(
12
+ global_quant_config=QLayerConfig(
13
+ weight=Int4PerGroupSpec(ch_axis=-1, group_size=128).to_quantization_spec()
14
+ ),
15
+ exclude=["model.visual.*", "lm_head", "mtp.*"],
16
+ algo_config=[
17
+ AWQConfig(
18
+ model_decoder_layers="model.language_model.layers",
19
+ scaling_layers=[
20
+ {
21
+ "prev_op": "post_attention_layernorm",
22
+ "layers": ["mlp.gate_proj", "mlp.up_proj"],
23
+ "inp": "mlp.gate_proj",
24
+ "module2inspect": "mlp",
25
+ },
26
+ {
27
+ "prev_op": "mlp.up_proj",
28
+ "layers": ["mlp.down_proj"],
29
+ "inp": "mlp.down_proj",
30
+ },
31
+ ],
32
+ )
33
+ ],
34
+ )
swift-speed-demo.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ebdc350da93f0f957d47674d62cf1308e9288d40790243045d95d1f88ae6eb79
3
+ size 7505194
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3
3
+ size 12809320
tokenizer_config.json ADDED
@@ -0,0 +1,305 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "248044": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": false,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ },
12
+ "248045": {
13
+ "content": "<|im_start|>",
14
+ "lstrip": false,
15
+ "normalized": false,
16
+ "rstrip": false,
17
+ "single_word": false,
18
+ "special": true
19
+ },
20
+ "248046": {
21
+ "content": "<|im_end|>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false,
26
+ "special": true
27
+ },
28
+ "248047": {
29
+ "content": "<|object_ref_start|>",
30
+ "lstrip": false,
31
+ "normalized": false,
32
+ "rstrip": false,
33
+ "single_word": false,
34
+ "special": true
35
+ },
36
+ "248048": {
37
+ "content": "<|object_ref_end|>",
38
+ "lstrip": false,
39
+ "normalized": false,
40
+ "rstrip": false,
41
+ "single_word": false,
42
+ "special": true
43
+ },
44
+ "248049": {
45
+ "content": "<|box_start|>",
46
+ "lstrip": false,
47
+ "normalized": false,
48
+ "rstrip": false,
49
+ "single_word": false,
50
+ "special": true
51
+ },
52
+ "248050": {
53
+ "content": "<|box_end|>",
54
+ "lstrip": false,
55
+ "normalized": false,
56
+ "rstrip": false,
57
+ "single_word": false,
58
+ "special": true
59
+ },
60
+ "248051": {
61
+ "content": "<|quad_start|>",
62
+ "lstrip": false,
63
+ "normalized": false,
64
+ "rstrip": false,
65
+ "single_word": false,
66
+ "special": true
67
+ },
68
+ "248052": {
69
+ "content": "<|quad_end|>",
70
+ "lstrip": false,
71
+ "normalized": false,
72
+ "rstrip": false,
73
+ "single_word": false,
74
+ "special": true
75
+ },
76
+ "248053": {
77
+ "content": "<|vision_start|>",
78
+ "lstrip": false,
79
+ "normalized": false,
80
+ "rstrip": false,
81
+ "single_word": false,
82
+ "special": true
83
+ },
84
+ "248054": {
85
+ "content": "<|vision_end|>",
86
+ "lstrip": false,
87
+ "normalized": false,
88
+ "rstrip": false,
89
+ "single_word": false,
90
+ "special": true
91
+ },
92
+ "248055": {
93
+ "content": "<|vision_pad|>",
94
+ "lstrip": false,
95
+ "normalized": false,
96
+ "rstrip": false,
97
+ "single_word": false,
98
+ "special": true
99
+ },
100
+ "248056": {
101
+ "content": "<|image_pad|>",
102
+ "lstrip": false,
103
+ "normalized": false,
104
+ "rstrip": false,
105
+ "single_word": false,
106
+ "special": true
107
+ },
108
+ "248057": {
109
+ "content": "<|video_pad|>",
110
+ "lstrip": false,
111
+ "normalized": false,
112
+ "rstrip": false,
113
+ "single_word": false,
114
+ "special": true
115
+ },
116
+ "248058": {
117
+ "content": "<tool_call>",
118
+ "lstrip": false,
119
+ "normalized": false,
120
+ "rstrip": false,
121
+ "single_word": false,
122
+ "special": false
123
+ },
124
+ "248059": {
125
+ "content": "</tool_call>",
126
+ "lstrip": false,
127
+ "normalized": false,
128
+ "rstrip": false,
129
+ "single_word": false,
130
+ "special": false
131
+ },
132
+ "248060": {
133
+ "content": "<|fim_prefix|>",
134
+ "lstrip": false,
135
+ "normalized": false,
136
+ "rstrip": false,
137
+ "single_word": false,
138
+ "special": false
139
+ },
140
+ "248061": {
141
+ "content": "<|fim_middle|>",
142
+ "lstrip": false,
143
+ "normalized": false,
144
+ "rstrip": false,
145
+ "single_word": false,
146
+ "special": false
147
+ },
148
+ "248062": {
149
+ "content": "<|fim_suffix|>",
150
+ "lstrip": false,
151
+ "normalized": false,
152
+ "rstrip": false,
153
+ "single_word": false,
154
+ "special": false
155
+ },
156
+ "248063": {
157
+ "content": "<|fim_pad|>",
158
+ "lstrip": false,
159
+ "normalized": false,
160
+ "rstrip": false,
161
+ "single_word": false,
162
+ "special": false
163
+ },
164
+ "248064": {
165
+ "content": "<|repo_name|>",
166
+ "lstrip": false,
167
+ "normalized": false,
168
+ "rstrip": false,
169
+ "single_word": false,
170
+ "special": false
171
+ },
172
+ "248065": {
173
+ "content": "<|file_sep|>",
174
+ "lstrip": false,
175
+ "normalized": false,
176
+ "rstrip": false,
177
+ "single_word": false,
178
+ "special": false
179
+ },
180
+ "248066": {
181
+ "content": "<tool_response>",
182
+ "lstrip": false,
183
+ "normalized": false,
184
+ "rstrip": false,
185
+ "single_word": false,
186
+ "special": false
187
+ },
188
+ "248067": {
189
+ "content": "</tool_response>",
190
+ "lstrip": false,
191
+ "normalized": false,
192
+ "rstrip": false,
193
+ "single_word": false,
194
+ "special": false
195
+ },
196
+ "248068": {
197
+ "content": "<think>",
198
+ "lstrip": false,
199
+ "normalized": false,
200
+ "rstrip": false,
201
+ "single_word": false,
202
+ "special": false
203
+ },
204
+ "248069": {
205
+ "content": "</think>",
206
+ "lstrip": false,
207
+ "normalized": false,
208
+ "rstrip": false,
209
+ "single_word": false,
210
+ "special": false
211
+ },
212
+ "248070": {
213
+ "content": "<|audio_start|>",
214
+ "lstrip": false,
215
+ "normalized": false,
216
+ "rstrip": false,
217
+ "single_word": false,
218
+ "special": true
219
+ },
220
+ "248071": {
221
+ "content": "<|audio_end|>",
222
+ "lstrip": false,
223
+ "normalized": false,
224
+ "rstrip": false,
225
+ "single_word": false,
226
+ "special": true
227
+ },
228
+ "248072": {
229
+ "content": "<tts_pad>",
230
+ "lstrip": false,
231
+ "normalized": false,
232
+ "rstrip": false,
233
+ "single_word": false,
234
+ "special": true
235
+ },
236
+ "248073": {
237
+ "content": "<tts_text_bos>",
238
+ "lstrip": false,
239
+ "normalized": false,
240
+ "rstrip": false,
241
+ "single_word": false,
242
+ "special": true
243
+ },
244
+ "248074": {
245
+ "content": "<tts_text_eod>",
246
+ "lstrip": false,
247
+ "normalized": false,
248
+ "rstrip": false,
249
+ "single_word": false,
250
+ "special": true
251
+ },
252
+ "248075": {
253
+ "content": "<tts_text_bos_single>",
254
+ "lstrip": false,
255
+ "normalized": false,
256
+ "rstrip": false,
257
+ "single_word": false,
258
+ "special": true
259
+ },
260
+ "248076": {
261
+ "content": "<|audio_pad|>",
262
+ "lstrip": false,
263
+ "normalized": false,
264
+ "rstrip": false,
265
+ "single_word": false,
266
+ "special": true
267
+ }
268
+ },
269
+ "additional_special_tokens": [
270
+ "<|im_start|>",
271
+ "<|im_end|>",
272
+ "<|object_ref_start|>",
273
+ "<|object_ref_end|>",
274
+ "<|box_start|>",
275
+ "<|box_end|>",
276
+ "<|quad_start|>",
277
+ "<|quad_end|>",
278
+ "<|vision_start|>",
279
+ "<|vision_end|>",
280
+ "<|vision_pad|>",
281
+ "<|image_pad|>",
282
+ "<|video_pad|>"
283
+ ],
284
+ "bos_token": null,
285
+ "chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- set reasoning_instructions = '' %}\n{%- if enable_thinking is undefined or enable_thinking is true %}\n {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}\n {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}\n {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}\n {%- endif %}\n {%- if resolved_reasoning_effort == 'xhigh' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}\n {%- elif resolved_reasoning_effort == 'low' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}\n {%- endif %}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {%- if reasoning_instructions %}\n {{- reasoning_instructions + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n<tool_call>\\n<function=example_function_name>\\n<parameter=example_parameter_1>\\nvalue_1\\n</parameter>\\n<parameter=example_parameter_2>\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n</parameter>\\n</function>\\n</tool_call>\\n\\n<IMPORTANT>\\nReminder:\\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n</IMPORTANT>' }}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {%- if content %}\n {{- '\\n\\n' + content }}\n {%- endif %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {%- if content %}\n {{- '<|im_start|>system\\n' + (reasoning_instructions + '\\n\\n' if reasoning_instructions else '') + content + '<|im_end|>\\n' }}\n {%- elif reasoning_instructions %}\n {{- '<|im_start|>system\\n' + reasoning_instructions + '<|im_end|>\\n' }}\n {%- endif %}\n {%- elif reasoning_instructions %}\n {{- '<|im_start|>system\\n' + reasoning_instructions + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if ns.multi_step_tool %}\n {{- raise_exception('No user query found in messages.') }}\n{%- endif %}\n{%- for message in messages %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" %}\n {%- if not loop.first %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- endif %}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n<think>\\n' + reasoning_content + '\\n</think>\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- else %}\n {{- '<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is defined and tool_call.arguments != '' %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '<parameter=' + args_name + '>\\n' }}\n {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}\n {{- args_value }}\n {{- '\\n</parameter>\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '</function>\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {{- content }}\n {{- '\\n</tool_response>' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '<think>\\n\\n</think>\\n\\n' }}\n {%- else %}\n {{- '<think>\\n' }}\n {%- endif %}\n{%- endif %}",
286
+ "clean_up_tokenization_spaces": false,
287
+ "eos_token": "<|im_end|>",
288
+ "errors": "replace",
289
+ "model_max_length": 262144,
290
+ "pad_token": "<|endoftext|>",
291
+ "split_special_tokens": false,
292
+ "tokenizer_class": "Qwen2Tokenizer",
293
+ "unk_token": null,
294
+ "add_bos_token": false,
295
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
296
+ "extra_special_tokens": {
297
+ "audio_bos_token": "<|audio_start|>",
298
+ "audio_eos_token": "<|audio_end|>",
299
+ "audio_token": "<|audio_pad|>",
300
+ "image_token": "<|image_pad|>",
301
+ "video_token": "<|video_pad|>",
302
+ "vision_bos_token": "<|vision_start|>",
303
+ "vision_eos_token": "<|vision_end|>"
304
+ }
305
+ }
ukisai-banner.png ADDED

Git LFS Details

  • SHA256: 3e9e39451b4586f86a8c7c12dfbba21a5adc16778543d93e39b139ca8627f0c7
  • Pointer size: 131 Bytes
  • Size of remote file: 180 kB
video_preprocessor_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "size": {
3
+ "longest_edge": 25165824,
4
+ "shortest_edge": 4096
5
+ },
6
+ "patch_size": 16,
7
+ "temporal_patch_size": 2,
8
+ "merge_size": 2,
9
+ "image_mean": [
10
+ 0.5,
11
+ 0.5,
12
+ 0.5
13
+ ],
14
+ "image_std": [
15
+ 0.5,
16
+ 0.5,
17
+ 0.5
18
+ ],
19
+ "processor_class": "Qwen3VLProcessor",
20
+ "video_processor_type": "Qwen3VLVideoProcessor"
21
+ }
vocab.json ADDED
The diff for this file is too large to render. See raw diff