daavidhauser commited on
Commit
c1a758b
·
verified ·
1 Parent(s): 6958143

Publish Swift HyperQwen collection with performance and quality comparisons

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +1 -0
  2. LICENSE +233 -0
  3. LICENSE-APACHE-2.0 +202 -0
  4. NOTICE +23 -0
  5. QUANTIZATION_MANIFEST.json +13 -0
  6. README.md +75 -0
  7. RELEASE_MANIFEST.json +433 -0
  8. RUNTIME.md +48 -0
  9. chat_template.jinja +170 -0
  10. config.json +546 -0
  11. evaluation/code/swift15/checkpoint.py +87 -0
  12. evaluation/code/swift15/code_runner.py +59 -0
  13. evaluation/code/swift15/common.py +49 -0
  14. evaluation/code/swift15/corpus.py +154 -0
  15. evaluation/code/swift15/eval_data.py +135 -0
  16. evaluation/code/swift15/evaluate.py +265 -0
  17. evaluation/code/swift15/generate.py +88 -0
  18. evaluation/code/swift15/ifbench_score.py +21 -0
  19. evaluation/code/swift15/measure.py +94 -0
  20. evaluation/code/swift15/quantize.py +183 -0
  21. evaluation/code/swift15/reference.py +41 -0
  22. evaluation/code/swift15/release.py +133 -0
  23. evaluation/code/swift15/report.py +68 -0
  24. evaluation/code/swift15/run.py +162 -0
  25. evaluation/code/swift15/runtime_compat.py +41 -0
  26. evaluation/code/swift15/serve.py +103 -0
  27. evaluation/code/swift15/setup.py +42 -0
  28. evaluation/code/swift15/smoke.py +55 -0
  29. evaluation/code/swift15/test_workflow.py +66 -0
  30. evaluation/environment.json +41 -0
  31. evaluation/manifest.json +72 -0
  32. evaluation/pilot-ids.json +22 -0
  33. evaluation/qwen-fast-full/manifest.json +693 -0
  34. evaluation/qwen-fast-full/summary.json +89 -0
  35. evaluation/qwen-fast-pilot/manifest.json +83 -0
  36. evaluation/qwen-fast-pilot/summary.json +86 -0
  37. evaluation/qwen-fast/perplexity.json +659 -0
  38. evaluation/qwen-fast/server.json +57 -0
  39. evaluation/qwen-fast/speed-c1.json +16 -0
  40. evaluation/swift10-full/manifest.json +697 -0
  41. evaluation/swift10-full/summary.json +88 -0
  42. evaluation/swift10-pilot/manifest.json +87 -0
  43. evaluation/swift10-pilot/summary.json +86 -0
  44. evaluation/swift10/perplexity.json +659 -0
  45. evaluation/swift10/server.json +61 -0
  46. evaluation/swift10/speed-c1.json +16 -0
  47. evaluation/swift15-baseline-full/manifest.json +693 -0
  48. evaluation/swift15-baseline-full/summary.json +88 -0
  49. evaluation/swift15-baseline-pilot/manifest.json +83 -0
  50. evaluation/swift15-baseline-pilot/summary.json +86 -0
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,233 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Swift Open License v1.0
2
+
3
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
4
+
5
+ 1. Definitions.
6
+
7
+ "License" shall mean the terms and conditions for use, reproduction, and
8
+ distribution as defined by this document.
9
+
10
+ "Licensor" shall mean UkisAI.
11
+
12
+ "Legal Entity" shall mean the union of the acting entity and all other entities
13
+ that control, are controlled by, or are under common control with that entity.
14
+ For the purposes of this definition, "control" means (i) the power, direct or
15
+ indirect, to cause the direction or management of such entity, whether by
16
+ contract or otherwise, or (ii) ownership of fifty percent (50%) or more of the
17
+ outstanding shares, or (iii) beneficial ownership of such entity.
18
+
19
+ "You" (or "Your") shall mean an individual or Legal Entity exercising
20
+ permissions granted by this License.
21
+
22
+ "Source" form shall mean the preferred form for making modifications,
23
+ including but not limited to software source code, documentation source,
24
+ configuration files, and model weights in an unquantized, trainable format.
25
+
26
+ "Object" form shall mean any form resulting from mechanical transformation or
27
+ translation of a Source form, including but not limited to compiled object
28
+ code, generated documentation, quantized or otherwise converted model weights,
29
+ and conversions to other media types or file formats.
30
+
31
+ "Base Model" shall mean the Qwen3.8-27B model, Copyright 2026 Alibaba Cloud,
32
+ made available at https://huggingface.co/Qwen/Qwen3.8-27B, including its
33
+ weights, configuration, tokenizer, and chat template, in any form.
34
+
35
+ "Base Model License" shall mean the Apache License, Version 2.0, under which
36
+ the Base Model is made available. A copy is distributed with the Work in the
37
+ file LICENSE-APACHE-2.0.
38
+
39
+ "Swift Contribution" shall mean the modifications to the Base Model authored
40
+ by Licensor, in any form, including without limitation trained weight adapters,
41
+ weight deltas, model weights to the extent they differ from the Base Model, and
42
+ any configuration, documentation, and evaluation materials created by Licensor
43
+ and distributed with the Work.
44
+
45
+ "Work" shall mean the Swift Contribution together with, to the extent of
46
+ Licensor's rights therein, the Derivative Work of the Base Model made available
47
+ by Licensor under this License (as indicated by a copyright notice that is
48
+ included in or attached to the work), in any format made available by Licensor.
49
+
50
+ "Derivative Works" shall mean any work, whether in Source or Object form, that
51
+ is based on (or derived from) the Work and for which the editorial revisions,
52
+ annotations, elaborations, or other modifications represent, as a whole, an
53
+ original work of authorship. For the purposes of this License, Derivative Works
54
+ shall not include works that remain separable from, or merely link (or bind by
55
+ name) to the interfaces of, the Work and Derivative Works thereof.
56
+
57
+ "Contribution" shall mean any work of authorship, including the original
58
+ version of the Work and any modifications or additions to that Work or
59
+ Derivative Works thereof, that is intentionally submitted to Licensor for
60
+ inclusion in the Work by the copyright owner or by an individual or Legal
61
+ Entity authorized to submit on behalf of the copyright owner. For the purposes
62
+ of this definition, "submitted" means any form of electronic, verbal, or
63
+ written communication sent to the Licensor or its representatives, including
64
+ but not limited to communication on electronic mailing lists, source code
65
+ control systems, and issue tracking systems that are managed by, or on behalf
66
+ of, the Licensor for the purpose of discussing and improving the Work, but
67
+ excluding communication that is conspicuously marked or otherwise designated in
68
+ writing by the copyright owner as "Not a Contribution."
69
+
70
+ "Contributor" shall mean Licensor and any individual or Legal Entity on behalf
71
+ of whom a Contribution has been received by Licensor and subsequently
72
+ incorporated within the Work.
73
+
74
+ "Commercial Use" shall mean any use of the Work or a Derivative Work for direct
75
+ or indirect commercial advantage or monetary compensation.
76
+
77
+ "Qualified Non-Profit Organization" shall mean a Legal Entity that is organized
78
+ and operated exclusively for religious, charitable, scientific, testing for
79
+ public safety, literary, or educational purposes, and which is exempt from
80
+ federal income tax under Section 501(c)(3) of the United States Internal
81
+ Revenue Code of 1986, as amended, or any equivalent non-profit or charitable
82
+ organization in a foreign jurisdiction.
83
+
84
+ "Non-Commercial or Research Purposes" shall mean purposes that do not involve
85
+ any use of the Work or a Derivative Work for Commercial Use.
86
+
87
+ "Threshold" shall mean gross revenue of one million United States dollars
88
+ (US$1,000,000) or more, measured over the most recently completed fiscal year
89
+ of You together with every Legal Entity that controls, is controlled by, or is
90
+ under common control with You.
91
+
92
+ 2. Grant of Copyright License. Subject to the terms and conditions of this
93
+ License, including the Commercial Use limitation set forth in Section 5, each
94
+ Contributor hereby grants to You a perpetual, worldwide, non-exclusive,
95
+ no-charge, royalty-free, irrevocable copyright license to reproduce, prepare
96
+ Derivative Works of, publicly display, publicly perform, sublicense, and
97
+ distribute the Work and such Derivative Works in Source or Object form.
98
+
99
+ 3. Grant of Patent License. Subject to the terms and conditions of this
100
+ License, including the Commercial Use limitation set forth in Section 5, each
101
+ Contributor hereby grants to You a perpetual, worldwide, non-exclusive,
102
+ no-charge, royalty-free, irrevocable (except as stated in this section) patent
103
+ license to make, have made, use, offer to sell, sell, import, and otherwise
104
+ transfer the Work, where such license applies only to those patent claims
105
+ licensable by such Contributor that are necessarily infringed by their
106
+ Contribution(s) alone or by combination of their Contribution(s) with the Work
107
+ to which such Contribution(s) was submitted. If You institute patent litigation
108
+ against any entity (including a cross-claim or counterclaim in a lawsuit)
109
+ alleging that the Work or a Contribution incorporated within the Work
110
+ constitutes direct or contributory patent infringement, then any patent
111
+ licenses granted to You under this License for that Work shall terminate as of
112
+ the date such litigation is filed.
113
+
114
+ 4. Redistribution. You may reproduce and distribute copies of the Work or
115
+ Derivative Works thereof in any medium, with or without modifications, and in
116
+ Source or Object form, provided that You meet the following conditions:
117
+
118
+ (a) You must give any other recipients of the Work or Derivative Works a copy
119
+ of this License; and
120
+
121
+ (b) You must cause any modified files to carry prominent notices stating that
122
+ You changed the files; and
123
+
124
+ (c) You must retain, in the Source form of any Derivative Works that You
125
+ distribute, all copyright, patent, trademark, and attribution notices from the
126
+ Source form of the Work, excluding those notices that do not pertain to any
127
+ part of the Derivative Works; and
128
+
129
+ (d) If the Work includes a "NOTICE" text file as part of its distribution, then
130
+ any Derivative Works that You distribute must include a readable copy of the
131
+ attribution notices contained within such NOTICE file, excluding those notices
132
+ that do not pertain to any part of the Derivative Works, in at least one of the
133
+ following places: within a NOTICE text file distributed as part of the
134
+ Derivative Works; within the Source form or documentation, if provided along
135
+ with the Derivative Works; or, within a display generated by the Derivative
136
+ Works, if and wherever such third-party notices normally appear. The contents
137
+ of the NOTICE file are for informational purposes only and do not modify the
138
+ License. You may add Your own attribution notices within Derivative Works that
139
+ You distribute, alongside or as an addendum to the NOTICE text from the Work,
140
+ provided that such additional attribution notices cannot be construed as
141
+ modifying the License; and
142
+
143
+ (e) If the copy You distribute contains any portion of the Base Model
144
+ (including merged, quantized, or otherwise converted weights that incorporate
145
+ the Base Model), You must also give recipients a copy of the Base Model License
146
+ and must comply with the Base Model License with respect to the Base Model.
147
+
148
+ You may add Your own copyright statement to Your modifications and may provide
149
+ additional or different license terms and conditions for use, reproduction, or
150
+ distribution of Your modifications, or for any such Derivative Works as a
151
+ whole, provided Your use, reproduction, and distribution of the Work otherwise
152
+ complies with the conditions stated in this License, and provided that Section
153
+ 5 continues to apply to the Swift Contribution contained in any such Derivative
154
+ Works.
155
+
156
+ 5. Commercial Use Limitation.
157
+
158
+ (a) The rights granted under this License for Commercial Use are conditioned
159
+ upon You or Your Legal Entity not exceeding the Threshold.
160
+
161
+ (b) Any Commercial Use of the Work or a Derivative Work by a Legal Entity that
162
+ exceeds the Threshold is not licensed under this License.
163
+
164
+ (c) The Threshold shall not apply to a Qualified Non-Profit Organization's use
165
+ of the Work or a Derivative Work for Non-Commercial or Research Purposes.
166
+
167
+ (d) A Legal Entity that exceeds the Threshold may obtain a separate written
168
+ license for Commercial Use from Licensor (the "Swift Enterprise License").
169
+ Contact: https://ukisai.com/contact.
170
+
171
+ 6. Base Model Rights. The Work incorporates the Base Model. Nothing in this
172
+ License limits, restricts, conditions, or modifies any rights You have in the
173
+ Base Model under the Base Model License, and the Base Model remains available
174
+ to You from its licensor under the Base Model License. Sections 5 and 12 of
175
+ this License apply solely to the Swift Contribution and to the Work or any
176
+ Derivative Work to the extent it contains, incorporates, or is derived from the
177
+ Swift Contribution.
178
+
179
+ 7. Submission of Contributions. Unless You explicitly state otherwise, any
180
+ Contribution intentionally submitted for inclusion in the Work by You to the
181
+ Licensor shall be under the terms and conditions of this License, without any
182
+ additional terms or conditions. Notwithstanding the above, nothing herein shall
183
+ supersede or modify the terms of any separate license agreement you may have
184
+ executed with Licensor regarding such Contributions.
185
+
186
+ 8. Trademarks. This License does not grant permission to use the trade names,
187
+ trademarks, service marks, or product names of the Licensor (including "UkisAI"
188
+ and "Swift"), except for the reasonable and customary use in describing the
189
+ origin of the Work and reproducing the content of the NOTICE file.
190
+
191
+ 9. Disclaimer of Warranty. Unless required by applicable law or agreed to in
192
+ writing, Licensor provides the Work (and each Contributor provides its
193
+ Contributions) on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
194
+ KIND, either express or implied, including, without limitation, any warranties
195
+ or conditions of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
196
+ PARTICULAR PURPOSE. You are solely responsible for determining the
197
+ appropriateness of using or redistributing the Work and assume any risks
198
+ associated with Your exercise of permissions under this License.
199
+
200
+ 10. Limitation of Liability. In no event and under no legal theory, whether in
201
+ tort (including negligence), contract, or otherwise, unless required by
202
+ applicable law (such as deliberate and grossly negligent acts) or agreed to in
203
+ writing, shall any Contributor be liable to You for damages, including any
204
+ direct, indirect, special, incidental, or consequential damages of any
205
+ character arising as a result of this License or out of the use or inability
206
+ to use the Work (including but not limited to damages for loss of goodwill,
207
+ work stoppage, computer failure or malfunction, or any and all other commercial
208
+ damages or losses), even if such Contributor has been advised of the
209
+ possibility of such damages.
210
+
211
+ 11. Accepting Warranty or Additional Liability. While redistributing the Work
212
+ or Derivative Works thereof, You may choose to offer, and charge a fee for,
213
+ acceptance of support, warranty, indemnity, or other liability obligations
214
+ and/or rights consistent with this License. However, in accepting such
215
+ obligations, You may act only on Your own behalf and on Your sole
216
+ responsibility, not on behalf of any other Contributor, and only if You agree
217
+ to indemnify, defend, and hold each Contributor harmless for any liability
218
+ incurred by, or claims asserted against, such Contributor by reason of your
219
+ accepting any such warranty or additional liability.
220
+
221
+ 12. Termination. This License will terminate automatically and immediately if
222
+ You fail to comply with any of its terms and conditions. Upon termination, You
223
+ must cease all use of the Swift Contribution and of any Work or Derivative
224
+ Works containing it, and delete all copies in Your possession. Termination does
225
+ not affect Your rights in the Base Model under the Base Model License.
226
+
227
+ END OF TERMS AND CONDITIONS
228
+
229
+ APPENDIX: Notice for redistributors (quantizations, conversions, merges).
230
+
231
+ Copyright 2026 UkisAI. Swift Contribution licensed under the Swift Open
232
+ License v1.0 (https://huggingface.co/ukisai/Swift-Qwen3.8-27b/blob/main/LICENSE).
233
+ Derivative of Qwen3.8-27B, Copyright 2026 Alibaba Cloud, Apache License 2.0.
LICENSE-APACHE-2.0 ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
NOTICE ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Swift-Qwen3.8-27B
2
+ Copyright 2026 UkisAI
3
+
4
+ UkisAI's contribution (the "Swift Contribution") is licensed under the
5
+ Swift Open License v1.0. See LICENSE.
6
+
7
+ This model is a Derivative Work of Qwen3.8-27B
8
+ https://huggingface.co/Qwen/Qwen3.8-27B
9
+ Copyright 2026 Alibaba Cloud
10
+ Licensed under the Apache License, Version 2.0. See LICENSE-APACHE-2.0.
11
+
12
+ Changes made by UkisAI (Apache License 2.0, Section 4(b) change notice):
13
+ - model-*.safetensors, model.safetensors.index.json: model weights were
14
+ fine-tuned by UkisAI (LoRA adapter trained by UkisAI and merged into the
15
+ Base Model weights).
16
+ - generation_config.json: added "min_p": 0 and "repetition_penalty": 1.0.
17
+ - README.md: replaced. ukisai-banner.png and swift-speed-demo.mp4 added.
18
+ - All other files (config.json, chat_template.jinja, tokenizer.json,
19
+ tokenizer_config.json, vocab.json, merges.txt, preprocessor_config.json,
20
+ video_preprocessor_config.json) are unmodified from Qwen3.8-27B and
21
+ remain under the Apache License, Version 2.0.
22
+
23
+ HyperQwen conversion by daavidhauser: INT8 embeddings, main output head and MTP linear weights; reference MTP draft shortlist. Upstream AWQ body retained.
QUANTIZATION_MANIFEST.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "source_repo": "TheUnderscore/Swift-Qwen3.8-27b-W4A16-AWQ",
3
+ "source_revision": "6ced337b9c99adddf9871ff84abc78a7f1acafad",
4
+ "revision_evidence": "Preserved Hugging Face local download metadata",
5
+ "license_source_repo": "ukisai/Swift-Qwen3.8-27b",
6
+ "license_source_revision": "6bc57e4eca31ee61d4e92a631978655a78bfa465",
7
+ "variant": "HyperQwen INT8 heads",
8
+ "body": "Preserved upstream asymmetric AWQ INT4 group128",
9
+ "embedding_bits": 8,
10
+ "lm_head_bits": 8,
11
+ "mtp_bits": 8,
12
+ "draft_vocabulary": "HyperQwen reference shortlist"
13
+ }
README.md ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: swift-open-license-1.0
4
+ license_link: LICENSE
5
+ base_model: ukisai/Swift-Qwen3.8-27b
6
+ base_model_relation: quantized
7
+ library_name: vllm
8
+ pipeline_tag: text-generation
9
+ tags:
10
+ - compressed-tensors
11
+ - awq
12
+ - hyperqwen
13
+ - efficient-thinking
14
+ - int8-heads
15
+ ---
16
+
17
+ Model card written by GPT-6 Astra:
18
+
19
+ # Swift 1.0 HyperQwen
20
+
21
+ Swift 1.0 adapted for **HyperQwen serving on a single RTX 3090 24 GB**, with FP8 KV cache, MTP speculative decoding, and a **150k-token configured context**.
22
+
23
+ In our 630-task comparison, this model used **42% fewer output tokens** and achieved **39% lower average request completion time** than the tested Qwen fast checkpoint. Swift's efficient-reasoning training reduces how much text the model generates; HyperQwen provides the serving runtime. Both matter for getting tasks finished quickly.
24
+
25
+ [Browse all three Swift HyperQwen variants](https://huggingface.co/collections/daavidhauser/swift-for-hyperqwen-rtx-3090-benchmarks-6abac232566f58d5e7c5a046).
26
+
27
+ ## Performance
28
+
29
+ | Measurement | Qwen fast | Swift 1.0 | Swift 1.5 INT8 heads | Swift 1.5 fast |
30
+ |---|---:|---:|---:|---:|
31
+ | **Average request time ↓** | **108.1 s** | **66.2 s** | **72.2 s** | **68.2 s** |
32
+ | **Average output tokens/task ↓** | **8,985** | **5,245** | **5,751** | **5,669** |
33
+ | Median decode tokens/s ↑ | 112.1 | 105.9 | 104.0 | 107.2 |
34
+ | Total output tokens, 630 tasks | 5.66M | 3.30M | 3.62M | 3.57M |
35
+
36
+ Request times come from the full quality suite at **two concurrent requests**, including failures. They are not single-user latency measurements. Output counts include reasoning. TPS is a separate **single-request** test: four short prompts repeated twice, with 512 greedy output tokens each. The configured context is 150k; these TPS figures are not measured at 150k occupied context.
37
+
38
+ ## Quality
39
+
40
+ | Test | Qwen fast | Swift 1.0 | Swift 1.5 INT8 heads | Swift 1.5 fast |
41
+ |---|---:|---:|---:|---:|
42
+ | **GSM8K — 200-question subset** | 97.5% | 98.0% | 98.0% | 97.5% |
43
+ | **IFBench — 300 prompts, strict** | 74.0% | 73.3% | 73.7% | 72.3% |
44
+ | **LiveCodeBench — 100-problem subset** | 90% | 89% | 89% | 91% |
45
+ | **Custom tool-call/JSON checks — 30 tasks** | 29/30 | 28/30 | 30/30 | 30/30 |
46
+ | Perplexity, English/Danish/Python ↓ | 8.143 | 8.215 | 8.252 | 8.318 |
47
+ | Truncated answers, counted wrong | 2 | 1 | 1 | 0 |
48
+
49
+ - **GSM8K:** the first 200 questions from the [test split](https://huggingface.co/datasets/openai/gsm8k), with thinking disabled.
50
+ - **IFBench:** all 300 prompts in the pinned [IFBench test dataset](https://huggingface.co/datasets/allenai/IFBench_test), scored with the official strict prompt-level verifier.
51
+ - **LiveCodeBench:** a frozen [v6-era dataset](https://huggingface.co/datasets/livecodebench/code_generation_lite) subset of Python stdin/stdout problems: 34 easy, 33 medium, 33 hard. Scored against supplied public/private tests with a custom judge; not a full official LiveCodeBench result.
52
+ - **Tool-call/JSON checks:** 20 custom weather-tool tasks checking the function name, arguments, Celsius-to-Fahrenheit conversion and final JSON; plus 10 JSON inventory-filtering tasks. These are integration checks, not an external agent benchmark.
53
+ - **Perplexity:** 32,646 scored tokens from English Wikipedia, Danish web text and Python source; lower is better.
54
+
55
+ All models used the same serving settings and task budgets. Thinking tests used xhigh effort, temperature 1.0, top_p 0.95, top_k 20 and seed 15027; GSM8K/tool checks were greedy. Outputs were capped at 128,000 tokens per call. These are single-seed local results; small accuracy differences do not establish a universal ranking. Qwen fast is a quantized AutoRound reference, not BF16 Qwen. Swift 1.0 was the most token-efficient model on this task mix.
56
+
57
+ ## Changes from upstream
58
+
59
+ The **upstream AWQ INT4 body is preserved**, and embeddings are converted to **INT8**. The main output head (`lm_head`) and MTP linear weights are converted to **INT8**. MTP uses HyperQwen's reference draft shortlist. This is the **INT8-head HyperQwen variant**, rather than the untouched upstream checkpoint. The target model retains its full vocabulary; the shortlist only limits speculative proposals. No additional fine-tuning is applied by this release.
60
+
61
+ The upstream AWQ quantization is [TheUnderscore/Swift-Qwen3.8-27b-W4A16-AWQ](https://huggingface.co/TheUnderscore/Swift-Qwen3.8-27b-W4A16-AWQ). Swift's training is by [UkisAI](https://huggingface.co/ukisai); the serving runtime is [HyperQwen](https://github.com/syv-ai/HyperQwen). This is an independent conversion with local evaluation by daavidhauser.
62
+
63
+ ## Setup
64
+
65
+ Requires the **patched HyperQwen runtime**, not stock vLLM or GGUF tools. From an installed HyperQwen checkout:
66
+
67
+ ```bash
68
+ hf download daavidhauser/Swift-1.0-Qwen3.8-27B-W4A16-HyperQwen --local-dir models/Swift-1.0-Qwen3.8-27B-W4A16-HyperQwen
69
+ MODEL="$PWD/models/Swift-1.0-Qwen3.8-27B-W4A16-HyperQwen" CTX=long MAX_LEN=150000 SPEC=mtp \
70
+ bash single-user/start_qwen.sh
71
+ ```
72
+
73
+ See [RUNTIME.md](RUNTIME.md) for the pinned runtime, launcher snapshot, complete evaluated settings and installation notes. The checkpoint is already converted: do not requantize its heads. Vision weights are retained, but the evaluation is text-only. Multi-user batch settings were not benchmarked in this campaign.
74
+
75
+ Detailed results and evaluation code are in [evaluation/](evaluation/). Quantization/source provenance is included with the model. The upstream [Swift Open License v1.0](LICENSE), [Apache 2.0 base-model license](LICENSE-APACHE-2.0), and [NOTICE](NOTICE) are retained.
RELEASE_MANIFEST.json ADDED
@@ -0,0 +1,433 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "repo": "daavidhauser/Swift-1.0-Qwen3.8-27B-W4A16-HyperQwen",
3
+ "files": {
4
+ "LICENSE": {
5
+ "size": 13304,
6
+ "sha256": "915ffe920f90088d15986bb6f7fab02e08b9a167cfbde65b012741ad88010e4f"
7
+ },
8
+ "LICENSE-APACHE-2.0": {
9
+ "size": 11544,
10
+ "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a"
11
+ },
12
+ "NOTICE": {
13
+ "size": 1172,
14
+ "sha256": "2fcf73df7395f673cad9b4ed6ae0de2d6b012d8e7d62cd64a361d45ac3cb4101"
15
+ },
16
+ "QUANTIZATION_MANIFEST.json": {
17
+ "size": 543,
18
+ "sha256": "9646e189bb9cfb1f35467eece46b59faf73c5eb0cc9800993343947cd875594d"
19
+ },
20
+ "README.md": {
21
+ "size": 5569,
22
+ "sha256": "d150cd7359b9b1923664c39438ad1a662ce5312d994961e5c2cca8e4ec86d6e0"
23
+ },
24
+ "RUNTIME.md": {
25
+ "size": 2588,
26
+ "sha256": "2f34c2a1daa4d900f39b961ad6179de530df2c5f44d853e43400514dfe805200"
27
+ },
28
+ "chat_template.jinja": {
29
+ "size": 8952,
30
+ "sha256": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041"
31
+ },
32
+ "config.json": {
33
+ "size": 22357,
34
+ "sha256": "13bcc57aaa12c046cf379d0680f6a8dd144db48696c294798daae0d485a1e255"
35
+ },
36
+ "evaluation/code/swift15/checkpoint.py": {
37
+ "size": 3656,
38
+ "sha256": "c43707f3623659cd882cfbb57506b152912f42f2a043dfd148c1cf829090f28d"
39
+ },
40
+ "evaluation/code/swift15/code_runner.py": {
41
+ "size": 2144,
42
+ "sha256": "23d52e36543766bb33d33f11adad3772e7836d5170e83f68e326cc2d00b55fd9"
43
+ },
44
+ "evaluation/code/swift15/common.py": {
45
+ "size": 1439,
46
+ "sha256": "0410dc972e74d8bbc05e090b0543426e4ea57ed707a80a114ea2e658822e1e8c"
47
+ },
48
+ "evaluation/code/swift15/corpus.py": {
49
+ "size": 7800,
50
+ "sha256": "d5b420de1f3a1d4be97a44321abdd393247b5d78a56199ad49a62058c98d9050"
51
+ },
52
+ "evaluation/code/swift15/eval_data.py": {
53
+ "size": 8097,
54
+ "sha256": "2a4497758d456a3cdd930aa711d0c510b24a2eaf6eb11be3da3e94481d4a166e"
55
+ },
56
+ "evaluation/code/swift15/evaluate.py": {
57
+ "size": 14113,
58
+ "sha256": "4d553cf10b42cbff0eeacfe72b3b6df87405aa2604fcbe19a5a0a78133cd18af"
59
+ },
60
+ "evaluation/code/swift15/generate.py": {
61
+ "size": 4412,
62
+ "sha256": "55e2659ce67295a4c57ffd12beed9dc0246808d908939c4113f7d1bb506af089"
63
+ },
64
+ "evaluation/code/swift15/ifbench_score.py": {
65
+ "size": 876,
66
+ "sha256": "1b97f5859020d8d992944a85357ee7477f8740e65574d5946d2305889244a04d"
67
+ },
68
+ "evaluation/code/swift15/measure.py": {
69
+ "size": 4726,
70
+ "sha256": "f955ace33396730c9fdf6fe5d424b6a2f77eb0fc58c83b3bd84b0c804c7a3b37"
71
+ },
72
+ "evaluation/code/swift15/quantize.py": {
73
+ "size": 8098,
74
+ "sha256": "a3f65ca5de15d54cf626cde340365e503462d1a33ae8525aa55f6159195886f7"
75
+ },
76
+ "evaluation/code/swift15/reference.py": {
77
+ "size": 1786,
78
+ "sha256": "0433adf523a676c0cb30c5ad25b90bdc5a4565c2b69180e171aa4a4d59c15bf8"
79
+ },
80
+ "evaluation/code/swift15/release.py": {
81
+ "size": 7409,
82
+ "sha256": "c5129121b4ba333b0c4dd2694fb69076cdc75e4b9cda8a6572b3a449c32c3b5e"
83
+ },
84
+ "evaluation/code/swift15/report.py": {
85
+ "size": 4938,
86
+ "sha256": "873db1c8e86fb281b15f0b3c2e66adfb3cf5da652cf0139788acd00f8bd2ac02"
87
+ },
88
+ "evaluation/code/swift15/run.py": {
89
+ "size": 9281,
90
+ "sha256": "e5d42332192e5bafdde88ae4acc20802f5c5477e6447ff5165fb6b836aa78ea2"
91
+ },
92
+ "evaluation/code/swift15/runtime_compat.py": {
93
+ "size": 1885,
94
+ "sha256": "bcbc7351a6a4719bf38c06658bb12cbcfc9eb31f9280fc9b916782832544dbe6"
95
+ },
96
+ "evaluation/code/swift15/serve.py": {
97
+ "size": 6020,
98
+ "sha256": "c705174eda86e57ae1a51c0b2e7a93f9b0e1f37fa32217a9739b569f52d2d56d"
99
+ },
100
+ "evaluation/code/swift15/setup.py": {
101
+ "size": 2219,
102
+ "sha256": "7de02ccb5575e0ac2b62d33842d28aacfc83d63fc96eedad9cbedf1129e90495"
103
+ },
104
+ "evaluation/code/swift15/smoke.py": {
105
+ "size": 3162,
106
+ "sha256": "e2d4a01d6613e667ceeef0d1a03ef1cec04edf3ab0518aaa4a006d419a91f1fe"
107
+ },
108
+ "evaluation/code/swift15/test_workflow.py": {
109
+ "size": 3468,
110
+ "sha256": "8d2dcb057318d482530ef9bf7b4bcf0d59347b594531e0df18b81dbea0beaddd"
111
+ },
112
+ "evaluation/environment.json": {
113
+ "size": 3282,
114
+ "sha256": "2e9ac10516648b0c634eaca007d58456d241e0c1ef1b6734fa51671bf315a841"
115
+ },
116
+ "evaluation/manifest.json": {
117
+ "size": 2685,
118
+ "sha256": "82c1c6eaad6c06ad91e20cac5f9dae6b74ceaba0a3063c868a6a4313366e7948"
119
+ },
120
+ "evaluation/pilot-ids.json": {
121
+ "size": 354,
122
+ "sha256": "44cffc01ab88b366f20da82d3150c7de8eb1d7801cd1381570e32313f120da41"
123
+ },
124
+ "evaluation/qwen-fast/perplexity.json": {
125
+ "size": 12570,
126
+ "sha256": "9db1599421dc916b97602a8921dbd219b815a5a76fc7def41f508500eb10e7e7"
127
+ },
128
+ "evaluation/qwen-fast/server.json": {
129
+ "size": 1942,
130
+ "sha256": "b2988235fd03e4cd30366b5ca418ba5bb687c1ab58c1eb29e3e53c57612ac0ae"
131
+ },
132
+ "evaluation/qwen-fast/speed-c1.json": {
133
+ "size": 630,
134
+ "sha256": "3d70c2868a2f2832e6bf7562abd3fa4589389baf87e02dbd89c76b154e49238d"
135
+ },
136
+ "evaluation/qwen-fast-full/manifest.json": {
137
+ "size": 14395,
138
+ "sha256": "3de5a8af19d01a53594d7f29349e19bf90f1bbe8d935e82fe33aa0cf8325391b"
139
+ },
140
+ "evaluation/qwen-fast-full/summary.json": {
141
+ "size": 3134,
142
+ "sha256": "a2ffcd9ed991279b5639274596a5139b39553d407753adc48ff465e55fd2eaf6"
143
+ },
144
+ "evaluation/qwen-fast-pilot/manifest.json": {
145
+ "size": 2663,
146
+ "sha256": "f06f4539feaa46b92d559427cd13809ecf3e8b0fc44b15768e54392a0ce9c34f"
147
+ },
148
+ "evaluation/qwen-fast-pilot/summary.json": {
149
+ "size": 2979,
150
+ "sha256": "8a0ec26f335c5a0c98bac1db6bd852a5292b0b5d1bfcc6a218a0b9aa1ce9b1c7"
151
+ },
152
+ "evaluation/swift10/perplexity.json": {
153
+ "size": 12560,
154
+ "sha256": "fe393bd4962d43f9cef2a08ea40c0dd2f15e6695b87a09a20e184d448d941775"
155
+ },
156
+ "evaluation/swift10/server.json": {
157
+ "size": 2041,
158
+ "sha256": "d6004a26531972be5407c0f542ec0b708c931e21f67801e5206906d4cb148547"
159
+ },
160
+ "evaluation/swift10/speed-c1.json": {
161
+ "size": 632,
162
+ "sha256": "a82f761f6a6ca5446808b1e8b129fee4e5b3b733787e286a7829b23f6dd1722f"
163
+ },
164
+ "evaluation/swift10-full/manifest.json": {
165
+ "size": 14501,
166
+ "sha256": "6af95e530b920a84a3c5f5dde5b76bc2c791d0a4cc5b3043686bd6946cac4acf"
167
+ },
168
+ "evaluation/swift10-full/summary.json": {
169
+ "size": 3125,
170
+ "sha256": "1c431ee8cc16e36090152526f8553c3e76cb5f66e5ddd5c6df1f90c3b3426add"
171
+ },
172
+ "evaluation/swift10-pilot/manifest.json": {
173
+ "size": 2769,
174
+ "sha256": "83df9bd1a41e0d96f62039ad1941593e4cafc57c52f63851eeec4523048f4315"
175
+ },
176
+ "evaluation/swift10-pilot/summary.json": {
177
+ "size": 2981,
178
+ "sha256": "3f8aed7d263dc65abb0095bc3af73515052645886f5f0565e6c21e82afa27bf8"
179
+ },
180
+ "evaluation/swift15-baseline/perplexity.json": {
181
+ "size": 12553,
182
+ "sha256": "6f63e857421957c76b3ca290ec8c95a62fd4374e3839adc2cdcdca54f14786fa"
183
+ },
184
+ "evaluation/swift15-baseline/server.json": {
185
+ "size": 1937,
186
+ "sha256": "9063a7b84cd89e83672d673d5fc30ff563d01adebd1a705fcae70c33b5ba47b4"
187
+ },
188
+ "evaluation/swift15-baseline/speed-c1.json": {
189
+ "size": 629,
190
+ "sha256": "079f1c870da2484a43ec214ee2ced3723b2636c83eaed4291636e121b4e81e3a"
191
+ },
192
+ "evaluation/swift15-baseline-full/manifest.json": {
193
+ "size": 14389,
194
+ "sha256": "b122ac2835f4c868522aa6982e1c44e2090ae176b27293cd8a71040108751d18"
195
+ },
196
+ "evaluation/swift15-baseline-full/summary.json": {
197
+ "size": 3091,
198
+ "sha256": "cd0ec9d98535c6c32c2ae9f32798f81c0dae628ebda0d1d56f6005df85edb91b"
199
+ },
200
+ "evaluation/swift15-baseline-pilot/manifest.json": {
201
+ "size": 2657,
202
+ "sha256": "588d9a88f35aa9207934bde6cbce4e0956868287316f964585b784b9d4f7cec3"
203
+ },
204
+ "evaluation/swift15-baseline-pilot/summary.json": {
205
+ "size": 2982,
206
+ "sha256": "511e69269b8d98d64876ab7033e124fc1d8aca196b022fbd27b34f65867ac8ef"
207
+ },
208
+ "evaluation/swift15-fast/perplexity.json": {
209
+ "size": 12561,
210
+ "sha256": "3fa2ac71c6c24072036837c3ae0afcf96ce03e85a09b276874c75432aeb091a1"
211
+ },
212
+ "evaluation/swift15-fast/server.json": {
213
+ "size": 1940,
214
+ "sha256": "53099847622dbdd7416196bfd8a7591aca1adb64db92049cf2be6ffdf489536d"
215
+ },
216
+ "evaluation/swift15-fast/speed-c1.json": {
217
+ "size": 629,
218
+ "sha256": "88d292f00ba50773e80c0dbf6ba9a1e18d3413737221d2ff8c5a969749747108"
219
+ },
220
+ "evaluation/swift15-fast-full/manifest.json": {
221
+ "size": 14393,
222
+ "sha256": "d242d0bd7f947c37a74430190ce4a1309c2ae8e2f5ba76313e1ba056044aa134"
223
+ },
224
+ "evaluation/swift15-fast-full/summary.json": {
225
+ "size": 2969,
226
+ "sha256": "3ffd7dc76fa4105089688731c6a8365519f80567acbca3b1bc15b64e93de7ab3"
227
+ },
228
+ "evaluation/swift15-fast-pilot/manifest.json": {
229
+ "size": 2661,
230
+ "sha256": "13203d3b56258d785ea74ddd3f89322edb6c58245f7d4ecf60756a7af66d4f60"
231
+ },
232
+ "evaluation/swift15-fast-pilot/summary.json": {
233
+ "size": 2979,
234
+ "sha256": "fe44f8972f40cf5236ca93d1ee7362cfea238a03e58f09c6e033312820fc8d89"
235
+ },
236
+ "generation_config.json": {
237
+ "size": 221,
238
+ "sha256": "0b65d40c797e19ad9bbd38538f1b45155b2832ffaffa7a69df96ddf57213fc9b"
239
+ },
240
+ "hyperqwen_provenance/upstream-README.md": {
241
+ "size": 3468,
242
+ "sha256": "17d82fcb87967b726b88c9e3cb47e0ef94815aae2abf26306ab9d6a053c86e37"
243
+ },
244
+ "merges.txt": {
245
+ "size": 3353259,
246
+ "sha256": "a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d"
247
+ },
248
+ "model-00001-of-00007.safetensors": {
249
+ "size": 2212761608,
250
+ "sha256": "6872c54bf590710482a55f2a4fc7648a55242e819785ea4f64a91064ee933ba4"
251
+ },
252
+ "model-00002-of-00007.safetensors": {
253
+ "size": 2186190664,
254
+ "sha256": "a44afcd2154f6bac8141a8af48ba7d4788584544a088da6a3112085d695acbfe"
255
+ },
256
+ "model-00003-of-00007.safetensors": {
257
+ "size": 2179699752,
258
+ "sha256": "07b0a5737261377859be9cb4fa19b0a0dbadb5b158bea0179367da9953de947b"
259
+ },
260
+ "model-00004-of-00007.safetensors": {
261
+ "size": 2179699760,
262
+ "sha256": "bf91ce93aa363b88392fbe32c4f785222de13c06e376768ed56ea60330a9b67e"
263
+ },
264
+ "model-00005-of-00007.safetensors": {
265
+ "size": 2179699760,
266
+ "sha256": "c3deb171898650ccc806f4a2e21c1aee5cea8efdce8fe149ab5f0bfc9049a83f"
267
+ },
268
+ "model-00006-of-00007.safetensors": {
269
+ "size": 2186211856,
270
+ "sha256": "007e4ccf91e5310b19a335be197342daffbe5bad46fc92dc4a0d2d37abb1c5b4"
271
+ },
272
+ "model-00007-of-00007.safetensors": {
273
+ "size": 3071134144,
274
+ "sha256": "a942b481417c1ecd24802576969ac66dd003d6bb1a734a1fde438e45e2e1bcb1"
275
+ },
276
+ "model-nonquant.safetensors": {
277
+ "size": 431364472,
278
+ "sha256": "9710e90c273e7cd36d22e8cdb635a81f8d78da5c79dd1bd217445137ece29b9d"
279
+ },
280
+ "model.safetensors.index.json": {
281
+ "size": 243539,
282
+ "sha256": "30bde9db796be88667575f52843af2a23538008cadfeb7d58cd1fd2201bd07b6"
283
+ },
284
+ "model_extra_tensors.safetensors": {
285
+ "size": 212992352,
286
+ "sha256": "e2c1a453b550d9e3c9d5c8df5c6fad58255128acdef570d493a5f376c2123e43"
287
+ },
288
+ "mtp_draft_vocab_ids.pt": {
289
+ "size": 329341,
290
+ "sha256": "8af9028616e41932e0c949cd5364d2da0a69d15375af9bb404cab1e4d723f581"
291
+ },
292
+ "preprocessor_config.json": {
293
+ "size": 390,
294
+ "sha256": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516"
295
+ },
296
+ "runtime/BASE_REVISION.txt": {
297
+ "size": 41,
298
+ "sha256": "c1f6fbc4227edb7daa83504c936afc8e23d8e31e8b62ff375fbbb5e29e81e546"
299
+ },
300
+ "runtime/LICENSE": {
301
+ "size": 11358,
302
+ "sha256": "cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30"
303
+ },
304
+ "runtime/batch/start_qwen.sh": {
305
+ "size": 11677,
306
+ "sha256": "ac6a84b455317537a230bc190ccc06dda169dcdbdeee4bebeb51e76032972944"
307
+ },
308
+ "runtime/patches/dflash2-backport.patch": {
309
+ "size": 40396,
310
+ "sha256": "2e937ef748c1942ab04ace05de884e7c24041d34b23e5be8428cac6939ac6875"
311
+ },
312
+ "runtime/patches/dflash2-lookup-drafting.patch": {
313
+ "size": 46770,
314
+ "sha256": "5df09ef03b592d4a2c3b47dd8d2dbf8862fa7383d68aef3dd6ef1d97d29ce196"
315
+ },
316
+ "runtime/patches/dflash2-ngram-chains.patch": {
317
+ "size": 19037,
318
+ "sha256": "555b5a75d9023c99b2cd17634ac5b1dc8505f9275d857b1eba08c2d9cbf2df6e"
319
+ },
320
+ "runtime/patches/dflash2-prewarm.patch": {
321
+ "size": 5181,
322
+ "sha256": "e1e8012fa2c304c6948c19fc0071332f80eebcff802b402255445a30d553ab92"
323
+ },
324
+ "runtime/patches/hybrid-kv-groups-v2-cudagraph.patch": {
325
+ "size": 6306,
326
+ "sha256": "143825dd9744ab81dabcb438ec0bf974b6e0be8d0d7b0ffbbdfbf7c16a549664"
327
+ },
328
+ "runtime/patches/hybrid-sw-block-promote.patch": {
329
+ "size": 7828,
330
+ "sha256": "10876ac706546e74fe74b8964b90576a7b36f90a2f35a856997c558d7817f154"
331
+ },
332
+ "runtime/patches/int4-kv-per-token-head.patch": {
333
+ "size": 7038,
334
+ "sha256": "c03ff10c8c997b355f04fc521827ee2fa99c332ce4127cc2d878b749b79ca3d9"
335
+ },
336
+ "runtime/patches/mamba-align-checkpoint-order.patch": {
337
+ "size": 12020,
338
+ "sha256": "515d9bf76e860c95d832d615bdab4a97b71e2b0412b4ccc2445f11535ddddf45"
339
+ },
340
+ "runtime/patches/marlin-int8-layer-select.patch": {
341
+ "size": 3048,
342
+ "sha256": "833405e2ed2916529eb20184c38243c843378935522b74bab8c39707bf3ee800"
343
+ },
344
+ "runtime/patches/marlin-int8-negative-scales.patch": {
345
+ "size": 2926,
346
+ "sha256": "4cb8a064c706cc62dc77224479898da58864fbfff6b65012d90aa8326035300d"
347
+ },
348
+ "runtime/patches/marlin-repack-staged-sm80.patch": {
349
+ "size": 6841,
350
+ "sha256": "1588e2f10b5f82194d449483e766c0b39c84ba522d1623d39caa9502040f1fab"
351
+ },
352
+ "runtime/patches/marlin-tune-table.patch": {
353
+ "size": 4422,
354
+ "sha256": "1cac17f12e4ce389b0cb0fc733ceceb0a8cf60c02696ee378dc1d8d0c1ef8914"
355
+ },
356
+ "runtime/patches/offload-dflash-eagle-groups.patch": {
357
+ "size": 5406,
358
+ "sha256": "f2791af64d8066b31e4250d866c4322f45670d75522aff221307281c97c73156"
359
+ },
360
+ "runtime/patches/offload-wsl2-devptr.patch": {
361
+ "size": 4778,
362
+ "sha256": "8c6f9e3e5723571d1f33925794435e55062b684b2706d9284d1c5b94778d34db"
363
+ },
364
+ "runtime/patches/qwen3_5-embed-quant.patch": {
365
+ "size": 1395,
366
+ "sha256": "0a1b9ca06798c1aef582995de5a0beb3ad9a22a54cdbd2361986563a9c7a980e"
367
+ },
368
+ "runtime/patches/qwen3_5-mtp-draft-vocab.patch": {
369
+ "size": 3943,
370
+ "sha256": "292ef662d17bcc10556b787d5bb5f2cd12d3d3fc1f3bd2fe982485ce0d916298"
371
+ },
372
+ "runtime/patches/sampler-small-topk-fast-softmax.patch": {
373
+ "size": 13679,
374
+ "sha256": "8828646ce1916c065282529d9c1bb52064f668397d8e1f33652aef2f308292b6"
375
+ },
376
+ "runtime/patches/spec-decode-attn.patch": {
377
+ "size": 15424,
378
+ "sha256": "007791047a1d60143f8e3fe4e0a56dabadfc0134cb288847338b76ba7d1f9fe7"
379
+ },
380
+ "runtime/patches/spec-decode-int4-kv-mq3d.patch": {
381
+ "size": 3876,
382
+ "sha256": "b93b186deba1513ad9c4803ee2c574c923b21ba5e88e23bca8b8594b6a6b20a0"
383
+ },
384
+ "runtime/patches/spec-decode-int8-kv.patch": {
385
+ "size": 11915,
386
+ "sha256": "3cfd31304353237af5277c861dc2b43ad33d0f9bff2700594300eff150f64382"
387
+ },
388
+ "runtime/patches/spec-sampler-prewarm.patch": {
389
+ "size": 4516,
390
+ "sha256": "18ed9608baa5a09ed2ccbb214d62763da3ce7bcf09c6aad9003778ea4e56d18f"
391
+ },
392
+ "runtime/patches/speed-knobs-envs.patch": {
393
+ "size": 2300,
394
+ "sha256": "841ec93021b1b90f90313a1880a791bbfbdb79cbe34d9b99fdb4c3a7e52ed7c0"
395
+ },
396
+ "runtime/patches/triton-prefill-attn-int8.patch": {
397
+ "size": 15661,
398
+ "sha256": "6bd36db0226dce7b92ebd1231b6f9718a25da4a933ab4b2a6a8be77860364fae"
399
+ },
400
+ "runtime/patches/vision-tower-cpu-offload.patch": {
401
+ "size": 8345,
402
+ "sha256": "81dff64a1177058783dcf7d4e8552f1547f74fa8a68019e349dacdb9d66e968e"
403
+ },
404
+ "runtime/patches/vllm-pr50021-gdn-spec-bounds.patch": {
405
+ "size": 8886,
406
+ "sha256": "cd6e00270fa28e37a8c7ad11f965662f4153b2fcaf840f2c4043710e7b078655"
407
+ },
408
+ "runtime/patches/xgrammar-spec-terminated.patch": {
409
+ "size": 3993,
410
+ "sha256": "37589b9a45d5ece37cc16e82b195362719011e52ea31b2e70cdc89845c825040"
411
+ },
412
+ "runtime/single-user/start_qwen.sh": {
413
+ "size": 42439,
414
+ "sha256": "6874eb0bc4306d61b57ebb2f2c7cab97b41f11a3e0ec2b1502d9c34d396c47d8"
415
+ },
416
+ "tokenizer.json": {
417
+ "size": 19989325,
418
+ "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
419
+ },
420
+ "tokenizer_config.json": {
421
+ "size": 1124,
422
+ "sha256": "66e427c470fe580fe8c7b5725d857af23d8417e37fae62667ec698306a19987b"
423
+ },
424
+ "video_preprocessor_config.json": {
425
+ "size": 385,
426
+ "sha256": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13"
427
+ },
428
+ "vocab.json": {
429
+ "size": 6722759,
430
+ "sha256": "ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003"
431
+ }
432
+ }
433
+ }
RUNTIME.md ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # HyperQwen runtime
2
+
3
+ This checkpoint requires the patched HyperQwen vLLM runtime. It is not a GGUF.
4
+ Tested HyperQwen base revision: `253c76aea0a240bf7cb2bd3ed92b672c0f258d0d`. Exact local launcher snapshots and
5
+ patch files are in `runtime/`; calibration/evaluation scripts are included separately.
6
+
7
+ Tested packages: vLLM 0.27.1, PyTorch 2.13.0, Transformers 5.15.0,
8
+ compressed-tensors 0.17.0, safetensors 0.8.0. Follow the
9
+ [pinned HyperQwen setup instructions](https://github.com/syv-ai/HyperQwen/blob/253c76aea0a240bf7cb2bd3ed92b672c0f258d0d/README.md#setup)
10
+ for the compiler, attention libraries, and patch installation. Use the supplied
11
+ `runtime/patches/` set when applying vLLM patches, and the supplied launcher for
12
+ the final serve command. Do not requantize this already prepared model.
13
+ An independent clean-machine installation has not been tested.
14
+
15
+ ## Download and launch from an installed HyperQwen checkout
16
+
17
+ ```bash
18
+ hf download daavidhauser/Swift-1.0-Qwen3.8-27B-W4A16-HyperQwen \
19
+ --local-dir models/Swift-1.0-Qwen3.8-27B-W4A16-HyperQwen
20
+
21
+ # In the HyperQwen checkout; use the launcher snapshot accompanying the model.
22
+ cp models/Swift-1.0-Qwen3.8-27B-W4A16-HyperQwen/runtime/single-user/start_qwen.sh single-user/start_qwen.sh
23
+
24
+ MODEL="$PWD/models/Swift-1.0-Qwen3.8-27B-W4A16-HyperQwen" \
25
+ CTX=long MAX_LEN=150000 SPEC=mtp DRAFT_TOKENS=3 \
26
+ PREFIX_CACHE=1 TOOLS=1 VISION=1 VISION_OFFLOAD=1 \
27
+ GPU_UTIL=0.93 MAX_SEQS=8 API_SERVERS=1 HOST=127.0.0.1 PORT=18020 \
28
+ VLLM_MAMBA_ALIGN_KEEP_CHECKPOINTS=1 FLASHINFER_DISABLE_VERSION_CHECK=1 \
29
+ EXTRA_ARGS='--limit-mm-per-prompt {"image":{"count":10}}' \
30
+ bash single-user/start_qwen.sh
31
+ ```
32
+
33
+ The tested run did not enable INT8 activation quantization. Remove conflicting
34
+ INT8_ACT, PREFILL_ATTN, or other performance overrides from your local environment
35
+ and `.env` when reproducing it. The launcher uses FP8 KV and FP16 recurrent state
36
+ in this profile. Test memory on your own stack before assuming 150k capacity.
37
+ Use model name `qwen3.8-27b` in API requests. Authentication follows HyperQwen's
38
+ normal `api_key.txt` / `VLLM_API_KEY` configuration.
39
+
40
+ ## Evaluation requests
41
+
42
+ Thinking: temperature 1.0, top_p 0.95, top_k 20, min_p 0, presence_penalty 0,
43
+ repetition_penalty 1.0, seed 15027, enable_thinking true, reasoning_effort xhigh.
44
+ Output limit: 128000 tokens. Nonthinking GSM8K/tool tests were greedy.
45
+ The quality suite used two concurrent requests; the speed test used one.
46
+
47
+ The final published campaign validates this single-user configuration only.
48
+ Separate batch launchers/INT8 activation settings are not benchmarked by these results.
chat_template.jinja ADDED
@@ -0,0 +1,170 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- set reasoning_instructions = '' %}
46
+ {%- if enable_thinking is undefined or enable_thinking is true %}
47
+ {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}
48
+ {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}
49
+ {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}
50
+ {%- endif %}
51
+ {%- if resolved_reasoning_effort == 'xhigh' %}
52
+ {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}
53
+ {%- elif resolved_reasoning_effort == 'low' %}
54
+ {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}
55
+ {%- endif %}
56
+ {%- endif %}
57
+ {%- if tools and tools is iterable and tools is not mapping %}
58
+ {{- '<|im_start|>system\n' }}
59
+ {%- if reasoning_instructions %}
60
+ {{- reasoning_instructions + '\n\n' }}
61
+ {%- endif %}
62
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
63
+ {%- for tool in tools %}
64
+ {{- "\n" }}
65
+ {{- tool | tojson }}
66
+ {%- endfor %}
67
+ {{- "\n</tools>" }}
68
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
69
+ {%- if messages[0].role == 'system' %}
70
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
71
+ {%- if content %}
72
+ {{- '\n\n' + content }}
73
+ {%- endif %}
74
+ {%- endif %}
75
+ {{- '<|im_end|>\n' }}
76
+ {%- else %}
77
+ {%- if messages[0].role == 'system' %}
78
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
79
+ {%- if content %}
80
+ {{- '<|im_start|>system\n' + (reasoning_instructions + '\n\n' if reasoning_instructions else '') + content + '<|im_end|>\n' }}
81
+ {%- elif reasoning_instructions %}
82
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
83
+ {%- endif %}
84
+ {%- elif reasoning_instructions %}
85
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
86
+ {%- endif %}
87
+ {%- endif %}
88
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
89
+ {%- for message in messages[::-1] %}
90
+ {%- set index = (messages|length - 1) - loop.index0 %}
91
+ {%- if ns.multi_step_tool and message.role == "user" %}
92
+ {%- set content = render_content(message.content, false)|trim %}
93
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
94
+ {%- set ns.multi_step_tool = false %}
95
+ {%- set ns.last_query_index = index %}
96
+ {%- endif %}
97
+ {%- endif %}
98
+ {%- endfor %}
99
+ {%- if ns.multi_step_tool %}
100
+ {{- raise_exception('No user query found in messages.') }}
101
+ {%- endif %}
102
+ {%- for message in messages %}
103
+ {%- set content = render_content(message.content, true)|trim %}
104
+ {%- if message.role == "system" %}
105
+ {%- if not loop.first %}
106
+ {{- raise_exception('System message must be at the beginning.') }}
107
+ {%- endif %}
108
+ {%- elif message.role == "user" %}
109
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
110
+ {%- elif message.role == "assistant" %}
111
+ {%- set reasoning_content = '' %}
112
+ {%- if message.reasoning_content is string %}
113
+ {%- set reasoning_content = message.reasoning_content %}
114
+ {%- endif %}
115
+ {%- set reasoning_content = reasoning_content|trim %}
116
+ {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}
117
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
118
+ {%- else %}
119
+ {{- '<|im_start|>' + message.role + '\n' + content }}
120
+ {%- endif %}
121
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
122
+ {%- for tool_call in message.tool_calls %}
123
+ {%- if tool_call.function is defined %}
124
+ {%- set tool_call = tool_call.function %}
125
+ {%- endif %}
126
+ {%- if loop.first %}
127
+ {%- if content|trim %}
128
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
129
+ {%- else %}
130
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
131
+ {%- endif %}
132
+ {%- else %}
133
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
134
+ {%- endif %}
135
+ {%- if tool_call.arguments is defined and tool_call.arguments != '' %}
136
+ {%- for args_name, args_value in tool_call.arguments|items %}
137
+ {{- '<parameter=' + args_name + '>\n' }}
138
+ {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}
139
+ {{- args_value }}
140
+ {{- '\n</parameter>\n' }}
141
+ {%- endfor %}
142
+ {%- endif %}
143
+ {{- '</function>\n</tool_call>' }}
144
+ {%- endfor %}
145
+ {%- endif %}
146
+ {{- '<|im_end|>\n' }}
147
+ {%- elif message.role == "tool" %}
148
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
149
+ {{- '<|im_start|>user' }}
150
+ {%- endif %}
151
+ {{- '\n<tool_response>\n' }}
152
+ {{- content }}
153
+ {{- '\n</tool_response>' }}
154
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
155
+ {{- '<|im_end|>\n' }}
156
+ {%- elif loop.last %}
157
+ {{- '<|im_end|>\n' }}
158
+ {%- endif %}
159
+ {%- else %}
160
+ {{- raise_exception('Unexpected message role.') }}
161
+ {%- endif %}
162
+ {%- endfor %}
163
+ {%- if add_generation_prompt %}
164
+ {{- '<|im_start|>assistant\n' }}
165
+ {%- if enable_thinking is defined and enable_thinking is false %}
166
+ {{- '<think>\n\n</think>\n\n' }}
167
+ {%- else %}
168
+ {{- '<think>\n' }}
169
+ {%- endif %}
170
+ {%- endif %}
config.json ADDED
@@ -0,0 +1,546 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForConditionalGeneration"
4
+ ],
5
+ "image_token_id": 248056,
6
+ "language_model_only": false,
7
+ "model_type": "qwen3_5",
8
+ "text_config": {
9
+ "attention_bias": false,
10
+ "attention_dropout": 0.0,
11
+ "attn_output_gate": true,
12
+ "bos_token_id": 248044,
13
+ "dtype": "bfloat16",
14
+ "eos_token_id": 248044,
15
+ "full_attention_interval": 4,
16
+ "head_dim": 256,
17
+ "hidden_act": "silu",
18
+ "hidden_size": 5120,
19
+ "initializer_range": 0.02,
20
+ "intermediate_size": 17408,
21
+ "layer_types": [
22
+ "linear_attention",
23
+ "linear_attention",
24
+ "linear_attention",
25
+ "full_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "linear_attention",
29
+ "full_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "linear_attention",
33
+ "full_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "linear_attention",
37
+ "full_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "linear_attention",
41
+ "full_attention",
42
+ "linear_attention",
43
+ "linear_attention",
44
+ "linear_attention",
45
+ "full_attention",
46
+ "linear_attention",
47
+ "linear_attention",
48
+ "linear_attention",
49
+ "full_attention",
50
+ "linear_attention",
51
+ "linear_attention",
52
+ "linear_attention",
53
+ "full_attention",
54
+ "linear_attention",
55
+ "linear_attention",
56
+ "linear_attention",
57
+ "full_attention",
58
+ "linear_attention",
59
+ "linear_attention",
60
+ "linear_attention",
61
+ "full_attention",
62
+ "linear_attention",
63
+ "linear_attention",
64
+ "linear_attention",
65
+ "full_attention",
66
+ "linear_attention",
67
+ "linear_attention",
68
+ "linear_attention",
69
+ "full_attention",
70
+ "linear_attention",
71
+ "linear_attention",
72
+ "linear_attention",
73
+ "full_attention",
74
+ "linear_attention",
75
+ "linear_attention",
76
+ "linear_attention",
77
+ "full_attention",
78
+ "linear_attention",
79
+ "linear_attention",
80
+ "linear_attention",
81
+ "full_attention",
82
+ "linear_attention",
83
+ "linear_attention",
84
+ "linear_attention",
85
+ "full_attention"
86
+ ],
87
+ "linear_conv_kernel_dim": 4,
88
+ "linear_key_head_dim": 128,
89
+ "linear_num_key_heads": 16,
90
+ "linear_num_value_heads": 48,
91
+ "linear_value_head_dim": 128,
92
+ "mamba_ssm_dtype": "float32",
93
+ "max_position_embeddings": 262144,
94
+ "model_type": "qwen3_5_text",
95
+ "mtp_num_hidden_layers": 1,
96
+ "mtp_use_dedicated_embeddings": false,
97
+ "num_attention_heads": 24,
98
+ "num_hidden_layers": 64,
99
+ "num_key_value_heads": 4,
100
+ "output_gate_type": "swish",
101
+ "pad_token_id": null,
102
+ "partial_rotary_factor": 0.25,
103
+ "rms_norm_eps": 1e-06,
104
+ "rope_parameters": {
105
+ "mrope_interleaved": true,
106
+ "mrope_section": [
107
+ 11,
108
+ 11,
109
+ 10
110
+ ],
111
+ "partial_rotary_factor": 0.25,
112
+ "rope_theta": 10000000,
113
+ "rope_type": "default"
114
+ },
115
+ "tie_word_embeddings": false,
116
+ "use_cache": true,
117
+ "vocab_size": 248320
118
+ },
119
+ "tie_word_embeddings": false,
120
+ "transformers_version": "5.8.0.dev0",
121
+ "video_token_id": 248057,
122
+ "vision_config": {
123
+ "deepstack_visual_indexes": [],
124
+ "depth": 27,
125
+ "hidden_act": "gelu_pytorch_tanh",
126
+ "hidden_size": 1152,
127
+ "in_channels": 3,
128
+ "initializer_range": 0.02,
129
+ "intermediate_size": 4304,
130
+ "model_type": "qwen3_5",
131
+ "num_heads": 16,
132
+ "num_position_embeddings": 2304,
133
+ "out_hidden_size": 5120,
134
+ "patch_size": 16,
135
+ "spatial_merge_size": 2,
136
+ "temporal_patch_size": 2,
137
+ "dtype": "bfloat16"
138
+ },
139
+ "vision_end_token_id": 248054,
140
+ "vision_start_token_id": 248053,
141
+ "quantization_config": {
142
+ "config_groups": {
143
+ "group_0": {
144
+ "format": "pack-quantized",
145
+ "input_activations": null,
146
+ "output_activations": null,
147
+ "targets": [
148
+ "Linear"
149
+ ],
150
+ "weights": {
151
+ "actorder": null,
152
+ "block_structure": null,
153
+ "dynamic": false,
154
+ "group_size": 128,
155
+ "num_bits": 4,
156
+ "observer": "memoryless_minmax",
157
+ "observer_kwargs": {},
158
+ "scale_dtype": null,
159
+ "strategy": "group",
160
+ "symmetric": false,
161
+ "type": "int",
162
+ "zp_dtype": "torch.int8"
163
+ }
164
+ },
165
+ "group_1": {
166
+ "format": "pack-quantized",
167
+ "input_activations": null,
168
+ "output_activations": null,
169
+ "targets": [
170
+ "re:.*lm_head$"
171
+ ],
172
+ "weights": {
173
+ "actorder": null,
174
+ "block_structure": null,
175
+ "dynamic": false,
176
+ "group_size": 128,
177
+ "num_bits": 8,
178
+ "observer": "memoryless_minmax",
179
+ "observer_kwargs": {},
180
+ "scale_dtype": null,
181
+ "strategy": "group",
182
+ "symmetric": true,
183
+ "type": "int",
184
+ "zp_dtype": null
185
+ }
186
+ },
187
+ "group_2": {
188
+ "format": "pack-quantized",
189
+ "input_activations": null,
190
+ "output_activations": null,
191
+ "targets": [
192
+ "re:.*embed_tokens$"
193
+ ],
194
+ "weights": {
195
+ "actorder": null,
196
+ "block_structure": null,
197
+ "dynamic": false,
198
+ "group_size": 128,
199
+ "num_bits": 8,
200
+ "observer": "memoryless_minmax",
201
+ "observer_kwargs": {},
202
+ "scale_dtype": null,
203
+ "strategy": "group",
204
+ "symmetric": true,
205
+ "type": "int",
206
+ "zp_dtype": null
207
+ }
208
+ },
209
+ "group_3": {
210
+ "format": "pack-quantized",
211
+ "input_activations": null,
212
+ "output_activations": null,
213
+ "targets": [
214
+ "re:^mtp\\..*"
215
+ ],
216
+ "weights": {
217
+ "actorder": null,
218
+ "block_structure": null,
219
+ "dynamic": false,
220
+ "group_size": 128,
221
+ "num_bits": 8,
222
+ "observer": "memoryless_minmax",
223
+ "observer_kwargs": {},
224
+ "scale_dtype": null,
225
+ "strategy": "group",
226
+ "symmetric": true,
227
+ "type": "int",
228
+ "zp_dtype": null
229
+ }
230
+ }
231
+ },
232
+ "format": "pack-quantized",
233
+ "global_compression_ratio": null,
234
+ "ignore": [
235
+ "model.visual.blocks.0.attn.qkv",
236
+ "model.visual.blocks.0.attn.proj",
237
+ "model.visual.blocks.0.mlp.linear_fc1",
238
+ "model.visual.blocks.0.mlp.linear_fc2",
239
+ "model.visual.blocks.1.attn.qkv",
240
+ "model.visual.blocks.1.attn.proj",
241
+ "model.visual.blocks.1.mlp.linear_fc1",
242
+ "model.visual.blocks.1.mlp.linear_fc2",
243
+ "model.visual.blocks.2.attn.qkv",
244
+ "model.visual.blocks.2.attn.proj",
245
+ "model.visual.blocks.2.mlp.linear_fc1",
246
+ "model.visual.blocks.2.mlp.linear_fc2",
247
+ "model.visual.blocks.3.attn.qkv",
248
+ "model.visual.blocks.3.attn.proj",
249
+ "model.visual.blocks.3.mlp.linear_fc1",
250
+ "model.visual.blocks.3.mlp.linear_fc2",
251
+ "model.visual.blocks.4.attn.qkv",
252
+ "model.visual.blocks.4.attn.proj",
253
+ "model.visual.blocks.4.mlp.linear_fc1",
254
+ "model.visual.blocks.4.mlp.linear_fc2",
255
+ "model.visual.blocks.5.attn.qkv",
256
+ "model.visual.blocks.5.attn.proj",
257
+ "model.visual.blocks.5.mlp.linear_fc1",
258
+ "model.visual.blocks.5.mlp.linear_fc2",
259
+ "model.visual.blocks.6.attn.qkv",
260
+ "model.visual.blocks.6.attn.proj",
261
+ "model.visual.blocks.6.mlp.linear_fc1",
262
+ "model.visual.blocks.6.mlp.linear_fc2",
263
+ "model.visual.blocks.7.attn.qkv",
264
+ "model.visual.blocks.7.attn.proj",
265
+ "model.visual.blocks.7.mlp.linear_fc1",
266
+ "model.visual.blocks.7.mlp.linear_fc2",
267
+ "model.visual.blocks.8.attn.qkv",
268
+ "model.visual.blocks.8.attn.proj",
269
+ "model.visual.blocks.8.mlp.linear_fc1",
270
+ "model.visual.blocks.8.mlp.linear_fc2",
271
+ "model.visual.blocks.9.attn.qkv",
272
+ "model.visual.blocks.9.attn.proj",
273
+ "model.visual.blocks.9.mlp.linear_fc1",
274
+ "model.visual.blocks.9.mlp.linear_fc2",
275
+ "model.visual.blocks.10.attn.qkv",
276
+ "model.visual.blocks.10.attn.proj",
277
+ "model.visual.blocks.10.mlp.linear_fc1",
278
+ "model.visual.blocks.10.mlp.linear_fc2",
279
+ "model.visual.blocks.11.attn.qkv",
280
+ "model.visual.blocks.11.attn.proj",
281
+ "model.visual.blocks.11.mlp.linear_fc1",
282
+ "model.visual.blocks.11.mlp.linear_fc2",
283
+ "model.visual.blocks.12.attn.qkv",
284
+ "model.visual.blocks.12.attn.proj",
285
+ "model.visual.blocks.12.mlp.linear_fc1",
286
+ "model.visual.blocks.12.mlp.linear_fc2",
287
+ "model.visual.blocks.13.attn.qkv",
288
+ "model.visual.blocks.13.attn.proj",
289
+ "model.visual.blocks.13.mlp.linear_fc1",
290
+ "model.visual.blocks.13.mlp.linear_fc2",
291
+ "model.visual.blocks.14.attn.qkv",
292
+ "model.visual.blocks.14.attn.proj",
293
+ "model.visual.blocks.14.mlp.linear_fc1",
294
+ "model.visual.blocks.14.mlp.linear_fc2",
295
+ "model.visual.blocks.15.attn.qkv",
296
+ "model.visual.blocks.15.attn.proj",
297
+ "model.visual.blocks.15.mlp.linear_fc1",
298
+ "model.visual.blocks.15.mlp.linear_fc2",
299
+ "model.visual.blocks.16.attn.qkv",
300
+ "model.visual.blocks.16.attn.proj",
301
+ "model.visual.blocks.16.mlp.linear_fc1",
302
+ "model.visual.blocks.16.mlp.linear_fc2",
303
+ "model.visual.blocks.17.attn.qkv",
304
+ "model.visual.blocks.17.attn.proj",
305
+ "model.visual.blocks.17.mlp.linear_fc1",
306
+ "model.visual.blocks.17.mlp.linear_fc2",
307
+ "model.visual.blocks.18.attn.qkv",
308
+ "model.visual.blocks.18.attn.proj",
309
+ "model.visual.blocks.18.mlp.linear_fc1",
310
+ "model.visual.blocks.18.mlp.linear_fc2",
311
+ "model.visual.blocks.19.attn.qkv",
312
+ "model.visual.blocks.19.attn.proj",
313
+ "model.visual.blocks.19.mlp.linear_fc1",
314
+ "model.visual.blocks.19.mlp.linear_fc2",
315
+ "model.visual.blocks.20.attn.qkv",
316
+ "model.visual.blocks.20.attn.proj",
317
+ "model.visual.blocks.20.mlp.linear_fc1",
318
+ "model.visual.blocks.20.mlp.linear_fc2",
319
+ "model.visual.blocks.21.attn.qkv",
320
+ "model.visual.blocks.21.attn.proj",
321
+ "model.visual.blocks.21.mlp.linear_fc1",
322
+ "model.visual.blocks.21.mlp.linear_fc2",
323
+ "model.visual.blocks.22.attn.qkv",
324
+ "model.visual.blocks.22.attn.proj",
325
+ "model.visual.blocks.22.mlp.linear_fc1",
326
+ "model.visual.blocks.22.mlp.linear_fc2",
327
+ "model.visual.blocks.23.attn.qkv",
328
+ "model.visual.blocks.23.attn.proj",
329
+ "model.visual.blocks.23.mlp.linear_fc1",
330
+ "model.visual.blocks.23.mlp.linear_fc2",
331
+ "model.visual.blocks.24.attn.qkv",
332
+ "model.visual.blocks.24.attn.proj",
333
+ "model.visual.blocks.24.mlp.linear_fc1",
334
+ "model.visual.blocks.24.mlp.linear_fc2",
335
+ "model.visual.blocks.25.attn.qkv",
336
+ "model.visual.blocks.25.attn.proj",
337
+ "model.visual.blocks.25.mlp.linear_fc1",
338
+ "model.visual.blocks.25.mlp.linear_fc2",
339
+ "model.visual.blocks.26.attn.qkv",
340
+ "model.visual.blocks.26.attn.proj",
341
+ "model.visual.blocks.26.mlp.linear_fc1",
342
+ "model.visual.blocks.26.mlp.linear_fc2",
343
+ "model.visual.merger.linear_fc1",
344
+ "model.visual.merger.linear_fc2",
345
+ "model.language_model.layers.0.linear_attn",
346
+ "model.language_model.layers.0.linear_attn.norm",
347
+ "model.language_model.layers.0.linear_attn.in_proj_b",
348
+ "model.language_model.layers.0.linear_attn.in_proj_a",
349
+ "model.language_model.layers.1.linear_attn",
350
+ "model.language_model.layers.1.linear_attn.norm",
351
+ "model.language_model.layers.1.linear_attn.in_proj_b",
352
+ "model.language_model.layers.1.linear_attn.in_proj_a",
353
+ "model.language_model.layers.2.linear_attn",
354
+ "model.language_model.layers.2.linear_attn.norm",
355
+ "model.language_model.layers.2.linear_attn.in_proj_b",
356
+ "model.language_model.layers.2.linear_attn.in_proj_a",
357
+ "model.language_model.layers.4.linear_attn",
358
+ "model.language_model.layers.4.linear_attn.norm",
359
+ "model.language_model.layers.4.linear_attn.in_proj_b",
360
+ "model.language_model.layers.4.linear_attn.in_proj_a",
361
+ "model.language_model.layers.5.linear_attn",
362
+ "model.language_model.layers.5.linear_attn.norm",
363
+ "model.language_model.layers.5.linear_attn.in_proj_b",
364
+ "model.language_model.layers.5.linear_attn.in_proj_a",
365
+ "model.language_model.layers.6.linear_attn",
366
+ "model.language_model.layers.6.linear_attn.norm",
367
+ "model.language_model.layers.6.linear_attn.in_proj_b",
368
+ "model.language_model.layers.6.linear_attn.in_proj_a",
369
+ "model.language_model.layers.8.linear_attn",
370
+ "model.language_model.layers.8.linear_attn.norm",
371
+ "model.language_model.layers.8.linear_attn.in_proj_b",
372
+ "model.language_model.layers.8.linear_attn.in_proj_a",
373
+ "model.language_model.layers.9.linear_attn",
374
+ "model.language_model.layers.9.linear_attn.norm",
375
+ "model.language_model.layers.9.linear_attn.in_proj_b",
376
+ "model.language_model.layers.9.linear_attn.in_proj_a",
377
+ "model.language_model.layers.10.linear_attn",
378
+ "model.language_model.layers.10.linear_attn.norm",
379
+ "model.language_model.layers.10.linear_attn.in_proj_b",
380
+ "model.language_model.layers.10.linear_attn.in_proj_a",
381
+ "model.language_model.layers.12.linear_attn",
382
+ "model.language_model.layers.12.linear_attn.norm",
383
+ "model.language_model.layers.12.linear_attn.in_proj_b",
384
+ "model.language_model.layers.12.linear_attn.in_proj_a",
385
+ "model.language_model.layers.13.linear_attn",
386
+ "model.language_model.layers.13.linear_attn.norm",
387
+ "model.language_model.layers.13.linear_attn.in_proj_b",
388
+ "model.language_model.layers.13.linear_attn.in_proj_a",
389
+ "model.language_model.layers.14.linear_attn",
390
+ "model.language_model.layers.14.linear_attn.norm",
391
+ "model.language_model.layers.14.linear_attn.in_proj_b",
392
+ "model.language_model.layers.14.linear_attn.in_proj_a",
393
+ "model.language_model.layers.16.linear_attn",
394
+ "model.language_model.layers.16.linear_attn.norm",
395
+ "model.language_model.layers.16.linear_attn.in_proj_b",
396
+ "model.language_model.layers.16.linear_attn.in_proj_a",
397
+ "model.language_model.layers.17.linear_attn",
398
+ "model.language_model.layers.17.linear_attn.norm",
399
+ "model.language_model.layers.17.linear_attn.in_proj_b",
400
+ "model.language_model.layers.17.linear_attn.in_proj_a",
401
+ "model.language_model.layers.18.linear_attn",
402
+ "model.language_model.layers.18.linear_attn.norm",
403
+ "model.language_model.layers.18.linear_attn.in_proj_b",
404
+ "model.language_model.layers.18.linear_attn.in_proj_a",
405
+ "model.language_model.layers.20.linear_attn",
406
+ "model.language_model.layers.20.linear_attn.norm",
407
+ "model.language_model.layers.20.linear_attn.in_proj_b",
408
+ "model.language_model.layers.20.linear_attn.in_proj_a",
409
+ "model.language_model.layers.21.linear_attn",
410
+ "model.language_model.layers.21.linear_attn.norm",
411
+ "model.language_model.layers.21.linear_attn.in_proj_b",
412
+ "model.language_model.layers.21.linear_attn.in_proj_a",
413
+ "model.language_model.layers.22.linear_attn",
414
+ "model.language_model.layers.22.linear_attn.norm",
415
+ "model.language_model.layers.22.linear_attn.in_proj_b",
416
+ "model.language_model.layers.22.linear_attn.in_proj_a",
417
+ "model.language_model.layers.24.linear_attn",
418
+ "model.language_model.layers.24.linear_attn.norm",
419
+ "model.language_model.layers.24.linear_attn.in_proj_b",
420
+ "model.language_model.layers.24.linear_attn.in_proj_a",
421
+ "model.language_model.layers.25.linear_attn",
422
+ "model.language_model.layers.25.linear_attn.norm",
423
+ "model.language_model.layers.25.linear_attn.in_proj_b",
424
+ "model.language_model.layers.25.linear_attn.in_proj_a",
425
+ "model.language_model.layers.26.linear_attn",
426
+ "model.language_model.layers.26.linear_attn.norm",
427
+ "model.language_model.layers.26.linear_attn.in_proj_b",
428
+ "model.language_model.layers.26.linear_attn.in_proj_a",
429
+ "model.language_model.layers.28.linear_attn",
430
+ "model.language_model.layers.28.linear_attn.norm",
431
+ "model.language_model.layers.28.linear_attn.in_proj_b",
432
+ "model.language_model.layers.28.linear_attn.in_proj_a",
433
+ "model.language_model.layers.29.linear_attn",
434
+ "model.language_model.layers.29.linear_attn.norm",
435
+ "model.language_model.layers.29.linear_attn.in_proj_b",
436
+ "model.language_model.layers.29.linear_attn.in_proj_a",
437
+ "model.language_model.layers.30.linear_attn",
438
+ "model.language_model.layers.30.linear_attn.norm",
439
+ "model.language_model.layers.30.linear_attn.in_proj_b",
440
+ "model.language_model.layers.30.linear_attn.in_proj_a",
441
+ "model.language_model.layers.32.linear_attn",
442
+ "model.language_model.layers.32.linear_attn.norm",
443
+ "model.language_model.layers.32.linear_attn.in_proj_b",
444
+ "model.language_model.layers.32.linear_attn.in_proj_a",
445
+ "model.language_model.layers.33.linear_attn",
446
+ "model.language_model.layers.33.linear_attn.norm",
447
+ "model.language_model.layers.33.linear_attn.in_proj_b",
448
+ "model.language_model.layers.33.linear_attn.in_proj_a",
449
+ "model.language_model.layers.34.linear_attn",
450
+ "model.language_model.layers.34.linear_attn.norm",
451
+ "model.language_model.layers.34.linear_attn.in_proj_b",
452
+ "model.language_model.layers.34.linear_attn.in_proj_a",
453
+ "model.language_model.layers.36.linear_attn",
454
+ "model.language_model.layers.36.linear_attn.norm",
455
+ "model.language_model.layers.36.linear_attn.in_proj_b",
456
+ "model.language_model.layers.36.linear_attn.in_proj_a",
457
+ "model.language_model.layers.37.linear_attn",
458
+ "model.language_model.layers.37.linear_attn.norm",
459
+ "model.language_model.layers.37.linear_attn.in_proj_b",
460
+ "model.language_model.layers.37.linear_attn.in_proj_a",
461
+ "model.language_model.layers.38.linear_attn",
462
+ "model.language_model.layers.38.linear_attn.norm",
463
+ "model.language_model.layers.38.linear_attn.in_proj_b",
464
+ "model.language_model.layers.38.linear_attn.in_proj_a",
465
+ "model.language_model.layers.40.linear_attn",
466
+ "model.language_model.layers.40.linear_attn.norm",
467
+ "model.language_model.layers.40.linear_attn.in_proj_b",
468
+ "model.language_model.layers.40.linear_attn.in_proj_a",
469
+ "model.language_model.layers.41.linear_attn",
470
+ "model.language_model.layers.41.linear_attn.norm",
471
+ "model.language_model.layers.41.linear_attn.in_proj_b",
472
+ "model.language_model.layers.41.linear_attn.in_proj_a",
473
+ "model.language_model.layers.42.linear_attn",
474
+ "model.language_model.layers.42.linear_attn.norm",
475
+ "model.language_model.layers.42.linear_attn.in_proj_b",
476
+ "model.language_model.layers.42.linear_attn.in_proj_a",
477
+ "model.language_model.layers.44.linear_attn",
478
+ "model.language_model.layers.44.linear_attn.norm",
479
+ "model.language_model.layers.44.linear_attn.in_proj_b",
480
+ "model.language_model.layers.44.linear_attn.in_proj_a",
481
+ "model.language_model.layers.45.linear_attn",
482
+ "model.language_model.layers.45.linear_attn.norm",
483
+ "model.language_model.layers.45.linear_attn.in_proj_b",
484
+ "model.language_model.layers.45.linear_attn.in_proj_a",
485
+ "model.language_model.layers.46.linear_attn",
486
+ "model.language_model.layers.46.linear_attn.norm",
487
+ "model.language_model.layers.46.linear_attn.in_proj_b",
488
+ "model.language_model.layers.46.linear_attn.in_proj_a",
489
+ "model.language_model.layers.48.linear_attn",
490
+ "model.language_model.layers.48.linear_attn.norm",
491
+ "model.language_model.layers.48.linear_attn.in_proj_b",
492
+ "model.language_model.layers.48.linear_attn.in_proj_a",
493
+ "model.language_model.layers.49.linear_attn",
494
+ "model.language_model.layers.49.linear_attn.norm",
495
+ "model.language_model.layers.49.linear_attn.in_proj_b",
496
+ "model.language_model.layers.49.linear_attn.in_proj_a",
497
+ "model.language_model.layers.50.linear_attn",
498
+ "model.language_model.layers.50.linear_attn.norm",
499
+ "model.language_model.layers.50.linear_attn.in_proj_b",
500
+ "model.language_model.layers.50.linear_attn.in_proj_a",
501
+ "model.language_model.layers.52.linear_attn",
502
+ "model.language_model.layers.52.linear_attn.norm",
503
+ "model.language_model.layers.52.linear_attn.in_proj_b",
504
+ "model.language_model.layers.52.linear_attn.in_proj_a",
505
+ "model.language_model.layers.53.linear_attn",
506
+ "model.language_model.layers.53.linear_attn.norm",
507
+ "model.language_model.layers.53.linear_attn.in_proj_b",
508
+ "model.language_model.layers.53.linear_attn.in_proj_a",
509
+ "model.language_model.layers.54.linear_attn",
510
+ "model.language_model.layers.54.linear_attn.norm",
511
+ "model.language_model.layers.54.linear_attn.in_proj_b",
512
+ "model.language_model.layers.54.linear_attn.in_proj_a",
513
+ "model.language_model.layers.56.linear_attn",
514
+ "model.language_model.layers.56.linear_attn.norm",
515
+ "model.language_model.layers.56.linear_attn.in_proj_b",
516
+ "model.language_model.layers.56.linear_attn.in_proj_a",
517
+ "model.language_model.layers.57.linear_attn",
518
+ "model.language_model.layers.57.linear_attn.norm",
519
+ "model.language_model.layers.57.linear_attn.in_proj_b",
520
+ "model.language_model.layers.57.linear_attn.in_proj_a",
521
+ "model.language_model.layers.58.linear_attn",
522
+ "model.language_model.layers.58.linear_attn.norm",
523
+ "model.language_model.layers.58.linear_attn.in_proj_b",
524
+ "model.language_model.layers.58.linear_attn.in_proj_a",
525
+ "model.language_model.layers.60.linear_attn",
526
+ "model.language_model.layers.60.linear_attn.norm",
527
+ "model.language_model.layers.60.linear_attn.in_proj_b",
528
+ "model.language_model.layers.60.linear_attn.in_proj_a",
529
+ "model.language_model.layers.61.linear_attn",
530
+ "model.language_model.layers.61.linear_attn.norm",
531
+ "model.language_model.layers.61.linear_attn.in_proj_b",
532
+ "model.language_model.layers.61.linear_attn.in_proj_a",
533
+ "model.language_model.layers.62.linear_attn",
534
+ "model.language_model.layers.62.linear_attn.norm",
535
+ "model.language_model.layers.62.linear_attn.in_proj_b",
536
+ "model.language_model.layers.62.linear_attn.in_proj_a"
537
+ ],
538
+ "kv_cache_scheme": null,
539
+ "quant_method": "compressed-tensors",
540
+ "quantization_status": "compressed",
541
+ "sparsity_config": {},
542
+ "transform_config": {},
543
+ "version": "0.18.0"
544
+ },
545
+ "dtype": "bfloat16"
546
+ }
evaluation/code/swift15/checkpoint.py ADDED
@@ -0,0 +1,87 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Prepare and audit a separate Swift baseline; never mutate the source checkpoint."""
2
+ import argparse
3
+ import copy
4
+ import json
5
+ import os
6
+ import shutil
7
+ import subprocess
8
+ import sys
9
+ from pathlib import Path
10
+
11
+ from common import ROOT, RUN, SOURCE, BASELINE, read_json, write_json, sha256, stamp
12
+
13
+
14
+ def clone(source, destination):
15
+ source, destination = Path(source), Path(destination)
16
+ destination.mkdir(parents=True, exist_ok=True)
17
+ for p in source.iterdir():
18
+ if not p.is_file() or ".bak" in p.name or p.name.endswith(".tmp"):
19
+ continue
20
+ target = destination / p.name
21
+ if target.exists():
22
+ continue
23
+ if p.suffix == ".safetensors":
24
+ os.link(p, target)
25
+ else:
26
+ shutil.copy2(p, target)
27
+
28
+
29
+ def audit(directory):
30
+ from safetensors import safe_open
31
+ directory = Path(directory)
32
+ index = read_json(directory / "model.safetensors.index.json")
33
+ mapping = index["weight_map"]
34
+ actual = {}
35
+ duplicates = []
36
+ total = 0
37
+ for name in sorted(set(mapping.values())):
38
+ p = directory / name
39
+ total += p.stat().st_size
40
+ with safe_open(p, framework="pt") as f:
41
+ for key in f.keys():
42
+ if key in actual:
43
+ duplicates.append(key)
44
+ actual[key] = name
45
+ missing = sorted(set(mapping) - set(actual))
46
+ orphaned = sorted(set(actual) - set(mapping))
47
+ wrong = [k for k in mapping if actual.get(k) != mapping[k]]
48
+ assert not (duplicates or missing or orphaned or wrong), (duplicates, missing, orphaned, wrong)
49
+ config = read_json(directory / "config.json")
50
+ result = {"time": stamp(), "directory": str(directory), "tensor_count": len(actual),
51
+ "shard_bytes": total, "duplicate_keys": duplicates,
52
+ "quantization": config["quantization_config"]}
53
+ write_json(directory / "hyperqwen-audit.json", result)
54
+ print("Verified", len(actual), "unique indexed tensors;", round(total / 2**30, 2), "GiB", flush=True)
55
+ return result
56
+
57
+
58
+ def baseline():
59
+ if (BASELINE / "hyperqwen-build.json").exists():
60
+ return audit(BASELINE)
61
+ clone(SOURCE, BASELINE)
62
+ index = read_json(BASELINE / "model.safetensors.index.json")
63
+ if "lm_head.weight" in index["weight_map"]:
64
+ subprocess.run([sys.executable, str(ROOT / "prepare/quant_heads_stream.py"), str(BASELINE)], check=True)
65
+ if not (BASELINE / "mtp_draft_vocab_ids.pt").exists():
66
+ subprocess.run([sys.executable, str(ROOT / "prepare/build_draft_vocab.py"), str(BASELINE),
67
+ "--ids", str(ROOT / "prepare/draft_vocab_ids.json")], check=True)
68
+ # config.json is authoritative. Some exports carry a second quantization file.
69
+ config = read_json(BASELINE / "config.json")
70
+ if (BASELINE / "quantization_config.json").exists():
71
+ write_json(BASELINE / "quantization_config.json", config["quantization_config"])
72
+ write_json(BASELINE / "hyperqwen-build.json", {
73
+ "created": stamp(), "source": read_json(RUN / "source.json"),
74
+ "variant": "baseline", "body": "unchanged asymmetric AWQ INT4 group128",
75
+ "embedding_bits": 8, "lm_head_bits": 8, "mtp_bits": 8,
76
+ "preparation_script_sha256": sha256(ROOT / "prepare/quant_heads_stream.py"),
77
+ "draft_vocabulary": "HyperQwen reference ids; rebuilt for Swift in fast variant",
78
+ })
79
+ audit(BASELINE)
80
+
81
+
82
+ if __name__ == "__main__":
83
+ ap = argparse.ArgumentParser()
84
+ ap.add_argument("action", choices=["baseline", "audit"])
85
+ ap.add_argument("--model", type=Path, default=BASELINE)
86
+ args = ap.parse_args()
87
+ baseline() if args.action == "baseline" else audit(args.model)
evaluation/code/swift15/code_runner.py ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Run only INSIDE the restricted Docker container, never on the host."""
2
+ import decimal
3
+ import json
4
+ import os
5
+ import resource
6
+ import subprocess
7
+ import sys
8
+
9
+
10
+ def equal(actual, expected):
11
+ a, b = actual.split(), expected.split()
12
+ if len(a) != len(b):
13
+ return False
14
+ for x, y in zip(a, b):
15
+ if x == y:
16
+ continue
17
+ try:
18
+ xx, yy = decimal.Decimal(x), decimal.Decimal(y)
19
+ if not (xx.is_finite() and yy.is_finite()):
20
+ return False
21
+ # Integer output must match exactly, without float precision loss.
22
+ if all(t.lstrip("+-").isdigit() for t in (x,y)):
23
+ if xx != yy:
24
+ return False
25
+ elif abs(xx-yy) > max(decimal.Decimal("0.000001"), abs(yy)*decimal.Decimal("0.000001")):
26
+ return False
27
+ except decimal.InvalidOperation:
28
+ return False
29
+ return True
30
+
31
+
32
+ def limits():
33
+ resource.setrlimit(resource.RLIMIT_CPU, (3, 3))
34
+ resource.setrlimit(resource.RLIMIT_FSIZE, (1 << 20, 1 << 20))
35
+
36
+
37
+ def main():
38
+ if os.environ.get("SWIFT15_CODE_SANDBOX") != "1":
39
+ raise SystemExit("Refusing to execute generated code outside the configured container")
40
+ task = json.load(sys.stdin)
41
+ with open("/tmp/solution.py", "w") as f:
42
+ f.write(task["code"])
43
+ passed = 0
44
+ for test in task["tests"]:
45
+ try:
46
+ r = subprocess.run([sys.executable, "-I", "/tmp/solution.py"], input=test["input"],
47
+ text=True, capture_output=True, timeout=4, preexec_fn=limits)
48
+ if r.returncode or not equal(r.stdout, test["output"]):
49
+ print(json.dumps({"correct": False, "passed": passed, "total": len(task["tests"]), "reason": "runtime_error" if r.returncode else "wrong_answer"}))
50
+ return
51
+ except subprocess.TimeoutExpired:
52
+ print(json.dumps({"correct": False, "passed": passed, "total": len(task["tests"]), "reason": "test_timeout"}))
53
+ return
54
+ passed += 1
55
+ print(json.dumps({"correct": True, "passed": passed, "total": passed}))
56
+
57
+
58
+ if __name__ == "__main__":
59
+ main()
evaluation/code/swift15/common.py ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Shared reproducibility helpers for the Swift 1.5 build."""
2
+ import hashlib
3
+ import json
4
+ import os
5
+ import time
6
+ from pathlib import Path
7
+
8
+ ROOT = Path(__file__).resolve().parents[1]
9
+ RUN = Path(os.environ.get("SWIFT15_RUN_DIR", ROOT / "runs/swift15")).resolve()
10
+ SOURCE = ROOT / "models/Swift-1.5-Qwen3.8-27B-AWQ-source"
11
+ BASELINE = ROOT / "models/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen"
12
+ FAST = Path(os.environ.get("SWIFT15_FAST_MODEL", ROOT / "models/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen-fast")).resolve()
13
+
14
+
15
+ def read_json(path):
16
+ return json.loads(Path(path).read_text())
17
+
18
+
19
+ def write_json(path, value):
20
+ path = Path(path)
21
+ path.parent.mkdir(parents=True, exist_ok=True)
22
+ temporary = path.with_name(path.name + ".tmp")
23
+ temporary.write_text(json.dumps(value, indent=2, ensure_ascii=False) + "\n")
24
+ os.replace(temporary, path)
25
+
26
+
27
+ def sha256(path):
28
+ h = hashlib.sha256()
29
+ with open(path, "rb") as f:
30
+ for chunk in iter(lambda: f.read(8 << 20), b""):
31
+ h.update(chunk)
32
+ return h.hexdigest()
33
+
34
+
35
+ def records(path):
36
+ with open(path) as f:
37
+ return [json.loads(line) for line in f if line.strip()]
38
+
39
+
40
+ def write_records(path, rows):
41
+ path = Path(path)
42
+ path.parent.mkdir(parents=True, exist_ok=True)
43
+ with path.open("w") as f:
44
+ for row in rows:
45
+ f.write(json.dumps(row, ensure_ascii=False) + "\n")
46
+
47
+
48
+ def stamp():
49
+ return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
evaluation/code/swift15/corpus.py ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Download pinned public training sources and select 6,000 calibration prompts."""
2
+ import argparse
3
+ import collections
4
+ import hashlib
5
+ import json
6
+ import random
7
+ from pathlib import Path
8
+
9
+ import pyarrow.parquet as pq
10
+ from huggingface_hub import hf_hub_download
11
+ from common import RUN, write_json, write_records, read_json, sha256
12
+
13
+ SOURCES = {
14
+ "chat": ("HuggingFaceH4/ultrachat_200k", "8049631c405ae6576f93f445c6b8166f76f5505a", "data/train_sft-00000-of-00003-a3ecf92756993583.parquet"),
15
+ "code": ("ise-uiuc/Magicoder-OSS-Instruct-75K", "5f839b1f368a76b161028bb9edff055db34022b2", "data-oss_instruct-decontaminated.jsonl"),
16
+ "tools": ("NousResearch/hermes-function-calling-v1", "dae3e1d28cfbcf4b915c04ea1e072030529b4bda", "func-calling-singleturn.json"),
17
+ "math": ("openai/gsm8k", "740312add88f781978c0658806c59bc2815b9866", "main/train-00000-of-00001.parquet"),
18
+ "multilingual": ("CohereLabs/aya_dataset", "f9ea04583f02a8f86404ff6c58bf75fe637df8a2", "data/train-00000-of-00001.parquet"),
19
+ }
20
+
21
+
22
+ def load_source(name):
23
+ repo, revision, filename = SOURCES[name]
24
+ path = hf_hub_download(repo, filename, repo_type="dataset", revision=revision,
25
+ local_dir=RUN / "datasets" / name)
26
+ if filename.endswith(".parquet"):
27
+ rows = pq.read_table(path).to_pylist()
28
+ elif filename.endswith(".jsonl"):
29
+ rows = [json.loads(s) for s in open(path) if s.strip()]
30
+ else:
31
+ rows = json.load(open(path))
32
+ assert isinstance(rows, list), type(rows)
33
+ return rows, {"repository": repo, "revision": revision, "file": filename,
34
+ "sha256": sha256(path), "rows_available": len(rows)}
35
+
36
+
37
+ def message(text):
38
+ return [{"role": "user", "content": text}]
39
+
40
+
41
+ def structured_prompts(n):
42
+ rows = []
43
+ for i in range(n):
44
+ a, b = 11 + i * 3, 7 + i % 37
45
+ variants = [
46
+ f'Return only JSON with keys "sum", "difference", "product" for the integers {a} and {b}.',
47
+ f'A tool returned {{"matches":[{{"name":"item-{i}","price":{a}}},{{"name":"item-{i+1}","price":{b}}}]}}. Return the cheaper item as JSON with keys name and price.',
48
+ f'Convert these records into CSV with columns name,count: item-{i} has {a}; item-{i+1} has {b}. Output only CSV.',
49
+ f'An API request for record {i} failed with HTTP 429 and Retry-After: {b}. Explain a safe retry policy and provide a Python implementation.',
50
+ f'Extract an object with fields city, nights and guests from: "Book accommodation in Vienna for {i%12+1} nights for {i%5+1} guests." Output JSON only.',
51
+ f'A file-reading tool returned a JSON parse error on line {i%20+1}. Give a debugging plan that preserves the original file and validates the repaired JSON.',
52
+ ]
53
+ rows.append({"src": "structured", "source_row": i, "messages": message(f"Request reference: calibration-{i}.\n" + variants[i % len(variants)])})
54
+ return rows
55
+
56
+
57
+ def main():
58
+ ap = argparse.ArgumentParser()
59
+ ap.add_argument("--seed", type=int, default=15027)
60
+ args = ap.parse_args()
61
+ rng = random.Random(args.seed)
62
+ counts = {"code": 2100, "tools": 900, "chat": 1200, "math": 900, "multilingual": 600}
63
+ all_rows, sources = [], {}
64
+ candidate_hashes = set()
65
+ for category, count in counts.items():
66
+ raw, info = load_source(category)
67
+ sources[category] = info
68
+ candidates = []
69
+ for index, r in enumerate(raw):
70
+ tools = None
71
+ language = None
72
+ if category == "chat":
73
+ msgs = r["messages"][:3] if len(r["messages"]) >= 3 and rng.random() < .25 else r["messages"][:1]
74
+ if not msgs or msgs[-1]["role"] != "user":
75
+ continue
76
+ elif category == "code":
77
+ msgs = message(r["problem"])
78
+ language = r.get("lang", "unknown")
79
+ elif category == "math":
80
+ msgs = message(r["question"])
81
+ elif category == "multilingual":
82
+ language = r["language_code"]
83
+ if language not in {"deu", "fra", "spa", "zho", "dan", "jpn", "arb", "hin", "por", "ita"}:
84
+ continue
85
+ msgs = message(r["inputs"])
86
+ else:
87
+ user_turns = [m["value"] for m in r["conversations"] if m["from"] in {"human", "user"}]
88
+ if not user_turns:
89
+ continue
90
+ msgs = message(user_turns[0])
91
+ raw_tools = json.loads(r["tools"]) if isinstance(r["tools"], str) else r["tools"]
92
+ tools = [{"type": "function", "function": t.get("function", t)} for t in raw_tools]
93
+ if not all(isinstance(m.get("content"), str) for m in msgs):
94
+ continue
95
+ length = sum(len(m["content"]) for m in msgs)
96
+ if not 20 <= length <= 16000:
97
+ continue
98
+ row = {"src": category, "source_row": index, "messages": msgs}
99
+ if tools:
100
+ row["tools"] = tools
101
+ if language:
102
+ row["language"] = language
103
+ fingerprint = hashlib.sha256(json.dumps([msgs, tools], sort_keys=True, ensure_ascii=False).encode()).hexdigest()
104
+ if fingerprint in candidate_hashes:
105
+ continue
106
+ candidate_hashes.add(fingerprint)
107
+ candidates.append(row)
108
+ rng.shuffle(candidates)
109
+ # Balance languages instead of allowing the largest source language to dominate.
110
+ if category in {"multilingual", "code"}:
111
+ groups = collections.defaultdict(list)
112
+ for row in candidates:
113
+ groups[row["language"]].append(row)
114
+ chosen = []
115
+ while len(chosen) < count and groups:
116
+ for lang in list(sorted(groups)):
117
+ chosen.append(groups[lang].pop())
118
+ if not groups[lang]:
119
+ del groups[lang]
120
+ if len(chosen) == count:
121
+ break
122
+ else:
123
+ chosen = candidates[:count]
124
+ assert len(chosen) == count, (category, len(chosen), count)
125
+ all_rows.extend(chosen)
126
+ print(category, len(chosen), flush=True)
127
+ all_rows.extend(structured_prompts(300))
128
+ rng.shuffle(all_rows)
129
+ seen = set()
130
+ for i, row in enumerate(all_rows):
131
+ digest = hashlib.sha256(json.dumps([row["messages"], row.get("tools")], sort_keys=True, ensure_ascii=False).encode()).hexdigest()
132
+ assert digest not in seen, "Duplicate calibration prompt: " + digest
133
+ seen.add(digest)
134
+ row.update(id=i, prompt_sha256=digest, think=rng.random() < (.7 if row["src"] == "math" else .5))
135
+ # Split whole examples before generation/capture, preserving category proportions.
136
+ groups = collections.defaultdict(list)
137
+ for row in all_rows:
138
+ groups[row["src"]].append(row)
139
+ holdout = {r["id"] for g in groups.values() for r in g[:max(1, len(g)//10)]}
140
+ for row in all_rows:
141
+ row["split"] = "holdout" if row["id"] in holdout else "calibration"
142
+ write_records(RUN / "calibration/prompts.jsonl", all_rows)
143
+ write_json(RUN / "calibration/manifest.json", {
144
+ "seed": args.seed, "sources": sources, "counts": dict(collections.Counter(r["src"] for r in all_rows)),
145
+ "holdout_examples": len(holdout), "total_examples": len(all_rows),
146
+ "prompts_sha256": sha256(RUN / "calibration/prompts.jsonl"),
147
+ "structured_source": "swift15/corpus.py deterministic templates, not benchmark test prompts",
148
+ "source_substitution": "Salesforce xLAM returned gated-access 403; use the independently released Apache-2.0 NousResearch source with native Swift tool formatting.",
149
+ })
150
+ print("Wrote", len(all_rows), "prompts;", len(holdout), "held out", flush=True)
151
+
152
+
153
+ if __name__ == "__main__":
154
+ main()
evaluation/code/swift15/eval_data.py ADDED
@@ -0,0 +1,135 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Freeze test data independently of calibration, including a reproducible LCB subset."""
2
+ import base64
3
+ import collections
4
+ import io
5
+ import json
6
+ import pickle
7
+ import random
8
+ import zlib
9
+ from pathlib import Path
10
+
11
+ import pyarrow.parquet as pq
12
+ from huggingface_hub import hf_hub_download
13
+ from common import ROOT, RUN, write_json, write_records, sha256
14
+
15
+
16
+ class DataOnlyUnpickler(pickle.Unpickler):
17
+ def find_class(self, module, name):
18
+ raise ValueError("Executable object in benchmark test data")
19
+
20
+
21
+ def private_tests(value):
22
+ try:
23
+ return json.loads(value)
24
+ except (ValueError, TypeError):
25
+ decoded = DataOnlyUnpickler(io.BytesIO(zlib.decompress(base64.b64decode(value)))).load()
26
+ return json.loads(decoded) if isinstance(decoded, str) else decoded
27
+
28
+
29
+ def download(repo, revision, filename):
30
+ return Path(hf_hub_download(repo, filename, repo_type="dataset", revision=revision,
31
+ local_dir=RUN / "evaluation/downloads" / repo.replace("/", "__")))
32
+
33
+
34
+ def tool_tasks():
35
+ rows = []
36
+ cities = ["Graz", "Linz", "Salzburg", "Innsbruck", "Klagenfurt", "Villach", "Wels", "Steyr", "Bregenz", "Eisenstadt"]
37
+ tool = {"type": "function", "function": {"name": "get_weather", "description": "Get current weather for a city.",
38
+ "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"], "additionalProperties": False}}}
39
+ for i in range(30):
40
+ if i < 20:
41
+ city = cities[i % len(cities)]
42
+ rows.append({"id": f"tool-{i}", "suite": "tools", "think": False, "max_tokens": 2048,
43
+ "messages": [{"role": "user", "content": f"Use get_weather to check {city}. Then return only a JSON object with keys city and fahrenheit. Convert the tool's Celsius temperature to Fahrenheit."}],
44
+ "tools": [tool], "expected_call": {"name": "get_weather", "arguments": {"city": city}},
45
+ "tool_result": {"city": city, "celsius": i - 7},
46
+ "expected": {"city": city, "fahrenheit": (i-7)*1.8+32}})
47
+ else:
48
+ codes = [f"SKU-{i}-{j}" for j in range(4)]
49
+ records = [{"sku": code, "stock": (i*j+3)%11} for j,code in enumerate(codes)]
50
+ expected = [r["sku"] for r in records if r["stock"] >= 5]
51
+ rows.append({"id": f"json-{i}", "suite": "tools", "think": False, "max_tokens": 2048,
52
+ "messages": [{"role": "user", "content": f'Return only JSON with one key "available" containing the SKUs with stock >= 5, preserving input order. Records: {json.dumps(records)}'}],
53
+ "expected": {"available": expected}})
54
+ return rows
55
+
56
+
57
+ def main():
58
+ out = RUN / "evaluation"
59
+ out.mkdir(parents=True, exist_ok=True)
60
+ rng = random.Random(15027)
61
+ tasks, sources = [], []
62
+ gsm = download("openai/gsm8k", "740312add88f781978c0658806c59bc2815b9866", "main/test-00000-of-00001.parquet")
63
+ for i, r in enumerate(pq.read_table(gsm).to_pylist()[:200]):
64
+ tasks.append({"id": f"gsm8k-{i}", "suite": "gsm8k", "think": False, "max_tokens": 768,
65
+ "messages": [{"role": "user", "content": r["question"] + "\n\nSolve step by step, then give the final answer as 'Final answer: <number>'."}],
66
+ "expected": r["answer"].split("####")[-1].strip()})
67
+ sources.append({"dataset": "openai/gsm8k", "split": "main/test", "file_sha256": sha256(gsm), "selection": "first 200, same as HyperQwen"})
68
+ iff = download("allenai/IFBench_test", "2e8a48de45ff3bf41242f927254ca81b59ca3ae2", "data/train-00000-of-00001.parquet")
69
+ for r in pq.read_table(iff).to_pylist():
70
+ tasks.append({"id": f"ifbench-{r['key']}", "suite": "ifbench", "think": True, "max_tokens": 4096,
71
+ "messages": [{"role": "user", "content": r["prompt"]}],
72
+ "instruction_id_list": r["instruction_id_list"], "kwargs": r["kwargs"]})
73
+ sources.append({"dataset": "allenai/IFBench_test", "file_sha256": sha256(iff),
74
+ "note": "Upstream names this split train, but it is the benchmark test set; never used for calibration."})
75
+ pool = {}
76
+ for filename in ["test.jsonl"] + [f"test{i}.jsonl" for i in range(2, 7)]:
77
+ p = download("livecodebench/code_generation_lite", "0fe84c3912ea0c4d4a78037083943e8f0c4dd505", filename)
78
+ sources.append({"dataset": "livecodebench/code_generation_lite", "file": filename, "sha256": sha256(p)})
79
+ for line in p.read_text().splitlines():
80
+ r = json.loads(line)
81
+ # v6 timeframe; stdin programs only, to keep execution protocol explicit.
82
+ if r["contest_date"][:10] > "2025-04-30" or r.get("starter_code"):
83
+ continue
84
+ public = json.loads(r["public_test_cases"])
85
+ private = private_tests(r["private_test_cases"])
86
+ tests = public + private
87
+ if not tests or any(t.get("testtype") != "stdin" for t in tests):
88
+ continue
89
+ pool[(r["platform"], r["question_id"])] = (r, tests)
90
+ grouped = collections.defaultdict(list)
91
+ for value in pool.values():
92
+ grouped[value[0]["difficulty"]].append(value)
93
+ for level, n in [("easy", 34), ("medium", 33), ("hard", 33)]:
94
+ group = sorted(grouped[level], key=lambda x: (x[0]["platform"], x[0]["question_id"]))
95
+ rng.shuffle(group)
96
+ assert len(group) >= n, (level, len(group))
97
+ for r, tests in group[:n]:
98
+ tasks.append({"id": f"lcb-{r['platform']}-{r['question_id']}", "suite": "livecodebench", "think": True,
99
+ "difficulty": level, "max_tokens": 4096,
100
+ "messages": [{"role": "user", "content": r["question_content"] + "\n\nWrite a complete Python 3 program that reads from standard input and writes to standard output. Put the final solution in a single ```python``` code block."}],
101
+ "tests": tests})
102
+ tasks.extend(tool_tasks())
103
+ write_records(out / "tasks.jsonl", tasks)
104
+ pilot = []
105
+ for suite in ["gsm8k", "ifbench", "livecodebench", "tools"]:
106
+ candidates = [t for t in tasks if t["suite"] == suite]
107
+ rng.shuffle(candidates)
108
+ pilot.extend(t["id"] for t in candidates[:5])
109
+ write_json(out / "pilot-ids.json", pilot)
110
+ # Freeze the old battery's input bytes, especially installed vLLM source text.
111
+ texts = []
112
+ p = ROOT / "bench/quality-data/wikitext/wikitext-2-raw-v1/test-00000-of-00001.parquet"
113
+ text = "".join(pq.read_table(p).column("text").to_pylist())
114
+ for i in range(0, min(len(text), 40*1200), 1200):
115
+ texts.append({"language": "en", "text": text[i:i+1200]})
116
+ p = ROOT / "bench/quality-data/fineweb2/data/dan_Latn/test/000_00000.parquet"
117
+ danish = pq.read_table(p, columns=["text"]).column("text").to_pylist()
118
+ random.Random(0).shuffle(danish)
119
+ texts.extend({"language": "da", "text": t[:1200]} for t in [t for t in danish if len(t)>1500][:40])
120
+ for p in sorted((ROOT / "venv/lib/python3.12/site-packages/vllm/v1/core").glob("*.py")):
121
+ text = p.read_text()
122
+ texts.extend({"language": "code", "text": text[i:i+1200]} for i in range(0,min(len(text),4800),1200) if len(text[i:i+1200])>800)
123
+ if sum(t["language"] == "code" for t in texts) >= 40:
124
+ break
125
+ write_records(out / "perplexity.jsonl", texts)
126
+ write_json(out / "manifest.json", {"seed": 15027, "sources": sources,
127
+ "tasks": dict(collections.Counter(t["suite"] for t in tasks)),
128
+ "task_file_sha256": sha256(out / "tasks.jsonl"), "ppl_file_sha256": sha256(out / "perplexity.jsonl"),
129
+ "protocol": "Single-seed bounded-budget local comparison, not official full-release leaderboard scores.",
130
+ "lcb_scoring": "100 stratified stdin-only v6-era tasks; all supplied public/private tests; whitespace token comparison with numeric tolerance; not the full official LCB runner."})
131
+ print("Frozen", len(tasks), "tasks and", len(texts), "perplexity windows", flush=True)
132
+
133
+
134
+ if __name__ == "__main__":
135
+ main()
evaluation/code/swift15/evaluate.py ADDED
@@ -0,0 +1,265 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stream naturally terminated tasks, preserve raw responses and score failures too."""
2
+ import argparse
3
+ import collections
4
+ import concurrent.futures
5
+ import copy
6
+ import json
7
+ import math
8
+ import os
9
+ import re
10
+ import statistics
11
+ import subprocess
12
+ import time
13
+ import urllib.request
14
+ import uuid
15
+ from pathlib import Path
16
+
17
+ from common import ROOT, RUN, records, read_json, write_json, sha256, stamp
18
+
19
+
20
+ def key():
21
+ return os.environ.get("VLLM_API_KEY") or (ROOT / "api_key.txt").read_text().strip()
22
+
23
+
24
+ def post(api, path, body, timeout=600):
25
+ req = urllib.request.Request(api + path, json.dumps(body).encode(),
26
+ headers={"Content-Type": "application/json", "Authorization": "Bearer " + key()})
27
+ return urllib.request.urlopen(req, timeout=timeout)
28
+
29
+
30
+ def stream(api, body):
31
+ body = dict(body, stream=True, stream_options={"include_usage": True})
32
+ start = time.monotonic()
33
+ first, first_answer, last = None, None, None
34
+ content, reasoning, calls = [], [], {}
35
+ usage, finish = {}, None
36
+ with post(api, "/chat/completions", body) as response:
37
+ for line in response:
38
+ if not line.startswith(b"data: "):
39
+ continue
40
+ raw = line[6:].strip()
41
+ if raw == b"[DONE]":
42
+ break
43
+ chunk = json.loads(raw)
44
+ if "error" in chunk:
45
+ raise RuntimeError(str(chunk["error"]))
46
+ if chunk.get("usage"):
47
+ usage = chunk["usage"]
48
+ for choice in chunk.get("choices", []):
49
+ delta = choice.get("delta", {})
50
+ now = time.monotonic()
51
+ think = delta.get("reasoning_content") or delta.get("reasoning") or ""
52
+ text = delta.get("content") or ""
53
+ tool = delta.get("tool_calls") or []
54
+ if think or text or tool:
55
+ first = first or now
56
+ last = now
57
+ if text:
58
+ first_answer = first_answer or now
59
+ content.append(text)
60
+ reasoning.append(think)
61
+ for call in tool:
62
+ c = calls.setdefault(call["index"], {"id": "", "type": "function", "function": {"name": "", "arguments": ""}})
63
+ if call.get("id"):
64
+ c["id"] = call["id"]
65
+ for k in ["name", "arguments"]:
66
+ c["function"][k] += call.get("function", {}).get(k) or ""
67
+ finish = choice.get("finish_reason") or finish
68
+ end = time.monotonic()
69
+ if not usage:
70
+ raise RuntimeError("Missing usage counters; refusing to report guessed token counts")
71
+ n = usage.get("completion_tokens", 0)
72
+ return {"content": "".join(content), "reasoning": "".join(reasoning),
73
+ "tool_calls": [calls[i] for i in sorted(calls)], "usage": usage, "finish_reason": finish,
74
+ "wall_seconds": end-start, "ttft_seconds": None if first is None else first-start,
75
+ "time_to_answer_seconds": None if first_answer is None else first_answer-start,
76
+ "decode_seconds": 0 if first is None or last is None else last-first,
77
+ "decode_tps": (n-1)/(last-first) if n>1 and last is not None and last>first else None}
78
+
79
+
80
+ def json_equal(a, b):
81
+ if isinstance(b, bool):
82
+ return isinstance(a, bool) and a == b
83
+ if isinstance(b, dict):
84
+ return isinstance(a, dict) and set(a) == set(b) and all(json_equal(a[k], v) for k,v in b.items())
85
+ if isinstance(b, list):
86
+ return isinstance(a, list) and len(a) == len(b) and all(json_equal(x,y) for x,y in zip(a,b))
87
+ if isinstance(b, (int,float)) and not isinstance(b, bool):
88
+ return isinstance(a, (int,float)) and not isinstance(a,bool) and math.isfinite(a) and abs(a-b)<1e-6
89
+ return a == b
90
+
91
+
92
+ def gsm_score(text, expected):
93
+ match = re.search(r"Final answer:\s*\**\s*\$?(-?[\d,]*\.?\d+)", text)
94
+ numbers = re.findall(r"-?\d[\d,]*\.?\d*", text.replace("$", ""))
95
+ pred = match.group(1) if match else (numbers[-1] if numbers else "")
96
+ try:
97
+ value, gold = float(pred.replace(",", "")), float(expected.replace(",", ""))
98
+ return math.isfinite(value) and abs(value-gold)<1e-6
99
+ except ValueError:
100
+ return False
101
+
102
+
103
+ def code_score(text, task):
104
+ blocks = re.findall(r"```(?:python|py)?\s*\n(.*?)```", text, re.S)
105
+ if not blocks:
106
+ return {"correct": False, "reason": "no_code_block"}
107
+ # This image is pinned locally in the run manifest before evaluation begins.
108
+ image = read_json(RUN / "evaluation/runtime.json")["code_image"]
109
+ container = "swift15-test-" + uuid.uuid4().hex
110
+ cmd = ["docker", "run", "--name", container, "--rm", "-i", "--network", "none", "--read-only", "--memory", "512m", "--cpus", "1",
111
+ "--pids-limit", "64", "--cap-drop", "ALL", "--security-opt", "no-new-privileges", "--user", "65534:65534",
112
+ "--tmpfs", "/tmp:rw,noexec,nosuid,size=64m", "-e", "SWIFT15_CODE_SANDBOX=1",
113
+ "-v", str(ROOT / "swift15/code_runner.py") + ":/runner.py:ro", image, "python", "-I", "/runner.py"]
114
+ try:
115
+ r = subprocess.run(cmd, input=json.dumps({"code": blocks[-1], "tests": task["tests"]}),
116
+ text=True, capture_output=True, timeout=180)
117
+ if r.returncode:
118
+ return {"correct": False, "reason": "sandbox_error", "detail": r.stderr[-1000:]}
119
+ return json.loads(r.stdout)
120
+ except subprocess.TimeoutExpired:
121
+ return {"correct": False, "reason": "task_test_timeout"}
122
+ finally:
123
+ subprocess.run(["docker", "rm", "-f", container], capture_output=True, timeout=30)
124
+
125
+
126
+ def execute(task, api):
127
+ start = time.monotonic()
128
+ result = {"id": task["id"], "suite": task["suite"], "calls": [], "correct": False, "error": None}
129
+ body = {"model": "qwen3.8-27b", "messages": copy.deepcopy(task["messages"]), "max_tokens": task["max_tokens"],
130
+ "temperature": 0, "seed": 15027, "top_p": 1.0,
131
+ "chat_template_kwargs": {"enable_thinking": task["think"], "reasoning_effort": "xhigh"}}
132
+ # HyperQwen evaluates thinking tasks at the model's recommended sampling.
133
+ # Greedy remains the repository's GSM8K protocol and our deterministic tool test.
134
+ if task["think"]:
135
+ body.update(temperature=1.0, top_p=.95, top_k=20, min_p=0,
136
+ presence_penalty=0, repetition_penalty=1.0)
137
+ result["sampling"] = {k:body[k] for k in ["temperature","top_p","seed"]}
138
+ if "top_k" in body:
139
+ result["sampling"]["top_k"] = body["top_k"]
140
+ if task.get("tools"):
141
+ body.update(tools=task["tools"], tool_choice="auto")
142
+ try:
143
+ first = stream(api, body)
144
+ result["calls"].append(first)
145
+ response = first
146
+ if task.get("tools"):
147
+ calls = first["tool_calls"]
148
+ expected = task["expected_call"]
149
+ if len(calls)!=1 or calls[0]["function"]["name"]!=expected["name"] or not json_equal(json.loads(calls[0]["function"]["arguments"]),expected["arguments"]):
150
+ result["error"] = "incorrect_tool_call"
151
+ else:
152
+ body["messages"].append({"role":"assistant","content":first["content"] or None,"tool_calls":calls})
153
+ body["messages"].append({"role":"tool","tool_call_id":calls[0]["id"],"content":json.dumps(task["tool_result"])})
154
+ body["tool_choice"] = "none"
155
+ response = stream(api, body)
156
+ result["calls"].append(response)
157
+ result["model_seconds"] = sum(c["wall_seconds"] for c in result["calls"])
158
+ result["response"] = response["content"]
159
+ if result["error"] is None and all(c["finish_reason"] != "length" for c in result["calls"]):
160
+ if task["suite"] == "gsm8k":
161
+ result["correct"] = gsm_score(response["content"],task["expected"])
162
+ elif task["suite"] == "tools":
163
+ try: result["correct"] = json_equal(json.loads(response["content"]),task["expected"])
164
+ except ValueError: pass
165
+ elif task["suite"] == "livecodebench":
166
+ result["code_score"] = code_score(response["content"],task)
167
+ result["correct"] = result["code_score"]["correct"]
168
+ else:
169
+ result["correct"] = None # official IFBench scoring, batched after generation
170
+ except Exception as e:
171
+ result["error"] = type(e).__name__ + ": " + str(e)[:500]
172
+ result["model_seconds"] = time.monotonic()-start
173
+ result["token_counts_incomplete"] = True
174
+ result["task_seconds"] = time.monotonic()-start
175
+ result.setdefault("model_seconds", sum(c["wall_seconds"] for c in result["calls"]))
176
+ result["input_tokens"] = sum(c["usage"].get("prompt_tokens",0) for c in result["calls"])
177
+ result["output_tokens"] = sum(c["usage"].get("completion_tokens",0) for c in result["calls"])
178
+ result["total_tokens"] = result["input_tokens"] + result["output_tokens"]
179
+ result["truncated"] = any(c["finish_reason"] == "length" for c in result["calls"])
180
+ if result["truncated"]:
181
+ result["correct"] = False # token-limit truncation counts as a wrong answer
182
+ return result
183
+
184
+
185
+ def aggregate(rows):
186
+ correct = sum(bool(r["correct"]) and not r["truncated"] for r in rows)
187
+ truncated = sum(r["truncated"] for r in rows)
188
+ return {"attempted":len(rows),"correct":correct,"accuracy":correct/len(rows),
189
+ "truncation_policy": "count_as_wrong",
190
+ "truncated_counted_as_wrong":truncated,
191
+ "errors":sum(r["error"] is not None for r in rows),"truncated":sum(r["truncated"] for r in rows),
192
+ "incomplete_token_counts":sum(r.get("token_counts_incomplete",False) for r in rows),
193
+ "mean_output_tokens":statistics.mean(r["output_tokens"] for r in rows),
194
+ "mean_input_tokens":statistics.mean(r["input_tokens"] for r in rows),
195
+ "mean_total_tokens":statistics.mean(r["input_tokens"] + r["output_tokens"] for r in rows),
196
+ "mean_model_seconds":statistics.mean(r["model_seconds"] for r in rows),
197
+ "median_model_seconds":statistics.median(r["model_seconds"] for r in rows),
198
+ "p95_model_seconds":sorted(r["model_seconds"] for r in rows)[math.ceil(.95*len(rows))-1],
199
+ "summed_request_seconds_per_correct":sum(r["model_seconds"] for r in rows)/correct if correct else None,
200
+ "note":"Summed request seconds are not GPU compute time when concurrency exceeds one."}
201
+
202
+
203
+ def main():
204
+ ap=argparse.ArgumentParser()
205
+ ap.add_argument("tag")
206
+ ap.add_argument("--api",default="http://127.0.0.1:18021/v1")
207
+ ap.add_argument("--pilot",action="store_true")
208
+ ap.add_argument("--suites",default="gsm8k,ifbench,livecodebench,tools")
209
+ ap.add_argument("--concurrency",type=int,default=1)
210
+ args=ap.parse_args()
211
+ tasks=records(RUN/"evaluation/tasks.jsonl")
212
+ wanted=set(read_json(RUN/"evaluation/pilot-ids.json")) if args.pilot else None
213
+ tasks=[t for t in tasks if t["suite"] in args.suites.split(",") and (wanted is None or t["id"] in wanted)]
214
+ folder=RUN/"results"/args.tag
215
+ folder.mkdir(parents=True,exist_ok=True)
216
+ manifest={"tasks_sha256":sha256(RUN/"evaluation/tasks.jsonl"),"task_ids":[t["id"] for t in tasks],
217
+ "concurrency":args.concurrency,"sampling":"thinking: temperature1/top_p0.95/top_k20/xhigh; nonthinking: greedy; seed15027", "api":args.api}
218
+ identity = read_json(RUN/"active-server.json")
219
+ assert not identity.get("stopped"), "No managed benchmark server is active"
220
+ manifest["server"] = {k:v for k,v in identity.items() if k not in {"created", "pid"}}
221
+ if (folder/"manifest.json").exists():
222
+ assert read_json(folder/"manifest.json")==manifest,"Cannot resume with different settings"
223
+ write_json(folder/"manifest.json",manifest)
224
+ output=folder/"tasks.jsonl"
225
+ old=records(output) if output.exists() else []
226
+ done={r["id"] for r in old}
227
+ todo=[t for t in tasks if t["id"] not in done]
228
+ start=time.monotonic()
229
+ with output.open("a") as f, concurrent.futures.ThreadPoolExecutor(args.concurrency) as pool:
230
+ futures=[pool.submit(execute,t,args.api) for t in todo]
231
+ for future in concurrent.futures.as_completed(futures):
232
+ result=future.result()
233
+ f.write(json.dumps(result)+"\n");f.flush()
234
+ print(result["id"],"correct=",result["correct"],"tokens=",result["output_tokens"],"seconds=",round(result["model_seconds"],2),"error=",result["error"],flush=True)
235
+ elapsed=time.monotonic()-start
236
+ rows=records(output)
237
+ for row in rows:
238
+ if row["truncated"]:
239
+ row["correct"] = False
240
+ lookup={t["id"]:t for t in tasks}
241
+ pending=[r for r in rows if r["suite"]=="ifbench" and r["correct"] is None and not r["truncated"]]
242
+ if pending:
243
+ env=dict(os.environ,NLTK_DATA=str(RUN/"nltk_data"))
244
+ r=subprocess.run([str(RUN/"eval-venv/bin/python"),str(ROOT/"swift15/ifbench_score.py")],
245
+ input=json.dumps([{"task":lookup[r["id"]],"response":r["response"]} for r in pending]),
246
+ text=True,capture_output=True,check=True,env=env)
247
+ scores=json.loads(r.stdout)
248
+ for row,score in zip(pending,scores):row.update(score)
249
+ from common import write_records
250
+ write_records(folder/"scored.jsonl",rows)
251
+ groups=collections.defaultdict(list)
252
+ for row in rows:groups[row["suite"]].append(row)
253
+ summary={"created":stamp(),"concurrency":args.concurrency,"resumed":bool(old),
254
+ "new_run_wall_seconds":elapsed,"new_tasks":len(todo),"suites":{k:aggregate(v) for k,v in groups.items()}}
255
+ summary["quality_comparison_ready"] = not any(r.get("token_counts_incomplete") for r in rows)
256
+ summary["truncation_policy"] = "count_as_wrong"
257
+ summary["truncated_task_ids"] = [r["id"] for r in rows if r["truncated"]]
258
+ if not old:
259
+ correct=sum(bool(r["correct"]) for r in rows)
260
+ summary.update(suite_wall_seconds=elapsed,wall_seconds_per_correct=elapsed/correct if correct else None)
261
+ write_json(folder/"summary.json",summary)
262
+ print(json.dumps(summary,indent=2),flush=True)
263
+
264
+
265
+ if __name__=="__main__":main()
evaluation/code/swift15/generate.py ADDED
@@ -0,0 +1,88 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Generate fresh Swift responses; resumable by prompt hash and model identity."""
2
+ import os
3
+ os.environ.setdefault("FLASHINFER_DISABLE_VERSION_CHECK", "1")
4
+ os.environ.setdefault("VLLM_USE_FLASHINFER_SAMPLER", "0")
5
+ os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True")
6
+ import argparse
7
+ import json
8
+ import time
9
+ from common import RUN, BASELINE, records, read_json, write_json, sha256, stamp
10
+
11
+
12
+ def main():
13
+ ap = argparse.ArgumentParser()
14
+ ap.add_argument("--model", default=str(BASELINE))
15
+ ap.add_argument("--limit", type=int)
16
+ ap.add_argument("--chunk", type=int, default=128)
17
+ args = ap.parse_args()
18
+ from transformers import AutoTokenizer
19
+ from vllm import LLM
20
+ data = RUN / "calibration"
21
+ prompts = records(data / "prompts.jsonl")
22
+ if args.limit:
23
+ prompts = prompts[:args.limit]
24
+ manifest = {"model": args.model, "config_sha256": sha256(os.path.join(args.model, "config.json")),
25
+ "prompt_manifest_sha256": sha256(data / "manifest.json"),
26
+ "thinking_cap": 2048, "nonthinking_cap": 1024, "seed": 15027,
27
+ "temperature": 1.0, "top_p": .95, "top_k": 20,
28
+ "max_model_len": 8192, "kv_cache_dtype": "fp8", "int8_activations": False}
29
+ mp = data / "generation-manifest.json"
30
+ if mp.exists():
31
+ assert read_json(mp) == manifest, "Generation settings changed; use a separate run directory"
32
+ else:
33
+ write_json(mp, manifest)
34
+ output = data / "gen.jsonl"
35
+ done = {r["id"]: r for r in records(output)} if output.exists() else {}
36
+ for p in prompts:
37
+ if p["id"] in done:
38
+ assert p["prompt_sha256"] == done[p["id"]]["prompt_sha256"]
39
+ todo = [p for p in prompts if p["id"] not in done]
40
+ if not todo:
41
+ print("All requested examples already generated")
42
+ return
43
+ tok = AutoTokenizer.from_pretrained(args.model)
44
+ llm = LLM(model=args.model, gpu_memory_utilization=.93, max_model_len=8192,
45
+ max_num_seqs=32, max_num_batched_tokens=2048, kv_cache_dtype="fp8",
46
+ mamba_ssm_cache_dtype="float16", language_model_only=True,
47
+ enable_prefix_caching=False,
48
+ compilation_config={"max_cudagraph_capture_size": 32, "custom_ops": ["+rms_norm", "+silu_and_mul"]})
49
+ base = llm.get_default_sampling_params()
50
+ t0 = time.monotonic()
51
+ token_count = 0
52
+ with output.open("a") as f:
53
+ for start in range(0, len(todo), args.chunk):
54
+ batch, inputs, params = [], [], []
55
+ for p in todo[start:start + args.chunk]:
56
+ kwargs = {"tools": p["tools"]} if p.get("tools") else {}
57
+ text = tok.apply_chat_template(p["messages"], tokenize=False, add_generation_prompt=True,
58
+ enable_thinking=p["think"], **kwargs)
59
+ ids = tok.encode(text, add_special_tokens=False)
60
+ if len(ids) > 6000:
61
+ raise ValueError(f"Prompt {p['id']} exceeds calibration context: {len(ids)}")
62
+ sp = base.clone()
63
+ sp.max_tokens = 2048 if p["think"] else 1024
64
+ sp.temperature, sp.top_p, sp.top_k = 1.0, .95, 20
65
+ sp.seed = 15027 + p["id"]
66
+ batch.append((p, ids))
67
+ inputs.append({"prompt_token_ids": ids})
68
+ params.append(sp)
69
+ results = llm.generate(inputs, params, use_tqdm=False)
70
+ for (p, ids), result in zip(batch, results):
71
+ c = result.outputs[0]
72
+ r = {k: p[k] for k in ["id", "src", "think", "split", "prompt_sha256"]}
73
+ r.update(prompt_ids=ids, output_ids=list(c.token_ids), finish=c.finish_reason)
74
+ f.write(json.dumps(r) + "\n")
75
+ token_count += len(c.token_ids)
76
+ f.flush()
77
+ elapsed = time.monotonic() - t0
78
+ print(f"{start+len(batch)}/{len(todo)} examples; {token_count} output tokens; {token_count/elapsed:.1f} tok/s; {elapsed/60:.1f} min", flush=True)
79
+ rows = records(output)
80
+ write_json(data / "generation-summary.json", {
81
+ "completed_at": stamp(), "examples": len(rows), "output_tokens": sum(len(r["output_ids"]) for r in rows),
82
+ "truncated": sum(r["finish"] == "length" for r in rows),
83
+ "note": "Calibration caps are not evaluation budgets; truncation rates are disclosed.",
84
+ })
85
+
86
+
87
+ if __name__ == "__main__":
88
+ main()
evaluation/code/swift15/ifbench_score.py ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Official IFBench strict prompt-level scoring in the isolated eval environment."""
2
+ import json
3
+ import sys
4
+ from ifbench import instructions_registry
5
+
6
+
7
+ def score(task, response):
8
+ passed = []
9
+ for name, kwargs in zip(task["instruction_id_list"], task["kwargs"]):
10
+ checker = instructions_registry.INSTRUCTION_DICT[name](name)
11
+ checker.build_description(**{k: v for k, v in kwargs.items() if v is not None})
12
+ needed = checker.get_instruction_args()
13
+ if needed and "prompt" in needed:
14
+ checker.build_description(prompt=task["messages"][-1]["content"])
15
+ passed.append(bool(response.strip()) and bool(checker.check_following(response)))
16
+ return {"correct": all(passed), "instructions": passed}
17
+
18
+
19
+ if __name__ == "__main__":
20
+ rows = json.load(sys.stdin)
21
+ print(json.dumps([score(r["task"], r["response"]) for r in rows]))
evaluation/code/swift15/measure.py ADDED
@@ -0,0 +1,94 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Frozen teacher-forced perplexity and fixed-output serving throughput."""
2
+ import argparse
3
+ import collections
4
+ import concurrent.futures
5
+ import json
6
+ import math
7
+ import statistics
8
+ import re
9
+ import time
10
+ import urllib.request
11
+ from common import RUN, records, read_json, write_json, sha256
12
+ from evaluate import stream, post, key
13
+
14
+
15
+ def perplexity(api):
16
+ windows = records(RUN / "evaluation/perplexity.jsonl")
17
+ rows = []
18
+ for i, window in enumerate(windows):
19
+ with post(api, "/completions", {"model": "qwen3.8-27b", "prompt": window["text"],
20
+ "max_tokens": 1, "temperature": 0, "prompt_logprobs": 0}) as response:
21
+ result = json.load(response)
22
+ values = []
23
+ for entry in result["choices"][0]["prompt_logprobs"][1:]:
24
+ if entry is None or len(entry) != 1:
25
+ raise ValueError("Expected exactly the observed token's log probability")
26
+ lp = next(iter(entry.values()))
27
+ lp = lp["logprob"] if isinstance(lp, dict) else lp
28
+ assert math.isfinite(lp)
29
+ values.append(lp)
30
+ assert values
31
+ rows.append({"id": i, "language": window["language"], "tokens": len(values), "logprob_sum": sum(values)})
32
+ groups = collections.defaultdict(list)
33
+ for r in rows:
34
+ groups[r["language"]].append(r)
35
+ groups["all"].append(r)
36
+ return {"input_sha256": sha256(RUN / "evaluation/perplexity.jsonl"), "windows": rows,
37
+ "scores": {name: {"tokens": sum(r["tokens"] for r in group),
38
+ "ppl": math.exp(-sum(r["logprob_sum"] for r in group)/sum(r["tokens"] for r in group))}
39
+ for name, group in groups.items()}}
40
+
41
+
42
+ def metrics(api):
43
+ req = urllib.request.Request(api.removesuffix("/v1") + "/metrics", headers={"Authorization": "Bearer " + key()})
44
+ with urllib.request.urlopen(req, timeout=10) as response:
45
+ return response.read().decode()
46
+
47
+
48
+ def throughput(api, concurrency):
49
+ prompts = ["Explain how a hash table handles collisions, with examples.",
50
+ "Write a Python implementation of merge sort and explain its complexity.",
51
+ "Describe how to design a reliable background job queue.",
52
+ "Explain photosynthesis and the role of chlorophyll."]
53
+ def one(i):
54
+ return stream(api, {"model": "qwen3.8-27b", "messages": [{"role": "user", "content": prompts[i % len(prompts)]}],
55
+ "temperature": 0, "seed": 15027, "max_tokens": 512, "ignore_eos": True,
56
+ "chat_template_kwargs": {"enable_thinking": False}})
57
+ one(0) # identical warm-up for every model
58
+ before = metrics(api)
59
+ start = time.monotonic()
60
+ with concurrent.futures.ThreadPoolExecutor(concurrency) as pool:
61
+ rows = list(pool.map(one, range(max(8, concurrency*2))))
62
+ seconds = time.monotonic()-start
63
+ after = metrics(api)
64
+ def counters(text):
65
+ totals = collections.Counter()
66
+ for line in text.splitlines():
67
+ match = re.match(r'(vllm:spec_decode_num_(?:drafts|draft_tokens|accepted_tokens)_total)(?:\{[^}]*\})?\s+([\d.eE+-]+)',line)
68
+ if match:
69
+ totals[match[1]] += float(match[2])
70
+ return totals
71
+ delta = counters(after)
72
+ delta.subtract(counters(before))
73
+ assert all(r["usage"]["completion_tokens"] == 512 for r in rows), "Fixed-length benchmark stopped early"
74
+ return {"concurrency": concurrency, "requests": len(rows), "output_tokens_per_request": 512,
75
+ "wall_seconds": seconds, "aggregate_output_tps": sum(r["usage"]["completion_tokens"] for r in rows)/seconds,
76
+ "median_decode_tps": statistics.median(r["decode_tps"] for r in rows),
77
+ "median_ttft_seconds": statistics.median(r["ttft_seconds"] for r in rows),
78
+ "sampling": "greedy; ignore_eos for this throughput test only", "calls": rows,
79
+ "speculation_counter_deltas": dict(delta),
80
+ "decode_tps_note": "Client stream timing estimate; speculative decoding may deliver several tokens in one event.",
81
+ "metrics_before": before, "metrics_after": after}
82
+
83
+
84
+ if __name__ == "__main__":
85
+ ap = argparse.ArgumentParser()
86
+ ap.add_argument("tag")
87
+ ap.add_argument("--api", default="http://127.0.0.1:18021/v1")
88
+ ap.add_argument("--kind", choices=["ppl", "speed"], required=True)
89
+ ap.add_argument("--concurrency", type=int, default=1)
90
+ args = ap.parse_args()
91
+ result = perplexity(args.api) if args.kind == "ppl" else throughput(args.api, args.concurrency)
92
+ target = RUN / "results" / args.tag / ("perplexity.json" if args.kind == "ppl" else f"speed-c{args.concurrency}.json")
93
+ write_json(target, result)
94
+ print(target, flush=True)
evaluation/code/swift15/quantize.py ADDED
@@ -0,0 +1,183 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Calibrate INT4 heads from pristine BF16 tensors and publish a local fast variant."""
2
+ import argparse
3
+ import collections
4
+ import copy
5
+ import json
6
+ import math
7
+ import os
8
+ import sys
9
+ from pathlib import Path
10
+
11
+ import numpy as np
12
+ import torch
13
+ from safetensors import safe_open
14
+ from safetensors.torch import save_file
15
+ from compressed_tensors.compressors.pack_quantized.base import pack_to_int32
16
+ from common import ROOT, RUN, SOURCE, BASELINE, FAST, read_json, write_json, records, sha256, stamp
17
+ from checkpoint import clone, audit
18
+ sys.path.insert(0, str(ROOT / "drafter"))
19
+ from gptq_utils import accumulate_hessian, gptq_quantize, rtn_quantize, dequant
20
+
21
+
22
+ def original(key):
23
+ index = read_json(SOURCE / "model.safetensors.index.json")["weight_map"]
24
+ with safe_open(SOURCE / index[key], "pt") as f:
25
+ return f.get_tensor(key)
26
+
27
+
28
+ def vocabulary():
29
+ from transformers import AutoTokenizer
30
+ rows = records(RUN / "calibration/gen.jsonl")
31
+ counts, held = collections.Counter(), collections.Counter()
32
+ for r in rows:
33
+ (held if r["split"] == "holdout" else counts).update(r["output_ids"])
34
+ assert counts and held
35
+ tok = AutoTokenizer.from_pretrained(BASELINE)
36
+ special = set(tok.all_special_ids)
37
+ total = sum(counts.values())
38
+ selected = set(special)
39
+ mass = sum(counts[t] for t in selected)
40
+ for token, count in counts.most_common():
41
+ if token not in selected:
42
+ selected.add(token)
43
+ mass += count
44
+ if mass / total >= .998 and len(selected) >= 16384:
45
+ break
46
+ if len(selected) >= 65536:
47
+ break
48
+ target_size = min(65536, max(16384, math.ceil(len(selected) / 128) * 128))
49
+ for token in list(counts) + torch.load(BASELINE / "mtp_draft_vocab_ids.pt", weights_only=True).tolist():
50
+ if len(selected) >= target_size:
51
+ break
52
+ selected.add(token)
53
+ assert len(selected) == target_size
54
+ ids = sorted(selected)
55
+ report = {"size": len(ids), "calibration_tokens": total,
56
+ "calibration_coverage": sum(counts[t] for t in ids) / total,
57
+ "holdout_tokens": sum(held.values()),
58
+ "holdout_coverage": sum(held[t] for t in ids) / sum(held.values()),
59
+ "selection": "99.8% calibration output-token coverage, min 16384 and max 65536; rounded to 128 rows; all special ids included; reference IDs only pad if calibration vocabulary is too small"}
60
+ write_json(RUN / "calibration/draft_vocab_ids.json", ids)
61
+ write_json(RUN / "calibration/vocabulary-report.json", report)
62
+ print(json.dumps(report), flush=True)
63
+
64
+
65
+ def replace_tensors(model, replacements, bits):
66
+ index = read_json(model / "model.safetensors.index.json")
67
+ grouped = collections.defaultdict(dict)
68
+ for module, tensors in replacements.items():
69
+ shard = index["weight_map"][module + ".weight_packed"]
70
+ grouped[shard][module] = tensors
71
+ for shard, modules in grouped.items():
72
+ # Never open a shared inode for writing: replace via a new file.
73
+ with safe_open(model / shard, "pt") as f:
74
+ data = {k: f.get_tensor(k) for k in f.keys()
75
+ if not any(k.startswith(m + ".") for m in modules)}
76
+ for m, tensors in modules.items():
77
+ data.update({m + "." + k: v for k, v in tensors.items()})
78
+ temporary = model / (shard + ".tmp")
79
+ save_file(data, temporary, metadata={"format": "pt"})
80
+ os.replace(temporary, model / shard)
81
+ del data
82
+ config = read_json(model / "config.json")
83
+ for group in bits:
84
+ config["quantization_config"]["config_groups"][group]["weights"]["num_bits"] = bits[group]
85
+ write_json(model / "config.json", config)
86
+ if (model / "quantization_config.json").exists():
87
+ write_json(model / "quantization_config.json", config["quantization_config"])
88
+
89
+
90
+ def packed(q, scale):
91
+ return {"weight_packed": pack_to_int32(q.cpu(), 4, packed_dim=1).contiguous(),
92
+ "weight_scale": scale.cpu().to(torch.float16).contiguous(),
93
+ "weight_shape": torch.tensor(list(q.shape), dtype=torch.int64)}
94
+
95
+
96
+ def output_head(calib_rows):
97
+ clone(BASELINE, FAST)
98
+ hiddens = np.load(RUN / "calibration/hidden.npy", mmap_mode="r")
99
+ seqs = read_json(RUN / "calibration/seqs.json")
100
+ rng = np.random.default_rng(15027)
101
+ pools = {"calibration": [], "holdout": []}
102
+ for seq in seqs:
103
+ # Include output-producing states; exclude prompt body and final nonpredicting row.
104
+ start = seq["off"] + max(0, seq["n_prompt"] - 1)
105
+ end = seq["off"] + seq["n"] - 1
106
+ if end > start:
107
+ pools[seq["split"]].append(np.arange(start, end, dtype=np.int64))
108
+ selected = {}
109
+ for split, size in [("calibration", calib_rows), ("holdout", 1024)]:
110
+ pool = np.concatenate(pools[split])
111
+ selected[split] = np.sort(rng.choice(pool, size=min(size, len(pool)), replace=False))
112
+ assert not np.intersect1d(selected["calibration"], selected["holdout"]).size
113
+ W = original("lm_head.weight").cuda()
114
+ H = torch.zeros(W.shape[1], W.shape[1], device="cuda")
115
+ seen = 0
116
+ for start in range(0, len(selected["calibration"]), 8192):
117
+ x = torch.from_numpy(np.array(hiddens[selected["calibration"][start:start+8192]])).view(torch.bfloat16).cuda()
118
+ H, seen = accumulate_hessian(H, x, seen)
119
+ del x
120
+ held = torch.from_numpy(np.array(hiddens[selected["holdout"]])).view(torch.bfloat16).cuda()
121
+
122
+ @torch.no_grad()
123
+ def kl(q, scale):
124
+ dq = dequant(q.cuda(), scale.cuda()).to(torch.bfloat16)
125
+ total = 0.
126
+ for start in range(0, len(held), 64):
127
+ x = held[start:start+64]
128
+ p = (x @ W.t()).float().log_softmax(-1)
129
+ lp = (x @ dq.t()).float().log_softmax(-1)
130
+ total += (p.exp() * (p-lp)).sum().item()
131
+ del dq
132
+ return total / len(held)
133
+
134
+ q, s = rtn_quantize(W, 4, 128)
135
+ rtn_kl = kl(q, s)
136
+ del q, s
137
+ torch.cuda.empty_cache()
138
+ qs, scales = [], []
139
+ for start in range(0, W.shape[0], 8192):
140
+ q, s = gptq_quantize(W[start:start+8192], H, bits=4, group=128)
141
+ qs.append(q.cpu()); scales.append(s.cpu())
142
+ print("lm_head rows", start + len(q), "/", W.shape[0], flush=True)
143
+ del q, s
144
+ q, s = torch.cat(qs), torch.cat(scales)
145
+ gptq_kl = kl(q, s)
146
+ assert math.isfinite(gptq_kl), gptq_kl
147
+ report = {"calibration_rows": seen, "holdout_rows": len(held),
148
+ "rtn_int4_kl": rtn_kl, "gptq_int4_kl": gptq_kl,
149
+ "original": "pristine source lm_head BF16", "row_split": "whole examples before generation"}
150
+ print(json.dumps(report), flush=True)
151
+ # A worse result must be reviewed rather than silently promoted.
152
+ if gptq_kl > rtn_kl:
153
+ raise RuntimeError("GPTQ failed to improve held-out KL over round-to-nearest")
154
+ replace_tensors(FAST, {"lm_head": packed(q, s)}, {"group_1": 4})
155
+ write_json(RUN / "calibration/lm-head-report.json", report)
156
+
157
+
158
+ def mtp():
159
+ hessians = torch.load(RUN / "calibration/mtp_hessians.pt", weights_only=True, map_location="cpu")
160
+ replacements, report = {}, {}
161
+ for module, hessian in hessians.items():
162
+ W = original(module + ".weight").cuda()
163
+ q, scale = gptq_quantize(W, hessian.cuda(), bits=4, group=128)
164
+ error = ((dequant(q, scale) - W.float()).norm() / W.float().norm()).item()
165
+ assert math.isfinite(error)
166
+ report[module] = {"relative_weight_error": error, "shape": list(W.shape)}
167
+ replacements[module] = packed(q, scale)
168
+ print(module, error, flush=True)
169
+ del W, q, scale
170
+ torch.cuda.empty_cache()
171
+ assert len(replacements) == 8, list(replacements)
172
+ replace_tensors(FAST, replacements, {"group_3": 4})
173
+ write_json(RUN / "calibration/mtp-report.json", report)
174
+
175
+
176
+ if __name__ == "__main__":
177
+ ap = argparse.ArgumentParser()
178
+ ap.add_argument("stage", choices=["vocabulary", "lm-head", "mtp"])
179
+ ap.add_argument("--calib-rows", type=int, default=300000)
180
+ args = ap.parse_args()
181
+ if args.stage == "vocabulary": vocabulary()
182
+ elif args.stage == "lm-head": output_head(args.calib_rows)
183
+ else: mtp()
evaluation/code/swift15/reference.py ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Repair the reference's embedding/index mismatch in a separate evaluation copy."""
2
+ import shutil
3
+ import os
4
+ import subprocess
5
+ import sys
6
+ from common import ROOT, RUN, read_json, write_json, stamp
7
+ from checkpoint import clone, audit
8
+
9
+ SOURCE = ROOT / "models/Qwen3.8-27B-W4A16-AutoRound-fast"
10
+ TARGET = ROOT / "models/Qwen3.8-27B-W4A16-AutoRound-fast-eval"
11
+
12
+
13
+ def main():
14
+ if (TARGET / "embedding-repair.json").exists():
15
+ audit(TARGET)
16
+ return
17
+ clone(SOURCE, TARGET)
18
+ index = read_json(TARGET / "model.safetensors.index.json")
19
+ prefix = "model.language_model.embed_tokens"
20
+ shard = index["weight_map"][prefix + ".weight_packed"]
21
+ from safetensors import safe_open
22
+ with safe_open(TARGET / shard, "pt") as file:
23
+ assert prefix + ".weight" in file.keys()
24
+ assert prefix + ".weight_packed" not in file.keys()
25
+ # The legacy converter writes in place; detach this shard from source inodes.
26
+ temporary = TARGET / (shard + ".private")
27
+ shutil.copyfile(TARGET / shard, temporary)
28
+ os.replace(temporary, TARGET / shard)
29
+ for suffix in ["weight_packed", "weight_scale", "weight_shape"]:
30
+ del index["weight_map"][prefix + "." + suffix]
31
+ index["weight_map"][prefix + ".weight"] = shard
32
+ write_json(TARGET / "model.safetensors.index.json", index)
33
+ subprocess.run([sys.executable, str(ROOT / "prepare/quant_embed.py"), str(TARGET)], check=True)
34
+ audit(TARGET)
35
+ write_json(TARGET / "embedding-repair.json", {"created": stamp(), "source": str(SOURCE),
36
+ "reason": "Source index declares INT8 embeddings but its shard contains BF16 embeddings.",
37
+ "change": "Apply the repository's standard INT8 group128 embedding conversion in an isolated copy; source preserved."})
38
+
39
+
40
+ if __name__ == "__main__":
41
+ main()
evaluation/code/swift15/release.py ADDED
@@ -0,0 +1,133 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Prepare an auditable local release; this script never uploads anything."""
2
+ import json
3
+ import shutil
4
+ from pathlib import Path
5
+ from safetensors import safe_open
6
+ from common import ROOT, RUN, SOURCE, BASELINE, FAST, read_json, write_json, sha256, stamp
7
+ from checkpoint import audit
8
+
9
+
10
+ def verify_body():
11
+ import torch
12
+ source = read_json(SOURCE / "model.safetensors.index.json")["weight_map"]
13
+ target = read_json(FAST / "model.safetensors.index.json")["weight_map"]
14
+ count = 0
15
+ for name, shard in source.items():
16
+ if name.startswith(("mtp.", "lm_head.", "model.language_model.embed_tokens.")):
17
+ continue
18
+ a, b = SOURCE / shard, FAST / target[name]
19
+ if a.stat().st_ino != b.stat().st_ino or a.stat().st_dev != b.stat().st_dev:
20
+ with safe_open(a,"pt") as left, safe_open(b,"pt") as right:
21
+ assert torch.equal(left.get_tensor(name),right.get_tensor(name)), f"Unintended body change: {name}"
22
+ count += 1
23
+ return count
24
+
25
+
26
+ def main():
27
+ unchanged = verify_body()
28
+ index = read_json(FAST / "model.safetensors.index.json")
29
+ sizes = {"F64": 8, "F32": 4, "F16": 2, "BF16": 2, "I64": 8, "I32": 4, "I16": 2, "I8": 1, "U8": 1, "BOOL": 1}
30
+ import math
31
+ total = 0
32
+ for shard in set(index["weight_map"].values()):
33
+ with safe_open(FAST / shard, "pt") as f:
34
+ for name in f.keys():
35
+ tensor = f.get_slice(name)
36
+ total += math.prod(tensor.get_shape()) * sizes[tensor.get_dtype()]
37
+ index.setdefault("metadata", {})["total_size"] = total
38
+ write_json(FAST / "model.safetensors.index.json", index)
39
+ result = audit(FAST)
40
+ provenance = FAST / "hyperqwen_provenance"
41
+ provenance.mkdir(exist_ok=True)
42
+ for name in ["LICENSE", "LICENSE-APACHE-2.0", "NOTICE"]:
43
+ assert (SOURCE / name).exists(), f"Missing upstream {name}"
44
+ shutil.copy2(SOURCE / name, FAST / name)
45
+ for name in ["README.md", "QUANTIZATION_MANIFEST.json", "UPLOAD_MANIFEST.json", "recipe.yaml"]:
46
+ if (SOURCE / name).exists():
47
+ shutil.copy2(SOURCE / name, provenance / ("upstream-" + name))
48
+ for name in ["manifest.json", "generation-manifest.json", "generation-summary.json", "vocabulary-report.json", "lm-head-report.json", "mtp-report.json", "mtp_hessians.pt.json"]:
49
+ shutil.copy2(RUN / "calibration" / name, provenance / name)
50
+ for source in list((ROOT/"swift15").glob("*.py")) + [ROOT/"swift15/README.md", ROOT/"drafter/gptq_utils.py",
51
+ ROOT/"drafter/capture.py", ROOT/"drafter/train_mtp.py", ROOT/"prepare/build_draft_vocab.py", ROOT/"prepare/quant_heads_stream.py"]:
52
+ destination = provenance/"recipe"/source.relative_to(ROOT)
53
+ destination.parent.mkdir(parents=True,exist_ok=True)
54
+ shutil.copy2(source,destination)
55
+ for name in ["runtime-compatibility.json", "preflight-validation.json"]:
56
+ if (RUN / name).exists():
57
+ shutil.copy2(RUN / name, provenance / name)
58
+ mtp_replay = read_json(RUN / "calibration/mtp_hessians.pt.json")
59
+ recipe = {"created": stamp(), "source": read_json(RUN / "source.json"), "variant": "fast",
60
+ "preflight_only": "preflight" in RUN.parts,
61
+ "unchanged_body_tensors_verified": unchanged,
62
+ "body": "unchanged asymmetric AWQ INT4 group128", "embedding_bits": 8,
63
+ "lm_head_bits": 4, "mtp_bits": 4, "group_size": 128, "act_order": False,
64
+ "gptq": {"damping": .01, "lm_head_rows": read_json(RUN / "calibration/lm-head-report.json")["calibration_rows"],
65
+ "mtp_examples": len(mtp_replay["examples"]), "mtp_depths": mtp_replay["depths"]},
66
+ "pristine_inputs": "Heads quantized directly from source BF16 tensors, not from INT8 tensors",
67
+ "calibration": read_json(RUN / "calibration/manifest.json"),
68
+ "runtime": read_json(RUN / "environment.json"),
69
+ "script_hashes": {str(p.relative_to(ROOT)): sha256(p) for p in list((ROOT / "swift15").glob("*.py")) +
70
+ [ROOT / "drafter/gptq_utils.py", ROOT / "drafter/capture.py", ROOT / "drafter/train_mtp.py", ROOT / "prepare/build_draft_vocab.py"]}}
71
+ write_json(FAST / "hyperqwen-build.json", recipe)
72
+ write_json(FAST / "QUANTIZATION_MANIFEST.json", recipe)
73
+ # The inherited upload list describes the upstream checkpoint, not this one.
74
+ if (FAST / "UPLOAD_MANIFEST.json").exists():
75
+ (FAST / "UPLOAD_MANIFEST.json").unlink()
76
+ with (FAST / "NOTICE").open("a") as out:
77
+ out.write("\nHyperQwen-compatible local derivative: INT8 embeddings; calibrated GPTQ INT4 output and MTP linear weights; reduced MTP draft vocabulary. The upstream AWQ body is unchanged. Quantization recipe and source attribution are included in hyperqwen-build.json.\n")
78
+ report = RUN / "REPORT.md"
79
+ measured = report.read_text() if report.exists() else "Evaluation is pending. No throughput or quality claim has been established for this derivative."
80
+ card = """---
81
+ license: other
82
+ license_name: swift-open-license-1.0
83
+ license_link: LICENSE
84
+ base_model: ukisai/Swift-1.5-Qwen3.8-27b
85
+ base_model_relation: quantized
86
+ library_name: vllm
87
+ pipeline_tag: text-generation
88
+ tags:
89
+ - compressed-tensors
90
+ - awq
91
+ - gptq
92
+ - hyperqwen
93
+ ---
94
+
95
+ # Swift 1.5 Qwen3.8 27B — HyperQwen fast derivative
96
+
97
+ Local release candidate. This is an independently prepared quantization of UkisAI's
98
+ Swift 1.5, not an official UkisAI or HyperQwen release.
99
+
100
+ The original asymmetric AWQ INT4 body is unchanged. Embeddings use INT8 group128;
101
+ the full output head and eight MTP linear matrices use calibrated GPTQ INT4 group128.
102
+ GPTQ starts from the upstream BF16 head tensors. The MTP draft head uses a subset
103
+ of the output head selected using fresh Swift responses. This subset only limits
104
+ draft proposals; the target retains its full vocabulary. This is not fine-tuning.
105
+
106
+ Requires the patched HyperQwen runtime, including INT8 embeddings and reduced MTP
107
+ vocabulary support. The tested package versions and recipe are in hyperqwen-build.json.
108
+ Do not assume an unpatched Transformers/vLLM installation can load this checkpoint.
109
+ Multi-user serving can use the W4A16 body without MTP. Optional INT8 activations are
110
+ a separate runtime setting and require asymmetric Marlin support; they are not a
111
+ property of the stored checkpoint.
112
+
113
+ The upstream vision tensors are retained, but the initial evaluation is text-only.
114
+ Calibration sources, pinned revisions and whole-example holdout separation are
115
+ documented under hyperqwen_provenance. Raw benchmark answers and calibration
116
+ examples are not included. See LICENSE, LICENSE-APACHE-2.0 and NOTICE for the
117
+ upstream license and attribution. See hyperqwen_provenance/upstream-README.md for
118
+ the original model information.
119
+
120
+ ## Local evaluation
121
+
122
+ """ + measured + "\n"
123
+ if recipe["preflight_only"]:
124
+ card = card.replace("Local release candidate.", "Integration-test checkpoint only; exclude this small-subset preflight model from publication and final comparisons.")
125
+ (FAST / "README.md").write_text(card)
126
+ files = sorted(p.relative_to(FAST).as_posix() for p in FAST.rglob("*") if p.is_file()
127
+ and ".bak" not in p.name and not p.name.endswith(".tmp") and ".cache" not in p.parts)
128
+ write_json(RUN / "release-files.json", files)
129
+ print("Prepared local release:", FAST, ";", len(files), "files; no upload performed")
130
+
131
+
132
+ if __name__ == "__main__":
133
+ main()
evaluation/code/swift15/report.py ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Generate the comparison only from completed measurements."""
2
+ import math
3
+ from common import RUN, read_json, stamp
4
+
5
+
6
+ def interval(correct, n):
7
+ z = 1.96
8
+ p = correct / n
9
+ center = (p + z*z/(2*n))/(1+z*z/n)
10
+ radius = z*math.sqrt(p*(1-p)/n+z*z/(4*n*n))/(1+z*z/n)
11
+ return f"{100*p:.1f}% ({100*(center-radius):.1f}–{100*(center+radius):.1f})"
12
+
13
+
14
+ def main():
15
+ lines = ["# Local Swift / HyperQwen comparison", "", "Updated: " + stamp(), "",
16
+ "RTX 3090 measurements. A task is one independently scored problem; tool tasks include every model call. "
17
+ "Output tokens include reasoning. Latency is observed serving time, not a GPU-kernel compute measurement. "
18
+ "Quality runs use bounded budgets and one seed (15027). Thinking tasks use temperature 1.0, top_p 0.95, top_k 20; nonthinking tasks use greedy decoding. These are not publisher leaderboard scores.", "",
19
+ "The Qwen reference is the existing AutoRound fast checkpoint; Swift 1.0 is the existing AWQ/INT8-head checkpoint. "
20
+ "The Swift 1.5 INT8-head baseline and INT4-head fast derivative share the same upstream AWQ body. "
21
+ "A BF16 reference measurement is outside this four-checkpoint comparison.", "",
22
+ "## Fixed-output speed and perplexity", "",
23
+ "512 output tokens/request, concurrency 1; natural task lengths are reported separately.", "",
24
+ "| Model | Decode tok/s (median) | TTFT (s) | EN PPL | DA PPL | Code PPL |",
25
+ "|---|---:|---:|---:|---:|---:|"]
26
+ tags = ["qwen-fast", "swift10", "swift15-baseline", "swift15-fast"]
27
+ for tag in tags:
28
+ directory = RUN / "results" / tag
29
+ speed = read_json(directory / "speed-c1.json") if (directory / "speed-c1.json").exists() else {}
30
+ ppl = read_json(directory / "perplexity.json").get("scores", {}) if (directory / "perplexity.json").exists() else {}
31
+ def f(v): return "pending" if v is None else f"{v:.3f}"
32
+ lines.append("| " + " | ".join([tag, f(speed.get("median_decode_tps")), f(speed.get("median_ttft_seconds")),
33
+ *[f(ppl.get(k, {}).get("ppl")) for k in ["en", "da", "code"]]]) + " |")
34
+ lines += ["", "## Swift 1.5 multi-user serving", "",
35
+ "Same fast checkpoint, no MTP, concurrency eight. INT8 applies to MLP activations only.", "",
36
+ "| Runtime | Aggregate output tok/s | Combined PPL |", "|---|---:|---:|"]
37
+ for tag in ["swift15-fast-batch", "swift15-fast-batch-int8"]:
38
+ directory = RUN / "results" / tag
39
+ if not (directory / "speed-c8.json").exists(): continue
40
+ speed = read_json(directory / "speed-c8.json")
41
+ ppl = read_json(directory / "perplexity.json")["scores"]["all"]["ppl"] if (directory / "perplexity.json").exists() else None
42
+ lines.append(f"| {tag} | {speed['aggregate_output_tps']:.2f} | {ppl if ppl is not None else 'pending'} |")
43
+ for mode in ["pilot", "full"]:
44
+ lines += ["", "## " + ("Sequential task latency pilot" if mode == "pilot" else "Quality suite (concurrency recorded per run)"), "",
45
+ "Accuracy includes all attempts. Truncated responses count as wrong; their count is reported separately. Parentheses show descriptive 95% Wilson intervals. Small differences are inconclusive.", "",
46
+ "| Model | Task suite | Correct/total | Accuracy (95% CI) | Output tokens/task | Total tokens/task | Request seconds/task | Truncated | Errors |",
47
+ "|---|---|---:|---:|---:|---:|---:|---:|---:|"]
48
+ for tag in tags:
49
+ path = RUN / "results" / (tag + "-" + mode) / "summary.json"
50
+ if not path.exists(): continue
51
+ summary = read_json(path)
52
+ for suite, values in summary["suites"].items():
53
+ v = values
54
+ accuracy = interval(v['correct'],v['attempted'])
55
+ lines.append(f"| {tag} | {suite} | {v['correct']}/{v['attempted']} | {accuracy} | "
56
+ f"{v['mean_output_tokens']:.1f} | {v['mean_total_tokens']:.1f} | {v['mean_model_seconds']:.2f} | {v['truncated']} | {v['errors']} |")
57
+ lines += ["", "Full-suite request latencies overlap and must not be summed as GPU time. "
58
+ "Per-run JSON summaries also report wall-clock makespan and wall seconds per correct task, including failed attempts. "
59
+ "Resumed runs preserve attempt timings and omit a misleading whole-suite makespan.", "",
60
+ "LiveCodeBench is a fixed 100-problem, stdin-only v6 subset with a custom deterministic judge, "
61
+ "not an official full LCB score. IFBench uses the official instruction verifier. "
62
+ "The 20-task latency pilot has five tasks per category and is too small to establish quality equivalence.", ""]
63
+ (RUN / "REPORT.md").write_text("\n".join(lines))
64
+ print(RUN / "REPORT.md")
65
+
66
+
67
+ if __name__ == "__main__":
68
+ main()
evaluation/code/swift15/run.py ADDED
@@ -0,0 +1,162 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Resumable build and four-checkpoint evaluation, one GPU owner at a time."""
2
+ import argparse
3
+ import fcntl
4
+ import importlib.metadata
5
+ import json
6
+ import os
7
+ import signal
8
+ import subprocess
9
+ import sys
10
+ import time
11
+ from common import ROOT, RUN, SOURCE, BASELINE, FAST, read_json, write_json, records, sha256, stamp
12
+ from serve import server
13
+
14
+ PYTHON = str(ROOT / "venv/bin/python")
15
+
16
+
17
+ def stage(name, command, outputs, env=None):
18
+ markers = RUN / "stages"
19
+ marker = markers / (name + ".json")
20
+ signature = {"command": command, "env": env or {}}
21
+ signature["script_hashes"] = {part: sha256(ROOT / part) for part in command if part.endswith(".py") and (ROOT / part).is_file()}
22
+ if marker.exists():
23
+ old = read_json(marker)
24
+ assert old["signature"] == signature, f"Stage settings changed: {name}"
25
+ assert all(p.exists() for p in outputs), f"Stage output missing: {name}"
26
+ print("Already complete:", name, flush=True)
27
+ return
28
+ write_json(RUN / "status.json", {"stage": name, "state": "running", "started": stamp()})
29
+ print("Starting:", name, flush=True)
30
+ log = RUN / "logs" / (name + ".log")
31
+ log.parent.mkdir(parents=True, exist_ok=True)
32
+ started = time.monotonic()
33
+ with log.open("a") as out:
34
+ process = subprocess.Popen(command, cwd=ROOT, env=dict(os.environ, OMP_NUM_THREADS="8", **(env or {})),
35
+ stdout=out, stderr=subprocess.STDOUT, start_new_session=True)
36
+ try:
37
+ code = process.wait()
38
+ if code:
39
+ raise subprocess.CalledProcessError(code, command)
40
+ except BaseException:
41
+ try: os.killpg(process.pid, signal.SIGTERM)
42
+ except ProcessLookupError: pass
43
+ try: process.wait(timeout=45)
44
+ except subprocess.TimeoutExpired:
45
+ try: os.killpg(process.pid, signal.SIGKILL)
46
+ except ProcessLookupError: pass
47
+ process.wait(timeout=15)
48
+ raise
49
+ assert all(p.exists() for p in outputs), f"Missing outputs after {name}"
50
+ write_json(marker, {"signature": signature, "completed": stamp(), "seconds": time.monotonic()-started,
51
+ "outputs": {str(p.relative_to(ROOT)): p.stat().st_size for p in outputs}})
52
+
53
+
54
+ def build():
55
+ cal = RUN / "calibration"
56
+ stage("baseline", [PYTHON, "swift15/checkpoint.py", "baseline"], [BASELINE / "hyperqwen-build.json"])
57
+ stage("generation", [PYTHON, "-u", "swift15/generate.py"], [cal / "generation-summary.json"])
58
+ assert len(records(cal / "gen.jsonl")) == len(records(cal / "prompts.jsonl")), "Incomplete generation"
59
+ stage("vocabulary", [PYTHON, "swift15/quantize.py", "vocabulary"], [cal / "draft_vocab_ids.json"])
60
+ stage("capture", [PYTHON, "-u", "drafter/capture.py"], [cal / "hidden.npy", cal / "seqs.json"],
61
+ {"MODEL": str(BASELINE), "CALIBRATION_DATA": str(cal)})
62
+ stage("lm-head", [PYTHON, "-u", "swift15/quantize.py", "lm-head"], [cal / "lm-head-report.json"])
63
+ stage("draft-head", [PYTHON, "prepare/build_draft_vocab.py", str(FAST), "--ids", str(cal / "draft_vocab_ids.json")],
64
+ [FAST / "mtp_draft_vocab_ids.pt"])
65
+ stage("mtp-hessians", [PYTHON, "-u", "drafter/train_mtp.py", "--model", str(BASELINE),
66
+ "--original-model", str(SOURCE), "--data", str(cal), "--out", str(cal / "mtp-replay"),
67
+ "--eval-only", "1", "--dump-hessians", str(cal / "mtp_hessians.pt"),
68
+ "--max-seqs", "400", "--val-frac", ".4", "--depths", "2", "--micro-tokens", "1",
69
+ "--head-chunk", "128", "--draft-ids", str(cal / "draft_vocab_ids.json")], [cal / "mtp_hessians.pt"])
70
+ stage("mtp-quant", [PYTHON, "-u", "swift15/quantize.py", "mtp"], [cal / "mtp-report.json"])
71
+ stage("release", [PYTHON, "swift15/release.py"], [FAST / "hyperqwen-build.json", FAST / "hyperqwen-audit.json"])
72
+
73
+
74
+ def evaluate(full):
75
+ evaluation_manifest = read_json(RUN / "evaluation/manifest.json")
76
+ quality_concurrency = int(evaluation_manifest.get(
77
+ "quality_concurrency", os.environ.get("SWIFT15_EVAL_CONCURRENCY", "8")))
78
+ if quality_concurrency < 1:
79
+ raise ValueError("Quality concurrency must be positive")
80
+ reference = ROOT / "models/Qwen3.8-27B-W4A16-AutoRound-fast-eval"
81
+ stage("reference-embedding-repair", [PYTHON, "swift15/reference.py"], [reference / "embedding-repair.json"])
82
+ models = {"qwen-fast": reference,
83
+ "swift10": ROOT / "models/Swift-Qwen3.8-27b-W4A16-AWQ",
84
+ "swift15-baseline": BASELINE, "swift15-fast": FAST}
85
+ for tag, model in models.items():
86
+ finished = RUN / "stages" / (tag + ("-full" if full else "-pilot") + ".json")
87
+ if finished.exists():
88
+ continue
89
+ with server(model, tag):
90
+ stage(tag + "-speed", [PYTHON, "swift15/measure.py", tag, "--kind", "speed"],
91
+ [RUN / "results" / tag / "speed-c1.json"])
92
+ stage(tag + "-ppl", [PYTHON, "swift15/measure.py", tag, "--kind", "ppl"],
93
+ [RUN / "results" / tag / "perplexity.json"])
94
+ stage(tag + "-pilot-tasks", [PYTHON, "-u", "swift15/evaluate.py", tag + "-pilot", "--pilot"],
95
+ [RUN / "results" / (tag + "-pilot") / "summary.json"])
96
+ if full:
97
+ stage(tag + "-full-tasks", [PYTHON, "-u", "swift15/evaluate.py", tag + "-full", "--concurrency", str(quality_concurrency)],
98
+ [RUN / "results" / (tag + "-full") / "summary.json"])
99
+ write_json(finished, {"completed": stamp()})
100
+ subprocess.run([PYTHON, "swift15/report.py"], cwd=ROOT, check=True)
101
+ if os.environ.get("SWIFT15_SINGLE_USER_ONLY") == "1":
102
+ return
103
+ stage("runtime-compat", [PYTHON, "swift15/runtime_compat.py"], [RUN / "runtime-compatibility.json"])
104
+ for tag, int8 in [("swift15-fast-batch", False), ("swift15-fast-batch-int8", True)]:
105
+ finished = RUN / "stages" / (tag + ".json")
106
+ if finished.exists():
107
+ continue
108
+ with server(FAST, tag, batch=True, int8=int8):
109
+ stage(tag + "-speed", [PYTHON, "swift15/measure.py", tag, "--kind", "speed", "--concurrency", "8"],
110
+ [RUN / "results" / tag / "speed-c8.json"])
111
+ stage(tag + "-ppl", [PYTHON, "swift15/measure.py", tag, "--kind", "ppl"],
112
+ [RUN / "results" / tag / "perplexity.json"])
113
+ stage(tag + "-tasks", [PYTHON, "-u", "swift15/evaluate.py", tag, "--pilot", "--concurrency", "8"],
114
+ [RUN / "results" / tag / "summary.json"])
115
+ write_json(finished, {"completed": stamp(), "activation_dtype": "INT8 MLP only" if int8 else "BF16"})
116
+ subprocess.run([PYTHON, "swift15/report.py"], cwd=ROOT, check=True)
117
+ if (RUN / "calibration/manifest.json").exists():
118
+ subprocess.run([PYTHON, "swift15/release.py"], cwd=ROOT, check=True)
119
+
120
+
121
+ def main():
122
+ ap = argparse.ArgumentParser()
123
+ ap.add_argument("mode", choices=["build", "pilot", "full"])
124
+ ap.add_argument("--evaluation-only", action="store_true", help="Evaluate existing checkpoints in a separately prepared run directory")
125
+ ap.add_argument("--wait-for-pid", type=int, help="Wait for an already-running calibration process before acquiring the GPU")
126
+ args = ap.parse_args()
127
+ def stop(signum, frame):
128
+ raise KeyboardInterrupt("Pipeline terminated")
129
+ signal.signal(signal.SIGTERM, stop)
130
+ RUN.mkdir(parents=True, exist_ok=True)
131
+ with (RUN / "pipeline.lock").open("w") as lock:
132
+ fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB)
133
+ if args.wait_for_pid:
134
+ print("Waiting for existing calibration process", args.wait_for_pid, flush=True)
135
+ while os.path.exists(f"/proc/{args.wait_for_pid}"):
136
+ time.sleep(5)
137
+ hardware = subprocess.check_output(["nvidia-smi", "--query-gpu=name,uuid,memory.total,power.limit,driver_version", "--format=csv"], text=True)
138
+ versions = {name: importlib.metadata.version(name) for name in ["torch", "vllm", "transformers", "safetensors", "compressed-tensors"]}
139
+ commit = subprocess.check_output(["git", "rev-parse", "HEAD"],cwd=ROOT,text=True).strip()
140
+ write_json(RUN / "environment.json", {"started": stamp(), "hardware": hardware, "versions": versions,
141
+ "hyperqwen_commit":commit,"python":sys.version,
142
+ "patches_sha256":{p.name:sha256(p) for p in (ROOT/"patches").glob("*.patch")}})
143
+ try:
144
+ if not args.evaluation_only:
145
+ build()
146
+ if args.mode != "build":
147
+ evaluate(full=args.mode == "full")
148
+ truncated = {}
149
+ for path in (RUN / "results").glob("*/summary.json"):
150
+ result = read_json(path)
151
+ if result.get("truncated_task_ids"):
152
+ truncated[path.parent.name] = result["truncated_task_ids"]
153
+ write_json(RUN / "status.json", {"state": "complete",
154
+ "mode": args.mode, "completed": stamp(), "truncated_counted_as_wrong": truncated})
155
+ except BaseException as error:
156
+ previous = read_json(RUN / "status.json") if (RUN / "status.json").exists() else {}
157
+ write_json(RUN / "status.json", dict(previous, state="failed", error=repr(error), failed=stamp()))
158
+ raise
159
+
160
+
161
+ if __name__ == "__main__":
162
+ main()
evaluation/code/swift15/runtime_compat.py ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Apply HyperQwen's two asymmetric INT4/INT8 Marlin guards idempotently.
2
+
3
+ Reference: https://github.com/syv-ai/HyperQwen/blob/main/patches/marlin-int8-asym-zp.patch
4
+ Backups and before/after hashes are recorded. No kernels are replaced.
5
+ """
6
+ import importlib.util
7
+ import shutil
8
+ from common import RUN, write_json, sha256
9
+ from pathlib import Path
10
+
11
+
12
+ def main():
13
+ root = Path(importlib.util.find_spec("vllm").origin).parent
14
+ changes = [
15
+ ("model_executor/kernels/linear/mixed_precision/marlin.py",
16
+ 'assert c.weight_type == scalar_types.uint4b8, (\n "W8A8 is not supported',
17
+ 'assert c.weight_type in (scalar_types.uint4b8, scalar_types.uint4), (\n "W8A8 is not supported'),
18
+ ("model_executor/layers/quantization/utils/marlin_utils.py",
19
+ 'assert wtype == scalar_types.uint4b8, (\n "W8A8-INT8 is not supported',
20
+ 'assert wtype in (scalar_types.uint4b8, scalar_types.uint4), (\n "W8A8-INT8 is not supported')]
21
+ plan = []
22
+ for relative, old, new in changes:
23
+ path = root / relative
24
+ text = path.read_text()
25
+ assert text.count(old) == 1 or text.count(new) == 1, f"Unexpected installed code: {path}"
26
+ plan.append((path, old, new, text))
27
+ report = []
28
+ for path, old, new, text in plan:
29
+ before = sha256(path)
30
+ backup = path.with_name(path.name + ".swift15-before")
31
+ if old in text:
32
+ assert not backup.exists(), f"Backup exists but patch is absent: {backup}"
33
+ shutil.copy2(path, backup)
34
+ path.write_text(text.replace(old, new))
35
+ report.append({"path": str(path), "before": before, "after": sha256(path), "backup": str(backup)})
36
+ write_json(RUN / "runtime-compatibility.json", report)
37
+ print("Asymmetric INT4 + INT8-activation guards installed")
38
+
39
+
40
+ if __name__ == "__main__":
41
+ main()
evaluation/code/swift15/serve.py ADDED
@@ -0,0 +1,103 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Own one benchmark server process group; leave unrelated services alone."""
2
+ import contextlib
3
+ import json
4
+ import os
5
+ import signal
6
+ import subprocess
7
+ import time
8
+ import urllib.request
9
+ from common import ROOT, RUN, write_json, sha256, stamp
10
+ from evaluate import key
11
+
12
+
13
+ @contextlib.contextmanager
14
+ def server(model, tag, batch=False, int8=False):
15
+ context_len = int(os.environ.get("SWIFT15_CONTEXT_LEN", "16384"))
16
+ speculation = os.environ.get("SWIFT15_SPEC", "mtp")
17
+ if speculation not in {"mtp", "off"}:
18
+ raise ValueError("SWIFT15_SPEC must be mtp or off")
19
+ write_json(RUN / "status.json", {"stage": tag + "-server", "state": "starting", "started": stamp()})
20
+ log = RUN / "logs" / (tag + "-server.log")
21
+ log.parent.mkdir(parents=True, exist_ok=True)
22
+ env = dict(os.environ, MODEL=str(model), HOST="127.0.0.1", PORT="18021",
23
+ VISION="0", CTX="fast", SPEC=speculation, MAX_LEN=str(context_len), MAX_SEQS="8",
24
+ GPU_UTIL="0.93", API_SERVERS="1", PREFIX_CACHE="0", INT8_ACT="int8" if int8 else "",
25
+ INT8_LAYERS="mlp", OMP_NUM_THREADS="8", EXTRA_ARGS="--generation-config vllm",
26
+ VLLM_OFFLOAD_KEEP_SHM="1")
27
+ for name in ["VLLM_MARLIN_INPUT_DTYPE", "VLLM_MARLIN_INT8_INCLUDE_RE", "VLLM_PREFILL_ATTN"]:
28
+ env.pop(name, None)
29
+ if not int8 and (ROOT / ".env").exists():
30
+ for line in (ROOT / ".env").read_text().splitlines():
31
+ if line.startswith("INT8_ACT=") and line.split("=", 1)[1].strip(' "'):
32
+ raise RuntimeError("Local .env enables INT8_ACT; remove that default before the controlled W4A16 comparison")
33
+ launcher = ROOT / ("batch/start_qwen.sh" if batch else "single-user/start_qwen.sh")
34
+ if batch:
35
+ env["EXTRA_ARGS"] += " --attention-backend FLASH_ATTN --kv-cache-dtype bfloat16 --no-enable-prefix-caching"
36
+ local_profile = os.environ.get("SWIFT15_SERVING_PROFILE") == "local-single-user" and not batch
37
+ if local_profile:
38
+ # Match the launcher's .env semantics, retaining all local runtime knobs.
39
+ env = dict(os.environ)
40
+ for line in (ROOT / ".env").read_text().splitlines():
41
+ if not line or line.startswith("#"):
42
+ continue
43
+ name, sep, value = line.removeprefix("export ").partition("=")
44
+ if sep and name.isidentifier() and not env.get(name):
45
+ env[name] = value.strip('"')
46
+ context_len = 150000
47
+ env.update(MODEL=str(model), HOST="127.0.0.1", PORT="18021",
48
+ CTX="long", MAX_LEN=str(context_len))
49
+ if env.get("SPEC", "mtp") != "mtp":
50
+ raise ValueError("The FP8 local profile requires SPEC=mtp")
51
+ speculation = "mtp"
52
+ import socket
53
+ with socket.socket() as s:
54
+ if s.connect_ex(("127.0.0.1", 18021)) == 0:
55
+ raise RuntimeError("Benchmark port 18021 is occupied; refusing to use or stop another server")
56
+ with log.open("w") as out:
57
+ p = subprocess.Popen(["bash", str(launcher)], cwd=ROOT, env=env, stdout=out, stderr=subprocess.STDOUT,
58
+ start_new_session=True)
59
+ identity = {"created": stamp(), "model": str(model), "config_sha256": sha256(model / "config.json"),
60
+ "index_sha256": sha256(model / "model.safetensors.index.json"), "launcher_sha256": sha256(launcher),
61
+ "tokenizer_sha256": sha256(model / "tokenizer.json"),
62
+ "chat_template_sha256": sha256(model / "chat_template.jinja") if (model / "chat_template.jinja").exists() else sha256(model / "tokenizer_config.json"),
63
+ "shards": {p.name: {"size": p.stat().st_size, "mtime_ns": p.stat().st_mtime_ns} for p in model.glob("*.safetensors")},
64
+ "mode": "batch" if batch else "single-user", "int8_activations": int8,
65
+ "speculation": speculation,
66
+ "max_model_len": context_len, "max_num_seqs": 8, "kv_cache_dtype": "bfloat16",
67
+ "prefix_cache": False, "gpu_memory_utilization": .93, "pid": p.pid}
68
+ if local_profile:
69
+ identity.update(serving_profile="local-single-user", kv_cache_dtype="fp8",
70
+ max_num_seqs=int(env.get("MAX_SEQS") or 8),
71
+ prefix_cache=env.get("PREFIX_CACHE") == "1",
72
+ vision=env.get("VISION") == "1",
73
+ gpu_memory_utilization=float(env.get("GPU_UTIL") or .93),
74
+ draft_tokens=int(env.get("DRAFT_TOKENS") or 3),
75
+ int8_activations=bool(env.get("INT8_ACT")),
76
+ local_env_sha256=sha256(ROOT / ".env"),
77
+ extra_args=env.get("EXTRA_ARGS", ""))
78
+ write_json(RUN / "active-server.json", identity)
79
+ write_json(RUN / "results" / tag / "server.json", identity)
80
+ try:
81
+ deadline = time.monotonic() + 900
82
+ while time.monotonic() < deadline:
83
+ if p.poll() is not None:
84
+ raise RuntimeError(f"Server exited {p.returncode}; see {log}")
85
+ try:
86
+ req = urllib.request.Request("http://127.0.0.1:18021/health", headers={"Authorization": "Bearer " + key()})
87
+ with urllib.request.urlopen(req, timeout=3) as response:
88
+ if response.status == 200:
89
+ break
90
+ except OSError:
91
+ time.sleep(2)
92
+ else:
93
+ raise TimeoutError(f"Server startup timed out; see {log}")
94
+ yield identity
95
+ finally:
96
+ try: os.killpg(p.pid, signal.SIGTERM)
97
+ except ProcessLookupError: pass
98
+ try: p.wait(timeout=45)
99
+ except subprocess.TimeoutExpired:
100
+ try: os.killpg(p.pid, signal.SIGKILL)
101
+ except ProcessLookupError: pass
102
+ p.wait(timeout=15)
103
+ write_json(RUN / "active-server.json", dict(identity, stopped=stamp(), returncode=p.returncode))
evaluation/code/swift15/setup.py ADDED
@@ -0,0 +1,42 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Download pinned public inputs and install the isolated official verifier."""
2
+ import hashlib
3
+ import io
4
+ import subprocess
5
+ import sys
6
+ import urllib.request
7
+ import zipfile
8
+ from pathlib import Path
9
+ from huggingface_hub import snapshot_download
10
+ from common import ROOT, RUN, SOURCE, write_json, stamp
11
+
12
+ REPO = "ukisai/Swift-1.5-Qwen3.8-27b-W4A16-AWQ"
13
+ REVISION = "9dba8a05877150d587215a519ce6befec3c9978a"
14
+ IFBENCH = "1c40f0c10d9b5c5c2f10a175a28007ebb64f7f4d"
15
+
16
+
17
+ def main():
18
+ RUN.mkdir(parents=True, exist_ok=True)
19
+ snapshot_download(REPO, revision=REVISION, local_dir=SOURCE,
20
+ allow_patterns=["*.json", "*.safetensors", "*.jinja", "*.txt", "LICENSE*", "NOTICE", "recipe.yaml", "README.md"])
21
+ if not (RUN / "source.json").exists():
22
+ write_json(RUN / "source.json", {"repo": REPO, "revision": REVISION, "downloaded_at": stamp()})
23
+ env = RUN / "eval-venv"
24
+ if not (env / "bin/python").exists():
25
+ subprocess.run([sys.executable, "-m", "venv", str(env)], check=True)
26
+ subprocess.run([str(env / "bin/pip"), "install", "https://github.com/allenai/IFBench/archive/" + IFBENCH + ".zip"], check=True)
27
+ url = "https://raw.githubusercontent.com/nltk/nltk_data/gh-pages/packages/tokenizers/punkt_tab.zip"
28
+ payload = urllib.request.urlopen(url).read()
29
+ with zipfile.ZipFile(io.BytesIO(payload)) as archive:
30
+ assert all(not n.startswith("/") and ".." not in n.split("/") for n in archive.namelist())
31
+ archive.extractall(RUN / "nltk_data/tokenizers")
32
+ subprocess.run(["docker", "image", "inspect", "python:3.12-slim"], check=True, stdout=subprocess.DEVNULL)
33
+ image = subprocess.check_output(["docker", "image", "inspect", "python:3.12-slim", "--format", "{{.Id}}"], text=True).strip()
34
+ write_json(RUN / "evaluation/runtime.json", {"code_image": image, "ifbench_revision": IFBENCH,
35
+ "punkt_tab_sha256": hashlib.sha256(payload).hexdigest()})
36
+ subprocess.run([sys.executable, "swift15/corpus.py"], cwd=ROOT, check=True)
37
+ subprocess.run([sys.executable, "swift15/eval_data.py"], cwd=ROOT, check=True)
38
+ print("Setup complete. Existing patched HyperQwen venv and Python Docker image are required.")
39
+
40
+
41
+ if __name__ == "__main__":
42
+ main()
evaluation/code/swift15/smoke.py ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Exercise real MTP serving, streaming, prompt logprobs, tools and scoring."""
2
+ import argparse
3
+ import json
4
+ import os
5
+ import subprocess
6
+ from pathlib import Path
7
+ from common import ROOT, RUN, records, write_json
8
+ from serve import server
9
+ from evaluate import execute, stream, post
10
+
11
+
12
+ def main():
13
+ ap = argparse.ArgumentParser()
14
+ ap.add_argument("model", type=Path)
15
+ ap.add_argument("--tag", default="smoke")
16
+ ap.add_argument("--batch-int8", action="store_true")
17
+ ap.add_argument("--check-harness", action="store_true")
18
+ args = ap.parse_args()
19
+ api = "http://127.0.0.1:18021/v1"
20
+ tasks = records(RUN / "evaluation/tasks.jsonl")
21
+ results = []
22
+ with server(args.model.resolve(), args.tag, batch=args.batch_int8, int8=args.batch_int8):
23
+ warm = stream(api, {"model": "qwen3.8-27b", "messages": [{"role":"user","content":"Reply with the word ready."}],
24
+ "max_tokens": 32, "temperature": 0, "chat_template_kwargs": {"enable_thinking": False}})
25
+ assert warm["usage"]["completion_tokens"] > 0
26
+ with post(api, "/completions", {"model":"qwen3.8-27b", "prompt":"A simple test of language model probabilities.",
27
+ "max_tokens":1,"temperature":0,"prompt_logprobs":0}) as response:
28
+ scored = json.load(response)["choices"][0]["prompt_logprobs"]
29
+ assert scored and all(len(entry)==1 for entry in scored[1:])
30
+ for suite in ["gsm8k", "tools", "ifbench", "livecodebench"]:
31
+ task = next(t for t in tasks if t["suite"] == suite)
32
+ result = execute(task, api)
33
+ assert result["output_tokens"] > 0, result
34
+ if result["error"] and result["error"] != "incorrect_tool_call":
35
+ raise RuntimeError(result["error"])
36
+ if result.get("code_score", {}).get("reason") == "sandbox_error":
37
+ raise RuntimeError(result["code_score"])
38
+ if suite == "ifbench" and result["correct"] is None:
39
+ scored = subprocess.run([str(RUN/"eval-venv/bin/python"),str(ROOT/"swift15/ifbench_score.py")],
40
+ input=json.dumps([{"task":task,"response":result["response"]}]), text=True,capture_output=True,check=True,
41
+ env=dict(os.environ,NLTK_DATA=str(RUN/"nltk_data")))
42
+ result.update(json.loads(scored.stdout)[0])
43
+ results.append(result)
44
+ print(suite, "correct:",result["correct"],"tokens:",result["output_tokens"],flush=True)
45
+ if args.check_harness:
46
+ subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/evaluate.py"),args.tag+"-harness",
47
+ "--pilot","--suites","tools","--concurrency","2"],check=True,cwd=ROOT)
48
+ subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/measure.py"),args.tag,
49
+ "--kind","speed","--concurrency","8" if args.batch_int8 else "1"],check=True,cwd=ROOT)
50
+ write_json(RUN/"results"/args.tag/"smoke.json",{"warmup":warm,"tasks":results,
51
+ "note":"Integration checks only; this small sample is not a quality benchmark."})
52
+
53
+
54
+ if __name__ == "__main__":
55
+ main()
evaluation/code/swift15/test_workflow.py ADDED
@@ -0,0 +1,66 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Regression checks for file isolation, calibration separation and task scoring."""
2
+ import json
3
+ import tempfile
4
+ import unittest
5
+ from pathlib import Path
6
+ import torch
7
+ from safetensors.torch import save_file, load_file
8
+ from checkpoint import clone
9
+ from quantize import replace_tensors
10
+ from evaluate import gsm_score, json_equal, aggregate
11
+ from code_runner import equal
12
+ from gptq_utils import accumulate_hessian
13
+
14
+
15
+ class WorkflowTests(unittest.TestCase):
16
+ def test_capped_answers_count_as_wrong_even_if_partial_answer_matches(self):
17
+ row={"input_tokens":10,"output_tokens":32768,"truncated":True,"error":None,"correct":True,"model_seconds":200}
18
+ result=aggregate([row])
19
+ self.assertEqual(result["accuracy"],0)
20
+ self.assertEqual(result["correct"],0)
21
+ self.assertEqual(result["attempted"],1)
22
+ self.assertEqual(result["truncated_counted_as_wrong"],1)
23
+
24
+ def test_hessian_remains_fp32_inside_bf16_autocast(self):
25
+ torch.manual_seed(42)
26
+ x = torch.randn(64,128)
27
+ h = torch.zeros(128,128)
28
+ with torch.autocast("cpu",dtype=torch.bfloat16):
29
+ h,n = accumulate_hessian(h,x,0)
30
+ reference = 2/len(x) * (x.T @ x)
31
+ self.assertEqual(n,64)
32
+ self.assertTrue(torch.allclose(h,reference,rtol=1e-5,atol=1e-6))
33
+
34
+ def test_quant_export_does_not_mutate_hardlinked_baseline(self):
35
+ with tempfile.TemporaryDirectory() as td:
36
+ source, target = Path(td)/"source", Path(td)/"fast"
37
+ source.mkdir()
38
+ old = torch.ones(2, 4, dtype=torch.int32)
39
+ save_file({"lm_head.weight_packed": old, "untouched.weight": torch.ones(2)}, source/"model.safetensors")
40
+ (source/"model.safetensors.index.json").write_text(json.dumps({"weight_map":{"lm_head.weight_packed":"model.safetensors"}}))
41
+ (source/"config.json").write_text(json.dumps({"quantization_config":{"config_groups":{"group_1":{"weights":{"num_bits":8}}}}}))
42
+ clone(source,target)
43
+ self.assertEqual((source/"model.safetensors").stat().st_ino,(target/"model.safetensors").stat().st_ino)
44
+ replace_tensors(target,{"lm_head":{"weight_packed":torch.zeros_like(old)}},{"group_1":4})
45
+ self.assertTrue(torch.equal(load_file(source/"model.safetensors")["lm_head.weight_packed"],old))
46
+ self.assertEqual(load_file(target/"model.safetensors")["lm_head.weight_packed"].sum().item(),0)
47
+ self.assertTrue(torch.equal(load_file(target/"model.safetensors")["untouched.weight"],torch.ones(2)))
48
+
49
+ def test_scoring_rejects_wrong_types_and_nonfinite_numbers(self):
50
+ self.assertFalse(json_equal({"x":True},{"x":1}))
51
+ self.assertFalse(json_equal(float("nan"),1))
52
+ self.assertFalse(equal("9007199254740993","9007199254740992"))
53
+ self.assertTrue(equal("1.0000001\n2","1 2"))
54
+ self.assertTrue(gsm_score("Final answer: 1,250", "1250"))
55
+ self.assertFalse(gsm_score("Final answer: 1251", "1250"))
56
+
57
+ def test_failures_remain_in_time_and_quality_denominators(self):
58
+ common={"input_tokens":10,"output_tokens":20,"truncated":False,"error":None}
59
+ result=aggregate([dict(common,correct=True,model_seconds=2),dict(common,correct=False,model_seconds=8)])
60
+ self.assertEqual(result["accuracy"],.5)
61
+ self.assertEqual(result["mean_model_seconds"],5)
62
+ self.assertEqual(result["summed_request_seconds_per_correct"],10)
63
+
64
+
65
+ if __name__ == "__main__":
66
+ unittest.main()
evaluation/environment.json ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "started": "2026-09-28T10:01:00Z",
3
+ "hardware": "name, uuid, memory.total [MiB], power.limit [W], driver_version\nNVIDIA GeForce RTX 3090, GPU-9724ba4e-72d2-f72e-69e7-c21e552fe118, 24576 MiB, 250.00 W, 580.173.02\n",
4
+ "versions": {
5
+ "torch": "2.13.0",
6
+ "vllm": "0.27.1",
7
+ "transformers": "5.15.0",
8
+ "safetensors": "0.8.0",
9
+ "compressed-tensors": "0.17.0"
10
+ },
11
+ "hyperqwen_commit": "253c76aea0a240bf7cb2bd3ed92b672c0f258d0d",
12
+ "python": "3.12.8 | packaged by conda-forge | (main, Dec 5 2024, 14:24:40) [GCC 13.3.0]",
13
+ "patches_sha256": {
14
+ "spec-decode-int8-kv.patch": "3cfd31304353237af5277c861dc2b43ad33d0f9bff2700594300eff150f64382",
15
+ "qwen3_5-mtp-draft-vocab.patch": "292ef662d17bcc10556b787d5bb5f2cd12d3d3fc1f3bd2fe982485ce0d916298",
16
+ "marlin-int8-layer-select.patch": "833405e2ed2916529eb20184c38243c843378935522b74bab8c39707bf3ee800",
17
+ "offload-wsl2-devptr.patch": "8c6f9e3e5723571d1f33925794435e55062b684b2706d9284d1c5b94778d34db",
18
+ "marlin-repack-staged-sm80.patch": "1588e2f10b5f82194d449483e766c0b39c84ba522d1623d39caa9502040f1fab",
19
+ "qwen3_5-embed-quant.patch": "0a1b9ca06798c1aef582995de5a0beb3ad9a22a54cdbd2361986563a9c7a980e",
20
+ "dflash2-lookup-drafting.patch": "5df09ef03b592d4a2c3b47dd8d2dbf8862fa7383d68aef3dd6ef1d97d29ce196",
21
+ "triton-prefill-attn-int8.patch": "6bd36db0226dce7b92ebd1231b6f9718a25da4a933ab4b2a6a8be77860364fae",
22
+ "dflash2-backport.patch": "2e937ef748c1942ab04ace05de884e7c24041d34b23e5be8428cac6939ac6875",
23
+ "sampler-small-topk-fast-softmax.patch": "8828646ce1916c065282529d9c1bb52064f668397d8e1f33652aef2f308292b6",
24
+ "dflash2-ngram-chains.patch": "555b5a75d9023c99b2cd17634ac5b1dc8505f9275d857b1eba08c2d9cbf2df6e",
25
+ "spec-decode-int4-kv-mq3d.patch": "b93b186deba1513ad9c4803ee2c574c923b21ba5e88e23bca8b8594b6a6b20a0",
26
+ "mamba-align-checkpoint-order.patch": "515d9bf76e860c95d832d615bdab4a97b71e2b0412b4ccc2445f11535ddddf45",
27
+ "offload-dflash-eagle-groups.patch": "f2791af64d8066b31e4250d866c4322f45670d75522aff221307281c97c73156",
28
+ "marlin-int8-negative-scales.patch": "4cb8a064c706cc62dc77224479898da58864fbfff6b65012d90aa8326035300d",
29
+ "vision-tower-cpu-offload.patch": "81dff64a1177058783dcf7d4e8552f1547f74fa8a68019e349dacdb9d66e968e",
30
+ "dflash2-prewarm.patch": "e1e8012fa2c304c6948c19fc0071332f80eebcff802b402255445a30d553ab92",
31
+ "hybrid-kv-groups-v2-cudagraph.patch": "143825dd9744ab81dabcb438ec0bf974b6e0be8d0d7b0ffbbdfbf7c16a549664",
32
+ "marlin-tune-table.patch": "1cac17f12e4ce389b0cb0fc733ceceb0a8cf60c02696ee378dc1d8d0c1ef8914",
33
+ "int4-kv-per-token-head.patch": "c03ff10c8c997b355f04fc521827ee2fa99c332ce4127cc2d878b749b79ca3d9",
34
+ "spec-sampler-prewarm.patch": "18ed9608baa5a09ed2ccbb214d62763da3ce7bcf09c6aad9003778ea4e56d18f",
35
+ "vllm-pr50021-gdn-spec-bounds.patch": "cd6e00270fa28e37a8c7ad11f965662f4153b2fcaf840f2c4043710e7b078655",
36
+ "spec-decode-attn.patch": "007791047a1d60143f8e3fe4e0a56dabadfc0134cb288847338b76ba7d1f9fe7",
37
+ "xgrammar-spec-terminated.patch": "37589b9a45d5ece37cc16e82b195362719011e52ea31b2e70cdc89845c825040",
38
+ "hybrid-sw-block-promote.patch": "10876ac706546e74fe74b8964b90576a7b36f90a2f35a856997c558d7817f154",
39
+ "speed-knobs-envs.patch": "841ec93021b1b90f90313a1880a791bbfbdb79cbe34d9b99fdb4c3a7e52ed7c0"
40
+ }
41
+ }
evaluation/manifest.json ADDED
@@ -0,0 +1,72 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "seed": 15027,
3
+ "sources": [
4
+ {
5
+ "dataset": "openai/gsm8k",
6
+ "split": "main/test",
7
+ "file_sha256": "ee7b8da9e381df27b9e3f7758a159ab2bdaa4dbaa910546cbbc47e0cb44e4f59",
8
+ "selection": "first 200, same as HyperQwen"
9
+ },
10
+ {
11
+ "dataset": "allenai/IFBench_test",
12
+ "file_sha256": "80037e4d99c39a55c1e2e7d5a863d8d9edeb5ebe136ba8c1e849f8c015027c6a",
13
+ "note": "Upstream names this split train, but it is the benchmark test set; never used for calibration."
14
+ },
15
+ {
16
+ "dataset": "livecodebench/code_generation_lite",
17
+ "file": "test.jsonl",
18
+ "sha256": "2bd02b38beb48e8c46b5b9987095d999ff38cd8efc255ea5d58974317c48f63f"
19
+ },
20
+ {
21
+ "dataset": "livecodebench/code_generation_lite",
22
+ "file": "test2.jsonl",
23
+ "sha256": "095df7c5daf15f882c51a9deb84085cff1e073495a5dbcf95015a564d485f3a3"
24
+ },
25
+ {
26
+ "dataset": "livecodebench/code_generation_lite",
27
+ "file": "test3.jsonl",
28
+ "sha256": "28ed26cc83363ce3f1fe2d5fad9f8393077beb1907b167a31bd3b32f80801b79"
29
+ },
30
+ {
31
+ "dataset": "livecodebench/code_generation_lite",
32
+ "file": "test4.jsonl",
33
+ "sha256": "d711138ddaebfcf5f8ec6a4283ee677298c0f5c5d374a235af92aaf0584510da"
34
+ },
35
+ {
36
+ "dataset": "livecodebench/code_generation_lite",
37
+ "file": "test5.jsonl",
38
+ "sha256": "7f77571c2a6df0c2a72a3277650309f67e01e0008e18117e624633df53f81214"
39
+ },
40
+ {
41
+ "dataset": "livecodebench/code_generation_lite",
42
+ "file": "test6.jsonl",
43
+ "sha256": "bb4c364f71921c4495a6ad15abe1a927350b720009f4933e2e71f8af0f6fd1f5"
44
+ }
45
+ ],
46
+ "tasks": {
47
+ "gsm8k": 200,
48
+ "ifbench": 300,
49
+ "livecodebench": 100,
50
+ "tools": 30
51
+ },
52
+ "task_file_sha256": "809c2c6b124579d41c379a6a436e9649b5e1d1ddf44a147a3d70d9b4b19d68f4",
53
+ "ppl_file_sha256": "57c83ffe7c0dfba2b869f6a379d2df3f0bc0d055069b49569a2154336151a718",
54
+ "protocol": "Local single-user launcher and .env, FP8 KV, 150000 context, 128000 output ceiling. Full quality concurrency two; speed and latency pilot concurrency one. Truncated answers count as wrong.",
55
+ "lcb_scoring": "100 stratified stdin-only v6-era tasks; all supplied public/private tests; whitespace token comparison with numeric tolerance; not the full official LCB runner.",
56
+ "max_output_tokens_per_call": 128000,
57
+ "context_len": 150000,
58
+ "quality_concurrency": 2,
59
+ "supersedes": "<WORKSPACE>/runs/swift15/evaluation/tasks.jsonl",
60
+ "truncation_policy": "count_as_wrong",
61
+ "sampling": {
62
+ "thinking": {
63
+ "temperature": 1.0,
64
+ "top_p": 0.95,
65
+ "top_k": 20,
66
+ "reasoning_effort": "xhigh"
67
+ },
68
+ "nonthinking": "greedy",
69
+ "seed": 15027
70
+ },
71
+ "serving_profile": "local-single-user"
72
+ }
evaluation/pilot-ids.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ "gsm8k-105",
3
+ "gsm8k-152",
4
+ "gsm8k-109",
5
+ "gsm8k-27",
6
+ "gsm8k-60",
7
+ "ifbench-268",
8
+ "ifbench-129",
9
+ "ifbench-76",
10
+ "ifbench-21",
11
+ "ifbench-130",
12
+ "lcb-atcoder-abc377_b",
13
+ "lcb-atcoder-abc390_d",
14
+ "lcb-atcoder-abc325_f",
15
+ "lcb-atcoder-abc385_e",
16
+ "lcb-atcoder-abc368_e",
17
+ "json-26",
18
+ "tool-16",
19
+ "tool-15",
20
+ "json-22",
21
+ "json-24"
22
+ ]
evaluation/qwen-fast-full/manifest.json ADDED
@@ -0,0 +1,693 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "tasks_sha256": "809c2c6b124579d41c379a6a436e9649b5e1d1ddf44a147a3d70d9b4b19d68f4",
3
+ "task_ids": [
4
+ "gsm8k-0",
5
+ "gsm8k-1",
6
+ "gsm8k-2",
7
+ "gsm8k-3",
8
+ "gsm8k-4",
9
+ "gsm8k-5",
10
+ "gsm8k-6",
11
+ "gsm8k-7",
12
+ "gsm8k-8",
13
+ "gsm8k-9",
14
+ "gsm8k-10",
15
+ "gsm8k-11",
16
+ "gsm8k-12",
17
+ "gsm8k-13",
18
+ "gsm8k-14",
19
+ "gsm8k-15",
20
+ "gsm8k-16",
21
+ "gsm8k-17",
22
+ "gsm8k-18",
23
+ "gsm8k-19",
24
+ "gsm8k-20",
25
+ "gsm8k-21",
26
+ "gsm8k-22",
27
+ "gsm8k-23",
28
+ "gsm8k-24",
29
+ "gsm8k-25",
30
+ "gsm8k-26",
31
+ "gsm8k-27",
32
+ "gsm8k-28",
33
+ "gsm8k-29",
34
+ "gsm8k-30",
35
+ "gsm8k-31",
36
+ "gsm8k-32",
37
+ "gsm8k-33",
38
+ "gsm8k-34",
39
+ "gsm8k-35",
40
+ "gsm8k-36",
41
+ "gsm8k-37",
42
+ "gsm8k-38",
43
+ "gsm8k-39",
44
+ "gsm8k-40",
45
+ "gsm8k-41",
46
+ "gsm8k-42",
47
+ "gsm8k-43",
48
+ "gsm8k-44",
49
+ "gsm8k-45",
50
+ "gsm8k-46",
51
+ "gsm8k-47",
52
+ "gsm8k-48",
53
+ "gsm8k-49",
54
+ "gsm8k-50",
55
+ "gsm8k-51",
56
+ "gsm8k-52",
57
+ "gsm8k-53",
58
+ "gsm8k-54",
59
+ "gsm8k-55",
60
+ "gsm8k-56",
61
+ "gsm8k-57",
62
+ "gsm8k-58",
63
+ "gsm8k-59",
64
+ "gsm8k-60",
65
+ "gsm8k-61",
66
+ "gsm8k-62",
67
+ "gsm8k-63",
68
+ "gsm8k-64",
69
+ "gsm8k-65",
70
+ "gsm8k-66",
71
+ "gsm8k-67",
72
+ "gsm8k-68",
73
+ "gsm8k-69",
74
+ "gsm8k-70",
75
+ "gsm8k-71",
76
+ "gsm8k-72",
77
+ "gsm8k-73",
78
+ "gsm8k-74",
79
+ "gsm8k-75",
80
+ "gsm8k-76",
81
+ "gsm8k-77",
82
+ "gsm8k-78",
83
+ "gsm8k-79",
84
+ "gsm8k-80",
85
+ "gsm8k-81",
86
+ "gsm8k-82",
87
+ "gsm8k-83",
88
+ "gsm8k-84",
89
+ "gsm8k-85",
90
+ "gsm8k-86",
91
+ "gsm8k-87",
92
+ "gsm8k-88",
93
+ "gsm8k-89",
94
+ "gsm8k-90",
95
+ "gsm8k-91",
96
+ "gsm8k-92",
97
+ "gsm8k-93",
98
+ "gsm8k-94",
99
+ "gsm8k-95",
100
+ "gsm8k-96",
101
+ "gsm8k-97",
102
+ "gsm8k-98",
103
+ "gsm8k-99",
104
+ "gsm8k-100",
105
+ "gsm8k-101",
106
+ "gsm8k-102",
107
+ "gsm8k-103",
108
+ "gsm8k-104",
109
+ "gsm8k-105",
110
+ "gsm8k-106",
111
+ "gsm8k-107",
112
+ "gsm8k-108",
113
+ "gsm8k-109",
114
+ "gsm8k-110",
115
+ "gsm8k-111",
116
+ "gsm8k-112",
117
+ "gsm8k-113",
118
+ "gsm8k-114",
119
+ "gsm8k-115",
120
+ "gsm8k-116",
121
+ "gsm8k-117",
122
+ "gsm8k-118",
123
+ "gsm8k-119",
124
+ "gsm8k-120",
125
+ "gsm8k-121",
126
+ "gsm8k-122",
127
+ "gsm8k-123",
128
+ "gsm8k-124",
129
+ "gsm8k-125",
130
+ "gsm8k-126",
131
+ "gsm8k-127",
132
+ "gsm8k-128",
133
+ "gsm8k-129",
134
+ "gsm8k-130",
135
+ "gsm8k-131",
136
+ "gsm8k-132",
137
+ "gsm8k-133",
138
+ "gsm8k-134",
139
+ "gsm8k-135",
140
+ "gsm8k-136",
141
+ "gsm8k-137",
142
+ "gsm8k-138",
143
+ "gsm8k-139",
144
+ "gsm8k-140",
145
+ "gsm8k-141",
146
+ "gsm8k-142",
147
+ "gsm8k-143",
148
+ "gsm8k-144",
149
+ "gsm8k-145",
150
+ "gsm8k-146",
151
+ "gsm8k-147",
152
+ "gsm8k-148",
153
+ "gsm8k-149",
154
+ "gsm8k-150",
155
+ "gsm8k-151",
156
+ "gsm8k-152",
157
+ "gsm8k-153",
158
+ "gsm8k-154",
159
+ "gsm8k-155",
160
+ "gsm8k-156",
161
+ "gsm8k-157",
162
+ "gsm8k-158",
163
+ "gsm8k-159",
164
+ "gsm8k-160",
165
+ "gsm8k-161",
166
+ "gsm8k-162",
167
+ "gsm8k-163",
168
+ "gsm8k-164",
169
+ "gsm8k-165",
170
+ "gsm8k-166",
171
+ "gsm8k-167",
172
+ "gsm8k-168",
173
+ "gsm8k-169",
174
+ "gsm8k-170",
175
+ "gsm8k-171",
176
+ "gsm8k-172",
177
+ "gsm8k-173",
178
+ "gsm8k-174",
179
+ "gsm8k-175",
180
+ "gsm8k-176",
181
+ "gsm8k-177",
182
+ "gsm8k-178",
183
+ "gsm8k-179",
184
+ "gsm8k-180",
185
+ "gsm8k-181",
186
+ "gsm8k-182",
187
+ "gsm8k-183",
188
+ "gsm8k-184",
189
+ "gsm8k-185",
190
+ "gsm8k-186",
191
+ "gsm8k-187",
192
+ "gsm8k-188",
193
+ "gsm8k-189",
194
+ "gsm8k-190",
195
+ "gsm8k-191",
196
+ "gsm8k-192",
197
+ "gsm8k-193",
198
+ "gsm8k-194",
199
+ "gsm8k-195",
200
+ "gsm8k-196",
201
+ "gsm8k-197",
202
+ "gsm8k-198",
203
+ "gsm8k-199",
204
+ "ifbench-0",
205
+ "ifbench-1",
206
+ "ifbench-2",
207
+ "ifbench-3",
208
+ "ifbench-4",
209
+ "ifbench-5",
210
+ "ifbench-6",
211
+ "ifbench-7",
212
+ "ifbench-8",
213
+ "ifbench-9",
214
+ "ifbench-10",
215
+ "ifbench-11",
216
+ "ifbench-12",
217
+ "ifbench-13",
218
+ "ifbench-14",
219
+ "ifbench-15",
220
+ "ifbench-16",
221
+ "ifbench-17",
222
+ "ifbench-18",
223
+ "ifbench-19",
224
+ "ifbench-20",
225
+ "ifbench-21",
226
+ "ifbench-22",
227
+ "ifbench-23",
228
+ "ifbench-24",
229
+ "ifbench-25",
230
+ "ifbench-26",
231
+ "ifbench-27",
232
+ "ifbench-28",
233
+ "ifbench-29",
234
+ "ifbench-30",
235
+ "ifbench-31",
236
+ "ifbench-32",
237
+ "ifbench-33",
238
+ "ifbench-34",
239
+ "ifbench-35",
240
+ "ifbench-36",
241
+ "ifbench-37",
242
+ "ifbench-38",
243
+ "ifbench-39",
244
+ "ifbench-40",
245
+ "ifbench-41",
246
+ "ifbench-42",
247
+ "ifbench-43",
248
+ "ifbench-44",
249
+ "ifbench-45",
250
+ "ifbench-46",
251
+ "ifbench-47",
252
+ "ifbench-48",
253
+ "ifbench-49",
254
+ "ifbench-50",
255
+ "ifbench-51",
256
+ "ifbench-52",
257
+ "ifbench-53",
258
+ "ifbench-54",
259
+ "ifbench-55",
260
+ "ifbench-56",
261
+ "ifbench-57",
262
+ "ifbench-58",
263
+ "ifbench-59",
264
+ "ifbench-60",
265
+ "ifbench-61",
266
+ "ifbench-62",
267
+ "ifbench-63",
268
+ "ifbench-64",
269
+ "ifbench-65",
270
+ "ifbench-66",
271
+ "ifbench-67",
272
+ "ifbench-68",
273
+ "ifbench-69",
274
+ "ifbench-70",
275
+ "ifbench-71",
276
+ "ifbench-72",
277
+ "ifbench-73",
278
+ "ifbench-74",
279
+ "ifbench-75",
280
+ "ifbench-76",
281
+ "ifbench-77",
282
+ "ifbench-78",
283
+ "ifbench-79",
284
+ "ifbench-80",
285
+ "ifbench-81",
286
+ "ifbench-82",
287
+ "ifbench-83",
288
+ "ifbench-84",
289
+ "ifbench-85",
290
+ "ifbench-86",
291
+ "ifbench-87",
292
+ "ifbench-88",
293
+ "ifbench-89",
294
+ "ifbench-90",
295
+ "ifbench-91",
296
+ "ifbench-92",
297
+ "ifbench-93",
298
+ "ifbench-94",
299
+ "ifbench-95",
300
+ "ifbench-96",
301
+ "ifbench-97",
302
+ "ifbench-98",
303
+ "ifbench-99",
304
+ "ifbench-100",
305
+ "ifbench-101",
306
+ "ifbench-102",
307
+ "ifbench-103",
308
+ "ifbench-104",
309
+ "ifbench-105",
310
+ "ifbench-106",
311
+ "ifbench-107",
312
+ "ifbench-108",
313
+ "ifbench-109",
314
+ "ifbench-110",
315
+ "ifbench-111",
316
+ "ifbench-112",
317
+ "ifbench-113",
318
+ "ifbench-114",
319
+ "ifbench-115",
320
+ "ifbench-116",
321
+ "ifbench-117",
322
+ "ifbench-118",
323
+ "ifbench-119",
324
+ "ifbench-120",
325
+ "ifbench-121",
326
+ "ifbench-122",
327
+ "ifbench-123",
328
+ "ifbench-124",
329
+ "ifbench-125",
330
+ "ifbench-126",
331
+ "ifbench-127",
332
+ "ifbench-128",
333
+ "ifbench-129",
334
+ "ifbench-130",
335
+ "ifbench-131",
336
+ "ifbench-132",
337
+ "ifbench-133",
338
+ "ifbench-134",
339
+ "ifbench-135",
340
+ "ifbench-136",
341
+ "ifbench-137",
342
+ "ifbench-138",
343
+ "ifbench-139",
344
+ "ifbench-140",
345
+ "ifbench-141",
346
+ "ifbench-142",
347
+ "ifbench-143",
348
+ "ifbench-144",
349
+ "ifbench-145",
350
+ "ifbench-146",
351
+ "ifbench-147",
352
+ "ifbench-148",
353
+ "ifbench-149",
354
+ "ifbench-150",
355
+ "ifbench-151",
356
+ "ifbench-152",
357
+ "ifbench-153",
358
+ "ifbench-154",
359
+ "ifbench-155",
360
+ "ifbench-156",
361
+ "ifbench-157",
362
+ "ifbench-158",
363
+ "ifbench-159",
364
+ "ifbench-160",
365
+ "ifbench-161",
366
+ "ifbench-162",
367
+ "ifbench-163",
368
+ "ifbench-164",
369
+ "ifbench-165",
370
+ "ifbench-166",
371
+ "ifbench-167",
372
+ "ifbench-168",
373
+ "ifbench-169",
374
+ "ifbench-170",
375
+ "ifbench-171",
376
+ "ifbench-172",
377
+ "ifbench-173",
378
+ "ifbench-174",
379
+ "ifbench-175",
380
+ "ifbench-176",
381
+ "ifbench-177",
382
+ "ifbench-178",
383
+ "ifbench-179",
384
+ "ifbench-180",
385
+ "ifbench-181",
386
+ "ifbench-182",
387
+ "ifbench-183",
388
+ "ifbench-184",
389
+ "ifbench-185",
390
+ "ifbench-186",
391
+ "ifbench-187",
392
+ "ifbench-188",
393
+ "ifbench-189",
394
+ "ifbench-190",
395
+ "ifbench-191",
396
+ "ifbench-192",
397
+ "ifbench-193",
398
+ "ifbench-194",
399
+ "ifbench-195",
400
+ "ifbench-196",
401
+ "ifbench-197",
402
+ "ifbench-198",
403
+ "ifbench-199",
404
+ "ifbench-200",
405
+ "ifbench-201",
406
+ "ifbench-202",
407
+ "ifbench-203",
408
+ "ifbench-204",
409
+ "ifbench-205",
410
+ "ifbench-206",
411
+ "ifbench-207",
412
+ "ifbench-208",
413
+ "ifbench-209",
414
+ "ifbench-210",
415
+ "ifbench-211",
416
+ "ifbench-212",
417
+ "ifbench-213",
418
+ "ifbench-214",
419
+ "ifbench-215",
420
+ "ifbench-216",
421
+ "ifbench-217",
422
+ "ifbench-218",
423
+ "ifbench-219",
424
+ "ifbench-220",
425
+ "ifbench-221",
426
+ "ifbench-222",
427
+ "ifbench-223",
428
+ "ifbench-224",
429
+ "ifbench-225",
430
+ "ifbench-226",
431
+ "ifbench-227",
432
+ "ifbench-228",
433
+ "ifbench-229",
434
+ "ifbench-230",
435
+ "ifbench-231",
436
+ "ifbench-232",
437
+ "ifbench-233",
438
+ "ifbench-234",
439
+ "ifbench-235",
440
+ "ifbench-236",
441
+ "ifbench-237",
442
+ "ifbench-238",
443
+ "ifbench-239",
444
+ "ifbench-240",
445
+ "ifbench-241",
446
+ "ifbench-242",
447
+ "ifbench-243",
448
+ "ifbench-244",
449
+ "ifbench-245",
450
+ "ifbench-246",
451
+ "ifbench-247",
452
+ "ifbench-248",
453
+ "ifbench-249",
454
+ "ifbench-250",
455
+ "ifbench-251",
456
+ "ifbench-252",
457
+ "ifbench-253",
458
+ "ifbench-254",
459
+ "ifbench-255",
460
+ "ifbench-256",
461
+ "ifbench-257",
462
+ "ifbench-258",
463
+ "ifbench-259",
464
+ "ifbench-260",
465
+ "ifbench-261",
466
+ "ifbench-262",
467
+ "ifbench-263",
468
+ "ifbench-264",
469
+ "ifbench-265",
470
+ "ifbench-266",
471
+ "ifbench-267",
472
+ "ifbench-268",
473
+ "ifbench-269",
474
+ "ifbench-270",
475
+ "ifbench-271",
476
+ "ifbench-272",
477
+ "ifbench-273",
478
+ "ifbench-274",
479
+ "ifbench-275",
480
+ "ifbench-276",
481
+ "ifbench-277",
482
+ "ifbench-278",
483
+ "ifbench-279",
484
+ "ifbench-280",
485
+ "ifbench-281",
486
+ "ifbench-282",
487
+ "ifbench-283",
488
+ "ifbench-284",
489
+ "ifbench-285",
490
+ "ifbench-286",
491
+ "ifbench-287",
492
+ "ifbench-288",
493
+ "ifbench-289",
494
+ "ifbench-290",
495
+ "ifbench-291",
496
+ "ifbench-292",
497
+ "ifbench-293",
498
+ "ifbench-294",
499
+ "ifbench-295",
500
+ "ifbench-296",
501
+ "ifbench-297",
502
+ "ifbench-298",
503
+ "ifbench-299",
504
+ "lcb-atcoder-abc322_b",
505
+ "lcb-atcoder-abc381_a",
506
+ "lcb-atcoder-abc393_a",
507
+ "lcb-atcoder-abc321_b",
508
+ "lcb-atcoder-abc359_a",
509
+ "lcb-atcoder-abc329_b",
510
+ "lcb-atcoder-abc353_a",
511
+ "lcb-atcoder-abc355_b",
512
+ "lcb-atcoder-abc326_b",
513
+ "lcb-codeforces-1873_B",
514
+ "lcb-atcoder-abc356_a",
515
+ "lcb-atcoder-abc356_b",
516
+ "lcb-atcoder-abc375_a",
517
+ "lcb-atcoder-abc377_b",
518
+ "lcb-atcoder-abc311_b",
519
+ "lcb-atcoder-abc378_b",
520
+ "lcb-atcoder-abc309_b",
521
+ "lcb-atcoder-abc325_a",
522
+ "lcb-atcoder-abc301_b",
523
+ "lcb-atcoder-abc391_a",
524
+ "lcb-atcoder-abc399_b",
525
+ "lcb-atcoder-abc354_a",
526
+ "lcb-atcoder-abc352_b",
527
+ "lcb-atcoder-abc382_b",
528
+ "lcb-atcoder-abc332_b",
529
+ "lcb-atcoder-abc343_b",
530
+ "lcb-atcoder-abc361_a",
531
+ "lcb-atcoder-abc362_a",
532
+ "lcb-atcoder-abc328_a",
533
+ "lcb-atcoder-abc393_b",
534
+ "lcb-atcoder-abc352_a",
535
+ "lcb-atcoder-abc310_a",
536
+ "lcb-atcoder-abc365_a",
537
+ "lcb-atcoder-abc371_b",
538
+ "lcb-atcoder-abc367_c",
539
+ "lcb-atcoder-abc334_b",
540
+ "lcb-atcoder-abc385_c",
541
+ "lcb-atcoder-abc307_c",
542
+ "lcb-atcoder-abc338_c",
543
+ "lcb-atcoder-abc303_d",
544
+ "lcb-atcoder-abc342_c",
545
+ "lcb-atcoder-abc319_d",
546
+ "lcb-atcoder-abc315_d",
547
+ "lcb-atcoder-abc309_c",
548
+ "lcb-atcoder-abc390_d",
549
+ "lcb-atcoder-abc343_d",
550
+ "lcb-atcoder-abc397_b",
551
+ "lcb-atcoder-abc370_c",
552
+ "lcb-atcoder-abc375_c",
553
+ "lcb-atcoder-abc368_c",
554
+ "lcb-atcoder-abc325_b",
555
+ "lcb-atcoder-abc323_c",
556
+ "lcb-atcoder-abc377_c",
557
+ "lcb-atcoder-abc383_d",
558
+ "lcb-atcoder-arc189_a",
559
+ "lcb-codeforces-1883_C",
560
+ "lcb-atcoder-abc371_c",
561
+ "lcb-atcoder-abc380_c",
562
+ "lcb-atcoder-abc378_c",
563
+ "lcb-atcoder-abc366_c",
564
+ "lcb-atcoder-abc397_c",
565
+ "lcb-atcoder-abc339_c",
566
+ "lcb-atcoder-abc324_c",
567
+ "lcb-atcoder-abc355_c",
568
+ "lcb-atcoder-abc358_c",
569
+ "lcb-atcoder-abc321_d",
570
+ "lcb-atcoder-abc334_c",
571
+ "lcb-atcoder-arc195_c",
572
+ "lcb-atcoder-arc184_c",
573
+ "lcb-atcoder-abc377_e",
574
+ "lcb-atcoder-arc186_e",
575
+ "lcb-atcoder-abc398_f",
576
+ "lcb-atcoder-abc330_e",
577
+ "lcb-atcoder-abc396_e",
578
+ "lcb-atcoder-arc192_b",
579
+ "lcb-atcoder-abc391_f",
580
+ "lcb-atcoder-arc195_b",
581
+ "lcb-atcoder-abc384_g",
582
+ "lcb-atcoder-abc343_e",
583
+ "lcb-atcoder-abc385_e",
584
+ "lcb-atcoder-abc333_e",
585
+ "lcb-atcoder-abc341_e",
586
+ "lcb-atcoder-abc400_g",
587
+ "lcb-atcoder-abc362_d",
588
+ "lcb-codeforces-1899_D",
589
+ "lcb-atcoder-abc363_f",
590
+ "lcb-atcoder-abc382_d",
591
+ "lcb-atcoder-abc331_e",
592
+ "lcb-atcoder-abc351_e",
593
+ "lcb-atcoder-arc194_b",
594
+ "lcb-atcoder-abc325_f",
595
+ "lcb-atcoder-arc188_d",
596
+ "lcb-atcoder-abc368_e",
597
+ "lcb-atcoder-abc301_e",
598
+ "lcb-atcoder-arc194_e",
599
+ "lcb-atcoder-abc379_e",
600
+ "lcb-atcoder-abc360_e",
601
+ "lcb-atcoder-abc305_e",
602
+ "lcb-atcoder-arc186_a",
603
+ "lcb-atcoder-abc362_e",
604
+ "tool-0",
605
+ "tool-1",
606
+ "tool-2",
607
+ "tool-3",
608
+ "tool-4",
609
+ "tool-5",
610
+ "tool-6",
611
+ "tool-7",
612
+ "tool-8",
613
+ "tool-9",
614
+ "tool-10",
615
+ "tool-11",
616
+ "tool-12",
617
+ "tool-13",
618
+ "tool-14",
619
+ "tool-15",
620
+ "tool-16",
621
+ "tool-17",
622
+ "tool-18",
623
+ "tool-19",
624
+ "json-20",
625
+ "json-21",
626
+ "json-22",
627
+ "json-23",
628
+ "json-24",
629
+ "json-25",
630
+ "json-26",
631
+ "json-27",
632
+ "json-28",
633
+ "json-29"
634
+ ],
635
+ "concurrency": 2,
636
+ "sampling": "thinking: temperature1/top_p0.95/top_k20/xhigh; nonthinking: greedy; seed15027",
637
+ "api": "http://127.0.0.1:18021/v1",
638
+ "server": {
639
+ "model": "<WORKSPACE>/models/Qwen3.8-27B-W4A16-AutoRound-fast-eval",
640
+ "config_sha256": "5a3e7312dee3a6c5909beb854b05ac818be836683730e4facba81f547b77fd62",
641
+ "index_sha256": "316f7411552d8e80df6018b5f01c521e2b1a60c81b5564ed84366e50c74e383f",
642
+ "launcher_sha256": "6874eb0bc4306d61b57ebb2f2c7cab97b41f11a3e0ec2b1502d9c34d396c47d8",
643
+ "tokenizer_sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
644
+ "chat_template_sha256": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
645
+ "shards": {
646
+ "model-00005-of-00007.safetensors": {
647
+ "size": 699275200,
648
+ "mtime_ns": 1789678521251603880
649
+ },
650
+ "model-00001-of-00007.safetensors": {
651
+ "size": 3211003472,
652
+ "mtime_ns": 1789678522484562485
653
+ },
654
+ "model-00006-of-00007.safetensors": {
655
+ "size": 1291274752,
656
+ "mtime_ns": 1790442052555391180
657
+ },
658
+ "model-00003-of-00007.safetensors": {
659
+ "size": 3195027104,
660
+ "mtime_ns": 1789678524567492723
661
+ },
662
+ "model-00002-of-00007.safetensors": {
663
+ "size": 3195027104,
664
+ "mtime_ns": 1789678522534560808
665
+ },
666
+ "model_extra_tensors.safetensors": {
667
+ "size": 327162544,
668
+ "mtime_ns": 1789678527579392230
669
+ },
670
+ "model-00007-of-00007.safetensors": {
671
+ "size": 655565120,
672
+ "mtime_ns": 1789678527180405514
673
+ },
674
+ "model-00004-of-00007.safetensors": {
675
+ "size": 3217442568,
676
+ "mtime_ns": 1789678522650556917
677
+ }
678
+ },
679
+ "mode": "single-user",
680
+ "int8_activations": false,
681
+ "speculation": "mtp",
682
+ "max_model_len": 150000,
683
+ "max_num_seqs": 8,
684
+ "kv_cache_dtype": "fp8",
685
+ "prefix_cache": true,
686
+ "gpu_memory_utilization": 0.93,
687
+ "serving_profile": "local-single-user",
688
+ "vision": true,
689
+ "draft_tokens": 3,
690
+ "local_env_sha256": "ce50b2e28ea929a620bd62aa7f94aacb844832f9e3dd5767b6952c2ffa0e74a8",
691
+ "extra_args": "--limit-mm-per-prompt {\"image\":{\"count\":10}}"
692
+ }
693
+ }
evaluation/qwen-fast-full/summary.json ADDED
@@ -0,0 +1,89 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created": "2026-09-27T06:45:29Z",
3
+ "concurrency": 2,
4
+ "resumed": false,
5
+ "new_run_wall_seconds": 34295.263359012024,
6
+ "new_tasks": 630,
7
+ "suites": {
8
+ "gsm8k": {
9
+ "attempted": 200,
10
+ "correct": 195,
11
+ "accuracy": 0.975,
12
+ "truncation_policy": "count_as_wrong",
13
+ "truncated_counted_as_wrong": 0,
14
+ "errors": 0,
15
+ "truncated": 0,
16
+ "incomplete_token_counts": 0,
17
+ "mean_output_tokens": 405.4,
18
+ "mean_input_tokens": 93.27,
19
+ "mean_total_tokens": 498.67,
20
+ "mean_model_seconds": 3.4959702649945394,
21
+ "median_model_seconds": 3.070697635994293,
22
+ "p95_model_seconds": 6.074415434035473,
23
+ "summed_request_seconds_per_correct": 3.5856105281995276,
24
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
25
+ },
26
+ "ifbench": {
27
+ "attempted": 300,
28
+ "correct": 222,
29
+ "accuracy": 0.74,
30
+ "truncation_policy": "count_as_wrong",
31
+ "truncated_counted_as_wrong": 1,
32
+ "errors": 0,
33
+ "truncated": 1,
34
+ "incomplete_token_counts": 0,
35
+ "mean_output_tokens": 10429.793333333333,
36
+ "mean_input_tokens": 125.44,
37
+ "mean_total_tokens": 10555.233333333334,
38
+ "mean_model_seconds": 114.29685370972729,
39
+ "median_model_seconds": 59.345528290461516,
40
+ "p95_model_seconds": 438.60096366197104,
41
+ "summed_request_seconds_per_correct": 154.4552077158477,
42
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
43
+ },
44
+ "livecodebench": {
45
+ "attempted": 100,
46
+ "correct": 90,
47
+ "accuracy": 0.9,
48
+ "truncation_policy": "count_as_wrong",
49
+ "truncated_counted_as_wrong": 1,
50
+ "errors": 0,
51
+ "truncated": 1,
52
+ "incomplete_token_counts": 0,
53
+ "mean_output_tokens": 24492.16,
54
+ "mean_input_tokens": 653.75,
55
+ "mean_total_tokens": 25145.91,
56
+ "mean_model_seconds": 331.0521950917202,
57
+ "median_model_seconds": 90.47728537750663,
58
+ "p95_model_seconds": 1340.8850151840015,
59
+ "summed_request_seconds_per_correct": 367.8357723241336,
60
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
61
+ },
62
+ "tools": {
63
+ "attempted": 30,
64
+ "correct": 29,
65
+ "accuracy": 0.9666666666666667,
66
+ "truncation_policy": "count_as_wrong",
67
+ "truncated_counted_as_wrong": 0,
68
+ "errors": 0,
69
+ "truncated": 0,
70
+ "incomplete_token_counts": 0,
71
+ "mean_output_tokens": 40.833333333333336,
72
+ "mean_input_tokens": 481.3333333333333,
73
+ "mean_total_tokens": 522.1666666666666,
74
+ "mean_model_seconds": 1.0324668154586107,
75
+ "median_model_seconds": 1.2877807764452882,
76
+ "p95_model_seconds": 1.3691221890621819,
77
+ "summed_request_seconds_per_correct": 1.068069119439942,
78
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
79
+ }
80
+ },
81
+ "quality_comparison_ready": true,
82
+ "truncation_policy": "count_as_wrong",
83
+ "truncated_task_ids": [
84
+ "ifbench-107",
85
+ "lcb-atcoder-arc184_c"
86
+ ],
87
+ "suite_wall_seconds": 34295.263359012024,
88
+ "wall_seconds_per_correct": 63.983700296664225
89
+ }
evaluation/qwen-fast-pilot/manifest.json ADDED
@@ -0,0 +1,83 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "tasks_sha256": "809c2c6b124579d41c379a6a436e9649b5e1d1ddf44a147a3d70d9b4b19d68f4",
3
+ "task_ids": [
4
+ "gsm8k-27",
5
+ "gsm8k-60",
6
+ "gsm8k-105",
7
+ "gsm8k-109",
8
+ "gsm8k-152",
9
+ "ifbench-21",
10
+ "ifbench-76",
11
+ "ifbench-129",
12
+ "ifbench-130",
13
+ "ifbench-268",
14
+ "lcb-atcoder-abc377_b",
15
+ "lcb-atcoder-abc390_d",
16
+ "lcb-atcoder-abc385_e",
17
+ "lcb-atcoder-abc325_f",
18
+ "lcb-atcoder-abc368_e",
19
+ "tool-15",
20
+ "tool-16",
21
+ "json-22",
22
+ "json-24",
23
+ "json-26"
24
+ ],
25
+ "concurrency": 1,
26
+ "sampling": "thinking: temperature1/top_p0.95/top_k20/xhigh; nonthinking: greedy; seed15027",
27
+ "api": "http://127.0.0.1:18021/v1",
28
+ "server": {
29
+ "model": "<WORKSPACE>/models/Qwen3.8-27B-W4A16-AutoRound-fast-eval",
30
+ "config_sha256": "5a3e7312dee3a6c5909beb854b05ac818be836683730e4facba81f547b77fd62",
31
+ "index_sha256": "316f7411552d8e80df6018b5f01c521e2b1a60c81b5564ed84366e50c74e383f",
32
+ "launcher_sha256": "6874eb0bc4306d61b57ebb2f2c7cab97b41f11a3e0ec2b1502d9c34d396c47d8",
33
+ "tokenizer_sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
34
+ "chat_template_sha256": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
35
+ "shards": {
36
+ "model-00005-of-00007.safetensors": {
37
+ "size": 699275200,
38
+ "mtime_ns": 1789678521251603880
39
+ },
40
+ "model-00001-of-00007.safetensors": {
41
+ "size": 3211003472,
42
+ "mtime_ns": 1789678522484562485
43
+ },
44
+ "model-00006-of-00007.safetensors": {
45
+ "size": 1291274752,
46
+ "mtime_ns": 1790442052555391180
47
+ },
48
+ "model-00003-of-00007.safetensors": {
49
+ "size": 3195027104,
50
+ "mtime_ns": 1789678524567492723
51
+ },
52
+ "model-00002-of-00007.safetensors": {
53
+ "size": 3195027104,
54
+ "mtime_ns": 1789678522534560808
55
+ },
56
+ "model_extra_tensors.safetensors": {
57
+ "size": 327162544,
58
+ "mtime_ns": 1789678527579392230
59
+ },
60
+ "model-00007-of-00007.safetensors": {
61
+ "size": 655565120,
62
+ "mtime_ns": 1789678527180405514
63
+ },
64
+ "model-00004-of-00007.safetensors": {
65
+ "size": 3217442568,
66
+ "mtime_ns": 1789678522650556917
67
+ }
68
+ },
69
+ "mode": "single-user",
70
+ "int8_activations": false,
71
+ "speculation": "mtp",
72
+ "max_model_len": 150000,
73
+ "max_num_seqs": 8,
74
+ "kv_cache_dtype": "fp8",
75
+ "prefix_cache": true,
76
+ "gpu_memory_utilization": 0.93,
77
+ "serving_profile": "local-single-user",
78
+ "vision": true,
79
+ "draft_tokens": 3,
80
+ "local_env_sha256": "ce50b2e28ea929a620bd62aa7f94aacb844832f9e3dd5767b6952c2ffa0e74a8",
81
+ "extra_args": "--limit-mm-per-prompt {\"image\":{\"count\":10}}"
82
+ }
83
+ }
evaluation/qwen-fast-pilot/summary.json ADDED
@@ -0,0 +1,86 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created": "2026-09-26T21:13:49Z",
3
+ "concurrency": 1,
4
+ "resumed": false,
5
+ "new_run_wall_seconds": 4374.499300124997,
6
+ "new_tasks": 20,
7
+ "suites": {
8
+ "gsm8k": {
9
+ "attempted": 5,
10
+ "correct": 5,
11
+ "accuracy": 1.0,
12
+ "truncation_policy": "count_as_wrong",
13
+ "truncated_counted_as_wrong": 0,
14
+ "errors": 0,
15
+ "truncated": 0,
16
+ "incomplete_token_counts": 0,
17
+ "mean_output_tokens": 275.2,
18
+ "mean_input_tokens": 77.4,
19
+ "mean_total_tokens": 352.6,
20
+ "mean_model_seconds": 2.2795148026081735,
21
+ "median_model_seconds": 2.2747438669903204,
22
+ "p95_model_seconds": 2.7609281560289674,
23
+ "summed_request_seconds_per_correct": 2.2795148026081735,
24
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
25
+ },
26
+ "ifbench": {
27
+ "attempted": 5,
28
+ "correct": 4,
29
+ "accuracy": 0.8,
30
+ "truncation_policy": "count_as_wrong",
31
+ "truncated_counted_as_wrong": 0,
32
+ "errors": 0,
33
+ "truncated": 0,
34
+ "incomplete_token_counts": 0,
35
+ "mean_output_tokens": 5641,
36
+ "mean_input_tokens": 134.6,
37
+ "mean_total_tokens": 5775.6,
38
+ "mean_model_seconds": 60.683199993602464,
39
+ "median_model_seconds": 33.55660462501692,
40
+ "p95_model_seconds": 188.0719450309989,
41
+ "summed_request_seconds_per_correct": 75.85399999200308,
42
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
43
+ },
44
+ "livecodebench": {
45
+ "attempted": 5,
46
+ "correct": 5,
47
+ "accuracy": 1.0,
48
+ "truncation_policy": "count_as_wrong",
49
+ "truncated_counted_as_wrong": 0,
50
+ "errors": 0,
51
+ "truncated": 0,
52
+ "incomplete_token_counts": 0,
53
+ "mean_output_tokens": 63239.6,
54
+ "mean_input_tokens": 742.4,
55
+ "mean_total_tokens": 63982,
56
+ "mean_model_seconds": 807.4264700495871,
57
+ "median_model_seconds": 839.0481605409877,
58
+ "p95_model_seconds": 1524.661622893007,
59
+ "summed_request_seconds_per_correct": 807.4264700495871,
60
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
61
+ },
62
+ "tools": {
63
+ "attempted": 5,
64
+ "correct": 5,
65
+ "accuracy": 1.0,
66
+ "truncation_policy": "count_as_wrong",
67
+ "truncated_counted_as_wrong": 0,
68
+ "errors": 0,
69
+ "truncated": 0,
70
+ "incomplete_token_counts": 0,
71
+ "mean_output_tokens": 32.8,
72
+ "mean_input_tokens": 330.6,
73
+ "mean_total_tokens": 363.4,
74
+ "mean_model_seconds": 0.6404867220087909,
75
+ "median_model_seconds": 0.4934498239890672,
76
+ "p95_model_seconds": 1.0912616860005073,
77
+ "summed_request_seconds_per_correct": 0.6404867220087909,
78
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
79
+ }
80
+ },
81
+ "quality_comparison_ready": true,
82
+ "truncation_policy": "count_as_wrong",
83
+ "truncated_task_ids": [],
84
+ "suite_wall_seconds": 4374.499300124997,
85
+ "wall_seconds_per_correct": 230.23680526973666
86
+ }
evaluation/qwen-fast/perplexity.json ADDED
@@ -0,0 +1,659 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "input_sha256": "57c83ffe7c0dfba2b869f6a379d2df3f0bc0d055069b49569a2154336151a718",
3
+ "windows": [
4
+ {
5
+ "id": 0,
6
+ "language": "en",
7
+ "tokens": 283,
8
+ "logprob_sum": -592.9908143195717
9
+ },
10
+ {
11
+ "id": 1,
12
+ "language": "en",
13
+ "tokens": 313,
14
+ "logprob_sum": -694.9407490367958
15
+ },
16
+ {
17
+ "id": 2,
18
+ "language": "en",
19
+ "tokens": 284,
20
+ "logprob_sum": -731.2198149585729
21
+ },
22
+ {
23
+ "id": 3,
24
+ "language": "en",
25
+ "tokens": 286,
26
+ "logprob_sum": -760.9622869625673
27
+ },
28
+ {
29
+ "id": 4,
30
+ "language": "en",
31
+ "tokens": 279,
32
+ "logprob_sum": -585.9759548976872
33
+ },
34
+ {
35
+ "id": 5,
36
+ "language": "en",
37
+ "tokens": 251,
38
+ "logprob_sum": -624.1354073971652
39
+ },
40
+ {
41
+ "id": 6,
42
+ "language": "en",
43
+ "tokens": 258,
44
+ "logprob_sum": -606.1599571104307
45
+ },
46
+ {
47
+ "id": 7,
48
+ "language": "en",
49
+ "tokens": 258,
50
+ "logprob_sum": -601.0718723634782
51
+ },
52
+ {
53
+ "id": 8,
54
+ "language": "en",
55
+ "tokens": 274,
56
+ "logprob_sum": -629.9130605636892
57
+ },
58
+ {
59
+ "id": 9,
60
+ "language": "en",
61
+ "tokens": 278,
62
+ "logprob_sum": -605.0941830798547
63
+ },
64
+ {
65
+ "id": 10,
66
+ "language": "en",
67
+ "tokens": 274,
68
+ "logprob_sum": -713.5701698549092
69
+ },
70
+ {
71
+ "id": 11,
72
+ "language": "en",
73
+ "tokens": 288,
74
+ "logprob_sum": -759.5278018228346
75
+ },
76
+ {
77
+ "id": 12,
78
+ "language": "en",
79
+ "tokens": 277,
80
+ "logprob_sum": -623.4307517245397
81
+ },
82
+ {
83
+ "id": 13,
84
+ "language": "en",
85
+ "tokens": 291,
86
+ "logprob_sum": -699.0976419129438
87
+ },
88
+ {
89
+ "id": 14,
90
+ "language": "en",
91
+ "tokens": 260,
92
+ "logprob_sum": -672.0876630296007
93
+ },
94
+ {
95
+ "id": 15,
96
+ "language": "en",
97
+ "tokens": 250,
98
+ "logprob_sum": -761.0354307279922
99
+ },
100
+ {
101
+ "id": 16,
102
+ "language": "en",
103
+ "tokens": 252,
104
+ "logprob_sum": -754.8780198292225
105
+ },
106
+ {
107
+ "id": 17,
108
+ "language": "en",
109
+ "tokens": 272,
110
+ "logprob_sum": -788.3398015776256
111
+ },
112
+ {
113
+ "id": 18,
114
+ "language": "en",
115
+ "tokens": 240,
116
+ "logprob_sum": -668.2802309252293
117
+ },
118
+ {
119
+ "id": 19,
120
+ "language": "en",
121
+ "tokens": 244,
122
+ "logprob_sum": -561.2377929796639
123
+ },
124
+ {
125
+ "id": 20,
126
+ "language": "en",
127
+ "tokens": 274,
128
+ "logprob_sum": -682.5377916035941
129
+ },
130
+ {
131
+ "id": 21,
132
+ "language": "en",
133
+ "tokens": 299,
134
+ "logprob_sum": -738.1227494915511
135
+ },
136
+ {
137
+ "id": 22,
138
+ "language": "en",
139
+ "tokens": 318,
140
+ "logprob_sum": -823.5101455471158
141
+ },
142
+ {
143
+ "id": 23,
144
+ "language": "en",
145
+ "tokens": 263,
146
+ "logprob_sum": -677.3689384464233
147
+ },
148
+ {
149
+ "id": 24,
150
+ "language": "en",
151
+ "tokens": 239,
152
+ "logprob_sum": -711.9839977540105
153
+ },
154
+ {
155
+ "id": 25,
156
+ "language": "en",
157
+ "tokens": 293,
158
+ "logprob_sum": -493.05563317897304
159
+ },
160
+ {
161
+ "id": 26,
162
+ "language": "en",
163
+ "tokens": 303,
164
+ "logprob_sum": -678.5333020727912
165
+ },
166
+ {
167
+ "id": 27,
168
+ "language": "en",
169
+ "tokens": 276,
170
+ "logprob_sum": -623.330151065903
171
+ },
172
+ {
173
+ "id": 28,
174
+ "language": "en",
175
+ "tokens": 282,
176
+ "logprob_sum": -729.7613100720046
177
+ },
178
+ {
179
+ "id": 29,
180
+ "language": "en",
181
+ "tokens": 298,
182
+ "logprob_sum": -691.1857994162092
183
+ },
184
+ {
185
+ "id": 30,
186
+ "language": "en",
187
+ "tokens": 338,
188
+ "logprob_sum": -478.5514251983941
189
+ },
190
+ {
191
+ "id": 31,
192
+ "language": "en",
193
+ "tokens": 280,
194
+ "logprob_sum": -639.465224127729
195
+ },
196
+ {
197
+ "id": 32,
198
+ "language": "en",
199
+ "tokens": 310,
200
+ "logprob_sum": -813.8291402243576
201
+ },
202
+ {
203
+ "id": 33,
204
+ "language": "en",
205
+ "tokens": 280,
206
+ "logprob_sum": -798.3652476328898
207
+ },
208
+ {
209
+ "id": 34,
210
+ "language": "en",
211
+ "tokens": 302,
212
+ "logprob_sum": -693.1609880588003
213
+ },
214
+ {
215
+ "id": 35,
216
+ "language": "en",
217
+ "tokens": 286,
218
+ "logprob_sum": -523.3296906485411
219
+ },
220
+ {
221
+ "id": 36,
222
+ "language": "en",
223
+ "tokens": 244,
224
+ "logprob_sum": -581.809817118934
225
+ },
226
+ {
227
+ "id": 37,
228
+ "language": "en",
229
+ "tokens": 280,
230
+ "logprob_sum": -495.3885064668866
231
+ },
232
+ {
233
+ "id": 38,
234
+ "language": "en",
235
+ "tokens": 315,
236
+ "logprob_sum": -599.4133885340561
237
+ },
238
+ {
239
+ "id": 39,
240
+ "language": "en",
241
+ "tokens": 283,
242
+ "logprob_sum": -647.8761923574384
243
+ },
244
+ {
245
+ "id": 40,
246
+ "language": "da",
247
+ "tokens": 319,
248
+ "logprob_sum": -518.2542151119087
249
+ },
250
+ {
251
+ "id": 41,
252
+ "language": "da",
253
+ "tokens": 334,
254
+ "logprob_sum": -796.8319925991527
255
+ },
256
+ {
257
+ "id": 42,
258
+ "language": "da",
259
+ "tokens": 321,
260
+ "logprob_sum": -657.0615293114461
261
+ },
262
+ {
263
+ "id": 43,
264
+ "language": "da",
265
+ "tokens": 355,
266
+ "logprob_sum": -765.0516552211648
267
+ },
268
+ {
269
+ "id": 44,
270
+ "language": "da",
271
+ "tokens": 372,
272
+ "logprob_sum": -793.2305373175477
273
+ },
274
+ {
275
+ "id": 45,
276
+ "language": "da",
277
+ "tokens": 360,
278
+ "logprob_sum": -591.0126144723617
279
+ },
280
+ {
281
+ "id": 46,
282
+ "language": "da",
283
+ "tokens": 348,
284
+ "logprob_sum": -1261.7429924435055
285
+ },
286
+ {
287
+ "id": 47,
288
+ "language": "da",
289
+ "tokens": 372,
290
+ "logprob_sum": -597.0048053516152
291
+ },
292
+ {
293
+ "id": 48,
294
+ "language": "da",
295
+ "tokens": 356,
296
+ "logprob_sum": -806.2568797468084
297
+ },
298
+ {
299
+ "id": 49,
300
+ "language": "da",
301
+ "tokens": 269,
302
+ "logprob_sum": -1434.5953341651184
303
+ },
304
+ {
305
+ "id": 50,
306
+ "language": "da",
307
+ "tokens": 334,
308
+ "logprob_sum": -586.6079792070204
309
+ },
310
+ {
311
+ "id": 51,
312
+ "language": "da",
313
+ "tokens": 364,
314
+ "logprob_sum": -621.1433348721202
315
+ },
316
+ {
317
+ "id": 52,
318
+ "language": "da",
319
+ "tokens": 364,
320
+ "logprob_sum": -678.3656660186107
321
+ },
322
+ {
323
+ "id": 53,
324
+ "language": "da",
325
+ "tokens": 378,
326
+ "logprob_sum": -453.8299491084478
327
+ },
328
+ {
329
+ "id": 54,
330
+ "language": "da",
331
+ "tokens": 400,
332
+ "logprob_sum": -827.0148732287234
333
+ },
334
+ {
335
+ "id": 55,
336
+ "language": "da",
337
+ "tokens": 319,
338
+ "logprob_sum": -1556.6795706446283
339
+ },
340
+ {
341
+ "id": 56,
342
+ "language": "da",
343
+ "tokens": 329,
344
+ "logprob_sum": -621.5343871314999
345
+ },
346
+ {
347
+ "id": 57,
348
+ "language": "da",
349
+ "tokens": 338,
350
+ "logprob_sum": -865.1125200971237
351
+ },
352
+ {
353
+ "id": 58,
354
+ "language": "da",
355
+ "tokens": 336,
356
+ "logprob_sum": -1569.0122353932675
357
+ },
358
+ {
359
+ "id": 59,
360
+ "language": "da",
361
+ "tokens": 348,
362
+ "logprob_sum": -990.5655090794025
363
+ },
364
+ {
365
+ "id": 60,
366
+ "language": "da",
367
+ "tokens": 338,
368
+ "logprob_sum": -484.93386594452477
369
+ },
370
+ {
371
+ "id": 61,
372
+ "language": "da",
373
+ "tokens": 349,
374
+ "logprob_sum": -868.9878758826126
375
+ },
376
+ {
377
+ "id": 62,
378
+ "language": "da",
379
+ "tokens": 349,
380
+ "logprob_sum": -569.4126007331457
381
+ },
382
+ {
383
+ "id": 63,
384
+ "language": "da",
385
+ "tokens": 356,
386
+ "logprob_sum": -669.0766140862625
387
+ },
388
+ {
389
+ "id": 64,
390
+ "language": "da",
391
+ "tokens": 388,
392
+ "logprob_sum": -739.1665408044555
393
+ },
394
+ {
395
+ "id": 65,
396
+ "language": "da",
397
+ "tokens": 342,
398
+ "logprob_sum": -1576.0850733005718
399
+ },
400
+ {
401
+ "id": 66,
402
+ "language": "da",
403
+ "tokens": 376,
404
+ "logprob_sum": -680.5289953708525
405
+ },
406
+ {
407
+ "id": 67,
408
+ "language": "da",
409
+ "tokens": 388,
410
+ "logprob_sum": -563.0824029279489
411
+ },
412
+ {
413
+ "id": 68,
414
+ "language": "da",
415
+ "tokens": 338,
416
+ "logprob_sum": -680.2388399759147
417
+ },
418
+ {
419
+ "id": 69,
420
+ "language": "da",
421
+ "tokens": 335,
422
+ "logprob_sum": -643.2929824862158
423
+ },
424
+ {
425
+ "id": 70,
426
+ "language": "da",
427
+ "tokens": 338,
428
+ "logprob_sum": -742.9423492672252
429
+ },
430
+ {
431
+ "id": 71,
432
+ "language": "da",
433
+ "tokens": 371,
434
+ "logprob_sum": -1402.213066599099
435
+ },
436
+ {
437
+ "id": 72,
438
+ "language": "da",
439
+ "tokens": 371,
440
+ "logprob_sum": -730.0679423760885
441
+ },
442
+ {
443
+ "id": 73,
444
+ "language": "da",
445
+ "tokens": 325,
446
+ "logprob_sum": -584.7675743625477
447
+ },
448
+ {
449
+ "id": 74,
450
+ "language": "da",
451
+ "tokens": 381,
452
+ "logprob_sum": -1003.3340749706913
453
+ },
454
+ {
455
+ "id": 75,
456
+ "language": "da",
457
+ "tokens": 265,
458
+ "logprob_sum": -1381.0563661176711
459
+ },
460
+ {
461
+ "id": 76,
462
+ "language": "da",
463
+ "tokens": 348,
464
+ "logprob_sum": -895.6867733937834
465
+ },
466
+ {
467
+ "id": 77,
468
+ "language": "da",
469
+ "tokens": 338,
470
+ "logprob_sum": -603.0461025548739
471
+ },
472
+ {
473
+ "id": 78,
474
+ "language": "da",
475
+ "tokens": 346,
476
+ "logprob_sum": -860.3451435220468
477
+ },
478
+ {
479
+ "id": 79,
480
+ "language": "da",
481
+ "tokens": 359,
482
+ "logprob_sum": -592.5020944437811
483
+ },
484
+ {
485
+ "id": 80,
486
+ "language": "code",
487
+ "tokens": 320,
488
+ "logprob_sum": -276.6699116966555
489
+ },
490
+ {
491
+ "id": 81,
492
+ "language": "code",
493
+ "tokens": 282,
494
+ "logprob_sum": -420.23796508818486
495
+ },
496
+ {
497
+ "id": 82,
498
+ "language": "code",
499
+ "tokens": 278,
500
+ "logprob_sum": -263.582931014716
501
+ },
502
+ {
503
+ "id": 83,
504
+ "language": "code",
505
+ "tokens": 273,
506
+ "logprob_sum": -418.2609401268255
507
+ },
508
+ {
509
+ "id": 84,
510
+ "language": "code",
511
+ "tokens": 253,
512
+ "logprob_sum": -199.9309399436836
513
+ },
514
+ {
515
+ "id": 85,
516
+ "language": "code",
517
+ "tokens": 239,
518
+ "logprob_sum": -465.6562252798758
519
+ },
520
+ {
521
+ "id": 86,
522
+ "language": "code",
523
+ "tokens": 272,
524
+ "logprob_sum": -374.48760858155583
525
+ },
526
+ {
527
+ "id": 87,
528
+ "language": "code",
529
+ "tokens": 253,
530
+ "logprob_sum": -291.8236611021075
531
+ },
532
+ {
533
+ "id": 88,
534
+ "language": "code",
535
+ "tokens": 317,
536
+ "logprob_sum": -284.14727575928714
537
+ },
538
+ {
539
+ "id": 89,
540
+ "language": "code",
541
+ "tokens": 298,
542
+ "logprob_sum": -303.1649754933601
543
+ },
544
+ {
545
+ "id": 90,
546
+ "language": "code",
547
+ "tokens": 280,
548
+ "logprob_sum": -329.05116950601234
549
+ },
550
+ {
551
+ "id": 91,
552
+ "language": "code",
553
+ "tokens": 265,
554
+ "logprob_sum": -394.77499111452255
555
+ },
556
+ {
557
+ "id": 92,
558
+ "language": "code",
559
+ "tokens": 305,
560
+ "logprob_sum": -219.57371839839698
561
+ },
562
+ {
563
+ "id": 93,
564
+ "language": "code",
565
+ "tokens": 310,
566
+ "logprob_sum": -331.6206139813031
567
+ },
568
+ {
569
+ "id": 94,
570
+ "language": "code",
571
+ "tokens": 278,
572
+ "logprob_sum": -251.4822827990156
573
+ },
574
+ {
575
+ "id": 95,
576
+ "language": "code",
577
+ "tokens": 301,
578
+ "logprob_sum": -308.4679439706525
579
+ },
580
+ {
581
+ "id": 96,
582
+ "language": "code",
583
+ "tokens": 311,
584
+ "logprob_sum": -184.7737082277912
585
+ },
586
+ {
587
+ "id": 97,
588
+ "language": "code",
589
+ "tokens": 312,
590
+ "logprob_sum": -226.99020833792702
591
+ },
592
+ {
593
+ "id": 98,
594
+ "language": "code",
595
+ "tokens": 325,
596
+ "logprob_sum": -197.7390554841084
597
+ },
598
+ {
599
+ "id": 99,
600
+ "language": "code",
601
+ "tokens": 291,
602
+ "logprob_sum": -325.6971021090387
603
+ },
604
+ {
605
+ "id": 100,
606
+ "language": "code",
607
+ "tokens": 321,
608
+ "logprob_sum": -453.9610157664529
609
+ },
610
+ {
611
+ "id": 101,
612
+ "language": "code",
613
+ "tokens": 318,
614
+ "logprob_sum": -451.3294726509339
615
+ },
616
+ {
617
+ "id": 102,
618
+ "language": "code",
619
+ "tokens": 316,
620
+ "logprob_sum": -502.03988517310944
621
+ },
622
+ {
623
+ "id": 103,
624
+ "language": "code",
625
+ "tokens": 307,
626
+ "logprob_sum": -297.40668610171537
627
+ },
628
+ {
629
+ "id": 104,
630
+ "language": "code",
631
+ "tokens": 254,
632
+ "logprob_sum": -418.13921085138236
633
+ },
634
+ {
635
+ "id": 105,
636
+ "language": "code",
637
+ "tokens": 275,
638
+ "logprob_sum": -457.8816399213333
639
+ }
640
+ ],
641
+ "scores": {
642
+ "en": {
643
+ "tokens": 11175,
644
+ "ppl": 10.764397834031014
645
+ },
646
+ "all": {
647
+ "tokens": 32646,
648
+ "ppl": 8.143312831984629
649
+ },
650
+ "da": {
651
+ "tokens": 13917,
652
+ "ppl": 10.913529905515743
653
+ },
654
+ "code": {
655
+ "tokens": 7554,
656
+ "ppl": 3.1422587970654146
657
+ }
658
+ }
659
+ }
evaluation/qwen-fast/server.json ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created": "2026-09-26T19:58:25Z",
3
+ "model": "<WORKSPACE>/models/Qwen3.8-27B-W4A16-AutoRound-fast-eval",
4
+ "config_sha256": "5a3e7312dee3a6c5909beb854b05ac818be836683730e4facba81f547b77fd62",
5
+ "index_sha256": "316f7411552d8e80df6018b5f01c521e2b1a60c81b5564ed84366e50c74e383f",
6
+ "launcher_sha256": "6874eb0bc4306d61b57ebb2f2c7cab97b41f11a3e0ec2b1502d9c34d396c47d8",
7
+ "tokenizer_sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
8
+ "chat_template_sha256": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
9
+ "shards": {
10
+ "model-00005-of-00007.safetensors": {
11
+ "size": 699275200,
12
+ "mtime_ns": 1789678521251603880
13
+ },
14
+ "model-00001-of-00007.safetensors": {
15
+ "size": 3211003472,
16
+ "mtime_ns": 1789678522484562485
17
+ },
18
+ "model-00006-of-00007.safetensors": {
19
+ "size": 1291274752,
20
+ "mtime_ns": 1790442052555391180
21
+ },
22
+ "model-00003-of-00007.safetensors": {
23
+ "size": 3195027104,
24
+ "mtime_ns": 1789678524567492723
25
+ },
26
+ "model-00002-of-00007.safetensors": {
27
+ "size": 3195027104,
28
+ "mtime_ns": 1789678522534560808
29
+ },
30
+ "model_extra_tensors.safetensors": {
31
+ "size": 327162544,
32
+ "mtime_ns": 1789678527579392230
33
+ },
34
+ "model-00007-of-00007.safetensors": {
35
+ "size": 655565120,
36
+ "mtime_ns": 1789678527180405514
37
+ },
38
+ "model-00004-of-00007.safetensors": {
39
+ "size": 3217442568,
40
+ "mtime_ns": 1789678522650556917
41
+ }
42
+ },
43
+ "mode": "single-user",
44
+ "int8_activations": false,
45
+ "speculation": "mtp",
46
+ "max_model_len": 150000,
47
+ "max_num_seqs": 8,
48
+ "kv_cache_dtype": "fp8",
49
+ "prefix_cache": true,
50
+ "gpu_memory_utilization": 0.93,
51
+ "pid": 926860,
52
+ "serving_profile": "local-single-user",
53
+ "vision": true,
54
+ "draft_tokens": 3,
55
+ "local_env_sha256": "ce50b2e28ea929a620bd62aa7f94aacb844832f9e3dd5767b6952c2ffa0e74a8",
56
+ "extra_args": "--limit-mm-per-prompt {\"image\":{\"count\":10}}"
57
+ }
evaluation/qwen-fast/speed-c1.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "concurrency": 1,
3
+ "requests": 8,
4
+ "output_tokens_per_request": 512,
5
+ "wall_seconds": 37.67895263497485,
6
+ "aggregate_output_tps": 108.7079049059861,
7
+ "median_decode_tps": 112.12193584833635,
8
+ "median_ttft_seconds": 0.07377005548914894,
9
+ "sampling": "greedy; ignore_eos for this throughput test only",
10
+ "speculation_counter_deltas": {
11
+ "vllm:spec_decode_num_drafts_total": 1320.0,
12
+ "vllm:spec_decode_num_draft_tokens_total": 3960.0,
13
+ "vllm:spec_decode_num_accepted_tokens_total": 2782.0
14
+ },
15
+ "decode_tps_note": "Client stream timing estimate; speculative decoding may deliver several tokens in one event."
16
+ }
evaluation/swift10-full/manifest.json ADDED
@@ -0,0 +1,697 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "tasks_sha256": "809c2c6b124579d41c379a6a436e9649b5e1d1ddf44a147a3d70d9b4b19d68f4",
3
+ "task_ids": [
4
+ "gsm8k-0",
5
+ "gsm8k-1",
6
+ "gsm8k-2",
7
+ "gsm8k-3",
8
+ "gsm8k-4",
9
+ "gsm8k-5",
10
+ "gsm8k-6",
11
+ "gsm8k-7",
12
+ "gsm8k-8",
13
+ "gsm8k-9",
14
+ "gsm8k-10",
15
+ "gsm8k-11",
16
+ "gsm8k-12",
17
+ "gsm8k-13",
18
+ "gsm8k-14",
19
+ "gsm8k-15",
20
+ "gsm8k-16",
21
+ "gsm8k-17",
22
+ "gsm8k-18",
23
+ "gsm8k-19",
24
+ "gsm8k-20",
25
+ "gsm8k-21",
26
+ "gsm8k-22",
27
+ "gsm8k-23",
28
+ "gsm8k-24",
29
+ "gsm8k-25",
30
+ "gsm8k-26",
31
+ "gsm8k-27",
32
+ "gsm8k-28",
33
+ "gsm8k-29",
34
+ "gsm8k-30",
35
+ "gsm8k-31",
36
+ "gsm8k-32",
37
+ "gsm8k-33",
38
+ "gsm8k-34",
39
+ "gsm8k-35",
40
+ "gsm8k-36",
41
+ "gsm8k-37",
42
+ "gsm8k-38",
43
+ "gsm8k-39",
44
+ "gsm8k-40",
45
+ "gsm8k-41",
46
+ "gsm8k-42",
47
+ "gsm8k-43",
48
+ "gsm8k-44",
49
+ "gsm8k-45",
50
+ "gsm8k-46",
51
+ "gsm8k-47",
52
+ "gsm8k-48",
53
+ "gsm8k-49",
54
+ "gsm8k-50",
55
+ "gsm8k-51",
56
+ "gsm8k-52",
57
+ "gsm8k-53",
58
+ "gsm8k-54",
59
+ "gsm8k-55",
60
+ "gsm8k-56",
61
+ "gsm8k-57",
62
+ "gsm8k-58",
63
+ "gsm8k-59",
64
+ "gsm8k-60",
65
+ "gsm8k-61",
66
+ "gsm8k-62",
67
+ "gsm8k-63",
68
+ "gsm8k-64",
69
+ "gsm8k-65",
70
+ "gsm8k-66",
71
+ "gsm8k-67",
72
+ "gsm8k-68",
73
+ "gsm8k-69",
74
+ "gsm8k-70",
75
+ "gsm8k-71",
76
+ "gsm8k-72",
77
+ "gsm8k-73",
78
+ "gsm8k-74",
79
+ "gsm8k-75",
80
+ "gsm8k-76",
81
+ "gsm8k-77",
82
+ "gsm8k-78",
83
+ "gsm8k-79",
84
+ "gsm8k-80",
85
+ "gsm8k-81",
86
+ "gsm8k-82",
87
+ "gsm8k-83",
88
+ "gsm8k-84",
89
+ "gsm8k-85",
90
+ "gsm8k-86",
91
+ "gsm8k-87",
92
+ "gsm8k-88",
93
+ "gsm8k-89",
94
+ "gsm8k-90",
95
+ "gsm8k-91",
96
+ "gsm8k-92",
97
+ "gsm8k-93",
98
+ "gsm8k-94",
99
+ "gsm8k-95",
100
+ "gsm8k-96",
101
+ "gsm8k-97",
102
+ "gsm8k-98",
103
+ "gsm8k-99",
104
+ "gsm8k-100",
105
+ "gsm8k-101",
106
+ "gsm8k-102",
107
+ "gsm8k-103",
108
+ "gsm8k-104",
109
+ "gsm8k-105",
110
+ "gsm8k-106",
111
+ "gsm8k-107",
112
+ "gsm8k-108",
113
+ "gsm8k-109",
114
+ "gsm8k-110",
115
+ "gsm8k-111",
116
+ "gsm8k-112",
117
+ "gsm8k-113",
118
+ "gsm8k-114",
119
+ "gsm8k-115",
120
+ "gsm8k-116",
121
+ "gsm8k-117",
122
+ "gsm8k-118",
123
+ "gsm8k-119",
124
+ "gsm8k-120",
125
+ "gsm8k-121",
126
+ "gsm8k-122",
127
+ "gsm8k-123",
128
+ "gsm8k-124",
129
+ "gsm8k-125",
130
+ "gsm8k-126",
131
+ "gsm8k-127",
132
+ "gsm8k-128",
133
+ "gsm8k-129",
134
+ "gsm8k-130",
135
+ "gsm8k-131",
136
+ "gsm8k-132",
137
+ "gsm8k-133",
138
+ "gsm8k-134",
139
+ "gsm8k-135",
140
+ "gsm8k-136",
141
+ "gsm8k-137",
142
+ "gsm8k-138",
143
+ "gsm8k-139",
144
+ "gsm8k-140",
145
+ "gsm8k-141",
146
+ "gsm8k-142",
147
+ "gsm8k-143",
148
+ "gsm8k-144",
149
+ "gsm8k-145",
150
+ "gsm8k-146",
151
+ "gsm8k-147",
152
+ "gsm8k-148",
153
+ "gsm8k-149",
154
+ "gsm8k-150",
155
+ "gsm8k-151",
156
+ "gsm8k-152",
157
+ "gsm8k-153",
158
+ "gsm8k-154",
159
+ "gsm8k-155",
160
+ "gsm8k-156",
161
+ "gsm8k-157",
162
+ "gsm8k-158",
163
+ "gsm8k-159",
164
+ "gsm8k-160",
165
+ "gsm8k-161",
166
+ "gsm8k-162",
167
+ "gsm8k-163",
168
+ "gsm8k-164",
169
+ "gsm8k-165",
170
+ "gsm8k-166",
171
+ "gsm8k-167",
172
+ "gsm8k-168",
173
+ "gsm8k-169",
174
+ "gsm8k-170",
175
+ "gsm8k-171",
176
+ "gsm8k-172",
177
+ "gsm8k-173",
178
+ "gsm8k-174",
179
+ "gsm8k-175",
180
+ "gsm8k-176",
181
+ "gsm8k-177",
182
+ "gsm8k-178",
183
+ "gsm8k-179",
184
+ "gsm8k-180",
185
+ "gsm8k-181",
186
+ "gsm8k-182",
187
+ "gsm8k-183",
188
+ "gsm8k-184",
189
+ "gsm8k-185",
190
+ "gsm8k-186",
191
+ "gsm8k-187",
192
+ "gsm8k-188",
193
+ "gsm8k-189",
194
+ "gsm8k-190",
195
+ "gsm8k-191",
196
+ "gsm8k-192",
197
+ "gsm8k-193",
198
+ "gsm8k-194",
199
+ "gsm8k-195",
200
+ "gsm8k-196",
201
+ "gsm8k-197",
202
+ "gsm8k-198",
203
+ "gsm8k-199",
204
+ "ifbench-0",
205
+ "ifbench-1",
206
+ "ifbench-2",
207
+ "ifbench-3",
208
+ "ifbench-4",
209
+ "ifbench-5",
210
+ "ifbench-6",
211
+ "ifbench-7",
212
+ "ifbench-8",
213
+ "ifbench-9",
214
+ "ifbench-10",
215
+ "ifbench-11",
216
+ "ifbench-12",
217
+ "ifbench-13",
218
+ "ifbench-14",
219
+ "ifbench-15",
220
+ "ifbench-16",
221
+ "ifbench-17",
222
+ "ifbench-18",
223
+ "ifbench-19",
224
+ "ifbench-20",
225
+ "ifbench-21",
226
+ "ifbench-22",
227
+ "ifbench-23",
228
+ "ifbench-24",
229
+ "ifbench-25",
230
+ "ifbench-26",
231
+ "ifbench-27",
232
+ "ifbench-28",
233
+ "ifbench-29",
234
+ "ifbench-30",
235
+ "ifbench-31",
236
+ "ifbench-32",
237
+ "ifbench-33",
238
+ "ifbench-34",
239
+ "ifbench-35",
240
+ "ifbench-36",
241
+ "ifbench-37",
242
+ "ifbench-38",
243
+ "ifbench-39",
244
+ "ifbench-40",
245
+ "ifbench-41",
246
+ "ifbench-42",
247
+ "ifbench-43",
248
+ "ifbench-44",
249
+ "ifbench-45",
250
+ "ifbench-46",
251
+ "ifbench-47",
252
+ "ifbench-48",
253
+ "ifbench-49",
254
+ "ifbench-50",
255
+ "ifbench-51",
256
+ "ifbench-52",
257
+ "ifbench-53",
258
+ "ifbench-54",
259
+ "ifbench-55",
260
+ "ifbench-56",
261
+ "ifbench-57",
262
+ "ifbench-58",
263
+ "ifbench-59",
264
+ "ifbench-60",
265
+ "ifbench-61",
266
+ "ifbench-62",
267
+ "ifbench-63",
268
+ "ifbench-64",
269
+ "ifbench-65",
270
+ "ifbench-66",
271
+ "ifbench-67",
272
+ "ifbench-68",
273
+ "ifbench-69",
274
+ "ifbench-70",
275
+ "ifbench-71",
276
+ "ifbench-72",
277
+ "ifbench-73",
278
+ "ifbench-74",
279
+ "ifbench-75",
280
+ "ifbench-76",
281
+ "ifbench-77",
282
+ "ifbench-78",
283
+ "ifbench-79",
284
+ "ifbench-80",
285
+ "ifbench-81",
286
+ "ifbench-82",
287
+ "ifbench-83",
288
+ "ifbench-84",
289
+ "ifbench-85",
290
+ "ifbench-86",
291
+ "ifbench-87",
292
+ "ifbench-88",
293
+ "ifbench-89",
294
+ "ifbench-90",
295
+ "ifbench-91",
296
+ "ifbench-92",
297
+ "ifbench-93",
298
+ "ifbench-94",
299
+ "ifbench-95",
300
+ "ifbench-96",
301
+ "ifbench-97",
302
+ "ifbench-98",
303
+ "ifbench-99",
304
+ "ifbench-100",
305
+ "ifbench-101",
306
+ "ifbench-102",
307
+ "ifbench-103",
308
+ "ifbench-104",
309
+ "ifbench-105",
310
+ "ifbench-106",
311
+ "ifbench-107",
312
+ "ifbench-108",
313
+ "ifbench-109",
314
+ "ifbench-110",
315
+ "ifbench-111",
316
+ "ifbench-112",
317
+ "ifbench-113",
318
+ "ifbench-114",
319
+ "ifbench-115",
320
+ "ifbench-116",
321
+ "ifbench-117",
322
+ "ifbench-118",
323
+ "ifbench-119",
324
+ "ifbench-120",
325
+ "ifbench-121",
326
+ "ifbench-122",
327
+ "ifbench-123",
328
+ "ifbench-124",
329
+ "ifbench-125",
330
+ "ifbench-126",
331
+ "ifbench-127",
332
+ "ifbench-128",
333
+ "ifbench-129",
334
+ "ifbench-130",
335
+ "ifbench-131",
336
+ "ifbench-132",
337
+ "ifbench-133",
338
+ "ifbench-134",
339
+ "ifbench-135",
340
+ "ifbench-136",
341
+ "ifbench-137",
342
+ "ifbench-138",
343
+ "ifbench-139",
344
+ "ifbench-140",
345
+ "ifbench-141",
346
+ "ifbench-142",
347
+ "ifbench-143",
348
+ "ifbench-144",
349
+ "ifbench-145",
350
+ "ifbench-146",
351
+ "ifbench-147",
352
+ "ifbench-148",
353
+ "ifbench-149",
354
+ "ifbench-150",
355
+ "ifbench-151",
356
+ "ifbench-152",
357
+ "ifbench-153",
358
+ "ifbench-154",
359
+ "ifbench-155",
360
+ "ifbench-156",
361
+ "ifbench-157",
362
+ "ifbench-158",
363
+ "ifbench-159",
364
+ "ifbench-160",
365
+ "ifbench-161",
366
+ "ifbench-162",
367
+ "ifbench-163",
368
+ "ifbench-164",
369
+ "ifbench-165",
370
+ "ifbench-166",
371
+ "ifbench-167",
372
+ "ifbench-168",
373
+ "ifbench-169",
374
+ "ifbench-170",
375
+ "ifbench-171",
376
+ "ifbench-172",
377
+ "ifbench-173",
378
+ "ifbench-174",
379
+ "ifbench-175",
380
+ "ifbench-176",
381
+ "ifbench-177",
382
+ "ifbench-178",
383
+ "ifbench-179",
384
+ "ifbench-180",
385
+ "ifbench-181",
386
+ "ifbench-182",
387
+ "ifbench-183",
388
+ "ifbench-184",
389
+ "ifbench-185",
390
+ "ifbench-186",
391
+ "ifbench-187",
392
+ "ifbench-188",
393
+ "ifbench-189",
394
+ "ifbench-190",
395
+ "ifbench-191",
396
+ "ifbench-192",
397
+ "ifbench-193",
398
+ "ifbench-194",
399
+ "ifbench-195",
400
+ "ifbench-196",
401
+ "ifbench-197",
402
+ "ifbench-198",
403
+ "ifbench-199",
404
+ "ifbench-200",
405
+ "ifbench-201",
406
+ "ifbench-202",
407
+ "ifbench-203",
408
+ "ifbench-204",
409
+ "ifbench-205",
410
+ "ifbench-206",
411
+ "ifbench-207",
412
+ "ifbench-208",
413
+ "ifbench-209",
414
+ "ifbench-210",
415
+ "ifbench-211",
416
+ "ifbench-212",
417
+ "ifbench-213",
418
+ "ifbench-214",
419
+ "ifbench-215",
420
+ "ifbench-216",
421
+ "ifbench-217",
422
+ "ifbench-218",
423
+ "ifbench-219",
424
+ "ifbench-220",
425
+ "ifbench-221",
426
+ "ifbench-222",
427
+ "ifbench-223",
428
+ "ifbench-224",
429
+ "ifbench-225",
430
+ "ifbench-226",
431
+ "ifbench-227",
432
+ "ifbench-228",
433
+ "ifbench-229",
434
+ "ifbench-230",
435
+ "ifbench-231",
436
+ "ifbench-232",
437
+ "ifbench-233",
438
+ "ifbench-234",
439
+ "ifbench-235",
440
+ "ifbench-236",
441
+ "ifbench-237",
442
+ "ifbench-238",
443
+ "ifbench-239",
444
+ "ifbench-240",
445
+ "ifbench-241",
446
+ "ifbench-242",
447
+ "ifbench-243",
448
+ "ifbench-244",
449
+ "ifbench-245",
450
+ "ifbench-246",
451
+ "ifbench-247",
452
+ "ifbench-248",
453
+ "ifbench-249",
454
+ "ifbench-250",
455
+ "ifbench-251",
456
+ "ifbench-252",
457
+ "ifbench-253",
458
+ "ifbench-254",
459
+ "ifbench-255",
460
+ "ifbench-256",
461
+ "ifbench-257",
462
+ "ifbench-258",
463
+ "ifbench-259",
464
+ "ifbench-260",
465
+ "ifbench-261",
466
+ "ifbench-262",
467
+ "ifbench-263",
468
+ "ifbench-264",
469
+ "ifbench-265",
470
+ "ifbench-266",
471
+ "ifbench-267",
472
+ "ifbench-268",
473
+ "ifbench-269",
474
+ "ifbench-270",
475
+ "ifbench-271",
476
+ "ifbench-272",
477
+ "ifbench-273",
478
+ "ifbench-274",
479
+ "ifbench-275",
480
+ "ifbench-276",
481
+ "ifbench-277",
482
+ "ifbench-278",
483
+ "ifbench-279",
484
+ "ifbench-280",
485
+ "ifbench-281",
486
+ "ifbench-282",
487
+ "ifbench-283",
488
+ "ifbench-284",
489
+ "ifbench-285",
490
+ "ifbench-286",
491
+ "ifbench-287",
492
+ "ifbench-288",
493
+ "ifbench-289",
494
+ "ifbench-290",
495
+ "ifbench-291",
496
+ "ifbench-292",
497
+ "ifbench-293",
498
+ "ifbench-294",
499
+ "ifbench-295",
500
+ "ifbench-296",
501
+ "ifbench-297",
502
+ "ifbench-298",
503
+ "ifbench-299",
504
+ "lcb-atcoder-abc322_b",
505
+ "lcb-atcoder-abc381_a",
506
+ "lcb-atcoder-abc393_a",
507
+ "lcb-atcoder-abc321_b",
508
+ "lcb-atcoder-abc359_a",
509
+ "lcb-atcoder-abc329_b",
510
+ "lcb-atcoder-abc353_a",
511
+ "lcb-atcoder-abc355_b",
512
+ "lcb-atcoder-abc326_b",
513
+ "lcb-codeforces-1873_B",
514
+ "lcb-atcoder-abc356_a",
515
+ "lcb-atcoder-abc356_b",
516
+ "lcb-atcoder-abc375_a",
517
+ "lcb-atcoder-abc377_b",
518
+ "lcb-atcoder-abc311_b",
519
+ "lcb-atcoder-abc378_b",
520
+ "lcb-atcoder-abc309_b",
521
+ "lcb-atcoder-abc325_a",
522
+ "lcb-atcoder-abc301_b",
523
+ "lcb-atcoder-abc391_a",
524
+ "lcb-atcoder-abc399_b",
525
+ "lcb-atcoder-abc354_a",
526
+ "lcb-atcoder-abc352_b",
527
+ "lcb-atcoder-abc382_b",
528
+ "lcb-atcoder-abc332_b",
529
+ "lcb-atcoder-abc343_b",
530
+ "lcb-atcoder-abc361_a",
531
+ "lcb-atcoder-abc362_a",
532
+ "lcb-atcoder-abc328_a",
533
+ "lcb-atcoder-abc393_b",
534
+ "lcb-atcoder-abc352_a",
535
+ "lcb-atcoder-abc310_a",
536
+ "lcb-atcoder-abc365_a",
537
+ "lcb-atcoder-abc371_b",
538
+ "lcb-atcoder-abc367_c",
539
+ "lcb-atcoder-abc334_b",
540
+ "lcb-atcoder-abc385_c",
541
+ "lcb-atcoder-abc307_c",
542
+ "lcb-atcoder-abc338_c",
543
+ "lcb-atcoder-abc303_d",
544
+ "lcb-atcoder-abc342_c",
545
+ "lcb-atcoder-abc319_d",
546
+ "lcb-atcoder-abc315_d",
547
+ "lcb-atcoder-abc309_c",
548
+ "lcb-atcoder-abc390_d",
549
+ "lcb-atcoder-abc343_d",
550
+ "lcb-atcoder-abc397_b",
551
+ "lcb-atcoder-abc370_c",
552
+ "lcb-atcoder-abc375_c",
553
+ "lcb-atcoder-abc368_c",
554
+ "lcb-atcoder-abc325_b",
555
+ "lcb-atcoder-abc323_c",
556
+ "lcb-atcoder-abc377_c",
557
+ "lcb-atcoder-abc383_d",
558
+ "lcb-atcoder-arc189_a",
559
+ "lcb-codeforces-1883_C",
560
+ "lcb-atcoder-abc371_c",
561
+ "lcb-atcoder-abc380_c",
562
+ "lcb-atcoder-abc378_c",
563
+ "lcb-atcoder-abc366_c",
564
+ "lcb-atcoder-abc397_c",
565
+ "lcb-atcoder-abc339_c",
566
+ "lcb-atcoder-abc324_c",
567
+ "lcb-atcoder-abc355_c",
568
+ "lcb-atcoder-abc358_c",
569
+ "lcb-atcoder-abc321_d",
570
+ "lcb-atcoder-abc334_c",
571
+ "lcb-atcoder-arc195_c",
572
+ "lcb-atcoder-arc184_c",
573
+ "lcb-atcoder-abc377_e",
574
+ "lcb-atcoder-arc186_e",
575
+ "lcb-atcoder-abc398_f",
576
+ "lcb-atcoder-abc330_e",
577
+ "lcb-atcoder-abc396_e",
578
+ "lcb-atcoder-arc192_b",
579
+ "lcb-atcoder-abc391_f",
580
+ "lcb-atcoder-arc195_b",
581
+ "lcb-atcoder-abc384_g",
582
+ "lcb-atcoder-abc343_e",
583
+ "lcb-atcoder-abc385_e",
584
+ "lcb-atcoder-abc333_e",
585
+ "lcb-atcoder-abc341_e",
586
+ "lcb-atcoder-abc400_g",
587
+ "lcb-atcoder-abc362_d",
588
+ "lcb-codeforces-1899_D",
589
+ "lcb-atcoder-abc363_f",
590
+ "lcb-atcoder-abc382_d",
591
+ "lcb-atcoder-abc331_e",
592
+ "lcb-atcoder-abc351_e",
593
+ "lcb-atcoder-arc194_b",
594
+ "lcb-atcoder-abc325_f",
595
+ "lcb-atcoder-arc188_d",
596
+ "lcb-atcoder-abc368_e",
597
+ "lcb-atcoder-abc301_e",
598
+ "lcb-atcoder-arc194_e",
599
+ "lcb-atcoder-abc379_e",
600
+ "lcb-atcoder-abc360_e",
601
+ "lcb-atcoder-abc305_e",
602
+ "lcb-atcoder-arc186_a",
603
+ "lcb-atcoder-abc362_e",
604
+ "tool-0",
605
+ "tool-1",
606
+ "tool-2",
607
+ "tool-3",
608
+ "tool-4",
609
+ "tool-5",
610
+ "tool-6",
611
+ "tool-7",
612
+ "tool-8",
613
+ "tool-9",
614
+ "tool-10",
615
+ "tool-11",
616
+ "tool-12",
617
+ "tool-13",
618
+ "tool-14",
619
+ "tool-15",
620
+ "tool-16",
621
+ "tool-17",
622
+ "tool-18",
623
+ "tool-19",
624
+ "json-20",
625
+ "json-21",
626
+ "json-22",
627
+ "json-23",
628
+ "json-24",
629
+ "json-25",
630
+ "json-26",
631
+ "json-27",
632
+ "json-28",
633
+ "json-29"
634
+ ],
635
+ "concurrency": 2,
636
+ "sampling": "thinking: temperature1/top_p0.95/top_k20/xhigh; nonthinking: greedy; seed15027",
637
+ "api": "http://127.0.0.1:18021/v1",
638
+ "server": {
639
+ "model": "<WORKSPACE>/models/Swift-Qwen3.8-27b-W4A16-AWQ",
640
+ "config_sha256": "13bcc57aaa12c046cf379d0680f6a8dd144db48696c294798daae0d485a1e255",
641
+ "index_sha256": "30bde9db796be88667575f52843af2a23538008cadfeb7d58cd1fd2201bd07b6",
642
+ "launcher_sha256": "6874eb0bc4306d61b57ebb2f2c7cab97b41f11a3e0ec2b1502d9c34d396c47d8",
643
+ "tokenizer_sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
644
+ "chat_template_sha256": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
645
+ "shards": {
646
+ "model-00005-of-00007.safetensors": {
647
+ "size": 2179699760,
648
+ "mtime_ns": 1789502506919812372
649
+ },
650
+ "model-00001-of-00007.safetensors": {
651
+ "size": 2212761608,
652
+ "mtime_ns": 1789544289096628097
653
+ },
654
+ "model-00006-of-00007.safetensors": {
655
+ "size": 2186211856,
656
+ "mtime_ns": 1789502504446803803
657
+ },
658
+ "model-00003-of-00007.safetensors": {
659
+ "size": 2179699752,
660
+ "mtime_ns": 1789502419990506742
661
+ },
662
+ "model-00002-of-00007.safetensors": {
663
+ "size": 2186190664,
664
+ "mtime_ns": 1789502424392522448
665
+ },
666
+ "model_extra_tensors.safetensors": {
667
+ "size": 212992352,
668
+ "mtime_ns": 1789544970849697727
669
+ },
670
+ "model-nonquant.safetensors": {
671
+ "size": 431364472,
672
+ "mtime_ns": 1789544293085638815
673
+ },
674
+ "model-00007-of-00007.safetensors": {
675
+ "size": 3071134144,
676
+ "mtime_ns": 1789544275606591269
677
+ },
678
+ "model-00004-of-00007.safetensors": {
679
+ "size": 2179699760,
680
+ "mtime_ns": 1789502463524660967
681
+ }
682
+ },
683
+ "mode": "single-user",
684
+ "int8_activations": false,
685
+ "speculation": "mtp",
686
+ "max_model_len": 150000,
687
+ "max_num_seqs": 8,
688
+ "kv_cache_dtype": "fp8",
689
+ "prefix_cache": true,
690
+ "gpu_memory_utilization": 0.93,
691
+ "serving_profile": "local-single-user",
692
+ "vision": true,
693
+ "draft_tokens": 3,
694
+ "local_env_sha256": "ce50b2e28ea929a620bd62aa7f94aacb844832f9e3dd5767b6952c2ffa0e74a8",
695
+ "extra_args": "--limit-mm-per-prompt {\"image\":{\"count\":10}}"
696
+ }
697
+ }
evaluation/swift10-full/summary.json ADDED
@@ -0,0 +1,88 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created": "2026-09-27T13:16:04Z",
3
+ "concurrency": 2,
4
+ "resumed": false,
5
+ "new_run_wall_seconds": 20949.800895433058,
6
+ "new_tasks": 630,
7
+ "suites": {
8
+ "gsm8k": {
9
+ "attempted": 200,
10
+ "correct": 196,
11
+ "accuracy": 0.98,
12
+ "truncation_policy": "count_as_wrong",
13
+ "truncated_counted_as_wrong": 0,
14
+ "errors": 0,
15
+ "truncated": 0,
16
+ "incomplete_token_counts": 0,
17
+ "mean_output_tokens": 380.925,
18
+ "mean_input_tokens": 93.27,
19
+ "mean_total_tokens": 474.195,
20
+ "mean_model_seconds": 3.4486288186194725,
21
+ "median_model_seconds": 2.997326802520547,
22
+ "p95_model_seconds": 5.85073896299582,
23
+ "summed_request_seconds_per_correct": 3.5190089985912985,
24
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
25
+ },
26
+ "ifbench": {
27
+ "attempted": 300,
28
+ "correct": 220,
29
+ "accuracy": 0.7333333333333333,
30
+ "truncation_policy": "count_as_wrong",
31
+ "truncated_counted_as_wrong": 0,
32
+ "errors": 0,
33
+ "truncated": 0,
34
+ "incomplete_token_counts": 0,
35
+ "mean_output_tokens": 4998.316666666667,
36
+ "mean_input_tokens": 125.44,
37
+ "mean_total_tokens": 5123.756666666667,
38
+ "mean_model_seconds": 55.58578871234068,
39
+ "median_model_seconds": 27.256246656470466,
40
+ "p95_model_seconds": 223.28189022897277,
41
+ "summed_request_seconds_per_correct": 75.79880278955547,
42
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
43
+ },
44
+ "livecodebench": {
45
+ "attempted": 100,
46
+ "correct": 89,
47
+ "accuracy": 0.89,
48
+ "truncation_policy": "count_as_wrong",
49
+ "truncated_counted_as_wrong": 1,
50
+ "errors": 0,
51
+ "truncated": 1,
52
+ "incomplete_token_counts": 0,
53
+ "mean_output_tokens": 17271.87,
54
+ "mean_input_tokens": 653.75,
55
+ "mean_total_tokens": 17925.62,
56
+ "mean_model_seconds": 243.10798836177565,
57
+ "median_model_seconds": 61.351035209954716,
58
+ "p95_model_seconds": 1182.111728347023,
59
+ "summed_request_seconds_per_correct": 273.1550431031187,
60
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
61
+ },
62
+ "tools": {
63
+ "attempted": 30,
64
+ "correct": 28,
65
+ "accuracy": 0.9333333333333333,
66
+ "truncation_policy": "count_as_wrong",
67
+ "truncated_counted_as_wrong": 0,
68
+ "errors": 0,
69
+ "truncated": 0,
70
+ "incomplete_token_counts": 0,
71
+ "mean_output_tokens": 40.333333333333336,
72
+ "mean_input_tokens": 481.3333333333333,
73
+ "mean_total_tokens": 521.6666666666666,
74
+ "mean_model_seconds": 0.994276461203117,
75
+ "median_model_seconds": 1.245341427042149,
76
+ "p95_model_seconds": 1.324655186966993,
77
+ "summed_request_seconds_per_correct": 1.065296208431911,
78
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
79
+ }
80
+ },
81
+ "quality_comparison_ready": true,
82
+ "truncation_policy": "count_as_wrong",
83
+ "truncated_task_ids": [
84
+ "lcb-atcoder-arc184_c"
85
+ ],
86
+ "suite_wall_seconds": 20949.800895433058,
87
+ "wall_seconds_per_correct": 39.30544258054983
88
+ }
evaluation/swift10-pilot/manifest.json ADDED
@@ -0,0 +1,87 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "tasks_sha256": "809c2c6b124579d41c379a6a436e9649b5e1d1ddf44a147a3d70d9b4b19d68f4",
3
+ "task_ids": [
4
+ "gsm8k-27",
5
+ "gsm8k-60",
6
+ "gsm8k-105",
7
+ "gsm8k-109",
8
+ "gsm8k-152",
9
+ "ifbench-21",
10
+ "ifbench-76",
11
+ "ifbench-129",
12
+ "ifbench-130",
13
+ "ifbench-268",
14
+ "lcb-atcoder-abc377_b",
15
+ "lcb-atcoder-abc390_d",
16
+ "lcb-atcoder-abc385_e",
17
+ "lcb-atcoder-abc325_f",
18
+ "lcb-atcoder-abc368_e",
19
+ "tool-15",
20
+ "tool-16",
21
+ "json-22",
22
+ "json-24",
23
+ "json-26"
24
+ ],
25
+ "concurrency": 1,
26
+ "sampling": "thinking: temperature1/top_p0.95/top_k20/xhigh; nonthinking: greedy; seed15027",
27
+ "api": "http://127.0.0.1:18021/v1",
28
+ "server": {
29
+ "model": "<WORKSPACE>/models/Swift-Qwen3.8-27b-W4A16-AWQ",
30
+ "config_sha256": "13bcc57aaa12c046cf379d0680f6a8dd144db48696c294798daae0d485a1e255",
31
+ "index_sha256": "30bde9db796be88667575f52843af2a23538008cadfeb7d58cd1fd2201bd07b6",
32
+ "launcher_sha256": "6874eb0bc4306d61b57ebb2f2c7cab97b41f11a3e0ec2b1502d9c34d396c47d8",
33
+ "tokenizer_sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
34
+ "chat_template_sha256": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
35
+ "shards": {
36
+ "model-00005-of-00007.safetensors": {
37
+ "size": 2179699760,
38
+ "mtime_ns": 1789502506919812372
39
+ },
40
+ "model-00001-of-00007.safetensors": {
41
+ "size": 2212761608,
42
+ "mtime_ns": 1789544289096628097
43
+ },
44
+ "model-00006-of-00007.safetensors": {
45
+ "size": 2186211856,
46
+ "mtime_ns": 1789502504446803803
47
+ },
48
+ "model-00003-of-00007.safetensors": {
49
+ "size": 2179699752,
50
+ "mtime_ns": 1789502419990506742
51
+ },
52
+ "model-00002-of-00007.safetensors": {
53
+ "size": 2186190664,
54
+ "mtime_ns": 1789502424392522448
55
+ },
56
+ "model_extra_tensors.safetensors": {
57
+ "size": 212992352,
58
+ "mtime_ns": 1789544970849697727
59
+ },
60
+ "model-nonquant.safetensors": {
61
+ "size": 431364472,
62
+ "mtime_ns": 1789544293085638815
63
+ },
64
+ "model-00007-of-00007.safetensors": {
65
+ "size": 3071134144,
66
+ "mtime_ns": 1789544275606591269
67
+ },
68
+ "model-00004-of-00007.safetensors": {
69
+ "size": 2179699760,
70
+ "mtime_ns": 1789502463524660967
71
+ }
72
+ },
73
+ "mode": "single-user",
74
+ "int8_activations": false,
75
+ "speculation": "mtp",
76
+ "max_model_len": 150000,
77
+ "max_num_seqs": 8,
78
+ "kv_cache_dtype": "fp8",
79
+ "prefix_cache": true,
80
+ "gpu_memory_utilization": 0.93,
81
+ "serving_profile": "local-single-user",
82
+ "vision": true,
83
+ "draft_tokens": 3,
84
+ "local_env_sha256": "ce50b2e28ea929a620bd62aa7f94aacb844832f9e3dd5767b6952c2ffa0e74a8",
85
+ "extra_args": "--limit-mm-per-prompt {\"image\":{\"count\":10}}"
86
+ }
87
+ }
evaluation/swift10-pilot/summary.json ADDED
@@ -0,0 +1,86 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created": "2026-09-27T07:26:50Z",
3
+ "concurrency": 1,
4
+ "resumed": false,
5
+ "new_run_wall_seconds": 2322.5720736039802,
6
+ "new_tasks": 20,
7
+ "suites": {
8
+ "gsm8k": {
9
+ "attempted": 5,
10
+ "correct": 5,
11
+ "accuracy": 1.0,
12
+ "truncation_policy": "count_as_wrong",
13
+ "truncated_counted_as_wrong": 0,
14
+ "errors": 0,
15
+ "truncated": 0,
16
+ "incomplete_token_counts": 0,
17
+ "mean_output_tokens": 262.6,
18
+ "mean_input_tokens": 77.4,
19
+ "mean_total_tokens": 340,
20
+ "mean_model_seconds": 2.3007265538093633,
21
+ "median_model_seconds": 2.3031491080182604,
22
+ "p95_model_seconds": 3.3395710720214993,
23
+ "summed_request_seconds_per_correct": 2.3007265538093633,
24
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
25
+ },
26
+ "ifbench": {
27
+ "attempted": 5,
28
+ "correct": 3,
29
+ "accuracy": 0.6,
30
+ "truncation_policy": "count_as_wrong",
31
+ "truncated_counted_as_wrong": 0,
32
+ "errors": 0,
33
+ "truncated": 0,
34
+ "incomplete_token_counts": 0,
35
+ "mean_output_tokens": 2349.2,
36
+ "mean_input_tokens": 134.6,
37
+ "mean_total_tokens": 2483.8,
38
+ "mean_model_seconds": 25.789347258815543,
39
+ "median_model_seconds": 11.154919354012236,
40
+ "p95_model_seconds": 89.14424104097998,
41
+ "summed_request_seconds_per_correct": 42.98224543135924,
42
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
43
+ },
44
+ "livecodebench": {
45
+ "attempted": 5,
46
+ "correct": 5,
47
+ "accuracy": 1.0,
48
+ "truncation_policy": "count_as_wrong",
49
+ "truncated_counted_as_wrong": 0,
50
+ "errors": 0,
51
+ "truncated": 0,
52
+ "incomplete_token_counts": 0,
53
+ "mean_output_tokens": 33674,
54
+ "mean_input_tokens": 742.4,
55
+ "mean_total_tokens": 34416.4,
56
+ "mean_model_seconds": 432.0463959699846,
57
+ "median_model_seconds": 277.599615551997,
58
+ "p95_model_seconds": 1155.1297870659619,
59
+ "summed_request_seconds_per_correct": 432.0463959699846,
60
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
61
+ },
62
+ "tools": {
63
+ "attempted": 5,
64
+ "correct": 4,
65
+ "accuracy": 0.8,
66
+ "truncation_policy": "count_as_wrong",
67
+ "truncated_counted_as_wrong": 0,
68
+ "errors": 0,
69
+ "truncated": 0,
70
+ "incomplete_token_counts": 0,
71
+ "mean_output_tokens": 34.6,
72
+ "mean_input_tokens": 330.6,
73
+ "mean_total_tokens": 365.2,
74
+ "mean_model_seconds": 0.6769871199969202,
75
+ "median_model_seconds": 0.5235941839637235,
76
+ "p95_model_seconds": 1.125315964978654,
77
+ "summed_request_seconds_per_correct": 0.8462338999961503,
78
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
79
+ }
80
+ },
81
+ "quality_comparison_ready": true,
82
+ "truncation_policy": "count_as_wrong",
83
+ "truncated_task_ids": [],
84
+ "suite_wall_seconds": 2322.5720736039802,
85
+ "wall_seconds_per_correct": 136.62188668258707
86
+ }
evaluation/swift10/perplexity.json ADDED
@@ -0,0 +1,659 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "input_sha256": "57c83ffe7c0dfba2b869f6a379d2df3f0bc0d055069b49569a2154336151a718",
3
+ "windows": [
4
+ {
5
+ "id": 0,
6
+ "language": "en",
7
+ "tokens": 283,
8
+ "logprob_sum": -593.3995295942841
9
+ },
10
+ {
11
+ "id": 1,
12
+ "language": "en",
13
+ "tokens": 313,
14
+ "logprob_sum": -700.5331317378659
15
+ },
16
+ {
17
+ "id": 2,
18
+ "language": "en",
19
+ "tokens": 284,
20
+ "logprob_sum": -724.3817458400572
21
+ },
22
+ {
23
+ "id": 3,
24
+ "language": "en",
25
+ "tokens": 286,
26
+ "logprob_sum": -774.8060930467109
27
+ },
28
+ {
29
+ "id": 4,
30
+ "language": "en",
31
+ "tokens": 279,
32
+ "logprob_sum": -573.8085490163357
33
+ },
34
+ {
35
+ "id": 5,
36
+ "language": "en",
37
+ "tokens": 251,
38
+ "logprob_sum": -647.1657810044508
39
+ },
40
+ {
41
+ "id": 6,
42
+ "language": "en",
43
+ "tokens": 258,
44
+ "logprob_sum": -613.3406054319057
45
+ },
46
+ {
47
+ "id": 7,
48
+ "language": "en",
49
+ "tokens": 258,
50
+ "logprob_sum": -598.2336515752831
51
+ },
52
+ {
53
+ "id": 8,
54
+ "language": "en",
55
+ "tokens": 274,
56
+ "logprob_sum": -655.9989099322702
57
+ },
58
+ {
59
+ "id": 9,
60
+ "language": "en",
61
+ "tokens": 278,
62
+ "logprob_sum": -620.1450639520117
63
+ },
64
+ {
65
+ "id": 10,
66
+ "language": "en",
67
+ "tokens": 274,
68
+ "logprob_sum": -750.830778833566
69
+ },
70
+ {
71
+ "id": 11,
72
+ "language": "en",
73
+ "tokens": 288,
74
+ "logprob_sum": -780.6144631060088
75
+ },
76
+ {
77
+ "id": 12,
78
+ "language": "en",
79
+ "tokens": 277,
80
+ "logprob_sum": -614.716890099935
81
+ },
82
+ {
83
+ "id": 13,
84
+ "language": "en",
85
+ "tokens": 291,
86
+ "logprob_sum": -711.6267314220386
87
+ },
88
+ {
89
+ "id": 14,
90
+ "language": "en",
91
+ "tokens": 260,
92
+ "logprob_sum": -671.5577233368549
93
+ },
94
+ {
95
+ "id": 15,
96
+ "language": "en",
97
+ "tokens": 250,
98
+ "logprob_sum": -775.6550934422994
99
+ },
100
+ {
101
+ "id": 16,
102
+ "language": "en",
103
+ "tokens": 252,
104
+ "logprob_sum": -756.1478926279815
105
+ },
106
+ {
107
+ "id": 17,
108
+ "language": "en",
109
+ "tokens": 272,
110
+ "logprob_sum": -783.3444251654037
111
+ },
112
+ {
113
+ "id": 18,
114
+ "language": "en",
115
+ "tokens": 240,
116
+ "logprob_sum": -676.2792851193226
117
+ },
118
+ {
119
+ "id": 19,
120
+ "language": "en",
121
+ "tokens": 244,
122
+ "logprob_sum": -561.7464151983149
123
+ },
124
+ {
125
+ "id": 20,
126
+ "language": "en",
127
+ "tokens": 274,
128
+ "logprob_sum": -689.3211253857444
129
+ },
130
+ {
131
+ "id": 21,
132
+ "language": "en",
133
+ "tokens": 299,
134
+ "logprob_sum": -719.0637272568165
135
+ },
136
+ {
137
+ "id": 22,
138
+ "language": "en",
139
+ "tokens": 318,
140
+ "logprob_sum": -828.7790999828721
141
+ },
142
+ {
143
+ "id": 23,
144
+ "language": "en",
145
+ "tokens": 263,
146
+ "logprob_sum": -680.0565696356061
147
+ },
148
+ {
149
+ "id": 24,
150
+ "language": "en",
151
+ "tokens": 239,
152
+ "logprob_sum": -715.9142811552592
153
+ },
154
+ {
155
+ "id": 25,
156
+ "language": "en",
157
+ "tokens": 293,
158
+ "logprob_sum": -489.55537099707726
159
+ },
160
+ {
161
+ "id": 26,
162
+ "language": "en",
163
+ "tokens": 303,
164
+ "logprob_sum": -682.4112273987994
165
+ },
166
+ {
167
+ "id": 27,
168
+ "language": "en",
169
+ "tokens": 276,
170
+ "logprob_sum": -620.8538292451813
171
+ },
172
+ {
173
+ "id": 28,
174
+ "language": "en",
175
+ "tokens": 282,
176
+ "logprob_sum": -738.9185054079862
177
+ },
178
+ {
179
+ "id": 29,
180
+ "language": "en",
181
+ "tokens": 298,
182
+ "logprob_sum": -686.1773185838956
183
+ },
184
+ {
185
+ "id": 30,
186
+ "language": "en",
187
+ "tokens": 338,
188
+ "logprob_sum": -473.0945813098033
189
+ },
190
+ {
191
+ "id": 31,
192
+ "language": "en",
193
+ "tokens": 280,
194
+ "logprob_sum": -630.0955357118728
195
+ },
196
+ {
197
+ "id": 32,
198
+ "language": "en",
199
+ "tokens": 310,
200
+ "logprob_sum": -810.4693004913643
201
+ },
202
+ {
203
+ "id": 33,
204
+ "language": "en",
205
+ "tokens": 280,
206
+ "logprob_sum": -799.6141130025499
207
+ },
208
+ {
209
+ "id": 34,
210
+ "language": "en",
211
+ "tokens": 302,
212
+ "logprob_sum": -712.025332148276
213
+ },
214
+ {
215
+ "id": 35,
216
+ "language": "en",
217
+ "tokens": 286,
218
+ "logprob_sum": -538.7282043778414
219
+ },
220
+ {
221
+ "id": 36,
222
+ "language": "en",
223
+ "tokens": 244,
224
+ "logprob_sum": -593.2613173677673
225
+ },
226
+ {
227
+ "id": 37,
228
+ "language": "en",
229
+ "tokens": 280,
230
+ "logprob_sum": -498.5803779190901
231
+ },
232
+ {
233
+ "id": 38,
234
+ "language": "en",
235
+ "tokens": 315,
236
+ "logprob_sum": -604.9404760405196
237
+ },
238
+ {
239
+ "id": 39,
240
+ "language": "en",
241
+ "tokens": 283,
242
+ "logprob_sum": -652.8633822322918
243
+ },
244
+ {
245
+ "id": 40,
246
+ "language": "da",
247
+ "tokens": 319,
248
+ "logprob_sum": -517.1994196986234
249
+ },
250
+ {
251
+ "id": 41,
252
+ "language": "da",
253
+ "tokens": 334,
254
+ "logprob_sum": -793.5113881794969
255
+ },
256
+ {
257
+ "id": 42,
258
+ "language": "da",
259
+ "tokens": 321,
260
+ "logprob_sum": -649.114615170889
261
+ },
262
+ {
263
+ "id": 43,
264
+ "language": "da",
265
+ "tokens": 355,
266
+ "logprob_sum": -777.4246189849518
267
+ },
268
+ {
269
+ "id": 44,
270
+ "language": "da",
271
+ "tokens": 372,
272
+ "logprob_sum": -790.426270207965
273
+ },
274
+ {
275
+ "id": 45,
276
+ "language": "da",
277
+ "tokens": 360,
278
+ "logprob_sum": -605.4395699449469
279
+ },
280
+ {
281
+ "id": 46,
282
+ "language": "da",
283
+ "tokens": 348,
284
+ "logprob_sum": -1271.5344694513333
285
+ },
286
+ {
287
+ "id": 47,
288
+ "language": "da",
289
+ "tokens": 372,
290
+ "logprob_sum": -609.4069087250737
291
+ },
292
+ {
293
+ "id": 48,
294
+ "language": "da",
295
+ "tokens": 356,
296
+ "logprob_sum": -803.8936290113306
297
+ },
298
+ {
299
+ "id": 49,
300
+ "language": "da",
301
+ "tokens": 269,
302
+ "logprob_sum": -1456.3805292455363
303
+ },
304
+ {
305
+ "id": 50,
306
+ "language": "da",
307
+ "tokens": 334,
308
+ "logprob_sum": -580.4934725780004
309
+ },
310
+ {
311
+ "id": 51,
312
+ "language": "da",
313
+ "tokens": 364,
314
+ "logprob_sum": -613.1963399831075
315
+ },
316
+ {
317
+ "id": 52,
318
+ "language": "da",
319
+ "tokens": 364,
320
+ "logprob_sum": -667.4508653348885
321
+ },
322
+ {
323
+ "id": 53,
324
+ "language": "da",
325
+ "tokens": 378,
326
+ "logprob_sum": -454.4978262218378
327
+ },
328
+ {
329
+ "id": 54,
330
+ "language": "da",
331
+ "tokens": 400,
332
+ "logprob_sum": -832.7972447402817
333
+ },
334
+ {
335
+ "id": 55,
336
+ "language": "da",
337
+ "tokens": 319,
338
+ "logprob_sum": -1565.9580685560359
339
+ },
340
+ {
341
+ "id": 56,
342
+ "language": "da",
343
+ "tokens": 329,
344
+ "logprob_sum": -603.6377663023723
345
+ },
346
+ {
347
+ "id": 57,
348
+ "language": "da",
349
+ "tokens": 338,
350
+ "logprob_sum": -870.4163886572023
351
+ },
352
+ {
353
+ "id": 58,
354
+ "language": "da",
355
+ "tokens": 336,
356
+ "logprob_sum": -1593.2642377602024
357
+ },
358
+ {
359
+ "id": 59,
360
+ "language": "da",
361
+ "tokens": 348,
362
+ "logprob_sum": -991.8582370316872
363
+ },
364
+ {
365
+ "id": 60,
366
+ "language": "da",
367
+ "tokens": 338,
368
+ "logprob_sum": -496.14374373139617
369
+ },
370
+ {
371
+ "id": 61,
372
+ "language": "da",
373
+ "tokens": 349,
374
+ "logprob_sum": -864.6478007899059
375
+ },
376
+ {
377
+ "id": 62,
378
+ "language": "da",
379
+ "tokens": 349,
380
+ "logprob_sum": -564.6994042088133
381
+ },
382
+ {
383
+ "id": 63,
384
+ "language": "da",
385
+ "tokens": 356,
386
+ "logprob_sum": -674.2995392707544
387
+ },
388
+ {
389
+ "id": 64,
390
+ "language": "da",
391
+ "tokens": 388,
392
+ "logprob_sum": -746.9267507952973
393
+ },
394
+ {
395
+ "id": 65,
396
+ "language": "da",
397
+ "tokens": 342,
398
+ "logprob_sum": -1581.7085867874302
399
+ },
400
+ {
401
+ "id": 66,
402
+ "language": "da",
403
+ "tokens": 376,
404
+ "logprob_sum": -684.6263309357491
405
+ },
406
+ {
407
+ "id": 67,
408
+ "language": "da",
409
+ "tokens": 388,
410
+ "logprob_sum": -585.0814573411387
411
+ },
412
+ {
413
+ "id": 68,
414
+ "language": "da",
415
+ "tokens": 338,
416
+ "logprob_sum": -679.226675238438
417
+ },
418
+ {
419
+ "id": 69,
420
+ "language": "da",
421
+ "tokens": 335,
422
+ "logprob_sum": -651.8282408997134
423
+ },
424
+ {
425
+ "id": 70,
426
+ "language": "da",
427
+ "tokens": 338,
428
+ "logprob_sum": -744.8929153228212
429
+ },
430
+ {
431
+ "id": 71,
432
+ "language": "da",
433
+ "tokens": 371,
434
+ "logprob_sum": -1395.236346740101
435
+ },
436
+ {
437
+ "id": 72,
438
+ "language": "da",
439
+ "tokens": 371,
440
+ "logprob_sum": -710.4150057600818
441
+ },
442
+ {
443
+ "id": 73,
444
+ "language": "da",
445
+ "tokens": 325,
446
+ "logprob_sum": -587.0278204807751
447
+ },
448
+ {
449
+ "id": 74,
450
+ "language": "da",
451
+ "tokens": 381,
452
+ "logprob_sum": -1015.3197047918511
453
+ },
454
+ {
455
+ "id": 75,
456
+ "language": "da",
457
+ "tokens": 265,
458
+ "logprob_sum": -1385.7349133007228
459
+ },
460
+ {
461
+ "id": 76,
462
+ "language": "da",
463
+ "tokens": 348,
464
+ "logprob_sum": -905.108547452197
465
+ },
466
+ {
467
+ "id": 77,
468
+ "language": "da",
469
+ "tokens": 338,
470
+ "logprob_sum": -602.2395354474211
471
+ },
472
+ {
473
+ "id": 78,
474
+ "language": "da",
475
+ "tokens": 346,
476
+ "logprob_sum": -875.796343427799
477
+ },
478
+ {
479
+ "id": 79,
480
+ "language": "da",
481
+ "tokens": 359,
482
+ "logprob_sum": -594.3750530699126
483
+ },
484
+ {
485
+ "id": 80,
486
+ "language": "code",
487
+ "tokens": 320,
488
+ "logprob_sum": -274.7489258961019
489
+ },
490
+ {
491
+ "id": 81,
492
+ "language": "code",
493
+ "tokens": 282,
494
+ "logprob_sum": -430.6838706552262
495
+ },
496
+ {
497
+ "id": 82,
498
+ "language": "code",
499
+ "tokens": 278,
500
+ "logprob_sum": -260.0947107755287
501
+ },
502
+ {
503
+ "id": 83,
504
+ "language": "code",
505
+ "tokens": 273,
506
+ "logprob_sum": -411.2611253242603
507
+ },
508
+ {
509
+ "id": 84,
510
+ "language": "code",
511
+ "tokens": 253,
512
+ "logprob_sum": -197.21853261078218
513
+ },
514
+ {
515
+ "id": 85,
516
+ "language": "code",
517
+ "tokens": 239,
518
+ "logprob_sum": -461.1512319083995
519
+ },
520
+ {
521
+ "id": 86,
522
+ "language": "code",
523
+ "tokens": 272,
524
+ "logprob_sum": -377.2601323130457
525
+ },
526
+ {
527
+ "id": 87,
528
+ "language": "code",
529
+ "tokens": 253,
530
+ "logprob_sum": -282.7043757491738
531
+ },
532
+ {
533
+ "id": 88,
534
+ "language": "code",
535
+ "tokens": 317,
536
+ "logprob_sum": -288.68833832504083
537
+ },
538
+ {
539
+ "id": 89,
540
+ "language": "code",
541
+ "tokens": 298,
542
+ "logprob_sum": -290.93484990602883
543
+ },
544
+ {
545
+ "id": 90,
546
+ "language": "code",
547
+ "tokens": 280,
548
+ "logprob_sum": -305.8559028699442
549
+ },
550
+ {
551
+ "id": 91,
552
+ "language": "code",
553
+ "tokens": 265,
554
+ "logprob_sum": -387.24696855193054
555
+ },
556
+ {
557
+ "id": 92,
558
+ "language": "code",
559
+ "tokens": 305,
560
+ "logprob_sum": -197.82955161913992
561
+ },
562
+ {
563
+ "id": 93,
564
+ "language": "code",
565
+ "tokens": 310,
566
+ "logprob_sum": -351.8639510574991
567
+ },
568
+ {
569
+ "id": 94,
570
+ "language": "code",
571
+ "tokens": 278,
572
+ "logprob_sum": -236.71723654100515
573
+ },
574
+ {
575
+ "id": 95,
576
+ "language": "code",
577
+ "tokens": 301,
578
+ "logprob_sum": -317.9189716539063
579
+ },
580
+ {
581
+ "id": 96,
582
+ "language": "code",
583
+ "tokens": 311,
584
+ "logprob_sum": -186.151134865902
585
+ },
586
+ {
587
+ "id": 97,
588
+ "language": "code",
589
+ "tokens": 312,
590
+ "logprob_sum": -227.76212669502684
591
+ },
592
+ {
593
+ "id": 98,
594
+ "language": "code",
595
+ "tokens": 325,
596
+ "logprob_sum": -195.6774061962218
597
+ },
598
+ {
599
+ "id": 99,
600
+ "language": "code",
601
+ "tokens": 291,
602
+ "logprob_sum": -313.3808719475601
603
+ },
604
+ {
605
+ "id": 100,
606
+ "language": "code",
607
+ "tokens": 321,
608
+ "logprob_sum": -436.92633147341314
609
+ },
610
+ {
611
+ "id": 101,
612
+ "language": "code",
613
+ "tokens": 318,
614
+ "logprob_sum": -442.086307042553
615
+ },
616
+ {
617
+ "id": 102,
618
+ "language": "code",
619
+ "tokens": 316,
620
+ "logprob_sum": -522.1931145482638
621
+ },
622
+ {
623
+ "id": 103,
624
+ "language": "code",
625
+ "tokens": 307,
626
+ "logprob_sum": -305.42324257145077
627
+ },
628
+ {
629
+ "id": 104,
630
+ "language": "code",
631
+ "tokens": 254,
632
+ "logprob_sum": -407.6405011889765
633
+ },
634
+ {
635
+ "id": 105,
636
+ "language": "code",
637
+ "tokens": 275,
638
+ "logprob_sum": -499.1410040079517
639
+ }
640
+ ],
641
+ "scores": {
642
+ "en": {
643
+ "tokens": 11175,
644
+ "ppl": 10.953418316428
645
+ },
646
+ "all": {
647
+ "tokens": 32646,
648
+ "ppl": 8.214905986747409
649
+ },
650
+ "da": {
651
+ "tokens": 13917,
652
+ "ppl": 11.017187284986234
653
+ },
654
+ "code": {
655
+ "tokens": 7554,
656
+ "ppl": 3.1255271414775376
657
+ }
658
+ }
659
+ }
evaluation/swift10/server.json ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created": "2026-09-27T06:45:30Z",
3
+ "model": "<WORKSPACE>/models/Swift-Qwen3.8-27b-W4A16-AWQ",
4
+ "config_sha256": "13bcc57aaa12c046cf379d0680f6a8dd144db48696c294798daae0d485a1e255",
5
+ "index_sha256": "30bde9db796be88667575f52843af2a23538008cadfeb7d58cd1fd2201bd07b6",
6
+ "launcher_sha256": "6874eb0bc4306d61b57ebb2f2c7cab97b41f11a3e0ec2b1502d9c34d396c47d8",
7
+ "tokenizer_sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
8
+ "chat_template_sha256": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
9
+ "shards": {
10
+ "model-00005-of-00007.safetensors": {
11
+ "size": 2179699760,
12
+ "mtime_ns": 1789502506919812372
13
+ },
14
+ "model-00001-of-00007.safetensors": {
15
+ "size": 2212761608,
16
+ "mtime_ns": 1789544289096628097
17
+ },
18
+ "model-00006-of-00007.safetensors": {
19
+ "size": 2186211856,
20
+ "mtime_ns": 1789502504446803803
21
+ },
22
+ "model-00003-of-00007.safetensors": {
23
+ "size": 2179699752,
24
+ "mtime_ns": 1789502419990506742
25
+ },
26
+ "model-00002-of-00007.safetensors": {
27
+ "size": 2186190664,
28
+ "mtime_ns": 1789502424392522448
29
+ },
30
+ "model_extra_tensors.safetensors": {
31
+ "size": 212992352,
32
+ "mtime_ns": 1789544970849697727
33
+ },
34
+ "model-nonquant.safetensors": {
35
+ "size": 431364472,
36
+ "mtime_ns": 1789544293085638815
37
+ },
38
+ "model-00007-of-00007.safetensors": {
39
+ "size": 3071134144,
40
+ "mtime_ns": 1789544275606591269
41
+ },
42
+ "model-00004-of-00007.safetensors": {
43
+ "size": 2179699760,
44
+ "mtime_ns": 1789502463524660967
45
+ }
46
+ },
47
+ "mode": "single-user",
48
+ "int8_activations": false,
49
+ "speculation": "mtp",
50
+ "max_model_len": 150000,
51
+ "max_num_seqs": 8,
52
+ "kv_cache_dtype": "fp8",
53
+ "prefix_cache": true,
54
+ "gpu_memory_utilization": 0.93,
55
+ "pid": 1665267,
56
+ "serving_profile": "local-single-user",
57
+ "vision": true,
58
+ "draft_tokens": 3,
59
+ "local_env_sha256": "ce50b2e28ea929a620bd62aa7f94aacb844832f9e3dd5767b6952c2ffa0e74a8",
60
+ "extra_args": "--limit-mm-per-prompt {\"image\":{\"count\":10}}"
61
+ }
evaluation/swift10/speed-c1.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "concurrency": 1,
3
+ "requests": 8,
4
+ "output_tokens_per_request": 512,
5
+ "wall_seconds": 39.691121668962296,
6
+ "aggregate_output_tps": 103.19688201714375,
7
+ "median_decode_tps": 105.94877596587726,
8
+ "median_ttft_seconds": 0.07828654450713657,
9
+ "sampling": "greedy; ignore_eos for this throughput test only",
10
+ "speculation_counter_deltas": {
11
+ "vllm:spec_decode_num_drafts_total": 1320.0,
12
+ "vllm:spec_decode_num_draft_tokens_total": 3960.0,
13
+ "vllm:spec_decode_num_accepted_tokens_total": 2776.0
14
+ },
15
+ "decode_tps_note": "Client stream timing estimate; speculative decoding may deliver several tokens in one event."
16
+ }
evaluation/swift15-baseline-full/manifest.json ADDED
@@ -0,0 +1,693 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "tasks_sha256": "809c2c6b124579d41c379a6a436e9649b5e1d1ddf44a147a3d70d9b4b19d68f4",
3
+ "task_ids": [
4
+ "gsm8k-0",
5
+ "gsm8k-1",
6
+ "gsm8k-2",
7
+ "gsm8k-3",
8
+ "gsm8k-4",
9
+ "gsm8k-5",
10
+ "gsm8k-6",
11
+ "gsm8k-7",
12
+ "gsm8k-8",
13
+ "gsm8k-9",
14
+ "gsm8k-10",
15
+ "gsm8k-11",
16
+ "gsm8k-12",
17
+ "gsm8k-13",
18
+ "gsm8k-14",
19
+ "gsm8k-15",
20
+ "gsm8k-16",
21
+ "gsm8k-17",
22
+ "gsm8k-18",
23
+ "gsm8k-19",
24
+ "gsm8k-20",
25
+ "gsm8k-21",
26
+ "gsm8k-22",
27
+ "gsm8k-23",
28
+ "gsm8k-24",
29
+ "gsm8k-25",
30
+ "gsm8k-26",
31
+ "gsm8k-27",
32
+ "gsm8k-28",
33
+ "gsm8k-29",
34
+ "gsm8k-30",
35
+ "gsm8k-31",
36
+ "gsm8k-32",
37
+ "gsm8k-33",
38
+ "gsm8k-34",
39
+ "gsm8k-35",
40
+ "gsm8k-36",
41
+ "gsm8k-37",
42
+ "gsm8k-38",
43
+ "gsm8k-39",
44
+ "gsm8k-40",
45
+ "gsm8k-41",
46
+ "gsm8k-42",
47
+ "gsm8k-43",
48
+ "gsm8k-44",
49
+ "gsm8k-45",
50
+ "gsm8k-46",
51
+ "gsm8k-47",
52
+ "gsm8k-48",
53
+ "gsm8k-49",
54
+ "gsm8k-50",
55
+ "gsm8k-51",
56
+ "gsm8k-52",
57
+ "gsm8k-53",
58
+ "gsm8k-54",
59
+ "gsm8k-55",
60
+ "gsm8k-56",
61
+ "gsm8k-57",
62
+ "gsm8k-58",
63
+ "gsm8k-59",
64
+ "gsm8k-60",
65
+ "gsm8k-61",
66
+ "gsm8k-62",
67
+ "gsm8k-63",
68
+ "gsm8k-64",
69
+ "gsm8k-65",
70
+ "gsm8k-66",
71
+ "gsm8k-67",
72
+ "gsm8k-68",
73
+ "gsm8k-69",
74
+ "gsm8k-70",
75
+ "gsm8k-71",
76
+ "gsm8k-72",
77
+ "gsm8k-73",
78
+ "gsm8k-74",
79
+ "gsm8k-75",
80
+ "gsm8k-76",
81
+ "gsm8k-77",
82
+ "gsm8k-78",
83
+ "gsm8k-79",
84
+ "gsm8k-80",
85
+ "gsm8k-81",
86
+ "gsm8k-82",
87
+ "gsm8k-83",
88
+ "gsm8k-84",
89
+ "gsm8k-85",
90
+ "gsm8k-86",
91
+ "gsm8k-87",
92
+ "gsm8k-88",
93
+ "gsm8k-89",
94
+ "gsm8k-90",
95
+ "gsm8k-91",
96
+ "gsm8k-92",
97
+ "gsm8k-93",
98
+ "gsm8k-94",
99
+ "gsm8k-95",
100
+ "gsm8k-96",
101
+ "gsm8k-97",
102
+ "gsm8k-98",
103
+ "gsm8k-99",
104
+ "gsm8k-100",
105
+ "gsm8k-101",
106
+ "gsm8k-102",
107
+ "gsm8k-103",
108
+ "gsm8k-104",
109
+ "gsm8k-105",
110
+ "gsm8k-106",
111
+ "gsm8k-107",
112
+ "gsm8k-108",
113
+ "gsm8k-109",
114
+ "gsm8k-110",
115
+ "gsm8k-111",
116
+ "gsm8k-112",
117
+ "gsm8k-113",
118
+ "gsm8k-114",
119
+ "gsm8k-115",
120
+ "gsm8k-116",
121
+ "gsm8k-117",
122
+ "gsm8k-118",
123
+ "gsm8k-119",
124
+ "gsm8k-120",
125
+ "gsm8k-121",
126
+ "gsm8k-122",
127
+ "gsm8k-123",
128
+ "gsm8k-124",
129
+ "gsm8k-125",
130
+ "gsm8k-126",
131
+ "gsm8k-127",
132
+ "gsm8k-128",
133
+ "gsm8k-129",
134
+ "gsm8k-130",
135
+ "gsm8k-131",
136
+ "gsm8k-132",
137
+ "gsm8k-133",
138
+ "gsm8k-134",
139
+ "gsm8k-135",
140
+ "gsm8k-136",
141
+ "gsm8k-137",
142
+ "gsm8k-138",
143
+ "gsm8k-139",
144
+ "gsm8k-140",
145
+ "gsm8k-141",
146
+ "gsm8k-142",
147
+ "gsm8k-143",
148
+ "gsm8k-144",
149
+ "gsm8k-145",
150
+ "gsm8k-146",
151
+ "gsm8k-147",
152
+ "gsm8k-148",
153
+ "gsm8k-149",
154
+ "gsm8k-150",
155
+ "gsm8k-151",
156
+ "gsm8k-152",
157
+ "gsm8k-153",
158
+ "gsm8k-154",
159
+ "gsm8k-155",
160
+ "gsm8k-156",
161
+ "gsm8k-157",
162
+ "gsm8k-158",
163
+ "gsm8k-159",
164
+ "gsm8k-160",
165
+ "gsm8k-161",
166
+ "gsm8k-162",
167
+ "gsm8k-163",
168
+ "gsm8k-164",
169
+ "gsm8k-165",
170
+ "gsm8k-166",
171
+ "gsm8k-167",
172
+ "gsm8k-168",
173
+ "gsm8k-169",
174
+ "gsm8k-170",
175
+ "gsm8k-171",
176
+ "gsm8k-172",
177
+ "gsm8k-173",
178
+ "gsm8k-174",
179
+ "gsm8k-175",
180
+ "gsm8k-176",
181
+ "gsm8k-177",
182
+ "gsm8k-178",
183
+ "gsm8k-179",
184
+ "gsm8k-180",
185
+ "gsm8k-181",
186
+ "gsm8k-182",
187
+ "gsm8k-183",
188
+ "gsm8k-184",
189
+ "gsm8k-185",
190
+ "gsm8k-186",
191
+ "gsm8k-187",
192
+ "gsm8k-188",
193
+ "gsm8k-189",
194
+ "gsm8k-190",
195
+ "gsm8k-191",
196
+ "gsm8k-192",
197
+ "gsm8k-193",
198
+ "gsm8k-194",
199
+ "gsm8k-195",
200
+ "gsm8k-196",
201
+ "gsm8k-197",
202
+ "gsm8k-198",
203
+ "gsm8k-199",
204
+ "ifbench-0",
205
+ "ifbench-1",
206
+ "ifbench-2",
207
+ "ifbench-3",
208
+ "ifbench-4",
209
+ "ifbench-5",
210
+ "ifbench-6",
211
+ "ifbench-7",
212
+ "ifbench-8",
213
+ "ifbench-9",
214
+ "ifbench-10",
215
+ "ifbench-11",
216
+ "ifbench-12",
217
+ "ifbench-13",
218
+ "ifbench-14",
219
+ "ifbench-15",
220
+ "ifbench-16",
221
+ "ifbench-17",
222
+ "ifbench-18",
223
+ "ifbench-19",
224
+ "ifbench-20",
225
+ "ifbench-21",
226
+ "ifbench-22",
227
+ "ifbench-23",
228
+ "ifbench-24",
229
+ "ifbench-25",
230
+ "ifbench-26",
231
+ "ifbench-27",
232
+ "ifbench-28",
233
+ "ifbench-29",
234
+ "ifbench-30",
235
+ "ifbench-31",
236
+ "ifbench-32",
237
+ "ifbench-33",
238
+ "ifbench-34",
239
+ "ifbench-35",
240
+ "ifbench-36",
241
+ "ifbench-37",
242
+ "ifbench-38",
243
+ "ifbench-39",
244
+ "ifbench-40",
245
+ "ifbench-41",
246
+ "ifbench-42",
247
+ "ifbench-43",
248
+ "ifbench-44",
249
+ "ifbench-45",
250
+ "ifbench-46",
251
+ "ifbench-47",
252
+ "ifbench-48",
253
+ "ifbench-49",
254
+ "ifbench-50",
255
+ "ifbench-51",
256
+ "ifbench-52",
257
+ "ifbench-53",
258
+ "ifbench-54",
259
+ "ifbench-55",
260
+ "ifbench-56",
261
+ "ifbench-57",
262
+ "ifbench-58",
263
+ "ifbench-59",
264
+ "ifbench-60",
265
+ "ifbench-61",
266
+ "ifbench-62",
267
+ "ifbench-63",
268
+ "ifbench-64",
269
+ "ifbench-65",
270
+ "ifbench-66",
271
+ "ifbench-67",
272
+ "ifbench-68",
273
+ "ifbench-69",
274
+ "ifbench-70",
275
+ "ifbench-71",
276
+ "ifbench-72",
277
+ "ifbench-73",
278
+ "ifbench-74",
279
+ "ifbench-75",
280
+ "ifbench-76",
281
+ "ifbench-77",
282
+ "ifbench-78",
283
+ "ifbench-79",
284
+ "ifbench-80",
285
+ "ifbench-81",
286
+ "ifbench-82",
287
+ "ifbench-83",
288
+ "ifbench-84",
289
+ "ifbench-85",
290
+ "ifbench-86",
291
+ "ifbench-87",
292
+ "ifbench-88",
293
+ "ifbench-89",
294
+ "ifbench-90",
295
+ "ifbench-91",
296
+ "ifbench-92",
297
+ "ifbench-93",
298
+ "ifbench-94",
299
+ "ifbench-95",
300
+ "ifbench-96",
301
+ "ifbench-97",
302
+ "ifbench-98",
303
+ "ifbench-99",
304
+ "ifbench-100",
305
+ "ifbench-101",
306
+ "ifbench-102",
307
+ "ifbench-103",
308
+ "ifbench-104",
309
+ "ifbench-105",
310
+ "ifbench-106",
311
+ "ifbench-107",
312
+ "ifbench-108",
313
+ "ifbench-109",
314
+ "ifbench-110",
315
+ "ifbench-111",
316
+ "ifbench-112",
317
+ "ifbench-113",
318
+ "ifbench-114",
319
+ "ifbench-115",
320
+ "ifbench-116",
321
+ "ifbench-117",
322
+ "ifbench-118",
323
+ "ifbench-119",
324
+ "ifbench-120",
325
+ "ifbench-121",
326
+ "ifbench-122",
327
+ "ifbench-123",
328
+ "ifbench-124",
329
+ "ifbench-125",
330
+ "ifbench-126",
331
+ "ifbench-127",
332
+ "ifbench-128",
333
+ "ifbench-129",
334
+ "ifbench-130",
335
+ "ifbench-131",
336
+ "ifbench-132",
337
+ "ifbench-133",
338
+ "ifbench-134",
339
+ "ifbench-135",
340
+ "ifbench-136",
341
+ "ifbench-137",
342
+ "ifbench-138",
343
+ "ifbench-139",
344
+ "ifbench-140",
345
+ "ifbench-141",
346
+ "ifbench-142",
347
+ "ifbench-143",
348
+ "ifbench-144",
349
+ "ifbench-145",
350
+ "ifbench-146",
351
+ "ifbench-147",
352
+ "ifbench-148",
353
+ "ifbench-149",
354
+ "ifbench-150",
355
+ "ifbench-151",
356
+ "ifbench-152",
357
+ "ifbench-153",
358
+ "ifbench-154",
359
+ "ifbench-155",
360
+ "ifbench-156",
361
+ "ifbench-157",
362
+ "ifbench-158",
363
+ "ifbench-159",
364
+ "ifbench-160",
365
+ "ifbench-161",
366
+ "ifbench-162",
367
+ "ifbench-163",
368
+ "ifbench-164",
369
+ "ifbench-165",
370
+ "ifbench-166",
371
+ "ifbench-167",
372
+ "ifbench-168",
373
+ "ifbench-169",
374
+ "ifbench-170",
375
+ "ifbench-171",
376
+ "ifbench-172",
377
+ "ifbench-173",
378
+ "ifbench-174",
379
+ "ifbench-175",
380
+ "ifbench-176",
381
+ "ifbench-177",
382
+ "ifbench-178",
383
+ "ifbench-179",
384
+ "ifbench-180",
385
+ "ifbench-181",
386
+ "ifbench-182",
387
+ "ifbench-183",
388
+ "ifbench-184",
389
+ "ifbench-185",
390
+ "ifbench-186",
391
+ "ifbench-187",
392
+ "ifbench-188",
393
+ "ifbench-189",
394
+ "ifbench-190",
395
+ "ifbench-191",
396
+ "ifbench-192",
397
+ "ifbench-193",
398
+ "ifbench-194",
399
+ "ifbench-195",
400
+ "ifbench-196",
401
+ "ifbench-197",
402
+ "ifbench-198",
403
+ "ifbench-199",
404
+ "ifbench-200",
405
+ "ifbench-201",
406
+ "ifbench-202",
407
+ "ifbench-203",
408
+ "ifbench-204",
409
+ "ifbench-205",
410
+ "ifbench-206",
411
+ "ifbench-207",
412
+ "ifbench-208",
413
+ "ifbench-209",
414
+ "ifbench-210",
415
+ "ifbench-211",
416
+ "ifbench-212",
417
+ "ifbench-213",
418
+ "ifbench-214",
419
+ "ifbench-215",
420
+ "ifbench-216",
421
+ "ifbench-217",
422
+ "ifbench-218",
423
+ "ifbench-219",
424
+ "ifbench-220",
425
+ "ifbench-221",
426
+ "ifbench-222",
427
+ "ifbench-223",
428
+ "ifbench-224",
429
+ "ifbench-225",
430
+ "ifbench-226",
431
+ "ifbench-227",
432
+ "ifbench-228",
433
+ "ifbench-229",
434
+ "ifbench-230",
435
+ "ifbench-231",
436
+ "ifbench-232",
437
+ "ifbench-233",
438
+ "ifbench-234",
439
+ "ifbench-235",
440
+ "ifbench-236",
441
+ "ifbench-237",
442
+ "ifbench-238",
443
+ "ifbench-239",
444
+ "ifbench-240",
445
+ "ifbench-241",
446
+ "ifbench-242",
447
+ "ifbench-243",
448
+ "ifbench-244",
449
+ "ifbench-245",
450
+ "ifbench-246",
451
+ "ifbench-247",
452
+ "ifbench-248",
453
+ "ifbench-249",
454
+ "ifbench-250",
455
+ "ifbench-251",
456
+ "ifbench-252",
457
+ "ifbench-253",
458
+ "ifbench-254",
459
+ "ifbench-255",
460
+ "ifbench-256",
461
+ "ifbench-257",
462
+ "ifbench-258",
463
+ "ifbench-259",
464
+ "ifbench-260",
465
+ "ifbench-261",
466
+ "ifbench-262",
467
+ "ifbench-263",
468
+ "ifbench-264",
469
+ "ifbench-265",
470
+ "ifbench-266",
471
+ "ifbench-267",
472
+ "ifbench-268",
473
+ "ifbench-269",
474
+ "ifbench-270",
475
+ "ifbench-271",
476
+ "ifbench-272",
477
+ "ifbench-273",
478
+ "ifbench-274",
479
+ "ifbench-275",
480
+ "ifbench-276",
481
+ "ifbench-277",
482
+ "ifbench-278",
483
+ "ifbench-279",
484
+ "ifbench-280",
485
+ "ifbench-281",
486
+ "ifbench-282",
487
+ "ifbench-283",
488
+ "ifbench-284",
489
+ "ifbench-285",
490
+ "ifbench-286",
491
+ "ifbench-287",
492
+ "ifbench-288",
493
+ "ifbench-289",
494
+ "ifbench-290",
495
+ "ifbench-291",
496
+ "ifbench-292",
497
+ "ifbench-293",
498
+ "ifbench-294",
499
+ "ifbench-295",
500
+ "ifbench-296",
501
+ "ifbench-297",
502
+ "ifbench-298",
503
+ "ifbench-299",
504
+ "lcb-atcoder-abc322_b",
505
+ "lcb-atcoder-abc381_a",
506
+ "lcb-atcoder-abc393_a",
507
+ "lcb-atcoder-abc321_b",
508
+ "lcb-atcoder-abc359_a",
509
+ "lcb-atcoder-abc329_b",
510
+ "lcb-atcoder-abc353_a",
511
+ "lcb-atcoder-abc355_b",
512
+ "lcb-atcoder-abc326_b",
513
+ "lcb-codeforces-1873_B",
514
+ "lcb-atcoder-abc356_a",
515
+ "lcb-atcoder-abc356_b",
516
+ "lcb-atcoder-abc375_a",
517
+ "lcb-atcoder-abc377_b",
518
+ "lcb-atcoder-abc311_b",
519
+ "lcb-atcoder-abc378_b",
520
+ "lcb-atcoder-abc309_b",
521
+ "lcb-atcoder-abc325_a",
522
+ "lcb-atcoder-abc301_b",
523
+ "lcb-atcoder-abc391_a",
524
+ "lcb-atcoder-abc399_b",
525
+ "lcb-atcoder-abc354_a",
526
+ "lcb-atcoder-abc352_b",
527
+ "lcb-atcoder-abc382_b",
528
+ "lcb-atcoder-abc332_b",
529
+ "lcb-atcoder-abc343_b",
530
+ "lcb-atcoder-abc361_a",
531
+ "lcb-atcoder-abc362_a",
532
+ "lcb-atcoder-abc328_a",
533
+ "lcb-atcoder-abc393_b",
534
+ "lcb-atcoder-abc352_a",
535
+ "lcb-atcoder-abc310_a",
536
+ "lcb-atcoder-abc365_a",
537
+ "lcb-atcoder-abc371_b",
538
+ "lcb-atcoder-abc367_c",
539
+ "lcb-atcoder-abc334_b",
540
+ "lcb-atcoder-abc385_c",
541
+ "lcb-atcoder-abc307_c",
542
+ "lcb-atcoder-abc338_c",
543
+ "lcb-atcoder-abc303_d",
544
+ "lcb-atcoder-abc342_c",
545
+ "lcb-atcoder-abc319_d",
546
+ "lcb-atcoder-abc315_d",
547
+ "lcb-atcoder-abc309_c",
548
+ "lcb-atcoder-abc390_d",
549
+ "lcb-atcoder-abc343_d",
550
+ "lcb-atcoder-abc397_b",
551
+ "lcb-atcoder-abc370_c",
552
+ "lcb-atcoder-abc375_c",
553
+ "lcb-atcoder-abc368_c",
554
+ "lcb-atcoder-abc325_b",
555
+ "lcb-atcoder-abc323_c",
556
+ "lcb-atcoder-abc377_c",
557
+ "lcb-atcoder-abc383_d",
558
+ "lcb-atcoder-arc189_a",
559
+ "lcb-codeforces-1883_C",
560
+ "lcb-atcoder-abc371_c",
561
+ "lcb-atcoder-abc380_c",
562
+ "lcb-atcoder-abc378_c",
563
+ "lcb-atcoder-abc366_c",
564
+ "lcb-atcoder-abc397_c",
565
+ "lcb-atcoder-abc339_c",
566
+ "lcb-atcoder-abc324_c",
567
+ "lcb-atcoder-abc355_c",
568
+ "lcb-atcoder-abc358_c",
569
+ "lcb-atcoder-abc321_d",
570
+ "lcb-atcoder-abc334_c",
571
+ "lcb-atcoder-arc195_c",
572
+ "lcb-atcoder-arc184_c",
573
+ "lcb-atcoder-abc377_e",
574
+ "lcb-atcoder-arc186_e",
575
+ "lcb-atcoder-abc398_f",
576
+ "lcb-atcoder-abc330_e",
577
+ "lcb-atcoder-abc396_e",
578
+ "lcb-atcoder-arc192_b",
579
+ "lcb-atcoder-abc391_f",
580
+ "lcb-atcoder-arc195_b",
581
+ "lcb-atcoder-abc384_g",
582
+ "lcb-atcoder-abc343_e",
583
+ "lcb-atcoder-abc385_e",
584
+ "lcb-atcoder-abc333_e",
585
+ "lcb-atcoder-abc341_e",
586
+ "lcb-atcoder-abc400_g",
587
+ "lcb-atcoder-abc362_d",
588
+ "lcb-codeforces-1899_D",
589
+ "lcb-atcoder-abc363_f",
590
+ "lcb-atcoder-abc382_d",
591
+ "lcb-atcoder-abc331_e",
592
+ "lcb-atcoder-abc351_e",
593
+ "lcb-atcoder-arc194_b",
594
+ "lcb-atcoder-abc325_f",
595
+ "lcb-atcoder-arc188_d",
596
+ "lcb-atcoder-abc368_e",
597
+ "lcb-atcoder-abc301_e",
598
+ "lcb-atcoder-arc194_e",
599
+ "lcb-atcoder-abc379_e",
600
+ "lcb-atcoder-abc360_e",
601
+ "lcb-atcoder-abc305_e",
602
+ "lcb-atcoder-arc186_a",
603
+ "lcb-atcoder-abc362_e",
604
+ "tool-0",
605
+ "tool-1",
606
+ "tool-2",
607
+ "tool-3",
608
+ "tool-4",
609
+ "tool-5",
610
+ "tool-6",
611
+ "tool-7",
612
+ "tool-8",
613
+ "tool-9",
614
+ "tool-10",
615
+ "tool-11",
616
+ "tool-12",
617
+ "tool-13",
618
+ "tool-14",
619
+ "tool-15",
620
+ "tool-16",
621
+ "tool-17",
622
+ "tool-18",
623
+ "tool-19",
624
+ "json-20",
625
+ "json-21",
626
+ "json-22",
627
+ "json-23",
628
+ "json-24",
629
+ "json-25",
630
+ "json-26",
631
+ "json-27",
632
+ "json-28",
633
+ "json-29"
634
+ ],
635
+ "concurrency": 2,
636
+ "sampling": "thinking: temperature1/top_p0.95/top_k20/xhigh; nonthinking: greedy; seed15027",
637
+ "api": "http://127.0.0.1:18021/v1",
638
+ "server": {
639
+ "model": "<WORKSPACE>/models/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen",
640
+ "config_sha256": "7a6b1bbea65bf844cd3d824845dd54d57fe8c92d46249233e1347e90c9f41b3f",
641
+ "index_sha256": "04a57f3004384efa2cb0aa67400454a478d3d0bd94cbad0dd27f199ee2b2e7e4",
642
+ "launcher_sha256": "6874eb0bc4306d61b57ebb2f2c7cab97b41f11a3e0ec2b1502d9c34d396c47d8",
643
+ "tokenizer_sha256": "0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3",
644
+ "chat_template_sha256": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
645
+ "shards": {
646
+ "model-00003-of-00006.safetensors": {
647
+ "size": 3966050624,
648
+ "mtime_ns": 1790420289637223015
649
+ },
650
+ "model-00002-of-00006.safetensors": {
651
+ "size": 2730582048,
652
+ "mtime_ns": 1790420535527155687
653
+ },
654
+ "model-mtp-bf16.safetensors": {
655
+ "size": 431364472,
656
+ "mtime_ns": 1790420539382140680
657
+ },
658
+ "model-00005-of-00006.safetensors": {
659
+ "size": 3993185016,
660
+ "mtime_ns": 1790420447135512937
661
+ },
662
+ "model-00006-of-00006.safetensors": {
663
+ "size": 248265904,
664
+ "mtime_ns": 1790420420446626139
665
+ },
666
+ "model-00001-of-00006.safetensors": {
667
+ "size": 1291264336,
668
+ "mtime_ns": 1790420521965208848
669
+ },
670
+ "model_extra_tensors.safetensors": {
671
+ "size": 212992352,
672
+ "mtime_ns": 1790420543620124231
673
+ },
674
+ "model-00004-of-00006.safetensors": {
675
+ "size": 3966050704,
676
+ "mtime_ns": 1790420327914040554
677
+ }
678
+ },
679
+ "mode": "single-user",
680
+ "int8_activations": false,
681
+ "speculation": "mtp",
682
+ "max_model_len": 150000,
683
+ "max_num_seqs": 8,
684
+ "kv_cache_dtype": "fp8",
685
+ "prefix_cache": true,
686
+ "gpu_memory_utilization": 0.93,
687
+ "serving_profile": "local-single-user",
688
+ "vision": true,
689
+ "draft_tokens": 3,
690
+ "local_env_sha256": "ce50b2e28ea929a620bd62aa7f94aacb844832f9e3dd5767b6952c2ffa0e74a8",
691
+ "extra_args": "--limit-mm-per-prompt {\"image\":{\"count\":10}}"
692
+ }
693
+ }
evaluation/swift15-baseline-full/summary.json ADDED
@@ -0,0 +1,88 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created": "2026-09-27T20:38:58Z",
3
+ "concurrency": 2,
4
+ "resumed": false,
5
+ "new_run_wall_seconds": 22960.825675303,
6
+ "new_tasks": 630,
7
+ "suites": {
8
+ "gsm8k": {
9
+ "attempted": 200,
10
+ "correct": 196,
11
+ "accuracy": 0.98,
12
+ "truncation_policy": "count_as_wrong",
13
+ "truncated_counted_as_wrong": 0,
14
+ "errors": 0,
15
+ "truncated": 0,
16
+ "incomplete_token_counts": 0,
17
+ "mean_output_tokens": 355.47,
18
+ "mean_input_tokens": 93.27,
19
+ "mean_total_tokens": 448.74,
20
+ "mean_model_seconds": 3.2202836997405395,
21
+ "median_model_seconds": 2.9490127855096944,
22
+ "p95_model_seconds": 5.4270519990241155,
23
+ "summed_request_seconds_per_correct": 3.2860037752454483,
24
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
25
+ },
26
+ "ifbench": {
27
+ "attempted": 300,
28
+ "correct": 221,
29
+ "accuracy": 0.7366666666666667,
30
+ "truncation_policy": "count_as_wrong",
31
+ "truncated_counted_as_wrong": 0,
32
+ "errors": 0,
33
+ "truncated": 0,
34
+ "incomplete_token_counts": 0,
35
+ "mean_output_tokens": 5598.606666666667,
36
+ "mean_input_tokens": 125.44,
37
+ "mean_total_tokens": 5724.046666666667,
38
+ "mean_model_seconds": 63.119914838518014,
39
+ "median_model_seconds": 28.452186586917378,
40
+ "p95_model_seconds": 249.57361784903333,
41
+ "summed_request_seconds_per_correct": 85.68314231473033,
42
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
43
+ },
44
+ "livecodebench": {
45
+ "attempted": 100,
46
+ "correct": 89,
47
+ "accuracy": 0.89,
48
+ "truncation_policy": "count_as_wrong",
49
+ "truncated_counted_as_wrong": 1,
50
+ "errors": 0,
51
+ "truncated": 1,
52
+ "incomplete_token_counts": 0,
53
+ "mean_output_tokens": 18712.25,
54
+ "mean_input_tokens": 653.75,
55
+ "mean_total_tokens": 19366,
56
+ "mean_model_seconds": 258.7643161940086,
57
+ "median_model_seconds": 62.81849751947448,
58
+ "p95_model_seconds": 1266.3205917470623,
59
+ "summed_request_seconds_per_correct": 290.74642268989726,
60
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
61
+ },
62
+ "tools": {
63
+ "attempted": 30,
64
+ "correct": 30,
65
+ "accuracy": 1.0,
66
+ "truncation_policy": "count_as_wrong",
67
+ "truncated_counted_as_wrong": 0,
68
+ "errors": 0,
69
+ "truncated": 0,
70
+ "incomplete_token_counts": 0,
71
+ "mean_output_tokens": 40,
72
+ "mean_input_tokens": 481.3333333333333,
73
+ "mean_total_tokens": 521.3333333333334,
74
+ "mean_model_seconds": 1.0695485257388404,
75
+ "median_model_seconds": 1.3348065734608099,
76
+ "p95_model_seconds": 1.4240245381370187,
77
+ "summed_request_seconds_per_correct": 1.0695485257388404,
78
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
79
+ }
80
+ },
81
+ "quality_comparison_ready": true,
82
+ "truncation_policy": "count_as_wrong",
83
+ "truncated_task_ids": [
84
+ "lcb-atcoder-arc184_c"
85
+ ],
86
+ "suite_wall_seconds": 22960.825675303,
87
+ "wall_seconds_per_correct": 42.83736133452052
88
+ }
evaluation/swift15-baseline-pilot/manifest.json ADDED
@@ -0,0 +1,83 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "tasks_sha256": "809c2c6b124579d41c379a6a436e9649b5e1d1ddf44a147a3d70d9b4b19d68f4",
3
+ "task_ids": [
4
+ "gsm8k-27",
5
+ "gsm8k-60",
6
+ "gsm8k-105",
7
+ "gsm8k-109",
8
+ "gsm8k-152",
9
+ "ifbench-21",
10
+ "ifbench-76",
11
+ "ifbench-129",
12
+ "ifbench-130",
13
+ "ifbench-268",
14
+ "lcb-atcoder-abc377_b",
15
+ "lcb-atcoder-abc390_d",
16
+ "lcb-atcoder-abc385_e",
17
+ "lcb-atcoder-abc325_f",
18
+ "lcb-atcoder-abc368_e",
19
+ "tool-15",
20
+ "tool-16",
21
+ "json-22",
22
+ "json-24",
23
+ "json-26"
24
+ ],
25
+ "concurrency": 1,
26
+ "sampling": "thinking: temperature1/top_p0.95/top_k20/xhigh; nonthinking: greedy; seed15027",
27
+ "api": "http://127.0.0.1:18021/v1",
28
+ "server": {
29
+ "model": "<WORKSPACE>/models/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen",
30
+ "config_sha256": "7a6b1bbea65bf844cd3d824845dd54d57fe8c92d46249233e1347e90c9f41b3f",
31
+ "index_sha256": "04a57f3004384efa2cb0aa67400454a478d3d0bd94cbad0dd27f199ee2b2e7e4",
32
+ "launcher_sha256": "6874eb0bc4306d61b57ebb2f2c7cab97b41f11a3e0ec2b1502d9c34d396c47d8",
33
+ "tokenizer_sha256": "0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3",
34
+ "chat_template_sha256": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
35
+ "shards": {
36
+ "model-00003-of-00006.safetensors": {
37
+ "size": 3966050624,
38
+ "mtime_ns": 1790420289637223015
39
+ },
40
+ "model-00002-of-00006.safetensors": {
41
+ "size": 2730582048,
42
+ "mtime_ns": 1790420535527155687
43
+ },
44
+ "model-mtp-bf16.safetensors": {
45
+ "size": 431364472,
46
+ "mtime_ns": 1790420539382140680
47
+ },
48
+ "model-00005-of-00006.safetensors": {
49
+ "size": 3993185016,
50
+ "mtime_ns": 1790420447135512937
51
+ },
52
+ "model-00006-of-00006.safetensors": {
53
+ "size": 248265904,
54
+ "mtime_ns": 1790420420446626139
55
+ },
56
+ "model-00001-of-00006.safetensors": {
57
+ "size": 1291264336,
58
+ "mtime_ns": 1790420521965208848
59
+ },
60
+ "model_extra_tensors.safetensors": {
61
+ "size": 212992352,
62
+ "mtime_ns": 1790420543620124231
63
+ },
64
+ "model-00004-of-00006.safetensors": {
65
+ "size": 3966050704,
66
+ "mtime_ns": 1790420327914040554
67
+ }
68
+ },
69
+ "mode": "single-user",
70
+ "int8_activations": false,
71
+ "speculation": "mtp",
72
+ "max_model_len": 150000,
73
+ "max_num_seqs": 8,
74
+ "kv_cache_dtype": "fp8",
75
+ "prefix_cache": true,
76
+ "gpu_memory_utilization": 0.93,
77
+ "serving_profile": "local-single-user",
78
+ "vision": true,
79
+ "draft_tokens": 3,
80
+ "local_env_sha256": "ce50b2e28ea929a620bd62aa7f94aacb844832f9e3dd5767b6952c2ffa0e74a8",
81
+ "extra_args": "--limit-mm-per-prompt {\"image\":{\"count\":10}}"
82
+ }
83
+ }
evaluation/swift15-baseline-pilot/summary.json ADDED
@@ -0,0 +1,86 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created": "2026-09-27T14:16:13Z",
3
+ "concurrency": 1,
4
+ "resumed": false,
5
+ "new_run_wall_seconds": 3445.778244882007,
6
+ "new_tasks": 20,
7
+ "suites": {
8
+ "gsm8k": {
9
+ "attempted": 5,
10
+ "correct": 5,
11
+ "accuracy": 1.0,
12
+ "truncation_policy": "count_as_wrong",
13
+ "truncated_counted_as_wrong": 0,
14
+ "errors": 0,
15
+ "truncated": 0,
16
+ "incomplete_token_counts": 0,
17
+ "mean_output_tokens": 275,
18
+ "mean_input_tokens": 77.4,
19
+ "mean_total_tokens": 352.4,
20
+ "mean_model_seconds": 2.3928021708037703,
21
+ "median_model_seconds": 2.220176461036317,
22
+ "p95_model_seconds": 3.1514280069386587,
23
+ "summed_request_seconds_per_correct": 2.3928021708037703,
24
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
25
+ },
26
+ "ifbench": {
27
+ "attempted": 5,
28
+ "correct": 3,
29
+ "accuracy": 0.6,
30
+ "truncation_policy": "count_as_wrong",
31
+ "truncated_counted_as_wrong": 0,
32
+ "errors": 0,
33
+ "truncated": 0,
34
+ "incomplete_token_counts": 0,
35
+ "mean_output_tokens": 1952.6,
36
+ "mean_input_tokens": 134.6,
37
+ "mean_total_tokens": 2087.2,
38
+ "mean_model_seconds": 21.503820910397916,
39
+ "median_model_seconds": 10.035008723963983,
40
+ "p95_model_seconds": 68.41221156402025,
41
+ "summed_request_seconds_per_correct": 35.839701517329864,
42
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
43
+ },
44
+ "livecodebench": {
45
+ "attempted": 5,
46
+ "correct": 5,
47
+ "accuracy": 1.0,
48
+ "truncation_policy": "count_as_wrong",
49
+ "truncated_counted_as_wrong": 0,
50
+ "errors": 0,
51
+ "truncated": 0,
52
+ "incomplete_token_counts": 0,
53
+ "mean_output_tokens": 49329,
54
+ "mean_input_tokens": 742.4,
55
+ "mean_total_tokens": 50071.4,
56
+ "mean_model_seconds": 660.0320765013806,
57
+ "median_model_seconds": 480.73357990791555,
58
+ "p95_model_seconds": 1812.1029692189768,
59
+ "summed_request_seconds_per_correct": 660.0320765013806,
60
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
61
+ },
62
+ "tools": {
63
+ "attempted": 5,
64
+ "correct": 5,
65
+ "accuracy": 1.0,
66
+ "truncation_policy": "count_as_wrong",
67
+ "truncated_counted_as_wrong": 0,
68
+ "errors": 0,
69
+ "truncated": 0,
70
+ "incomplete_token_counts": 0,
71
+ "mean_output_tokens": 33.6,
72
+ "mean_input_tokens": 330.6,
73
+ "mean_total_tokens": 364.2,
74
+ "mean_model_seconds": 0.6659889166243375,
75
+ "median_model_seconds": 0.5145661529386416,
76
+ "p95_model_seconds": 1.1192404899047688,
77
+ "summed_request_seconds_per_correct": 0.6659889166243375,
78
+ "note": "Summed request seconds are not GPU compute time when concurrency exceeds one."
79
+ }
80
+ },
81
+ "quality_comparison_ready": true,
82
+ "truncation_policy": "count_as_wrong",
83
+ "truncated_task_ids": [],
84
+ "suite_wall_seconds": 3445.778244882007,
85
+ "wall_seconds_per_correct": 191.43212471566707
86
+ }