ukisai commited on
Commit
9fd3d5f
·
0 Parent(s):

initial release

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +38 -0
  2. LICENSE +233 -0
  3. LICENSE-APACHE-2.0 +202 -0
  4. NOTICE +24 -0
  5. QUANTIZATION_MANIFEST.json +240 -0
  6. README.md +146 -0
  7. UPLOAD_MANIFEST.json +333 -0
  8. USAGE.md +92 -0
  9. chat_template.jinja +170 -0
  10. compatibility/MLX-LM-LICENSE +21 -0
  11. compatibility/SOURCE_EXPORT_MANIFEST.json +1730 -0
  12. compatibility/architecture-compatibility-report.md +40 -0
  13. compatibility/aws-actual-source-structural-results.json +53 -0
  14. compatibility/aws-source-verification.json +0 -0
  15. compatibility/compatibility-tests-linux.log +2 -0
  16. compatibility/conversion-command.json +25 -0
  17. compatibility/conversion-result.json +28 -0
  18. compatibility/conversion.log +4 -0
  19. compatibility/cpu-quantized-matmul-diagnostic.json +19 -0
  20. compatibility/environment-linux.json +20 -0
  21. compatibility/fixed-affine-test.log +2 -0
  22. compatibility/mac-check/checkpoint-headers.json +0 -0
  23. compatibility/mac-check/config.json +154 -0
  24. compatibility/mac-check/model.safetensors.index.json +0 -0
  25. compatibility/mac-check/real-checkpoint-samples.safetensors +3 -0
  26. compatibility/mac-check/samples.json +45 -0
  27. compatibility/mac-compatibility-results.json +72 -0
  28. compatibility/missing-file-recovery-report.md +16 -0
  29. compatibility/quant-tensor-mapping-manifest.json +0 -0
  30. compatibility/quant-validation-results.json +103 -0
  31. compatibility/quant-validation.log +110 -0
  32. compatibility/requirements-linux.txt +44 -0
  33. compatibility/run-compatibility.sh +21 -0
  34. compatibility/run-conversion.py +40 -0
  35. compatibility/setup-linux.sh +32 -0
  36. compatibility/source-file-manifest.csv +31 -0
  37. compatibility/swift15-mlx-lm.patch +962 -0
  38. compatibility/unmatched-tensor-analysis.csv +0 -0
  39. compatibility/validate-mac-format.py +57 -0
  40. compatibility/validate-quant.py +134 -0
  41. compatibility/validate_aws_source_linux.py +104 -0
  42. compatibility/verify_source_readonly.py +92 -0
  43. config.json +154 -0
  44. generation_config.json +12 -0
  45. merges.txt +0 -0
  46. model-00001-of-00003.safetensors +3 -0
  47. model-00002-of-00003.safetensors +3 -0
  48. model-00003-of-00003.safetensors +3 -0
  49. model.safetensors.index.json +0 -0
  50. preprocessor_config.json +21 -0
.gitattributes ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ ukisai-banner.png filter=lfs diff=lfs merge=lfs -text
38
+ swift-1.5-planet-demo.mp4 filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,233 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Swift Open License v1.0
2
+
3
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
4
+
5
+ 1. Definitions.
6
+
7
+ "License" shall mean the terms and conditions for use, reproduction, and
8
+ distribution as defined by this document.
9
+
10
+ "Licensor" shall mean UkisAI.
11
+
12
+ "Legal Entity" shall mean the union of the acting entity and all other entities
13
+ that control, are controlled by, or are under common control with that entity.
14
+ For the purposes of this definition, "control" means (i) the power, direct or
15
+ indirect, to cause the direction or management of such entity, whether by
16
+ contract or otherwise, or (ii) ownership of fifty percent (50%) or more of the
17
+ outstanding shares, or (iii) beneficial ownership of such entity.
18
+
19
+ "You" (or "Your") shall mean an individual or Legal Entity exercising
20
+ permissions granted by this License.
21
+
22
+ "Source" form shall mean the preferred form for making modifications,
23
+ including but not limited to software source code, documentation source,
24
+ configuration files, and model weights in an unquantized, trainable format.
25
+
26
+ "Object" form shall mean any form resulting from mechanical transformation or
27
+ translation of a Source form, including but not limited to compiled object
28
+ code, generated documentation, quantized or otherwise converted model weights,
29
+ and conversions to other media types or file formats.
30
+
31
+ "Base Model" shall mean the Qwen3.8-27B model, Copyright 2026 Alibaba Cloud,
32
+ made available at https://huggingface.co/Qwen/Qwen3.8-27B, including its
33
+ weights, configuration, tokenizer, and chat template, in any form.
34
+
35
+ "Base Model License" shall mean the Apache License, Version 2.0, under which
36
+ the Base Model is made available. A copy is distributed with the Work in the
37
+ file LICENSE-APACHE-2.0.
38
+
39
+ "Swift Contribution" shall mean the modifications to the Base Model authored
40
+ by Licensor, in any form, including without limitation adapted model weights,
41
+ weight deltas, model weights to the extent they differ from the Base Model, and
42
+ any configuration, documentation, and evaluation materials created by Licensor
43
+ and distributed with the Work.
44
+
45
+ "Work" shall mean the Swift Contribution together with, to the extent of
46
+ Licensor's rights therein, the Derivative Work of the Base Model made available
47
+ by Licensor under this License (as indicated by a copyright notice that is
48
+ included in or attached to the work), in any format made available by Licensor.
49
+
50
+ "Derivative Works" shall mean any work, whether in Source or Object form, that
51
+ is based on (or derived from) the Work and for which the editorial revisions,
52
+ annotations, elaborations, or other modifications represent, as a whole, an
53
+ original work of authorship. For the purposes of this License, Derivative Works
54
+ shall not include works that remain separable from, or merely link (or bind by
55
+ name) to the interfaces of, the Work and Derivative Works thereof.
56
+
57
+ "Contribution" shall mean any work of authorship, including the original
58
+ version of the Work and any modifications or additions to that Work or
59
+ Derivative Works thereof, that is intentionally submitted to Licensor for
60
+ inclusion in the Work by the copyright owner or by an individual or Legal
61
+ Entity authorized to submit on behalf of the copyright owner. For the purposes
62
+ of this definition, "submitted" means any form of electronic, verbal, or
63
+ written communication sent to the Licensor or its representatives, including
64
+ but not limited to communication on electronic mailing lists, source code
65
+ control systems, and issue tracking systems that are managed by, or on behalf
66
+ of, the Licensor for the purpose of discussing and improving the Work, but
67
+ excluding communication that is conspicuously marked or otherwise designated in
68
+ writing by the copyright owner as "Not a Contribution."
69
+
70
+ "Contributor" shall mean Licensor and any individual or Legal Entity on behalf
71
+ of whom a Contribution has been received by Licensor and subsequently
72
+ incorporated within the Work.
73
+
74
+ "Commercial Use" shall mean any use of the Work or a Derivative Work for direct
75
+ or indirect commercial advantage or monetary compensation.
76
+
77
+ "Qualified Non-Profit Organization" shall mean a Legal Entity that is organized
78
+ and operated exclusively for religious, charitable, scientific, testing for
79
+ public safety, literary, or educational purposes, and which is exempt from
80
+ federal income tax under Section 501(c)(3) of the United States Internal
81
+ Revenue Code of 1986, as amended, or any equivalent non-profit or charitable
82
+ organization in a foreign jurisdiction.
83
+
84
+ "Non-Commercial or Research Purposes" shall mean purposes that do not involve
85
+ any use of the Work or a Derivative Work for Commercial Use.
86
+
87
+ "Threshold" shall mean gross revenue of one million United States dollars
88
+ (US$1,000,000) or more, measured over the most recently completed fiscal year
89
+ of You together with every Legal Entity that controls, is controlled by, or is
90
+ under common control with You.
91
+
92
+ 2. Grant of Copyright License. Subject to the terms and conditions of this
93
+ License, including the Commercial Use limitation set forth in Section 5, each
94
+ Contributor hereby grants to You a perpetual, worldwide, non-exclusive,
95
+ no-charge, royalty-free, irrevocable copyright license to reproduce, prepare
96
+ Derivative Works of, publicly display, publicly perform, sublicense, and
97
+ distribute the Work and such Derivative Works in Source or Object form.
98
+
99
+ 3. Grant of Patent License. Subject to the terms and conditions of this
100
+ License, including the Commercial Use limitation set forth in Section 5, each
101
+ Contributor hereby grants to You a perpetual, worldwide, non-exclusive,
102
+ no-charge, royalty-free, irrevocable (except as stated in this section) patent
103
+ license to make, have made, use, offer to sell, sell, import, and otherwise
104
+ transfer the Work, where such license applies only to those patent claims
105
+ licensable by such Contributor that are necessarily infringed by their
106
+ Contribution(s) alone or by combination of their Contribution(s) with the Work
107
+ to which such Contribution(s) was submitted. If You institute patent litigation
108
+ against any entity (including a cross-claim or counterclaim in a lawsuit)
109
+ alleging that the Work or a Contribution incorporated within the Work
110
+ constitutes direct or contributory patent infringement, then any patent
111
+ licenses granted to You under this License for that Work shall terminate as of
112
+ the date such litigation is filed.
113
+
114
+ 4. Redistribution. You may reproduce and distribute copies of the Work or
115
+ Derivative Works thereof in any medium, with or without modifications, and in
116
+ Source or Object form, provided that You meet the following conditions:
117
+
118
+ (a) You must give any other recipients of the Work or Derivative Works a copy
119
+ of this License; and
120
+
121
+ (b) You must cause any modified files to carry prominent notices stating that
122
+ You changed the files; and
123
+
124
+ (c) You must retain, in the Source form of any Derivative Works that You
125
+ distribute, all copyright, patent, trademark, and attribution notices from the
126
+ Source form of the Work, excluding those notices that do not pertain to any
127
+ part of the Derivative Works; and
128
+
129
+ (d) If the Work includes a "NOTICE" text file as part of its distribution, then
130
+ any Derivative Works that You distribute must include a readable copy of the
131
+ attribution notices contained within such NOTICE file, excluding those notices
132
+ that do not pertain to any part of the Derivative Works, in at least one of the
133
+ following places: within a NOTICE text file distributed as part of the
134
+ Derivative Works; within the Source form or documentation, if provided along
135
+ with the Derivative Works; or, within a display generated by the Derivative
136
+ Works, if and wherever such third-party notices normally appear. The contents
137
+ of the NOTICE file are for informational purposes only and do not modify the
138
+ License. You may add Your own attribution notices within Derivative Works that
139
+ You distribute, alongside or as an addendum to the NOTICE text from the Work,
140
+ provided that such additional attribution notices cannot be construed as
141
+ modifying the License; and
142
+
143
+ (e) If the copy You distribute contains any portion of the Base Model
144
+ (including merged, quantized, or otherwise converted weights that incorporate
145
+ the Base Model), You must also give recipients a copy of the Base Model License
146
+ and must comply with the Base Model License with respect to the Base Model.
147
+
148
+ You may add Your own copyright statement to Your modifications and may provide
149
+ additional or different license terms and conditions for use, reproduction, or
150
+ distribution of Your modifications, or for any such Derivative Works as a
151
+ whole, provided Your use, reproduction, and distribution of the Work otherwise
152
+ complies with the conditions stated in this License, and provided that Section
153
+ 5 continues to apply to the Swift Contribution contained in any such
154
+ Derivative Works.
155
+
156
+ 5. Commercial Use Limitation.
157
+
158
+ (a) The rights granted under this License for Commercial Use are conditioned
159
+ upon You or Your Legal Entity not exceeding the Threshold.
160
+
161
+ (b) Any Commercial Use of the Work or a Derivative Work by a Legal Entity that
162
+ exceeds the Threshold is not licensed under this License.
163
+
164
+ (c) The Threshold shall not apply to a Qualified Non-Profit Organization's use
165
+ of the Work or a Derivative Work for Non-Commercial or Research Purposes.
166
+
167
+ (d) A Legal Entity that exceeds the Threshold may obtain a separate written
168
+ license for Commercial Use from Licensor (the "Swift Enterprise License").
169
+ Contact: https://ukisai.com/contact.
170
+
171
+ 6. Base Model Rights. The Work incorporates the Base Model. Nothing in this
172
+ License limits, restricts, conditions, or modifies any rights You have in the
173
+ Base Model under the Base Model License, and the Base Model remains available
174
+ to You from its licensor under the Base Model License. Sections 5 and 12 of
175
+ this License apply solely to the Swift Contribution and to the Work or any
176
+ Derivative Work to the extent it contains, incorporates, or is derived from the
177
+ Swift Contribution.
178
+
179
+ 7. Submission of Contributions. Unless You explicitly state otherwise, any
180
+ Contribution intentionally submitted for inclusion in the Work by You to the
181
+ Licensor shall be under the terms and conditions of this License, without any
182
+ additional terms or conditions. Notwithstanding the above, nothing herein shall
183
+ supersede or modify the terms of any separate license agreement you may have
184
+ executed with Licensor regarding such Contributions.
185
+
186
+ 8. Trademarks. This License does not grant permission to use the trade names,
187
+ trademarks, service marks, or product names of the Licensor (including "UkisAI"
188
+ and "Swift"), except for the reasonable and customary use in describing the
189
+ origin of the Work and reproducing the content of the NOTICE file.
190
+
191
+ 9. Disclaimer of Warranty. Unless required by applicable law or agreed to in
192
+ writing, Licensor provides the Work (and each Contributor provides its
193
+ Contributions) on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
194
+ KIND, either express or implied, including, without limitation, any warranties
195
+ or conditions of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
196
+ PARTICULAR PURPOSE. You are solely responsible for determining the
197
+ appropriateness of using or redistributing the Work and assume any risks
198
+ associated with Your exercise of permissions under this License.
199
+
200
+ 10. Limitation of Liability. In no event and under no legal theory, whether in
201
+ tort (including negligence), contract, or otherwise, unless required by
202
+ applicable law (such as deliberate and grossly negligent acts) or agreed to in
203
+ writing, shall any Contributor be liable to You for damages, including any
204
+ direct, indirect, special, incidental, or consequential damages of any
205
+ character arising as a result of this License or out of the use or inability to
206
+ use the Work (including but not limited to damages for loss of goodwill, work
207
+ stoppage, computer failure or malfunction, or any and all other commercial
208
+ damages or losses), even if such Contributor has been advised of the
209
+ possibility of such damages.
210
+
211
+ 11. Accepting Warranty or Additional Liability. While redistributing the Work
212
+ or Derivative Works thereof, You may choose to offer, and charge a fee for,
213
+ acceptance of support, warranty, indemnity, or other liability obligations
214
+ and/or rights consistent with this License. However, in accepting such
215
+ obligations, You may act only on Your own behalf and on Your sole
216
+ responsibility, not on behalf of any other Contributor, and only if You agree
217
+ to indemnify, defend, and hold each Contributor harmless for any liability
218
+ incurred by, or claims asserted against, such Contributor by reason of your
219
+ accepting any such warranty or additional liability.
220
+
221
+ 12. Termination. This License will terminate automatically and immediately if
222
+ You fail to comply with any of its terms and conditions. Upon termination, You
223
+ must cease all use of the Swift Contribution and of any Work or Derivative
224
+ Works containing it, and delete all copies in Your possession. Termination does
225
+ not affect Your rights in the Base Model under the Base Model License.
226
+
227
+ END OF TERMS AND CONDITIONS
228
+
229
+ APPENDIX: Notice for redistributors (quantizations, conversions, merges).
230
+
231
+ Copyright 2026 UkisAI. Swift Contribution licensed under the Swift Open
232
+ License v1.0 (https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b/blob/main/LICENSE).
233
+ Derivative of Qwen3.8-27B, Copyright 2026 Alibaba Cloud, Apache License 2.0.
LICENSE-APACHE-2.0 ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
NOTICE ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Swift 1.5 Qwen3.8-27B
2
+ Copyright 2026 UkisAI
3
+
4
+ UkisAI's contribution (the "Swift Contribution") is licensed under the
5
+ Swift Open License v1.0. See LICENSE.
6
+
7
+ This model is a Derivative Work of Qwen3.8-27B
8
+ https://huggingface.co/Qwen/Qwen3.8-27B
9
+ Copyright 2026 Alibaba Cloud
10
+ Licensed under the Apache License, Version 2.0. See LICENSE-APACHE-2.0.
11
+
12
+ Swift 1.5 continues UkisAI's Swift 1.0 model, itself a Qwen3.8-27B
13
+ derivative. The full release incorporates the Qwen3.8-27B Base Model and
14
+ UkisAI's Swift 1.0 and Swift 1.5 contributions.
15
+
16
+ Changes made by UkisAI (Apache License 2.0, Section 4(b) change notice):
17
+ - model-*.safetensors, model.safetensors.index.json: the model weights from
18
+ Swift 1.0 were further adapted by UkisAI using additional post-training
19
+ methods.
20
+ - README.md: replaced. LICENSE and NOTICE added.
21
+ - All other files (config.json, generation_config.json, chat_template.jinja,
22
+ tokenizer.json, tokenizer_config.json, vocab.json, merges.txt,
23
+ preprocessor_config.json, video_preprocessor_config.json) are retained from
24
+ the parent model and remain available under Apache License, Version 2.0.
QUANTIZATION_MANIFEST.json ADDED
@@ -0,0 +1,240 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "VALIDATED",
3
+ "created_at": "2026-09-21T18:16:33.223816+00:00",
4
+ "source_repository": "ukisai/Swift-1.5-Qwen3.8-27b",
5
+ "source_revision": "00ccd14e006897d28cb0ed5bf26390e60d274251",
6
+ "source_manifest_sha256": "0a00065b88ab003281853a7fb9bd5ce0086bc3781b36136d8c39da19933923ae",
7
+ "source_manifest_note": "SHA-256 of the original pre-publication-redaction export manifest.",
8
+ "source_weight_shards": 18,
9
+ "source_weight_bytes": 55563006776,
10
+ "source_file_bytes": 55586102681,
11
+ "source_recovery": "The pinned Hub revision was incomplete at conversion time. All 18 original shards and runtime assets were supplied by the verified project BF16 export and checked against the original export manifest; no weights were reconstructed or borrowed.",
12
+ "quantization": {
13
+ "mode": "affine",
14
+ "bits": 4,
15
+ "group_size": 64
16
+ },
17
+ "official_mlx_lm_repository": "https://github.com/ml-explore/mlx-lm",
18
+ "official_mlx_lm_base_commit": "c69d1288440a0dc4e6401fc417098b07598dccd5",
19
+ "architecture_patch_sha256": "f6f1d0bdafa45863bfbf93dac0398c481c993ea04fdf38b9bae98c643f89eaec",
20
+ "packages": {
21
+ "mlx": "0.32.2",
22
+ "mlx-cpu": "0.32.2",
23
+ "mlx-lm": "0.32.0",
24
+ "transformers": "5.14.1",
25
+ "huggingface_hub": "1.31.0",
26
+ "torch": "2.11.0+cpu",
27
+ "torchvision": "0.26.0+cpu",
28
+ "safetensors": "0.8.0"
29
+ },
30
+ "python": "3.12.3",
31
+ "platform": "Linux x86_64",
32
+ "validation_runtime": {
33
+ "device": "CPU"
34
+ },
35
+ "conversion_command": [
36
+ "mlx_lm.convert",
37
+ "--hf-path",
38
+ "<SOURCE_MODEL_DIR>",
39
+ "--mlx-path",
40
+ "<OUTPUT_MODEL_DIR>",
41
+ "--quantize",
42
+ "--q-mode",
43
+ "affine",
44
+ "--q-bits",
45
+ "4",
46
+ "--q-group-size",
47
+ "64"
48
+ ],
49
+ "conversion_elapsed_seconds": 120.018799242,
50
+ "output_weight_shards": 3,
51
+ "output_weight_bytes": 15826764635,
52
+ "source_tensors": 1199,
53
+ "mapped_source_tensors": 1199,
54
+ "saved_tensors": 2379,
55
+ "categories": {
56
+ "text": 851,
57
+ "MTP": 15,
58
+ "vision": 333
59
+ },
60
+ "quantized_weight_tensors": 590,
61
+ "original_bf16_tensors": 609,
62
+ "ignored_tensors": 0,
63
+ "unexplained_tensors": 0,
64
+ "validation": {
65
+ "status": "PASS",
66
+ "recorded_at": "2026-09-21T18:14:21.887850+00:00",
67
+ "source_repo": "ukisai/Swift-1.5-Qwen3.8-27b",
68
+ "source_revision": "00ccd14e006897d28cb0ed5bf26390e60d274251",
69
+ "source_shards": 18,
70
+ "source_shard_bytes": 55563006776,
71
+ "source_tensors": 1199,
72
+ "mapped_source_tensors": 1199,
73
+ "saved_tensors": 2379,
74
+ "categories": {
75
+ "text": 851,
76
+ "MTP": 15,
77
+ "vision": 333
78
+ },
79
+ "ignored_tensors": 0,
80
+ "unexplained_tensors": 0,
81
+ "exact_unquantized_tensors": 609,
82
+ "quantization": {
83
+ "group_size": 64,
84
+ "bits": 4,
85
+ "mode": "affine"
86
+ },
87
+ "all_floating_tensors_finite": true,
88
+ "tokenizer": "Qwen2Tokenizer",
89
+ "processor": "Qwen3VLProcessor",
90
+ "assets_sha256": {
91
+ "generation_config.json": "e70c136c1b78ddc1fb0905bac8e733a4dc448d4f852a5dd75143fffc70be550e",
92
+ "preprocessor_config.json": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516",
93
+ "video_preprocessor_config.json": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13",
94
+ "tokenizer.json": "0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3",
95
+ "tokenizer_config.json": "b11349aafa7cdc6a320767cf7ceb29ed82f7eda5d65e8e0819e76f0ce947bf27",
96
+ "vocab.json": "ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003",
97
+ "merges.txt": "a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d",
98
+ "chat_template.jinja": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041"
99
+ },
100
+ "chat_templates": [
101
+ {
102
+ "options": {
103
+ "enable_thinking": false
104
+ },
105
+ "rendered": "<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\n"
106
+ },
107
+ {
108
+ "options": {
109
+ "reasoning_effort": "low"
110
+ },
111
+ "rendered": "<|im_start|>system\nReasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.<|im_end|>\n<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n"
112
+ },
113
+ {
114
+ "options": {
115
+ "reasoning_effort": "xhigh"
116
+ },
117
+ "rendered": "<|im_start|>system\nReasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.<|im_end|>\n<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n"
118
+ }
119
+ ],
120
+ "load_seconds": 3.351265648000208,
121
+ "load_memory_bytes": 15826466152,
122
+ "process_peak_rss_bytes": 21142360064,
123
+ "inference_floating_dtype": "float32, CPU runtime only; stored floating tensors remain BF16",
124
+ "native_bf16_cpu_inference": "Aborted after reproducing incorrect accumulation in the official Linux BF16 quantized matmul. See cpu-quantized-matmul-diagnostic.json.",
125
+ "generation": {
126
+ "prompt": "<|im_start|>user\nReply with exactly: Hello from Swift.<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\n",
127
+ "text": "Hello from Swift.",
128
+ "token_ids": [
129
+ 9419,
130
+ 494,
131
+ 22929,
132
+ 13,
133
+ 248046
134
+ ],
135
+ "tokens": 5,
136
+ "tokens_per_second": 0.0797672292904007,
137
+ "prompt_tokens_per_second": 0.06443886297275572,
138
+ "elapsed_seconds": 374.00736365800003,
139
+ "finish_reason": "stop"
140
+ },
141
+ "mtp": {
142
+ "status": "PASS",
143
+ "shape": [
144
+ 1,
145
+ 1,
146
+ 248320
147
+ ],
148
+ "path": "Explicit MTP step with real text hidden states and shared LM head; speculative generation is not integrated"
149
+ },
150
+ "vision": {
151
+ "status": "PASS",
152
+ "shape": [
153
+ 64,
154
+ 5120
155
+ ],
156
+ "grid": [
157
+ [
158
+ 1,
159
+ 16,
160
+ 16
161
+ ]
162
+ ],
163
+ "path": "Vision encoder only; image/video insertion and multimodal text generation are not implemented"
164
+ },
165
+ "total_validation_seconds": 467.15810263900016
166
+ },
167
+ "private_repository": "ukisai/Swift-1.5-4bit-MLX",
168
+ "apple_silicon_verification": {
169
+ "status": "PASS_MAC_NATIVE_MLX_FORMAT_AND_REAL_METAL_SAMPLES",
170
+ "platform": "macOS-26.6-arm64-arm-64bit",
171
+ "machine": "arm64",
172
+ "mlx": "0.32.2",
173
+ "mlx_lm": "0.32.0",
174
+ "device": "Device(gpu, 0)",
175
+ "quantization": {
176
+ "group_size": 64,
177
+ "bits": 4,
178
+ "mode": "affine"
179
+ },
180
+ "all_saved_tensor_headers_validated": 2379,
181
+ "all_source_parameters_accounted_for": 1199,
182
+ "strict_complete_parameter_tree": "PASS using unevaluated header fixtures; no fabricated weights saved",
183
+ "actual_checkpoint_samples": [
184
+ {
185
+ "sample": 0,
186
+ "category": "text",
187
+ "checkpoint_weight": "language_model.model.layers.0.linear_attn.in_proj_a.weight",
188
+ "original_shape": [
189
+ 48,
190
+ 5120
191
+ ],
192
+ "packed_shape": [
193
+ 48,
194
+ 640
195
+ ],
196
+ "native_metal_bf16": "PASS",
197
+ "metal_fp32": "PASS",
198
+ "bf16_max_absolute_error": 0.0004401206970214844,
199
+ "fp32_max_absolute_error": 0.0
200
+ },
201
+ {
202
+ "sample": 1,
203
+ "category": "vision",
204
+ "checkpoint_weight": "visual.blocks.0.attn.proj.weight",
205
+ "original_shape": [
206
+ 1152,
207
+ 1152
208
+ ],
209
+ "packed_shape": [
210
+ 1152,
211
+ 144
212
+ ],
213
+ "native_metal_bf16": "PASS",
214
+ "metal_fp32": "PASS",
215
+ "bf16_max_absolute_error": 0.0002315044403076172,
216
+ "fp32_max_absolute_error": 0.0
217
+ },
218
+ {
219
+ "sample": 2,
220
+ "category": "MTP",
221
+ "checkpoint_weight": "mtp.layers.0.self_attn.k_proj.weight",
222
+ "original_shape": [
223
+ 1024,
224
+ 5120
225
+ ],
226
+ "packed_shape": [
227
+ 1024,
228
+ 640
229
+ ],
230
+ "native_metal_bf16": "PASS",
231
+ "metal_fp32": "PASS",
232
+ "bf16_max_absolute_error": 0.0009589195251464844,
233
+ "fp32_max_absolute_error": 0.0
234
+ }
235
+ ],
236
+ "full_model_mac_generation": "NOT_RUN: targeted native Metal format and real-weight component checks only",
237
+ "peak_mlx_bytes": 4597960,
238
+ "peak_process_rss_bytes": 360693760
239
+ }
240
+ }
README.md ADDED
@@ -0,0 +1,146 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: swift-open-license-1.0
4
+ license_link: https://huggingface.co/ukisai/Swift-1.5-4bit-MLX/blob/main/LICENSE
5
+ base_model: ukisai/Swift-1.5-Qwen3.8-27b
6
+ base_model_relation: quantized
7
+ library_name: mlx
8
+ pipeline_tag: text-generation
9
+ tags:
10
+ - mlx
11
+ - quantized
12
+ - 4-bit
13
+ - affine
14
+ - qwen3_8
15
+ ---
16
+
17
+ <div align="center">
18
+ <a href="https://ukisai.com"><img src="ukisai-banner.png" alt="UkisAI" style="width:100%;max-width:100%;height:auto;display:block;margin-bottom:0.6em;" /></a>
19
+ <div style="display:flex;justify-content:center;gap:0.6em;margin-bottom:1em;">
20
+ <a href="https://ukisai.com"><strong>Website</strong></a> &nbsp;&bull;&nbsp;
21
+ <a href="https://ukisai.com/products/swift"><strong>Learn more</strong></a> &nbsp;&bull;&nbsp;
22
+ <a href="https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b"><strong>BF16 model</strong></a> &nbsp;&bull;&nbsp;
23
+ <a href="https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GGUF"><strong>GGUF</strong></a> &nbsp;&bull;&nbsp;
24
+ <a href="https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF"><strong>GSQ-RCO GGUF</strong></a> &nbsp;&bull;&nbsp;
25
+ <a href="#evaluation"><strong>Evaluation</strong></a> &nbsp;&bull;&nbsp;
26
+ <a href="#license-and-access"><strong>Enterprise licensing</strong></a>
27
+ </div>
28
+ </div>
29
+
30
+ # Swift 1.5 Qwen3.8-27B — 4-bit MLX
31
+
32
+ **Apple MLX 4-bit affine quantization.** This is Swift 1.5 in native MLX format,
33
+ converted with the official Apple MLX-LM converter using 4 bits and group size 64.
34
+ Swift 1.5 is UkisAI's reasoning-efficient Qwen3.8-27B derivative, focused on stronger
35
+ long-horizon, agentic and coding performance while using fewer thinking tokens.
36
+
37
+ Swift 1.5 uses **58.5% fewer thinking tokens** than base Qwen3.8-27B while scoring **0.35% higher**, for a **9.18× speed-up** on several tasks.
38
+
39
+ ## Demo
40
+
41
+ We gave base Qwen3.8-27B and Swift 1.5 27B the same prompt:
42
+
43
+ > create a 3d little planet globe where I (player can walk around) and it has all these biomes to explore, the globe doesn't have to be too big, but still fun to go around. It's about a boy scout who is camping and goes around exploring.
44
+
45
+ <video src="https://huggingface.co/ukisai/Swift-1.5-4bit-MLX/resolve/main/swift-1.5-planet-demo.mp4" controls autoplay muted loop playsinline style="width:100%;height:auto;border-radius:12px;"></video>
46
+
47
+ Try the game yourself here: [https://ukisai.com/swift-games/27b](https://ukisai.com/swift-games/27b)
48
+
49
+ Base Qwen3.8-27B took 104.6 minutes to build its game. Swift 1.5 took 11.39 minutes.
50
+
51
+ ## Source and quantization
52
+
53
+ The conversion used the merged Swift 1.5 BF16 export associated with
54
+ [`ukisai/Swift-1.5-Qwen3.8-27b` revision `00ccd14e006897d28cb0ed5bf26390e60d274251`](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b/tree/00ccd14e006897d28cb0ed5bf26390e60d274251).
55
+ The source repository was subsequently completed with all 18 BF16 shards and runtime
56
+ assets at [revision `5ad04445d2686f525e9fbe5c077e6fa0c7df4200`](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b/tree/5ad04445d2686f525e9fbe5c077e6fa0c7df4200).
57
+ The later complete-revision link does not change the actual conversion provenance.
58
+
59
+ All 1,199 source tensors are accounted for, including 333 vision and 15 MTP tensors.
60
+ Eligible linear and embedding weights use 4-bit affine storage; 609 remaining tensors
61
+ retain their original BF16 values after the documented layout mapping. All 18 source
62
+ shards and runtime assets were verified by SHA-256. No base Qwen or alternate derived
63
+ checkpoint weights were substituted.
64
+
65
+ The saved checkpoint contains three weight shards. The original tokenizer, chat
66
+ template, processor/config assets, license and notices are included.
67
+ `QUANTIZATION_MANIFEST.json` records the fixed settings and validation results, while
68
+ `UPLOAD_MANIFEST.json` records release-file checksums.
69
+
70
+ ## Evaluation
71
+
72
+ See the [Swift 1.5 source model card](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b#evaluation)
73
+ for the source model's evaluations and methodology. Those results were not independently
74
+ re-run on this MLX quantization. No broad accuracy or long-context benchmark was run for
75
+ this release.
76
+
77
+ ## Validation and use
78
+
79
+ The validation results below are preserved historical build/component evidence,
80
+ not a new full-model Apple run. The approximately 15.83 GB tensor payload
81
+ requires additional runtime/cache and OS memory; do not force it onto a
82
+ 16 GiB Mac or raise system limits.
83
+
84
+ **Install the included MLX-LM architecture patch before loading this model.**
85
+ [`USAGE.md`](USAGE.md) provides the pinned official revision, patch commands and a text
86
+ generation example. The source configuration declares
87
+ `Qwen3_5ForConditionalGeneration` / `qwen3_5`; the patch preserves that configuration
88
+ and the inherited Swift text behavior.
89
+
90
+ Validation passed on Linux CPU with MLX 0.32.2 and patched MLX-LM 0.32.0: complete
91
+ source hashing, strict mapping and reload, finite floating tensors, exact BF16 remainder
92
+ preservation, tokenizer/chat-template/processor loading, and a short text-generation
93
+ smoke test that returned `Hello from Swift.`.
94
+
95
+ CPU smoke tests use FP32 floating-point arithmetic with the original packed 4-bit
96
+ tensors. This avoids a reproduced accumulation issue in MLX 0.32.2's Linux BF16
97
+ quantized-matmul path; checkpoint files and stored BF16 values are unchanged. Follow
98
+ the Linux branch in `USAGE.md`.
99
+
100
+ Apple Silicon checks passed for all 2,379 native tensor headers. Real packed text,
101
+ vision and MTP weight samples passed native BF16 Metal execution and matched the FP32
102
+ reference within BF16 tolerance. These checks verify native Mac MLX compatibility,
103
+ but **full 27B Apple Silicon generation was not tested**.
104
+
105
+ The vision encoder and an explicit MTP step passed real-weight component checks.
106
+ Integrated image/video chat and speculative generation are not implemented in this
107
+ patch. Component validation does not establish those end-to-end runtime features.
108
+
109
+ The `compatibility/` directory retains the MLX-LM patch, reproducible instructions,
110
+ source/tensor validation evidence, and the documented runtime limitations. Internal
111
+ project names, machine paths and cloud-instance details have been redacted from the
112
+ current published copies; older commits remain unchanged.
113
+
114
+ ## License and access
115
+
116
+ Swift 1.5 is a derivative of [Qwen3.8-27B](https://huggingface.co/Qwen/Qwen3.8-27B)
117
+ (Copyright 2026 Alibaba Cloud, [Apache License 2.0](https://huggingface.co/ukisai/Swift-1.5-4bit-MLX/blob/main/LICENSE-APACHE-2.0)). UkisAI's
118
+ contribution, including the adapted weights, is licensed under the
119
+ **[Swift Open License v1.0](https://huggingface.co/ukisai/Swift-1.5-4bit-MLX/blob/main/LICENSE)**. See [NOTICE](https://huggingface.co/ukisai/Swift-1.5-4bit-MLX/blob/main/NOTICE) for the change notice and
120
+ attribution details.
121
+
122
+ Personal, research, educational, evaluation and commercial use are free for individuals
123
+ and organizations with gross annual revenue, including affiliates, of up to US$1,000,000.
124
+ Above that threshold, commercial use requires a separate Swift Enterprise License.
125
+ Contact [UkisAI](https://ukisai.com/contact) for terms.
126
+
127
+ Nothing in the Swift Open License limits rights in Qwen3.8-27B itself under Apache 2.0.
128
+ The accompanying Apple MLX-LM code has a separate [MIT notice](compatibility/MLX-LM-LICENSE).
129
+
130
+ ## Citation
131
+
132
+ ```bibtex
133
+ @misc{swift-1.5-qwen3.8-27b,
134
+ title = {Swift 1.5 Qwen3.8-27B},
135
+ author = {UkisAI},
136
+ year = {2026},
137
+ url = {https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b}
138
+ }
139
+ ```
140
+
141
+ ## Acknowledgements
142
+
143
+ We acknowledge the [NVIDIA Innovation Lab](https://www.nvidia.com/en-us/data-center/innovation-lab/),
144
+ [Amazon Web Services](https://aws.amazon.com/), and
145
+ [Google Cloud](https://cloud.google.com/) for providing compute credits and
146
+ infrastructure support for Swift's development, training, and evaluation.
UPLOAD_MANIFEST.json ADDED
@@ -0,0 +1,333 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "repository": "ukisai/Swift-1.5-4bit-MLX",
3
+ "private": true,
4
+ "source_repository": "ukisai/Swift-1.5-Qwen3.8-27b",
5
+ "source_revision": "00ccd14e006897d28cb0ed5bf26390e60d274251",
6
+ "files": [
7
+ {
8
+ "path": "LICENSE",
9
+ "bytes": 13306,
10
+ "sha256": "1367057bf17041aa1d69286a1400be5f464d2f8b8f54f600088b209eac1850be",
11
+ "git_blob_sha1": "209a5720f7e4a747e658b49808a933309eeecceb"
12
+ },
13
+ {
14
+ "path": "LICENSE-APACHE-2.0",
15
+ "bytes": 11544,
16
+ "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a",
17
+ "git_blob_sha1": "f938136e3adacfd92be087f6e113b5d6d97f678f"
18
+ },
19
+ {
20
+ "path": "NOTICE",
21
+ "bytes": 1133,
22
+ "sha256": "be30f3d464974990e40e9833bc6f356fd89fe16733c6c1a581d97106b8ae741d",
23
+ "git_blob_sha1": "c4ad1a7188a4191cb5efd0f5109bf6e9ba8caee8"
24
+ },
25
+ {
26
+ "path": "QUANTIZATION_MANIFEST.json",
27
+ "bytes": 8230,
28
+ "sha256": "9f061d15042d9368e2c6a406ee20071ba6d78ce8da4c3ca53bfc13f5c2f95dd1",
29
+ "git_blob_sha1": "9726e2a5a93ee4abe32cb1032a38c01695b056d8"
30
+ },
31
+ {
32
+ "path": "README.md",
33
+ "bytes": 7084,
34
+ "sha256": "f9b9dd2366f69000bf70d47e393581a51a3ef69866e702be94259f1cc0261f4c",
35
+ "git_blob_sha1": "f4c63e395c2b6bf9b473a42dcb260d6d0a1fa834"
36
+ },
37
+ {
38
+ "path": "USAGE.md",
39
+ "bytes": 4074,
40
+ "sha256": "2d50b7882228b72dadb7b506580d4e57cc4f8b482f9afd9262c5240da161b53b",
41
+ "git_blob_sha1": "8037d79a8570d801f9c8f22fd4afa4bf53d1eca8"
42
+ },
43
+ {
44
+ "path": "chat_template.jinja",
45
+ "bytes": 8952,
46
+ "sha256": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
47
+ "git_blob_sha1": "c0c686f9c38d70d179fb7b5f5aa7530bc913dda3"
48
+ },
49
+ {
50
+ "path": "compatibility/MLX-LM-LICENSE",
51
+ "bytes": 1066,
52
+ "sha256": "ccfab7ccb2ea306f71531c8ca77bb55507606cd90768b1e32b8b52ab5b48cf01",
53
+ "git_blob_sha1": "98ff47b9ef9d4ac9f1a4bde3db13dd27456c5ea1"
54
+ },
55
+ {
56
+ "path": "compatibility/SOURCE_EXPORT_MANIFEST.json",
57
+ "bytes": 62808,
58
+ "sha256": "77dd1e156ac7f0c3f5e241bbbb2a0d721fb0e3513d8dd6870862509a1812275e",
59
+ "git_blob_sha1": "2c1632f6cf958f5b43169a4dfda3a5e4e813b9b6"
60
+ },
61
+ {
62
+ "path": "compatibility/architecture-compatibility-report.md",
63
+ "bytes": 2465,
64
+ "sha256": "39d19fae47331b0e70f661a839c7b320d540985f8ed56b8d2d5f171128d0eaf4",
65
+ "git_blob_sha1": "a7769aead59da94d0406ec0fe24e65b5dfc1f1de"
66
+ },
67
+ {
68
+ "path": "compatibility/aws-actual-source-structural-results.json",
69
+ "bytes": 2144,
70
+ "sha256": "a869d8e7a5048f59089077f37135cdc63ec955ad38738bc1b6bbc6460cc9d7bc",
71
+ "git_blob_sha1": "dea1d92bcf6cef93a611e19bf034dfc1891467ba"
72
+ },
73
+ {
74
+ "path": "compatibility/aws-source-verification.json",
75
+ "bytes": 379707,
76
+ "sha256": "ad944ae15bedaaec80abf9cd5e9e945f478397328cf3145f19517ff96295ad9c",
77
+ "git_blob_sha1": "c0de55933daec5c6a5f794b9b9a9c4e3bc2bdd7d"
78
+ },
79
+ {
80
+ "path": "compatibility/compatibility-tests-linux.log",
81
+ "bytes": 99,
82
+ "sha256": "e8735c26c6801337a00665881341818701e196a0c69ebd90f7702d98535f3179",
83
+ "git_blob_sha1": "eb3133aba134e5c5d8378aad30409dd4d4afa0d3"
84
+ },
85
+ {
86
+ "path": "compatibility/conversion-command.json",
87
+ "bytes": 581,
88
+ "sha256": "94e711a0a931855979a4db1d83a43cd54167f55c829c74e24135c8ad583a7f2b",
89
+ "git_blob_sha1": "a98ccccdb019212183d29b6374974ce63d1b8050"
90
+ },
91
+ {
92
+ "path": "compatibility/conversion-result.json",
93
+ "bytes": 689,
94
+ "sha256": "aa74632f5e0ebe7cfaccb9fd8251ccfd7dbe9d81e9384658e7d313b611acafc7",
95
+ "git_blob_sha1": "af1aa936ff6f2f2a5986e6d68ff3a748b72ab4da"
96
+ },
97
+ {
98
+ "path": "compatibility/conversion.log",
99
+ "bytes": 113,
100
+ "sha256": "927e01909126b96eea5faf9692d99f02fdaf99d1aa61d14d257f3ebdb0afd1a0",
101
+ "git_blob_sha1": "e933301e9777876e2ccac21fc49ae6a2d4879660"
102
+ },
103
+ {
104
+ "path": "compatibility/cpu-quantized-matmul-diagnostic.json",
105
+ "bytes": 290,
106
+ "sha256": "5adefdff0a3efbc54c1abc92cd05aa18e63a39771fe9d6b8972bcd2fea1cf3d3",
107
+ "git_blob_sha1": "7c452317ed3da7cdd2e6ee9742bd91534ba25c28"
108
+ },
109
+ {
110
+ "path": "compatibility/environment-linux.json",
111
+ "bytes": 367,
112
+ "sha256": "efdd012bfc5ad59f07f9dc6e8a6bd67bb992cd9363174171c271ce1ad5ee0cec",
113
+ "git_blob_sha1": "972ebef9b09def952f0880ec1e60a2f192df1416"
114
+ },
115
+ {
116
+ "path": "compatibility/fixed-affine-test.log",
117
+ "bytes": 113,
118
+ "sha256": "a40772b0d08d9f5ce659ee06cd5e5573ec0697073295d819a5dc9b6b1f4302bd",
119
+ "git_blob_sha1": "7e3746c1857e2e5dc808589514ca56d65d699a11"
120
+ },
121
+ {
122
+ "path": "compatibility/mac-check/checkpoint-headers.json",
123
+ "bytes": 561445,
124
+ "sha256": "141b5b6193da63332328bb0f8ad8aac129f51232ab345021dd6f8e4319c5640c",
125
+ "git_blob_sha1": "20ed71f7444de28ca236be0da3bbd70319ee083c"
126
+ },
127
+ {
128
+ "path": "compatibility/mac-check/config.json",
129
+ "bytes": 4577,
130
+ "sha256": "d0bbe7655fdaa589acbdf3739e2f2378d2c7462da05338ad5161c70025540b98",
131
+ "git_blob_sha1": "9143ac44bb812f357882d503ae90149661616fc1"
132
+ },
133
+ {
134
+ "path": "compatibility/mac-check/model.safetensors.index.json",
135
+ "bytes": 232537,
136
+ "sha256": "937800b65e142989cf9f973de5c0f42b0a5ad0659871dd1b67d917d78a195921",
137
+ "git_blob_sha1": "c943542ddfee76de8e5b0ec14e7fbb86243adbb6"
138
+ },
139
+ {
140
+ "path": "compatibility/mac-check/real-checkpoint-samples.safetensors",
141
+ "bytes": 3866825,
142
+ "sha256": "c0e3f757947e7ffdaf96b90eb8e0f0ae72e83b9685d3462b99cd6bf193a2cd5b",
143
+ "git_blob_sha1": "63afc5a8192c9726d15abe3955846fc0dd9c663a"
144
+ },
145
+ {
146
+ "path": "compatibility/mac-check/samples.json",
147
+ "bytes": 975,
148
+ "sha256": "e1d9c0062005f8204d01b52099deaa97dcda44f62bb44c7dba6ffaf8f21c72d7",
149
+ "git_blob_sha1": "fad93b5e0ebffe6fc096b157f30dd4794f661130"
150
+ },
151
+ {
152
+ "path": "compatibility/mac-compatibility-results.json",
153
+ "bytes": 1924,
154
+ "sha256": "45f665d5fb442861c40631b78d1134a888b318e30382d729e7e8a703b94434e2",
155
+ "git_blob_sha1": "af7ac5d2d3d9ec83b6aae7db3ede7ee61c730303"
156
+ },
157
+ {
158
+ "path": "compatibility/missing-file-recovery-report.md",
159
+ "bytes": 1046,
160
+ "sha256": "78fe1768db6b2c95354638cf257bea795389c14b38e4872fb3150df851b9d3ec",
161
+ "git_blob_sha1": "7c45edf5a824384bea8c9cc4b0dd17ea1bbbb096"
162
+ },
163
+ {
164
+ "path": "compatibility/quant-tensor-mapping-manifest.json",
165
+ "bytes": 646822,
166
+ "sha256": "bf99693a9747c476af80b306bf61df546824815fb934cc6bfa1ed150fa967587",
167
+ "git_blob_sha1": "431b551674ec9669b6f04e3f2b5736f93b1f91ba"
168
+ },
169
+ {
170
+ "path": "compatibility/quant-validation-results.json",
171
+ "bytes": 3783,
172
+ "sha256": "ecb99453d4c2b23a818f121d0e2c141328dded5f7fac205af766221ec973ae1d",
173
+ "git_blob_sha1": "1e6eba8e7d3c785284c79d7603d3db13fef158fe"
174
+ },
175
+ {
176
+ "path": "compatibility/quant-validation.log",
177
+ "bytes": 4092,
178
+ "sha256": "8616f0367709a655a549a458c7d337ba326681646de9ce01b61439a270b2b493",
179
+ "git_blob_sha1": "a3c99e8901464f3149b8f9103604f12ce54182cb"
180
+ },
181
+ {
182
+ "path": "compatibility/requirements-linux.txt",
183
+ "bytes": 810,
184
+ "sha256": "89a4ef8054809e043b1e0b1c7b0f41847666a7fd7d263c325311b30be9eb798f",
185
+ "git_blob_sha1": "513937cb32426c806dd60bb83b54ccb402aa81c0"
186
+ },
187
+ {
188
+ "path": "compatibility/run-compatibility.sh",
189
+ "bytes": 1059,
190
+ "sha256": "1024e009047bf97ba4876b8543f4151728b1c85f2c93abf09b73f7811c1a25fb",
191
+ "git_blob_sha1": "551f80902537a58768a170d53134fc0559e131f1"
192
+ },
193
+ {
194
+ "path": "compatibility/run-conversion.py",
195
+ "bytes": 2860,
196
+ "sha256": "6d612bbfd438abdecf393f16143f724c1d37f60903ca8ce46da486c2f3671a41",
197
+ "git_blob_sha1": "2d7e9a1453a489eeb72aecce44a5171e1fadf27f"
198
+ },
199
+ {
200
+ "path": "compatibility/setup-linux.sh",
201
+ "bytes": 1734,
202
+ "sha256": "e253306f08eb86d66e1c15d02733e349a5f82c00abb15ff0656f5ce40acfb775",
203
+ "git_blob_sha1": "ce35740f29f974529d27fde03ac348dabb83efb9"
204
+ },
205
+ {
206
+ "path": "compatibility/source-file-manifest.csv",
207
+ "bytes": 4842,
208
+ "sha256": "f4fec6fc3318813395745f4d68700ebaf13e13bbf9bbdf2b90454faf6bbe721e",
209
+ "git_blob_sha1": "8275d62522a8da3aafbcfc412f23bfb7abf4739c"
210
+ },
211
+ {
212
+ "path": "compatibility/swift15-mlx-lm.patch",
213
+ "bytes": 38976,
214
+ "sha256": "f6f1d0bdafa45863bfbf93dac0398c481c993ea04fdf38b9bae98c643f89eaec",
215
+ "git_blob_sha1": "b739acc2e3b4cb82f073b7280255bcb898bb2a33"
216
+ },
217
+ {
218
+ "path": "compatibility/unmatched-tensor-analysis.csv",
219
+ "bytes": 102593,
220
+ "sha256": "57f6537c39953a2f02971c2ef09ccfb7806f0ea729a6143fa90157e1b961ed05",
221
+ "git_blob_sha1": "43011b66fcf2b390f1c821a21a80d4e84c84fb19"
222
+ },
223
+ {
224
+ "path": "compatibility/validate-mac-format.py",
225
+ "bytes": 3576,
226
+ "sha256": "1f8a04cf27919daad342365b26c86c02ceb7b97b71752bc1cf8ac66309743554",
227
+ "git_blob_sha1": "2abc61b81ddb01ab0754c76ec15d14bd1670bb28"
228
+ },
229
+ {
230
+ "path": "compatibility/validate-quant.py",
231
+ "bytes": 8617,
232
+ "sha256": "484255b5a4a8f71119312ed1ae54b1246f5b4dd70ef698fb32d4cc78c75ffba9",
233
+ "git_blob_sha1": "a52f8c19a3f91fcb09701758d5664bfa453ed8d7"
234
+ },
235
+ {
236
+ "path": "compatibility/validate_aws_source_linux.py",
237
+ "bytes": 4798,
238
+ "sha256": "40c8f2678e7efa1ec6df9acbefc7e0c65def8e707d70d78257b0baff02d9b3d8",
239
+ "git_blob_sha1": "ca986cbfa2a4ba480ef9e7a0986209fae0256831"
240
+ },
241
+ {
242
+ "path": "compatibility/verify_source_readonly.py",
243
+ "bytes": 4093,
244
+ "sha256": "2d02c6d7c7e81f6029faf3f7cfcd43ba6b0ca7d3858dd3e5e6c1a82a5631e765",
245
+ "git_blob_sha1": "5dee1520d86f961e4ceee7e694fbe3515075c6f2"
246
+ },
247
+ {
248
+ "path": "config.json",
249
+ "bytes": 4577,
250
+ "sha256": "d0bbe7655fdaa589acbdf3739e2f2378d2c7462da05338ad5161c70025540b98",
251
+ "git_blob_sha1": "9143ac44bb812f357882d503ae90149661616fc1"
252
+ },
253
+ {
254
+ "path": "generation_config.json",
255
+ "bytes": 202,
256
+ "sha256": "e70c136c1b78ddc1fb0905bac8e733a4dc448d4f852a5dd75143fffc70be550e",
257
+ "git_blob_sha1": "023756cfadf88e5bf69eefeee3e172f38c448d64"
258
+ },
259
+ {
260
+ "path": "merges.txt",
261
+ "bytes": 3353259,
262
+ "sha256": "a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d",
263
+ "git_blob_sha1": "a494e019ca1502219fd0128658b979e5f05ae8e8"
264
+ },
265
+ {
266
+ "path": "model-00001-of-00003.safetensors",
267
+ "bytes": 5328325554,
268
+ "sha256": "d91b70ca84dff315addeeb8a599dce2beccfec33c7483686ede2e8ff3cc459dc",
269
+ "git_blob_sha1": "972fbab4cd3302fad7f76024f95765a35c6f0c70"
270
+ },
271
+ {
272
+ "path": "model-00002-of-00003.safetensors",
273
+ "bytes": 5354185158,
274
+ "sha256": "eddcff1a6ef0971990f7cd01dba77ab9a8c20d59821bb4cd7e25234fe3809346",
275
+ "git_blob_sha1": "6ab9eeb3981fe2f243a5c6c9b18efb23ad42f879"
276
+ },
277
+ {
278
+ "path": "model-00003-of-00003.safetensors",
279
+ "bytes": 5144253923,
280
+ "sha256": "644cc5322ff706c9608b42ea1ff83e0beb12b828d49b36c468a366f44593b8eb",
281
+ "git_blob_sha1": "7127bb75e507028d7684e48815cfeb11f74dcaa5"
282
+ },
283
+ {
284
+ "path": "model.safetensors.index.json",
285
+ "bytes": 232537,
286
+ "sha256": "937800b65e142989cf9f973de5c0f42b0a5ad0659871dd1b67d917d78a195921",
287
+ "git_blob_sha1": "c943542ddfee76de8e5b0ec14e7fbb86243adbb6"
288
+ },
289
+ {
290
+ "path": "preprocessor_config.json",
291
+ "bytes": 390,
292
+ "sha256": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516",
293
+ "git_blob_sha1": "2ea84a437d448ff71b08df68fdd949d5cc4ebb64"
294
+ },
295
+ {
296
+ "path": "tokenizer.json",
297
+ "bytes": 12809320,
298
+ "sha256": "0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3",
299
+ "git_blob_sha1": "b438cf847e527641b288c91745740dfa0f9da294"
300
+ },
301
+ {
302
+ "path": "tokenizer_config.json",
303
+ "bytes": 17928,
304
+ "sha256": "b11349aafa7cdc6a320767cf7ceb29ed82f7eda5d65e8e0819e76f0ce947bf27",
305
+ "git_blob_sha1": "5de744b3fca2129d7186979ae47c06be33903243"
306
+ },
307
+ {
308
+ "path": "ukisai-banner.png",
309
+ "bytes": 307739,
310
+ "sha256": "8577252b8b37e331f06f7ddecdf3bfa1c497e164f9eb36d78c37232d1b8492ca",
311
+ "git_blob_sha1": "efd86f93bda3ac7da9628c5284a07706ce798a5b"
312
+ },
313
+ {
314
+ "path": "video_preprocessor_config.json",
315
+ "bytes": 385,
316
+ "sha256": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13",
317
+ "git_blob_sha1": "3ba673a5ad7d4d13f54155ecd38b2a94a6dac8fe"
318
+ },
319
+ {
320
+ "path": "vocab.json",
321
+ "bytes": 6722759,
322
+ "sha256": "ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003",
323
+ "git_blob_sha1": "0aa0ce0658d60ac4a5d609f4eadb0e8e43514176"
324
+ },
325
+ {
326
+ "path": "swift-1.5-planet-demo.mp4",
327
+ "bytes": 4738825,
328
+ "sha256": "1cda6924169e8b83e5d6d7299baf8329ed1b1a182a481aa0e06344e0f5c00111"
329
+ }
330
+ ],
331
+ "total_bytes_excluding_manifest": 15860955305,
332
+ "manifest_self_excluded": true
333
+ }
USAGE.md ADDED
@@ -0,0 +1,92 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Load Swift 1.5 with its complete MLX architecture
2
+
3
+ Use the included patch with the pinned official Apple MLX-LM revision. Unpatched
4
+ text-only Qwen support does not preserve this checkpoint's complete parameter tree.
5
+
6
+ Use Python 3.12 in a new working directory. Install the HF CLI before using it,
7
+ and pin the complete model revision as well as the MLX-LM source revision.
8
+ This repository is private: after installing the CLI, run `hf auth login`
9
+ interactively if not already signed in with an account that has access.
10
+
11
+ ```bash
12
+ python3.12 -m venv .venv
13
+ source .venv/bin/activate
14
+ python -m pip install 'huggingface_hub==1.31.0'
15
+ SWIFT_MLX_REVISION=d2140379e1fd593002c92fa552b2b37fb6eb1159
16
+ hf download ukisai/Swift-1.5-4bit-MLX --revision "$SWIFT_MLX_REVISION" --local-dir Swift-1.5-4bit-MLX
17
+ hf cache verify ukisai/Swift-1.5-4bit-MLX --revision "$SWIFT_MLX_REVISION" --local-dir Swift-1.5-4bit-MLX --fail-on-missing-files
18
+ git clone https://github.com/ml-explore/mlx-lm.git swift15-mlx-lm
19
+ git -C swift15-mlx-lm checkout --detach c69d1288440a0dc4e6401fc417098b07598dccd5
20
+ git -C swift15-mlx-lm apply --check ../Swift-1.5-4bit-MLX/compatibility/swift15-mlx-lm.patch
21
+ git -C swift15-mlx-lm apply ../Swift-1.5-4bit-MLX/compatibility/swift15-mlx-lm.patch
22
+ ```
23
+
24
+ Stop after any missing-file or checksum failure. The checkpoint contains about
25
+ 15.83 GB of tensor data, before runtime, cache and OS overhead. Do not force the
26
+ full model onto a 16 GiB Mac or increase system memory limits. Full-model Apple
27
+ generation remains unverified. The recorded small Metal samples are not a full run.
28
+
29
+ On Apple Silicon:
30
+
31
+ ```bash
32
+ pip install 'mlx==0.32.2' 'transformers==5.14.1' 'huggingface_hub==1.31.0' pillow
33
+ pip install -e ./swift15-mlx-lm
34
+ ```
35
+
36
+ On Linux CPU (Python 3.12 and glibc 2.35 or newer):
37
+
38
+ ```bash
39
+ pip install 'mlx[cpu]==0.32.2' 'transformers==5.14.1' 'huggingface_hub==1.31.0' pillow
40
+ pip install -e ./swift15-mlx-lm
41
+ ```
42
+
43
+ Text generation:
44
+
45
+ ```python
46
+ import mlx.core as mx
47
+ from mlx_lm import load, generate
48
+
49
+ model, tokenizer = load("Swift-1.5-4bit-MLX")
50
+ if mx.default_device() == mx.cpu:
51
+ model.apply(
52
+ lambda value: value.astype(mx.float32)
53
+ if mx.issubdtype(value.dtype, mx.floating) else value
54
+ )
55
+ prompt = tokenizer.apply_chat_template(
56
+ [{"role": "user", "content": "Say hello."}],
57
+ tokenize=False,
58
+ add_generation_prompt=True,
59
+ enable_thinking=False,
60
+ )
61
+ print(generate(model, tokenizer, prompt=prompt, max_tokens=32))
62
+ ```
63
+
64
+ The Linux CPU branch promotes only in-memory floating parameters to FP32. Packed
65
+ 4-bit weights and all files remain unchanged. This avoids the official MLX 0.32.2
66
+ Linux scalar BF16 quantized-matmul accumulation bug reproduced in
67
+ `compatibility/cpu-quantized-matmul-diagnostic.json` (8,192 exact ones summed to
68
+ 256 in BF16, versus the correct 8,192 in FP32). The release's CPU generation,
69
+ MTP and vision smoke tests use this FP32 runtime. Apple Silicon inference does
70
+ not use this CPU workaround; full-model Apple Silicon execution was not tested.
71
+
72
+ The original chat template also accepts `reasoning_effort="low"`, `"medium"`,
73
+ and `"xhigh"`; this release validates the original low and xhigh formats.
74
+ This structural smoke test does not establish long-context or benchmark accuracy.
75
+
76
+ The patch implements an explicit MTP step (`model.mtp_logits`) and the vision
77
+ encoder (`model.visual`). Their weights are retained and the release validation
78
+ records their component execution. Speculative generation and integrated image/video
79
+ chat are not implemented. Unsupported multimodal generation calls raise an error.
80
+
81
+ Reproduce conversion only from the complete original Swift BF16 export identified
82
+ in `QUANTIZATION_MANIFEST.json`, after verifying its 18 shards and original assets:
83
+
84
+ ```bash
85
+ mlx_lm.convert --hf-path /path/to/Swift-1.5-BF16 \
86
+ --mlx-path Swift-1.5-4bit-MLX \
87
+ --quantize --q-mode affine --q-bits 4 --q-group-size 64
88
+ ```
89
+
90
+ The converter refuses an existing output directory. It uses official MLX-LM lazy
91
+ loading, quantization, sharding, and saving; the patch supplies the complete model
92
+ and strict parameter/asset mapping.
chat_template.jinja ADDED
@@ -0,0 +1,170 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- set reasoning_instructions = '' %}
46
+ {%- if enable_thinking is undefined or enable_thinking is true %}
47
+ {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}
48
+ {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}
49
+ {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}
50
+ {%- endif %}
51
+ {%- if resolved_reasoning_effort == 'xhigh' %}
52
+ {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}
53
+ {%- elif resolved_reasoning_effort == 'low' %}
54
+ {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}
55
+ {%- endif %}
56
+ {%- endif %}
57
+ {%- if tools and tools is iterable and tools is not mapping %}
58
+ {{- '<|im_start|>system\n' }}
59
+ {%- if reasoning_instructions %}
60
+ {{- reasoning_instructions + '\n\n' }}
61
+ {%- endif %}
62
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
63
+ {%- for tool in tools %}
64
+ {{- "\n" }}
65
+ {{- tool | tojson }}
66
+ {%- endfor %}
67
+ {{- "\n</tools>" }}
68
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
69
+ {%- if messages[0].role == 'system' %}
70
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
71
+ {%- if content %}
72
+ {{- '\n\n' + content }}
73
+ {%- endif %}
74
+ {%- endif %}
75
+ {{- '<|im_end|>\n' }}
76
+ {%- else %}
77
+ {%- if messages[0].role == 'system' %}
78
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
79
+ {%- if content %}
80
+ {{- '<|im_start|>system\n' + (reasoning_instructions + '\n\n' if reasoning_instructions else '') + content + '<|im_end|>\n' }}
81
+ {%- elif reasoning_instructions %}
82
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
83
+ {%- endif %}
84
+ {%- elif reasoning_instructions %}
85
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
86
+ {%- endif %}
87
+ {%- endif %}
88
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
89
+ {%- for message in messages[::-1] %}
90
+ {%- set index = (messages|length - 1) - loop.index0 %}
91
+ {%- if ns.multi_step_tool and message.role == "user" %}
92
+ {%- set content = render_content(message.content, false)|trim %}
93
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
94
+ {%- set ns.multi_step_tool = false %}
95
+ {%- set ns.last_query_index = index %}
96
+ {%- endif %}
97
+ {%- endif %}
98
+ {%- endfor %}
99
+ {%- if ns.multi_step_tool %}
100
+ {{- raise_exception('No user query found in messages.') }}
101
+ {%- endif %}
102
+ {%- for message in messages %}
103
+ {%- set content = render_content(message.content, true)|trim %}
104
+ {%- if message.role == "system" %}
105
+ {%- if not loop.first %}
106
+ {{- raise_exception('System message must be at the beginning.') }}
107
+ {%- endif %}
108
+ {%- elif message.role == "user" %}
109
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
110
+ {%- elif message.role == "assistant" %}
111
+ {%- set reasoning_content = '' %}
112
+ {%- if message.reasoning_content is string %}
113
+ {%- set reasoning_content = message.reasoning_content %}
114
+ {%- endif %}
115
+ {%- set reasoning_content = reasoning_content|trim %}
116
+ {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}
117
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
118
+ {%- else %}
119
+ {{- '<|im_start|>' + message.role + '\n' + content }}
120
+ {%- endif %}
121
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
122
+ {%- for tool_call in message.tool_calls %}
123
+ {%- if tool_call.function is defined %}
124
+ {%- set tool_call = tool_call.function %}
125
+ {%- endif %}
126
+ {%- if loop.first %}
127
+ {%- if content|trim %}
128
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
129
+ {%- else %}
130
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
131
+ {%- endif %}
132
+ {%- else %}
133
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
134
+ {%- endif %}
135
+ {%- if tool_call.arguments is defined and tool_call.arguments != '' %}
136
+ {%- for args_name, args_value in tool_call.arguments|items %}
137
+ {{- '<parameter=' + args_name + '>\n' }}
138
+ {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}
139
+ {{- args_value }}
140
+ {{- '\n</parameter>\n' }}
141
+ {%- endfor %}
142
+ {%- endif %}
143
+ {{- '</function>\n</tool_call>' }}
144
+ {%- endfor %}
145
+ {%- endif %}
146
+ {{- '<|im_end|>\n' }}
147
+ {%- elif message.role == "tool" %}
148
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
149
+ {{- '<|im_start|>user' }}
150
+ {%- endif %}
151
+ {{- '\n<tool_response>\n' }}
152
+ {{- content }}
153
+ {{- '\n</tool_response>' }}
154
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
155
+ {{- '<|im_end|>\n' }}
156
+ {%- elif loop.last %}
157
+ {{- '<|im_end|>\n' }}
158
+ {%- endif %}
159
+ {%- else %}
160
+ {{- raise_exception('Unexpected message role.') }}
161
+ {%- endif %}
162
+ {%- endfor %}
163
+ {%- if add_generation_prompt %}
164
+ {{- '<|im_start|>assistant\n' }}
165
+ {%- if enable_thinking is defined and enable_thinking is false %}
166
+ {{- '<think>\n\n</think>\n\n' }}
167
+ {%- else %}
168
+ {{- '<think>\n' }}
169
+ {%- endif %}
170
+ {%- endif %}
compatibility/MLX-LM-LICENSE ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ MIT License
2
+
3
+ Copyright © 2023 Apple Inc.
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
compatibility/SOURCE_EXPORT_MANIFEST.json ADDED
@@ -0,0 +1,1730 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "Swift-1.5",
3
+ "publication_redaction": "Internal training names, paths and timestamps were removed from the current published copy. The original manifest SHA-256 remains recorded in QUANTIZATION_MANIFEST.json.",
4
+ "source_checkpoint": "<INTERNAL_TRAINING_CHECKPOINT_REDACTED>",
5
+ "parent_model": "<INTERNAL_PARENT_MODEL_REDACTED>",
6
+ "dtype": "bfloat16",
7
+ "source_training_state": "Internal training-state inventory redacted from the current publication.",
8
+ "converter_path": "<INTERNAL_CONVERTER_PATH_REDACTED>",
9
+ "converter_sha256": "e83cd55ba2cf81936ab72a6b0a8ae3c2cbd02de6a232094aec5dc215939ee639",
10
+ "export_script_sha256": "c8a097372b7f7eb5f35470697d37857e72b695d12309400eaea834c867e5d35d",
11
+ "text_tensors": 851,
12
+ "replicated_tensors_checked_equal": 454,
13
+ "preserved_parent_tensors": [
14
+ "model.visual.blocks.0.attn.proj.bias",
15
+ "model.visual.blocks.0.attn.proj.weight",
16
+ "model.visual.blocks.0.attn.qkv.bias",
17
+ "model.visual.blocks.0.attn.qkv.weight",
18
+ "model.visual.blocks.0.mlp.linear_fc1.bias",
19
+ "model.visual.blocks.0.mlp.linear_fc1.weight",
20
+ "model.visual.blocks.0.mlp.linear_fc2.bias",
21
+ "model.visual.blocks.0.mlp.linear_fc2.weight",
22
+ "model.visual.blocks.0.norm1.bias",
23
+ "model.visual.blocks.0.norm1.weight",
24
+ "model.visual.blocks.0.norm2.bias",
25
+ "model.visual.blocks.0.norm2.weight",
26
+ "model.visual.blocks.1.attn.proj.bias",
27
+ "model.visual.blocks.1.attn.proj.weight",
28
+ "model.visual.blocks.1.attn.qkv.bias",
29
+ "model.visual.blocks.1.attn.qkv.weight",
30
+ "model.visual.blocks.1.mlp.linear_fc1.bias",
31
+ "model.visual.blocks.1.mlp.linear_fc1.weight",
32
+ "model.visual.blocks.1.mlp.linear_fc2.bias",
33
+ "model.visual.blocks.1.mlp.linear_fc2.weight",
34
+ "model.visual.blocks.1.norm1.bias",
35
+ "model.visual.blocks.1.norm1.weight",
36
+ "model.visual.blocks.1.norm2.bias",
37
+ "model.visual.blocks.1.norm2.weight",
38
+ "model.visual.blocks.10.attn.proj.bias",
39
+ "model.visual.blocks.10.attn.proj.weight",
40
+ "model.visual.blocks.10.attn.qkv.bias",
41
+ "model.visual.blocks.10.attn.qkv.weight",
42
+ "model.visual.blocks.10.mlp.linear_fc1.bias",
43
+ "model.visual.blocks.10.mlp.linear_fc1.weight",
44
+ "model.visual.blocks.10.mlp.linear_fc2.bias",
45
+ "model.visual.blocks.10.mlp.linear_fc2.weight",
46
+ "model.visual.blocks.10.norm1.bias",
47
+ "model.visual.blocks.10.norm1.weight",
48
+ "model.visual.blocks.10.norm2.bias",
49
+ "model.visual.blocks.10.norm2.weight",
50
+ "model.visual.blocks.11.attn.proj.bias",
51
+ "model.visual.blocks.11.attn.proj.weight",
52
+ "model.visual.blocks.11.attn.qkv.bias",
53
+ "model.visual.blocks.11.attn.qkv.weight",
54
+ "model.visual.blocks.11.mlp.linear_fc1.bias",
55
+ "model.visual.blocks.11.mlp.linear_fc1.weight",
56
+ "model.visual.blocks.11.mlp.linear_fc2.bias",
57
+ "model.visual.blocks.11.mlp.linear_fc2.weight",
58
+ "model.visual.blocks.11.norm1.bias",
59
+ "model.visual.blocks.11.norm1.weight",
60
+ "model.visual.blocks.11.norm2.bias",
61
+ "model.visual.blocks.11.norm2.weight",
62
+ "model.visual.blocks.12.attn.proj.bias",
63
+ "model.visual.blocks.12.attn.proj.weight",
64
+ "model.visual.blocks.12.attn.qkv.bias",
65
+ "model.visual.blocks.12.attn.qkv.weight",
66
+ "model.visual.blocks.12.mlp.linear_fc1.bias",
67
+ "model.visual.blocks.12.mlp.linear_fc1.weight",
68
+ "model.visual.blocks.12.mlp.linear_fc2.bias",
69
+ "model.visual.blocks.12.mlp.linear_fc2.weight",
70
+ "model.visual.blocks.12.norm1.bias",
71
+ "model.visual.blocks.12.norm1.weight",
72
+ "model.visual.blocks.12.norm2.bias",
73
+ "model.visual.blocks.12.norm2.weight",
74
+ "model.visual.blocks.13.attn.proj.bias",
75
+ "model.visual.blocks.13.attn.proj.weight",
76
+ "model.visual.blocks.13.attn.qkv.bias",
77
+ "model.visual.blocks.13.attn.qkv.weight",
78
+ "model.visual.blocks.13.mlp.linear_fc1.bias",
79
+ "model.visual.blocks.13.mlp.linear_fc1.weight",
80
+ "model.visual.blocks.13.mlp.linear_fc2.bias",
81
+ "model.visual.blocks.13.mlp.linear_fc2.weight",
82
+ "model.visual.blocks.13.norm1.bias",
83
+ "model.visual.blocks.13.norm1.weight",
84
+ "model.visual.blocks.13.norm2.bias",
85
+ "model.visual.blocks.13.norm2.weight",
86
+ "model.visual.blocks.14.attn.proj.bias",
87
+ "model.visual.blocks.14.attn.proj.weight",
88
+ "model.visual.blocks.14.attn.qkv.bias",
89
+ "model.visual.blocks.14.attn.qkv.weight",
90
+ "model.visual.blocks.14.mlp.linear_fc1.bias",
91
+ "model.visual.blocks.14.mlp.linear_fc1.weight",
92
+ "model.visual.blocks.14.mlp.linear_fc2.bias",
93
+ "model.visual.blocks.14.mlp.linear_fc2.weight",
94
+ "model.visual.blocks.14.norm1.bias",
95
+ "model.visual.blocks.14.norm1.weight",
96
+ "model.visual.blocks.14.norm2.bias",
97
+ "model.visual.blocks.14.norm2.weight",
98
+ "model.visual.blocks.15.attn.proj.bias",
99
+ "model.visual.blocks.15.attn.proj.weight",
100
+ "model.visual.blocks.15.attn.qkv.bias",
101
+ "model.visual.blocks.15.attn.qkv.weight",
102
+ "model.visual.blocks.15.mlp.linear_fc1.bias",
103
+ "model.visual.blocks.15.mlp.linear_fc1.weight",
104
+ "model.visual.blocks.15.mlp.linear_fc2.bias",
105
+ "model.visual.blocks.15.mlp.linear_fc2.weight",
106
+ "model.visual.blocks.15.norm1.bias",
107
+ "model.visual.blocks.15.norm1.weight",
108
+ "model.visual.blocks.15.norm2.bias",
109
+ "model.visual.blocks.15.norm2.weight",
110
+ "model.visual.blocks.16.attn.proj.bias",
111
+ "model.visual.blocks.16.attn.proj.weight",
112
+ "model.visual.blocks.16.attn.qkv.bias",
113
+ "model.visual.blocks.16.attn.qkv.weight",
114
+ "model.visual.blocks.16.mlp.linear_fc1.bias",
115
+ "model.visual.blocks.16.mlp.linear_fc1.weight",
116
+ "model.visual.blocks.16.mlp.linear_fc2.bias",
117
+ "model.visual.blocks.16.mlp.linear_fc2.weight",
118
+ "model.visual.blocks.16.norm1.bias",
119
+ "model.visual.blocks.16.norm1.weight",
120
+ "model.visual.blocks.16.norm2.bias",
121
+ "model.visual.blocks.16.norm2.weight",
122
+ "model.visual.blocks.17.attn.proj.bias",
123
+ "model.visual.blocks.17.attn.proj.weight",
124
+ "model.visual.blocks.17.attn.qkv.bias",
125
+ "model.visual.blocks.17.attn.qkv.weight",
126
+ "model.visual.blocks.17.mlp.linear_fc1.bias",
127
+ "model.visual.blocks.17.mlp.linear_fc1.weight",
128
+ "model.visual.blocks.17.mlp.linear_fc2.bias",
129
+ "model.visual.blocks.17.mlp.linear_fc2.weight",
130
+ "model.visual.blocks.17.norm1.bias",
131
+ "model.visual.blocks.17.norm1.weight",
132
+ "model.visual.blocks.17.norm2.bias",
133
+ "model.visual.blocks.17.norm2.weight",
134
+ "model.visual.blocks.18.attn.proj.bias",
135
+ "model.visual.blocks.18.attn.proj.weight",
136
+ "model.visual.blocks.18.attn.qkv.bias",
137
+ "model.visual.blocks.18.attn.qkv.weight",
138
+ "model.visual.blocks.18.mlp.linear_fc1.bias",
139
+ "model.visual.blocks.18.mlp.linear_fc1.weight",
140
+ "model.visual.blocks.18.mlp.linear_fc2.bias",
141
+ "model.visual.blocks.18.mlp.linear_fc2.weight",
142
+ "model.visual.blocks.18.norm1.bias",
143
+ "model.visual.blocks.18.norm1.weight",
144
+ "model.visual.blocks.18.norm2.bias",
145
+ "model.visual.blocks.18.norm2.weight",
146
+ "model.visual.blocks.19.attn.proj.bias",
147
+ "model.visual.blocks.19.attn.proj.weight",
148
+ "model.visual.blocks.19.attn.qkv.bias",
149
+ "model.visual.blocks.19.attn.qkv.weight",
150
+ "model.visual.blocks.19.mlp.linear_fc1.bias",
151
+ "model.visual.blocks.19.mlp.linear_fc1.weight",
152
+ "model.visual.blocks.19.mlp.linear_fc2.bias",
153
+ "model.visual.blocks.19.mlp.linear_fc2.weight",
154
+ "model.visual.blocks.19.norm1.bias",
155
+ "model.visual.blocks.19.norm1.weight",
156
+ "model.visual.blocks.19.norm2.bias",
157
+ "model.visual.blocks.19.norm2.weight",
158
+ "model.visual.blocks.2.attn.proj.bias",
159
+ "model.visual.blocks.2.attn.proj.weight",
160
+ "model.visual.blocks.2.attn.qkv.bias",
161
+ "model.visual.blocks.2.attn.qkv.weight",
162
+ "model.visual.blocks.2.mlp.linear_fc1.bias",
163
+ "model.visual.blocks.2.mlp.linear_fc1.weight",
164
+ "model.visual.blocks.2.mlp.linear_fc2.bias",
165
+ "model.visual.blocks.2.mlp.linear_fc2.weight",
166
+ "model.visual.blocks.2.norm1.bias",
167
+ "model.visual.blocks.2.norm1.weight",
168
+ "model.visual.blocks.2.norm2.bias",
169
+ "model.visual.blocks.2.norm2.weight",
170
+ "model.visual.blocks.20.attn.proj.bias",
171
+ "model.visual.blocks.20.attn.proj.weight",
172
+ "model.visual.blocks.20.attn.qkv.bias",
173
+ "model.visual.blocks.20.attn.qkv.weight",
174
+ "model.visual.blocks.20.mlp.linear_fc1.bias",
175
+ "model.visual.blocks.20.mlp.linear_fc1.weight",
176
+ "model.visual.blocks.20.mlp.linear_fc2.bias",
177
+ "model.visual.blocks.20.mlp.linear_fc2.weight",
178
+ "model.visual.blocks.20.norm1.bias",
179
+ "model.visual.blocks.20.norm1.weight",
180
+ "model.visual.blocks.20.norm2.bias",
181
+ "model.visual.blocks.20.norm2.weight",
182
+ "model.visual.blocks.21.attn.proj.bias",
183
+ "model.visual.blocks.21.attn.proj.weight",
184
+ "model.visual.blocks.21.attn.qkv.bias",
185
+ "model.visual.blocks.21.attn.qkv.weight",
186
+ "model.visual.blocks.21.mlp.linear_fc1.bias",
187
+ "model.visual.blocks.21.mlp.linear_fc1.weight",
188
+ "model.visual.blocks.21.mlp.linear_fc2.bias",
189
+ "model.visual.blocks.21.mlp.linear_fc2.weight",
190
+ "model.visual.blocks.21.norm1.bias",
191
+ "model.visual.blocks.21.norm1.weight",
192
+ "model.visual.blocks.21.norm2.bias",
193
+ "model.visual.blocks.21.norm2.weight",
194
+ "model.visual.blocks.22.attn.proj.bias",
195
+ "model.visual.blocks.22.attn.proj.weight",
196
+ "model.visual.blocks.22.attn.qkv.bias",
197
+ "model.visual.blocks.22.attn.qkv.weight",
198
+ "model.visual.blocks.22.mlp.linear_fc1.bias",
199
+ "model.visual.blocks.22.mlp.linear_fc1.weight",
200
+ "model.visual.blocks.22.mlp.linear_fc2.bias",
201
+ "model.visual.blocks.22.mlp.linear_fc2.weight",
202
+ "model.visual.blocks.22.norm1.bias",
203
+ "model.visual.blocks.22.norm1.weight",
204
+ "model.visual.blocks.22.norm2.bias",
205
+ "model.visual.blocks.22.norm2.weight",
206
+ "model.visual.blocks.23.attn.proj.bias",
207
+ "model.visual.blocks.23.attn.proj.weight",
208
+ "model.visual.blocks.23.attn.qkv.bias",
209
+ "model.visual.blocks.23.attn.qkv.weight",
210
+ "model.visual.blocks.23.mlp.linear_fc1.bias",
211
+ "model.visual.blocks.23.mlp.linear_fc1.weight",
212
+ "model.visual.blocks.23.mlp.linear_fc2.bias",
213
+ "model.visual.blocks.23.mlp.linear_fc2.weight",
214
+ "model.visual.blocks.23.norm1.bias",
215
+ "model.visual.blocks.23.norm1.weight",
216
+ "model.visual.blocks.23.norm2.bias",
217
+ "model.visual.blocks.23.norm2.weight",
218
+ "model.visual.blocks.24.attn.proj.bias",
219
+ "model.visual.blocks.24.attn.proj.weight",
220
+ "model.visual.blocks.24.attn.qkv.bias",
221
+ "model.visual.blocks.24.attn.qkv.weight",
222
+ "model.visual.blocks.24.mlp.linear_fc1.bias",
223
+ "model.visual.blocks.24.mlp.linear_fc1.weight",
224
+ "model.visual.blocks.24.mlp.linear_fc2.bias",
225
+ "model.visual.blocks.24.mlp.linear_fc2.weight",
226
+ "model.visual.blocks.24.norm1.bias",
227
+ "model.visual.blocks.24.norm1.weight",
228
+ "model.visual.blocks.24.norm2.bias",
229
+ "model.visual.blocks.24.norm2.weight",
230
+ "model.visual.blocks.25.attn.proj.bias",
231
+ "model.visual.blocks.25.attn.proj.weight",
232
+ "model.visual.blocks.25.attn.qkv.bias",
233
+ "model.visual.blocks.25.attn.qkv.weight",
234
+ "model.visual.blocks.25.mlp.linear_fc1.bias",
235
+ "model.visual.blocks.25.mlp.linear_fc1.weight",
236
+ "model.visual.blocks.25.mlp.linear_fc2.bias",
237
+ "model.visual.blocks.25.mlp.linear_fc2.weight",
238
+ "model.visual.blocks.25.norm1.bias",
239
+ "model.visual.blocks.25.norm1.weight",
240
+ "model.visual.blocks.25.norm2.bias",
241
+ "model.visual.blocks.25.norm2.weight",
242
+ "model.visual.blocks.26.attn.proj.bias",
243
+ "model.visual.blocks.26.attn.proj.weight",
244
+ "model.visual.blocks.26.attn.qkv.bias",
245
+ "model.visual.blocks.26.attn.qkv.weight",
246
+ "model.visual.blocks.26.mlp.linear_fc1.bias",
247
+ "model.visual.blocks.26.mlp.linear_fc1.weight",
248
+ "model.visual.blocks.26.mlp.linear_fc2.bias",
249
+ "model.visual.blocks.26.mlp.linear_fc2.weight",
250
+ "model.visual.blocks.26.norm1.bias",
251
+ "model.visual.blocks.26.norm1.weight",
252
+ "model.visual.blocks.26.norm2.bias",
253
+ "model.visual.blocks.26.norm2.weight",
254
+ "model.visual.blocks.3.attn.proj.bias",
255
+ "model.visual.blocks.3.attn.proj.weight",
256
+ "model.visual.blocks.3.attn.qkv.bias",
257
+ "model.visual.blocks.3.attn.qkv.weight",
258
+ "model.visual.blocks.3.mlp.linear_fc1.bias",
259
+ "model.visual.blocks.3.mlp.linear_fc1.weight",
260
+ "model.visual.blocks.3.mlp.linear_fc2.bias",
261
+ "model.visual.blocks.3.mlp.linear_fc2.weight",
262
+ "model.visual.blocks.3.norm1.bias",
263
+ "model.visual.blocks.3.norm1.weight",
264
+ "model.visual.blocks.3.norm2.bias",
265
+ "model.visual.blocks.3.norm2.weight",
266
+ "model.visual.blocks.4.attn.proj.bias",
267
+ "model.visual.blocks.4.attn.proj.weight",
268
+ "model.visual.blocks.4.attn.qkv.bias",
269
+ "model.visual.blocks.4.attn.qkv.weight",
270
+ "model.visual.blocks.4.mlp.linear_fc1.bias",
271
+ "model.visual.blocks.4.mlp.linear_fc1.weight",
272
+ "model.visual.blocks.4.mlp.linear_fc2.bias",
273
+ "model.visual.blocks.4.mlp.linear_fc2.weight",
274
+ "model.visual.blocks.4.norm1.bias",
275
+ "model.visual.blocks.4.norm1.weight",
276
+ "model.visual.blocks.4.norm2.bias",
277
+ "model.visual.blocks.4.norm2.weight",
278
+ "model.visual.blocks.5.attn.proj.bias",
279
+ "model.visual.blocks.5.attn.proj.weight",
280
+ "model.visual.blocks.5.attn.qkv.bias",
281
+ "model.visual.blocks.5.attn.qkv.weight",
282
+ "model.visual.blocks.5.mlp.linear_fc1.bias",
283
+ "model.visual.blocks.5.mlp.linear_fc1.weight",
284
+ "model.visual.blocks.5.mlp.linear_fc2.bias",
285
+ "model.visual.blocks.5.mlp.linear_fc2.weight",
286
+ "model.visual.blocks.5.norm1.bias",
287
+ "model.visual.blocks.5.norm1.weight",
288
+ "model.visual.blocks.5.norm2.bias",
289
+ "model.visual.blocks.5.norm2.weight",
290
+ "model.visual.blocks.6.attn.proj.bias",
291
+ "model.visual.blocks.6.attn.proj.weight",
292
+ "model.visual.blocks.6.attn.qkv.bias",
293
+ "model.visual.blocks.6.attn.qkv.weight",
294
+ "model.visual.blocks.6.mlp.linear_fc1.bias",
295
+ "model.visual.blocks.6.mlp.linear_fc1.weight",
296
+ "model.visual.blocks.6.mlp.linear_fc2.bias",
297
+ "model.visual.blocks.6.mlp.linear_fc2.weight",
298
+ "model.visual.blocks.6.norm1.bias",
299
+ "model.visual.blocks.6.norm1.weight",
300
+ "model.visual.blocks.6.norm2.bias",
301
+ "model.visual.blocks.6.norm2.weight",
302
+ "model.visual.blocks.7.attn.proj.bias",
303
+ "model.visual.blocks.7.attn.proj.weight",
304
+ "model.visual.blocks.7.attn.qkv.bias",
305
+ "model.visual.blocks.7.attn.qkv.weight",
306
+ "model.visual.blocks.7.mlp.linear_fc1.bias",
307
+ "model.visual.blocks.7.mlp.linear_fc1.weight",
308
+ "model.visual.blocks.7.mlp.linear_fc2.bias",
309
+ "model.visual.blocks.7.mlp.linear_fc2.weight",
310
+ "model.visual.blocks.7.norm1.bias",
311
+ "model.visual.blocks.7.norm1.weight",
312
+ "model.visual.blocks.7.norm2.bias",
313
+ "model.visual.blocks.7.norm2.weight",
314
+ "model.visual.blocks.8.attn.proj.bias",
315
+ "model.visual.blocks.8.attn.proj.weight",
316
+ "model.visual.blocks.8.attn.qkv.bias",
317
+ "model.visual.blocks.8.attn.qkv.weight",
318
+ "model.visual.blocks.8.mlp.linear_fc1.bias",
319
+ "model.visual.blocks.8.mlp.linear_fc1.weight",
320
+ "model.visual.blocks.8.mlp.linear_fc2.bias",
321
+ "model.visual.blocks.8.mlp.linear_fc2.weight",
322
+ "model.visual.blocks.8.norm1.bias",
323
+ "model.visual.blocks.8.norm1.weight",
324
+ "model.visual.blocks.8.norm2.bias",
325
+ "model.visual.blocks.8.norm2.weight",
326
+ "model.visual.blocks.9.attn.proj.bias",
327
+ "model.visual.blocks.9.attn.proj.weight",
328
+ "model.visual.blocks.9.attn.qkv.bias",
329
+ "model.visual.blocks.9.attn.qkv.weight",
330
+ "model.visual.blocks.9.mlp.linear_fc1.bias",
331
+ "model.visual.blocks.9.mlp.linear_fc1.weight",
332
+ "model.visual.blocks.9.mlp.linear_fc2.bias",
333
+ "model.visual.blocks.9.mlp.linear_fc2.weight",
334
+ "model.visual.blocks.9.norm1.bias",
335
+ "model.visual.blocks.9.norm1.weight",
336
+ "model.visual.blocks.9.norm2.bias",
337
+ "model.visual.blocks.9.norm2.weight",
338
+ "model.visual.merger.linear_fc1.bias",
339
+ "model.visual.merger.linear_fc1.weight",
340
+ "model.visual.merger.linear_fc2.bias",
341
+ "model.visual.merger.linear_fc2.weight",
342
+ "model.visual.merger.norm.bias",
343
+ "model.visual.merger.norm.weight",
344
+ "model.visual.patch_embed.proj.bias",
345
+ "model.visual.patch_embed.proj.weight",
346
+ "model.visual.pos_embed.weight",
347
+ "mtp.fc.weight",
348
+ "mtp.layers.0.input_layernorm.weight",
349
+ "mtp.layers.0.mlp.down_proj.weight",
350
+ "mtp.layers.0.mlp.gate_proj.weight",
351
+ "mtp.layers.0.mlp.up_proj.weight",
352
+ "mtp.layers.0.post_attention_layernorm.weight",
353
+ "mtp.layers.0.self_attn.k_norm.weight",
354
+ "mtp.layers.0.self_attn.k_proj.weight",
355
+ "mtp.layers.0.self_attn.o_proj.weight",
356
+ "mtp.layers.0.self_attn.q_norm.weight",
357
+ "mtp.layers.0.self_attn.q_proj.weight",
358
+ "mtp.layers.0.self_attn.v_proj.weight",
359
+ "mtp.norm.weight",
360
+ "mtp.pre_fc_norm_embedding.weight",
361
+ "mtp.pre_fc_norm_hidden.weight"
362
+ ],
363
+ "replicated_tensor_differences": [
364
+ {
365
+ "pp_rank": 1,
366
+ "parameter": "decoder.layers.0.self_attention.linear_attn.conv1d.weight",
367
+ "different_elements": 45,
368
+ "numel": 40960,
369
+ "max_abs_difference": 1.52587890625e-05,
370
+ "mean_abs_difference": 1.8214023622675768e-09,
371
+ "selected_tp_rank": 0
372
+ },
373
+ {
374
+ "pp_rank": 1,
375
+ "parameter": "decoder.layers.0.self_attention.linear_attn.in_proj_qkv.weight",
376
+ "different_elements": 18609,
377
+ "numel": 52428800,
378
+ "max_abs_difference": 6.103515625e-05,
379
+ "mean_abs_difference": 1.291975748607399e-09,
380
+ "selected_tp_rank": 0
381
+ },
382
+ {
383
+ "pp_rank": 1,
384
+ "parameter": "decoder.layers.0.self_attention.linear_attn.in_proj_z.weight",
385
+ "different_elements": 9639,
386
+ "numel": 31457280,
387
+ "max_abs_difference": 6.103515625e-05,
388
+ "mean_abs_difference": 1.112669623104523e-09,
389
+ "selected_tp_rank": 0
390
+ },
391
+ {
392
+ "pp_rank": 1,
393
+ "parameter": "decoder.layers.0.self_attention.linear_attn.in_proj_b.weight",
394
+ "different_elements": 213,
395
+ "numel": 245760,
396
+ "max_abs_difference": 3.0517578125e-05,
397
+ "mean_abs_difference": 3.895282318922e-09,
398
+ "selected_tp_rank": 0
399
+ },
400
+ {
401
+ "pp_rank": 1,
402
+ "parameter": "decoder.layers.0.self_attention.linear_attn.in_proj_a.weight",
403
+ "different_elements": 115,
404
+ "numel": 245760,
405
+ "max_abs_difference": 3.0517578125e-05,
406
+ "mean_abs_difference": 1.531059501402865e-09,
407
+ "selected_tp_rank": 0
408
+ },
409
+ {
410
+ "pp_rank": 1,
411
+ "parameter": "decoder.layers.0.self_attention.linear_attn.out_proj.weight",
412
+ "different_elements": 9183,
413
+ "numel": 31457280,
414
+ "max_abs_difference": 6.103515625e-05,
415
+ "mean_abs_difference": 1.0142108264332705e-09,
416
+ "selected_tp_rank": 0
417
+ },
418
+ {
419
+ "pp_rank": 1,
420
+ "parameter": "decoder.layers.1.self_attention.linear_attn.conv1d.weight",
421
+ "different_elements": 49,
422
+ "numel": 40960,
423
+ "max_abs_difference": 1.52587890625e-05,
424
+ "mean_abs_difference": 2.0707602299552264e-09,
425
+ "selected_tp_rank": 0
426
+ },
427
+ {
428
+ "pp_rank": 1,
429
+ "parameter": "decoder.layers.1.self_attention.linear_attn.in_proj_qkv.weight",
430
+ "different_elements": 15243,
431
+ "numel": 52428800,
432
+ "max_abs_difference": 6.103515625e-05,
433
+ "mean_abs_difference": 1.0479778156380348e-09,
434
+ "selected_tp_rank": 0
435
+ },
436
+ {
437
+ "pp_rank": 1,
438
+ "parameter": "decoder.layers.1.self_attention.linear_attn.in_proj_z.weight",
439
+ "different_elements": 8172,
440
+ "numel": 31457280,
441
+ "max_abs_difference": 6.103515625e-05,
442
+ "mean_abs_difference": 9.054810790054546e-10,
443
+ "selected_tp_rank": 0
444
+ },
445
+ {
446
+ "pp_rank": 1,
447
+ "parameter": "decoder.layers.1.self_attention.linear_attn.in_proj_b.weight",
448
+ "different_elements": 182,
449
+ "numel": 245760,
450
+ "max_abs_difference": 3.0517578125e-05,
451
+ "mean_abs_difference": 2.5461306396579175e-09,
452
+ "selected_tp_rank": 0
453
+ },
454
+ {
455
+ "pp_rank": 1,
456
+ "parameter": "decoder.layers.1.self_attention.linear_attn.in_proj_a.weight",
457
+ "different_elements": 102,
458
+ "numel": 245760,
459
+ "max_abs_difference": 3.0517578125e-05,
460
+ "mean_abs_difference": 1.203668831273319e-09,
461
+ "selected_tp_rank": 0
462
+ },
463
+ {
464
+ "pp_rank": 1,
465
+ "parameter": "decoder.layers.1.self_attention.linear_attn.out_proj.weight",
466
+ "different_elements": 7573,
467
+ "numel": 31457280,
468
+ "max_abs_difference": 6.103515625e-05,
469
+ "mean_abs_difference": 8.400473094916094e-10,
470
+ "selected_tp_rank": 0
471
+ },
472
+ {
473
+ "pp_rank": 1,
474
+ "parameter": "decoder.layers.2.self_attention.linear_attn.conv1d.weight",
475
+ "different_elements": 19,
476
+ "numel": 40960,
477
+ "max_abs_difference": 3.814697265625e-06,
478
+ "mean_abs_difference": 3.8039615901652724e-10,
479
+ "selected_tp_rank": 0
480
+ },
481
+ {
482
+ "pp_rank": 1,
483
+ "parameter": "decoder.layers.2.self_attention.linear_attn.in_proj_qkv.weight",
484
+ "different_elements": 6412,
485
+ "numel": 52428800,
486
+ "max_abs_difference": 6.103515625e-05,
487
+ "mean_abs_difference": 4.1276040918525325e-10,
488
+ "selected_tp_rank": 0
489
+ },
490
+ {
491
+ "pp_rank": 1,
492
+ "parameter": "decoder.layers.2.self_attention.linear_attn.in_proj_z.weight",
493
+ "different_elements": 3067,
494
+ "numel": 31457280,
495
+ "max_abs_difference": 6.103515625e-05,
496
+ "mean_abs_difference": 3.217904831487317e-10,
497
+ "selected_tp_rank": 0
498
+ },
499
+ {
500
+ "pp_rank": 1,
501
+ "parameter": "decoder.layers.2.self_attention.linear_attn.in_proj_b.weight",
502
+ "different_elements": 85,
503
+ "numel": 245760,
504
+ "max_abs_difference": 3.0517578125e-05,
505
+ "mean_abs_difference": 1.0481007173268608e-09,
506
+ "selected_tp_rank": 0
507
+ },
508
+ {
509
+ "pp_rank": 1,
510
+ "parameter": "decoder.layers.2.self_attention.linear_attn.in_proj_a.weight",
511
+ "different_elements": 48,
512
+ "numel": 245760,
513
+ "max_abs_difference": 3.0517578125e-05,
514
+ "mean_abs_difference": 4.6057949121269814e-10,
515
+ "selected_tp_rank": 0
516
+ },
517
+ {
518
+ "pp_rank": 1,
519
+ "parameter": "decoder.layers.2.self_attention.linear_attn.out_proj.weight",
520
+ "different_elements": 2922,
521
+ "numel": 31457280,
522
+ "max_abs_difference": 6.103515625e-05,
523
+ "mean_abs_difference": 2.940546695029411e-10,
524
+ "selected_tp_rank": 0
525
+ },
526
+ {
527
+ "pp_rank": 1,
528
+ "parameter": "decoder.layers.4.self_attention.linear_attn.conv1d.weight",
529
+ "different_elements": 22,
530
+ "numel": 40960,
531
+ "max_abs_difference": 1.52587890625e-05,
532
+ "mean_abs_difference": 1.0222720447927713e-09,
533
+ "selected_tp_rank": 0
534
+ },
535
+ {
536
+ "pp_rank": 1,
537
+ "parameter": "decoder.layers.4.self_attention.linear_attn.in_proj_qkv.weight",
538
+ "different_elements": 6353,
539
+ "numel": 52428800,
540
+ "max_abs_difference": 6.103515625e-05,
541
+ "mean_abs_difference": 5.091326249484496e-10,
542
+ "selected_tp_rank": 0
543
+ },
544
+ {
545
+ "pp_rank": 1,
546
+ "parameter": "decoder.layers.4.self_attention.linear_attn.in_proj_z.weight",
547
+ "different_elements": 2880,
548
+ "numel": 31457280,
549
+ "max_abs_difference": 6.103515625e-05,
550
+ "mean_abs_difference": 3.73936021036414e-10,
551
+ "selected_tp_rank": 0
552
+ },
553
+ {
554
+ "pp_rank": 1,
555
+ "parameter": "decoder.layers.4.self_attention.linear_attn.in_proj_b.weight",
556
+ "different_elements": 57,
557
+ "numel": 245760,
558
+ "max_abs_difference": 3.0517578125e-05,
559
+ "mean_abs_difference": 5.159601079718357e-10,
560
+ "selected_tp_rank": 0
561
+ },
562
+ {
563
+ "pp_rank": 1,
564
+ "parameter": "decoder.layers.4.self_attention.linear_attn.in_proj_a.weight",
565
+ "different_elements": 52,
566
+ "numel": 245760,
567
+ "max_abs_difference": 3.0517578125e-05,
568
+ "mean_abs_difference": 6.798018259424055e-10,
569
+ "selected_tp_rank": 0
570
+ },
571
+ {
572
+ "pp_rank": 1,
573
+ "parameter": "decoder.layers.4.self_attention.linear_attn.out_proj.weight",
574
+ "different_elements": 2780,
575
+ "numel": 31457280,
576
+ "max_abs_difference": 6.103515625e-05,
577
+ "mean_abs_difference": 3.477997612133521e-10,
578
+ "selected_tp_rank": 0
579
+ },
580
+ {
581
+ "pp_rank": 1,
582
+ "parameter": "decoder.layers.5.self_attention.linear_attn.conv1d.weight",
583
+ "different_elements": 2,
584
+ "numel": 40960,
585
+ "max_abs_difference": 3.0517578125e-05,
586
+ "mean_abs_difference": 7.465132401129893e-10,
587
+ "selected_tp_rank": 0
588
+ },
589
+ {
590
+ "pp_rank": 1,
591
+ "parameter": "decoder.layers.5.self_attention.linear_attn.in_proj_qkv.weight",
592
+ "different_elements": 4450,
593
+ "numel": 52428800,
594
+ "max_abs_difference": 6.103515625e-05,
595
+ "mean_abs_difference": 3.5795788555503805e-10,
596
+ "selected_tp_rank": 0
597
+ },
598
+ {
599
+ "pp_rank": 1,
600
+ "parameter": "decoder.layers.5.self_attention.linear_attn.in_proj_z.weight",
601
+ "different_elements": 1443,
602
+ "numel": 31457280,
603
+ "max_abs_difference": 6.103515625e-05,
604
+ "mean_abs_difference": 1.8462815998265825e-10,
605
+ "selected_tp_rank": 0
606
+ },
607
+ {
608
+ "pp_rank": 1,
609
+ "parameter": "decoder.layers.5.self_attention.linear_attn.in_proj_b.weight",
610
+ "different_elements": 44,
611
+ "numel": 245760,
612
+ "max_abs_difference": 3.0517578125e-05,
613
+ "mean_abs_difference": 6.661214912995206e-10,
614
+ "selected_tp_rank": 0
615
+ },
616
+ {
617
+ "pp_rank": 1,
618
+ "parameter": "decoder.layers.5.self_attention.linear_attn.in_proj_a.weight",
619
+ "different_elements": 25,
620
+ "numel": 245760,
621
+ "max_abs_difference": 3.0517578125e-05,
622
+ "mean_abs_difference": 3.0735236578038894e-10,
623
+ "selected_tp_rank": 0
624
+ },
625
+ {
626
+ "pp_rank": 1,
627
+ "parameter": "decoder.layers.5.self_attention.linear_attn.out_proj.weight",
628
+ "different_elements": 1449,
629
+ "numel": 31457280,
630
+ "max_abs_difference": 3.0517578125e-05,
631
+ "mean_abs_difference": 1.7835337373650617e-10,
632
+ "selected_tp_rank": 0
633
+ },
634
+ {
635
+ "pp_rank": 1,
636
+ "parameter": "decoder.layers.6.self_attention.linear_attn.conv1d.weight",
637
+ "different_elements": 6,
638
+ "numel": 40960,
639
+ "max_abs_difference": 3.814697265625e-06,
640
+ "mean_abs_difference": 1.6470949604219243e-10,
641
+ "selected_tp_rank": 0
642
+ },
643
+ {
644
+ "pp_rank": 1,
645
+ "parameter": "decoder.layers.6.self_attention.linear_attn.in_proj_qkv.weight",
646
+ "different_elements": 423,
647
+ "numel": 52428800,
648
+ "max_abs_difference": 6.103515625e-05,
649
+ "mean_abs_difference": 2.7659968065973928e-11,
650
+ "selected_tp_rank": 0
651
+ },
652
+ {
653
+ "pp_rank": 1,
654
+ "parameter": "decoder.layers.6.self_attention.linear_attn.in_proj_z.weight",
655
+ "different_elements": 165,
656
+ "numel": 31457280,
657
+ "max_abs_difference": 3.0517578125e-05,
658
+ "mean_abs_difference": 1.594046075692468e-11,
659
+ "selected_tp_rank": 0
660
+ },
661
+ {
662
+ "pp_rank": 1,
663
+ "parameter": "decoder.layers.6.self_attention.linear_attn.in_proj_b.weight",
664
+ "different_elements": 2,
665
+ "numel": 245760,
666
+ "max_abs_difference": 2.9802322387695312e-08,
667
+ "mean_abs_difference": 1.3452941831273296e-13,
668
+ "selected_tp_rank": 0
669
+ },
670
+ {
671
+ "pp_rank": 1,
672
+ "parameter": "decoder.layers.6.self_attention.linear_attn.in_proj_a.weight",
673
+ "different_elements": 3,
674
+ "numel": 245760,
675
+ "max_abs_difference": 2.384185791015625e-07,
676
+ "mean_abs_difference": 1.515824466814808e-12,
677
+ "selected_tp_rank": 0
678
+ },
679
+ {
680
+ "pp_rank": 1,
681
+ "parameter": "decoder.layers.6.self_attention.linear_attn.out_proj.weight",
682
+ "different_elements": 143,
683
+ "numel": 31457280,
684
+ "max_abs_difference": 3.0517578125e-05,
685
+ "mean_abs_difference": 1.4855916843914407e-11,
686
+ "selected_tp_rank": 0
687
+ },
688
+ {
689
+ "pp_rank": 1,
690
+ "parameter": "decoder.layers.8.self_attention.linear_attn.conv1d.weight",
691
+ "different_elements": 93,
692
+ "numel": 40960,
693
+ "max_abs_difference": 3.0517578125e-05,
694
+ "mean_abs_difference": 6.272486530178867e-09,
695
+ "selected_tp_rank": 0
696
+ },
697
+ {
698
+ "pp_rank": 1,
699
+ "parameter": "decoder.layers.8.self_attention.linear_attn.in_proj_qkv.weight",
700
+ "different_elements": 51265,
701
+ "numel": 52428800,
702
+ "max_abs_difference": 6.103515625e-05,
703
+ "mean_abs_difference": 4.324009594824929e-09,
704
+ "selected_tp_rank": 0
705
+ },
706
+ {
707
+ "pp_rank": 1,
708
+ "parameter": "decoder.layers.8.self_attention.linear_attn.in_proj_z.weight",
709
+ "different_elements": 23917,
710
+ "numel": 31457280,
711
+ "max_abs_difference": 6.103515625e-05,
712
+ "mean_abs_difference": 3.2533382654520437e-09,
713
+ "selected_tp_rank": 0
714
+ },
715
+ {
716
+ "pp_rank": 1,
717
+ "parameter": "decoder.layers.8.self_attention.linear_attn.in_proj_b.weight",
718
+ "different_elements": 518,
719
+ "numel": 245760,
720
+ "max_abs_difference": 3.0517578125e-05,
721
+ "mean_abs_difference": 9.054241800754426e-09,
722
+ "selected_tp_rank": 0
723
+ },
724
+ {
725
+ "pp_rank": 1,
726
+ "parameter": "decoder.layers.8.self_attention.linear_attn.in_proj_a.weight",
727
+ "different_elements": 387,
728
+ "numel": 245760,
729
+ "max_abs_difference": 3.0517578125e-05,
730
+ "mean_abs_difference": 6.4282117406833095e-09,
731
+ "selected_tp_rank": 0
732
+ },
733
+ {
734
+ "pp_rank": 1,
735
+ "parameter": "decoder.layers.8.self_attention.linear_attn.out_proj.weight",
736
+ "different_elements": 22821,
737
+ "numel": 31457280,
738
+ "max_abs_difference": 6.103515625e-05,
739
+ "mean_abs_difference": 3.0835232145420832e-09,
740
+ "selected_tp_rank": 0
741
+ },
742
+ {
743
+ "pp_rank": 1,
744
+ "parameter": "decoder.layers.9.self_attention.linear_attn.conv1d.weight",
745
+ "different_elements": 32,
746
+ "numel": 40960,
747
+ "max_abs_difference": 7.62939453125e-06,
748
+ "mean_abs_difference": 1.059993315344343e-09,
749
+ "selected_tp_rank": 0
750
+ },
751
+ {
752
+ "pp_rank": 1,
753
+ "parameter": "decoder.layers.9.self_attention.linear_attn.in_proj_qkv.weight",
754
+ "different_elements": 12825,
755
+ "numel": 52428800,
756
+ "max_abs_difference": 6.103515625e-05,
757
+ "mean_abs_difference": 9.968682546102059e-10,
758
+ "selected_tp_rank": 0
759
+ },
760
+ {
761
+ "pp_rank": 1,
762
+ "parameter": "decoder.layers.9.self_attention.linear_attn.in_proj_z.weight",
763
+ "different_elements": 5561,
764
+ "numel": 31457280,
765
+ "max_abs_difference": 6.103515625e-05,
766
+ "mean_abs_difference": 6.905871430262778e-10,
767
+ "selected_tp_rank": 0
768
+ },
769
+ {
770
+ "pp_rank": 1,
771
+ "parameter": "decoder.layers.9.self_attention.linear_attn.in_proj_b.weight",
772
+ "different_elements": 152,
773
+ "numel": 245760,
774
+ "max_abs_difference": 3.0517578125e-05,
775
+ "mean_abs_difference": 1.8903487664090335e-09,
776
+ "selected_tp_rank": 0
777
+ },
778
+ {
779
+ "pp_rank": 1,
780
+ "parameter": "decoder.layers.9.self_attention.linear_attn.in_proj_a.weight",
781
+ "different_elements": 116,
782
+ "numel": 245760,
783
+ "max_abs_difference": 3.0517578125e-05,
784
+ "mean_abs_difference": 1.5164772770859258e-09,
785
+ "selected_tp_rank": 0
786
+ },
787
+ {
788
+ "pp_rank": 1,
789
+ "parameter": "decoder.layers.9.self_attention.linear_attn.out_proj.weight",
790
+ "different_elements": 5352,
791
+ "numel": 31457280,
792
+ "max_abs_difference": 6.103515625e-05,
793
+ "mean_abs_difference": 6.321395629171889e-10,
794
+ "selected_tp_rank": 0
795
+ },
796
+ {
797
+ "pp_rank": 1,
798
+ "parameter": "decoder.layers.10.self_attention.linear_attn.conv1d.weight",
799
+ "different_elements": 97,
800
+ "numel": 40960,
801
+ "max_abs_difference": 3.0517578125e-05,
802
+ "mean_abs_difference": 6.235148841682303e-09,
803
+ "selected_tp_rank": 0
804
+ },
805
+ {
806
+ "pp_rank": 1,
807
+ "parameter": "decoder.layers.10.self_attention.linear_attn.in_proj_qkv.weight",
808
+ "different_elements": 53292,
809
+ "numel": 52428800,
810
+ "max_abs_difference": 6.103515625e-05,
811
+ "mean_abs_difference": 4.463299507762031e-09,
812
+ "selected_tp_rank": 0
813
+ },
814
+ {
815
+ "pp_rank": 1,
816
+ "parameter": "decoder.layers.10.self_attention.linear_attn.in_proj_z.weight",
817
+ "different_elements": 26066,
818
+ "numel": 31457280,
819
+ "max_abs_difference": 6.103515625e-05,
820
+ "mean_abs_difference": 3.531776870957515e-09,
821
+ "selected_tp_rank": 0
822
+ },
823
+ {
824
+ "pp_rank": 1,
825
+ "parameter": "decoder.layers.10.self_attention.linear_attn.in_proj_b.weight",
826
+ "different_elements": 536,
827
+ "numel": 245760,
828
+ "max_abs_difference": 3.0517578125e-05,
829
+ "mean_abs_difference": 9.588347893441096e-09,
830
+ "selected_tp_rank": 0
831
+ },
832
+ {
833
+ "pp_rank": 1,
834
+ "parameter": "decoder.layers.10.self_attention.linear_attn.in_proj_a.weight",
835
+ "different_elements": 358,
836
+ "numel": 245760,
837
+ "max_abs_difference": 3.0517578125e-05,
838
+ "mean_abs_difference": 6.20299145381864e-09,
839
+ "selected_tp_rank": 0
840
+ },
841
+ {
842
+ "pp_rank": 1,
843
+ "parameter": "decoder.layers.10.self_attention.linear_attn.out_proj.weight",
844
+ "different_elements": 23819,
845
+ "numel": 31457280,
846
+ "max_abs_difference": 6.103515625e-05,
847
+ "mean_abs_difference": 3.1701967717623347e-09,
848
+ "selected_tp_rank": 0
849
+ },
850
+ {
851
+ "pp_rank": 1,
852
+ "parameter": "decoder.layers.10.self_attention.input_layernorm.weight",
853
+ "different_elements": 3,
854
+ "numel": 5120,
855
+ "max_abs_difference": 1.9073486328125e-06,
856
+ "mean_abs_difference": 3.978129770043637e-10,
857
+ "selected_tp_rank": 0
858
+ },
859
+ {
860
+ "pp_rank": 1,
861
+ "parameter": "decoder.layers.12.self_attention.linear_attn.conv1d.weight",
862
+ "different_elements": 74,
863
+ "numel": 40960,
864
+ "max_abs_difference": 3.0517578125e-05,
865
+ "mean_abs_difference": 5.36947464269133e-09,
866
+ "selected_tp_rank": 0
867
+ },
868
+ {
869
+ "pp_rank": 1,
870
+ "parameter": "decoder.layers.12.self_attention.linear_attn.in_proj_qkv.weight",
871
+ "different_elements": 38407,
872
+ "numel": 52428800,
873
+ "max_abs_difference": 6.103515625e-05,
874
+ "mean_abs_difference": 3.054222652565386e-09,
875
+ "selected_tp_rank": 0
876
+ },
877
+ {
878
+ "pp_rank": 1,
879
+ "parameter": "decoder.layers.12.self_attention.linear_attn.in_proj_z.weight",
880
+ "different_elements": 18097,
881
+ "numel": 31457280,
882
+ "max_abs_difference": 6.103515625e-05,
883
+ "mean_abs_difference": 2.2853243741849383e-09,
884
+ "selected_tp_rank": 0
885
+ },
886
+ {
887
+ "pp_rank": 1,
888
+ "parameter": "decoder.layers.12.self_attention.linear_attn.in_proj_b.weight",
889
+ "different_elements": 523,
890
+ "numel": 245760,
891
+ "max_abs_difference": 3.0517578125e-05,
892
+ "mean_abs_difference": 9.139074386155244e-09,
893
+ "selected_tp_rank": 0
894
+ },
895
+ {
896
+ "pp_rank": 1,
897
+ "parameter": "decoder.layers.12.self_attention.linear_attn.in_proj_a.weight",
898
+ "different_elements": 283,
899
+ "numel": 245760,
900
+ "max_abs_difference": 3.0517578125e-05,
901
+ "mean_abs_difference": 4.652696450335725e-09,
902
+ "selected_tp_rank": 0
903
+ },
904
+ {
905
+ "pp_rank": 1,
906
+ "parameter": "decoder.layers.12.self_attention.linear_attn.out_proj.weight",
907
+ "different_elements": 16178,
908
+ "numel": 31457280,
909
+ "max_abs_difference": 6.103515625e-05,
910
+ "mean_abs_difference": 1.9580772558924764e-09,
911
+ "selected_tp_rank": 0
912
+ },
913
+ {
914
+ "pp_rank": 1,
915
+ "parameter": "decoder.layers.12.self_attention.input_layernorm.weight",
916
+ "different_elements": 3,
917
+ "numel": 5120,
918
+ "max_abs_difference": 9.5367431640625e-07,
919
+ "mean_abs_difference": 2.561137135703717e-10,
920
+ "selected_tp_rank": 0
921
+ },
922
+ {
923
+ "pp_rank": 1,
924
+ "parameter": "decoder.layers.13.self_attention.linear_attn.conv1d.weight",
925
+ "different_elements": 76,
926
+ "numel": 40960,
927
+ "max_abs_difference": 3.0517578125e-05,
928
+ "mean_abs_difference": 4.16889633925166e-09,
929
+ "selected_tp_rank": 0
930
+ },
931
+ {
932
+ "pp_rank": 1,
933
+ "parameter": "decoder.layers.13.self_attention.linear_attn.in_proj_qkv.weight",
934
+ "different_elements": 31336,
935
+ "numel": 52428800,
936
+ "max_abs_difference": 6.103515625e-05,
937
+ "mean_abs_difference": 2.4398696396588093e-09,
938
+ "selected_tp_rank": 0
939
+ },
940
+ {
941
+ "pp_rank": 1,
942
+ "parameter": "decoder.layers.13.self_attention.linear_attn.in_proj_z.weight",
943
+ "different_elements": 15366,
944
+ "numel": 31457280,
945
+ "max_abs_difference": 6.103515625e-05,
946
+ "mean_abs_difference": 1.9322343725036717e-09,
947
+ "selected_tp_rank": 0
948
+ },
949
+ {
950
+ "pp_rank": 1,
951
+ "parameter": "decoder.layers.13.self_attention.linear_attn.in_proj_b.weight",
952
+ "different_elements": 279,
953
+ "numel": 245760,
954
+ "max_abs_difference": 3.0517578125e-05,
955
+ "mean_abs_difference": 5.049010542990118e-09,
956
+ "selected_tp_rank": 0
957
+ },
958
+ {
959
+ "pp_rank": 1,
960
+ "parameter": "decoder.layers.13.self_attention.linear_attn.in_proj_a.weight",
961
+ "different_elements": 221,
962
+ "numel": 245760,
963
+ "max_abs_difference": 3.0517578125e-05,
964
+ "mean_abs_difference": 3.640221235556851e-09,
965
+ "selected_tp_rank": 0
966
+ },
967
+ {
968
+ "pp_rank": 1,
969
+ "parameter": "decoder.layers.13.self_attention.linear_attn.out_proj.weight",
970
+ "different_elements": 13685,
971
+ "numel": 31457280,
972
+ "max_abs_difference": 6.103515625e-05,
973
+ "mean_abs_difference": 1.660277915149777e-09,
974
+ "selected_tp_rank": 0
975
+ },
976
+ {
977
+ "pp_rank": 1,
978
+ "parameter": "decoder.layers.14.self_attention.linear_attn.conv1d.weight",
979
+ "different_elements": 84,
980
+ "numel": 40960,
981
+ "max_abs_difference": 3.0517578125e-05,
982
+ "mean_abs_difference": 4.1558565477828324e-09,
983
+ "selected_tp_rank": 0
984
+ },
985
+ {
986
+ "pp_rank": 1,
987
+ "parameter": "decoder.layers.14.self_attention.linear_attn.in_proj_qkv.weight",
988
+ "different_elements": 40416,
989
+ "numel": 52428800,
990
+ "max_abs_difference": 6.103515625e-05,
991
+ "mean_abs_difference": 3.276448889977246e-09,
992
+ "selected_tp_rank": 0
993
+ },
994
+ {
995
+ "pp_rank": 1,
996
+ "parameter": "decoder.layers.14.self_attention.linear_attn.in_proj_z.weight",
997
+ "different_elements": 19017,
998
+ "numel": 31457280,
999
+ "max_abs_difference": 6.103515625e-05,
1000
+ "mean_abs_difference": 2.4159578781990376e-09,
1001
+ "selected_tp_rank": 0
1002
+ },
1003
+ {
1004
+ "pp_rank": 1,
1005
+ "parameter": "decoder.layers.14.self_attention.linear_attn.in_proj_b.weight",
1006
+ "different_elements": 406,
1007
+ "numel": 245760,
1008
+ "max_abs_difference": 3.0517578125e-05,
1009
+ "mean_abs_difference": 5.500220279230916e-09,
1010
+ "selected_tp_rank": 0
1011
+ },
1012
+ {
1013
+ "pp_rank": 1,
1014
+ "parameter": "decoder.layers.14.self_attention.linear_attn.in_proj_a.weight",
1015
+ "different_elements": 273,
1016
+ "numel": 245760,
1017
+ "max_abs_difference": 6.103515625e-05,
1018
+ "mean_abs_difference": 4.591712787771485e-09,
1019
+ "selected_tp_rank": 0
1020
+ },
1021
+ {
1022
+ "pp_rank": 1,
1023
+ "parameter": "decoder.layers.14.self_attention.linear_attn.out_proj.weight",
1024
+ "different_elements": 17752,
1025
+ "numel": 31457280,
1026
+ "max_abs_difference": 6.103515625e-05,
1027
+ "mean_abs_difference": 2.2255928211478704e-09,
1028
+ "selected_tp_rank": 0
1029
+ },
1030
+ {
1031
+ "pp_rank": 1,
1032
+ "parameter": "decoder.layers.14.self_attention.input_layernorm.weight",
1033
+ "different_elements": 2,
1034
+ "numel": 5120,
1035
+ "max_abs_difference": 1.1920928955078125e-07,
1036
+ "mean_abs_difference": 3.4924597935859225e-11,
1037
+ "selected_tp_rank": 0
1038
+ },
1039
+ {
1040
+ "pp_rank": 1,
1041
+ "parameter": "decoder.layers.16.self_attention.linear_attn.conv1d.weight",
1042
+ "different_elements": 57,
1043
+ "numel": 40960,
1044
+ "max_abs_difference": 3.0517578125e-05,
1045
+ "mean_abs_difference": 2.780075281094696e-09,
1046
+ "selected_tp_rank": 0
1047
+ },
1048
+ {
1049
+ "pp_rank": 1,
1050
+ "parameter": "decoder.layers.16.self_attention.linear_attn.in_proj_qkv.weight",
1051
+ "different_elements": 20152,
1052
+ "numel": 52428800,
1053
+ "max_abs_difference": 6.103515625e-05,
1054
+ "mean_abs_difference": 1.5492070959410853e-09,
1055
+ "selected_tp_rank": 0
1056
+ },
1057
+ {
1058
+ "pp_rank": 1,
1059
+ "parameter": "decoder.layers.16.self_attention.linear_attn.in_proj_z.weight",
1060
+ "different_elements": 9562,
1061
+ "numel": 31457280,
1062
+ "max_abs_difference": 6.103515625e-05,
1063
+ "mean_abs_difference": 1.1243013187112183e-09,
1064
+ "selected_tp_rank": 0
1065
+ },
1066
+ {
1067
+ "pp_rank": 1,
1068
+ "parameter": "decoder.layers.16.self_attention.linear_attn.in_proj_b.weight",
1069
+ "different_elements": 222,
1070
+ "numel": 245760,
1071
+ "max_abs_difference": 3.0517578125e-05,
1072
+ "mean_abs_difference": 4.512121343225317e-09,
1073
+ "selected_tp_rank": 0
1074
+ },
1075
+ {
1076
+ "pp_rank": 1,
1077
+ "parameter": "decoder.layers.16.self_attention.linear_attn.in_proj_a.weight",
1078
+ "different_elements": 134,
1079
+ "numel": 245760,
1080
+ "max_abs_difference": 3.0517578125e-05,
1081
+ "mean_abs_difference": 2.5048989549247835e-09,
1082
+ "selected_tp_rank": 0
1083
+ },
1084
+ {
1085
+ "pp_rank": 1,
1086
+ "parameter": "decoder.layers.16.self_attention.linear_attn.out_proj.weight",
1087
+ "different_elements": 9301,
1088
+ "numel": 31457280,
1089
+ "max_abs_difference": 6.103515625e-05,
1090
+ "mean_abs_difference": 1.0786788129379943e-09,
1091
+ "selected_tp_rank": 0
1092
+ },
1093
+ {
1094
+ "pp_rank": 1,
1095
+ "parameter": "decoder.layers.17.self_attention.linear_attn.conv1d.weight",
1096
+ "different_elements": 1,
1097
+ "numel": 40960,
1098
+ "max_abs_difference": 1.862645149230957e-09,
1099
+ "mean_abs_difference": 4.547473576627277e-14,
1100
+ "selected_tp_rank": 0
1101
+ },
1102
+ {
1103
+ "pp_rank": 1,
1104
+ "parameter": "decoder.layers.17.self_attention.linear_attn.in_proj_qkv.weight",
1105
+ "different_elements": 195,
1106
+ "numel": 52428800,
1107
+ "max_abs_difference": 3.0517578125e-05,
1108
+ "mean_abs_difference": 1.0620981524822604e-11,
1109
+ "selected_tp_rank": 0
1110
+ },
1111
+ {
1112
+ "pp_rank": 1,
1113
+ "parameter": "decoder.layers.17.self_attention.linear_attn.in_proj_z.weight",
1114
+ "different_elements": 67,
1115
+ "numel": 31457280,
1116
+ "max_abs_difference": 3.0517578125e-05,
1117
+ "mean_abs_difference": 5.297081349942001e-12,
1118
+ "selected_tp_rank": 0
1119
+ },
1120
+ {
1121
+ "pp_rank": 1,
1122
+ "parameter": "decoder.layers.17.self_attention.linear_attn.in_proj_b.weight",
1123
+ "different_elements": 1,
1124
+ "numel": 245760,
1125
+ "max_abs_difference": 1.4901161193847656e-08,
1126
+ "mean_abs_difference": 6.063298328045155e-14,
1127
+ "selected_tp_rank": 0
1128
+ },
1129
+ {
1130
+ "pp_rank": 1,
1131
+ "parameter": "decoder.layers.17.self_attention.linear_attn.out_proj.weight",
1132
+ "different_elements": 65,
1133
+ "numel": 31457280,
1134
+ "max_abs_difference": 1.52587890625e-05,
1135
+ "mean_abs_difference": 3.132455449542104e-12,
1136
+ "selected_tp_rank": 0
1137
+ },
1138
+ {
1139
+ "pp_rank": 1,
1140
+ "parameter": "decoder.layers.18.self_attention.linear_attn.conv1d.weight",
1141
+ "different_elements": 68,
1142
+ "numel": 40960,
1143
+ "max_abs_difference": 3.0517578125e-05,
1144
+ "mean_abs_difference": 4.081948556944326e-09,
1145
+ "selected_tp_rank": 0
1146
+ },
1147
+ {
1148
+ "pp_rank": 1,
1149
+ "parameter": "decoder.layers.18.self_attention.linear_attn.in_proj_qkv.weight",
1150
+ "different_elements": 36762,
1151
+ "numel": 52428800,
1152
+ "max_abs_difference": 6.103515625e-05,
1153
+ "mean_abs_difference": 2.761990192112762e-09,
1154
+ "selected_tp_rank": 0
1155
+ },
1156
+ {
1157
+ "pp_rank": 1,
1158
+ "parameter": "decoder.layers.18.self_attention.linear_attn.in_proj_z.weight",
1159
+ "different_elements": 17856,
1160
+ "numel": 31457280,
1161
+ "max_abs_difference": 6.103515625e-05,
1162
+ "mean_abs_difference": 2.11608064404345e-09,
1163
+ "selected_tp_rank": 0
1164
+ },
1165
+ {
1166
+ "pp_rank": 1,
1167
+ "parameter": "decoder.layers.18.self_attention.linear_attn.in_proj_b.weight",
1168
+ "different_elements": 374,
1169
+ "numel": 245760,
1170
+ "max_abs_difference": 3.0517578125e-05,
1171
+ "mean_abs_difference": 6.4634009255826186e-09,
1172
+ "selected_tp_rank": 0
1173
+ },
1174
+ {
1175
+ "pp_rank": 1,
1176
+ "parameter": "decoder.layers.18.self_attention.linear_attn.in_proj_a.weight",
1177
+ "different_elements": 275,
1178
+ "numel": 245760,
1179
+ "max_abs_difference": 3.0517578125e-05,
1180
+ "mean_abs_difference": 4.500869454915346e-09,
1181
+ "selected_tp_rank": 0
1182
+ },
1183
+ {
1184
+ "pp_rank": 1,
1185
+ "parameter": "decoder.layers.18.self_attention.linear_attn.out_proj.weight",
1186
+ "different_elements": 17512,
1187
+ "numel": 31457280,
1188
+ "max_abs_difference": 6.103515625e-05,
1189
+ "mean_abs_difference": 2.1355404111744747e-09,
1190
+ "selected_tp_rank": 0
1191
+ },
1192
+ {
1193
+ "pp_rank": 1,
1194
+ "parameter": "decoder.layers.20.self_attention.linear_attn.in_proj_qkv.weight",
1195
+ "different_elements": 1241,
1196
+ "numel": 52428800,
1197
+ "max_abs_difference": 6.103515625e-05,
1198
+ "mean_abs_difference": 8.9016530258057e-11,
1199
+ "selected_tp_rank": 0
1200
+ },
1201
+ {
1202
+ "pp_rank": 1,
1203
+ "parameter": "decoder.layers.20.self_attention.linear_attn.in_proj_z.weight",
1204
+ "different_elements": 567,
1205
+ "numel": 31457280,
1206
+ "max_abs_difference": 3.0517578125e-05,
1207
+ "mean_abs_difference": 6.48919945556159e-11,
1208
+ "selected_tp_rank": 0
1209
+ },
1210
+ {
1211
+ "pp_rank": 1,
1212
+ "parameter": "decoder.layers.20.self_attention.linear_attn.in_proj_b.weight",
1213
+ "different_elements": 14,
1214
+ "numel": 245760,
1215
+ "max_abs_difference": 3.0517578125e-05,
1216
+ "mean_abs_difference": 2.3502857993129567e-10,
1217
+ "selected_tp_rank": 0
1218
+ },
1219
+ {
1220
+ "pp_rank": 1,
1221
+ "parameter": "decoder.layers.20.self_attention.linear_attn.in_proj_a.weight",
1222
+ "different_elements": 23,
1223
+ "numel": 245760,
1224
+ "max_abs_difference": 3.0517578125e-05,
1225
+ "mean_abs_difference": 2.7907845479013815e-10,
1226
+ "selected_tp_rank": 0
1227
+ },
1228
+ {
1229
+ "pp_rank": 1,
1230
+ "parameter": "decoder.layers.20.self_attention.linear_attn.out_proj.weight",
1231
+ "different_elements": 536,
1232
+ "numel": 31457280,
1233
+ "max_abs_difference": 6.103515625e-05,
1234
+ "mean_abs_difference": 5.636174513212744e-11,
1235
+ "selected_tp_rank": 0
1236
+ },
1237
+ {
1238
+ "pp_rank": 1,
1239
+ "parameter": "decoder.layers.21.self_attention.linear_attn.conv1d.weight",
1240
+ "different_elements": 6,
1241
+ "numel": 40960,
1242
+ "max_abs_difference": 1.52587890625e-05,
1243
+ "mean_abs_difference": 5.768924782323381e-10,
1244
+ "selected_tp_rank": 0
1245
+ },
1246
+ {
1247
+ "pp_rank": 1,
1248
+ "parameter": "decoder.layers.21.self_attention.linear_attn.in_proj_qkv.weight",
1249
+ "different_elements": 2887,
1250
+ "numel": 52428800,
1251
+ "max_abs_difference": 6.103515625e-05,
1252
+ "mean_abs_difference": 2.173750207612457e-10,
1253
+ "selected_tp_rank": 0
1254
+ },
1255
+ {
1256
+ "pp_rank": 1,
1257
+ "parameter": "decoder.layers.21.self_attention.linear_attn.in_proj_z.weight",
1258
+ "different_elements": 932,
1259
+ "numel": 31457280,
1260
+ "max_abs_difference": 6.103515625e-05,
1261
+ "mean_abs_difference": 1.1033134822424628e-10,
1262
+ "selected_tp_rank": 0
1263
+ },
1264
+ {
1265
+ "pp_rank": 1,
1266
+ "parameter": "decoder.layers.21.self_attention.linear_attn.in_proj_b.weight",
1267
+ "different_elements": 22,
1268
+ "numel": 245760,
1269
+ "max_abs_difference": 3.0517578125e-05,
1270
+ "mean_abs_difference": 3.046504160053587e-10,
1271
+ "selected_tp_rank": 0
1272
+ },
1273
+ {
1274
+ "pp_rank": 1,
1275
+ "parameter": "decoder.layers.21.self_attention.linear_attn.in_proj_a.weight",
1276
+ "different_elements": 19,
1277
+ "numel": 245760,
1278
+ "max_abs_difference": 7.62939453125e-06,
1279
+ "mean_abs_difference": 7.048583938740194e-11,
1280
+ "selected_tp_rank": 0
1281
+ },
1282
+ {
1283
+ "pp_rank": 1,
1284
+ "parameter": "decoder.layers.21.self_attention.linear_attn.out_proj.weight",
1285
+ "different_elements": 993,
1286
+ "numel": 31457280,
1287
+ "max_abs_difference": 3.0517578125e-05,
1288
+ "mean_abs_difference": 1.1451626452663177e-10,
1289
+ "selected_tp_rank": 0
1290
+ },
1291
+ {
1292
+ "pp_rank": 1,
1293
+ "parameter": "decoder.layers.22.self_attention.linear_attn.conv1d.weight",
1294
+ "different_elements": 4,
1295
+ "numel": 40960,
1296
+ "max_abs_difference": 3.814697265625e-06,
1297
+ "mean_abs_difference": 1.9244908444626674e-10,
1298
+ "selected_tp_rank": 0
1299
+ },
1300
+ {
1301
+ "pp_rank": 1,
1302
+ "parameter": "decoder.layers.22.self_attention.linear_attn.in_proj_qkv.weight",
1303
+ "different_elements": 2806,
1304
+ "numel": 52428800,
1305
+ "max_abs_difference": 3.0517578125e-05,
1306
+ "mean_abs_difference": 1.8561535641836713e-10,
1307
+ "selected_tp_rank": 0
1308
+ },
1309
+ {
1310
+ "pp_rank": 1,
1311
+ "parameter": "decoder.layers.22.self_attention.linear_attn.in_proj_z.weight",
1312
+ "different_elements": 1136,
1313
+ "numel": 31457280,
1314
+ "max_abs_difference": 3.0517578125e-05,
1315
+ "mean_abs_difference": 1.3165928069991395e-10,
1316
+ "selected_tp_rank": 0
1317
+ },
1318
+ {
1319
+ "pp_rank": 1,
1320
+ "parameter": "decoder.layers.22.self_attention.linear_attn.in_proj_b.weight",
1321
+ "different_elements": 22,
1322
+ "numel": 245760,
1323
+ "max_abs_difference": 3.0517578125e-05,
1324
+ "mean_abs_difference": 3.7560615728793323e-10,
1325
+ "selected_tp_rank": 0
1326
+ },
1327
+ {
1328
+ "pp_rank": 1,
1329
+ "parameter": "decoder.layers.22.self_attention.linear_attn.in_proj_a.weight",
1330
+ "different_elements": 16,
1331
+ "numel": 245760,
1332
+ "max_abs_difference": 3.0517578125e-05,
1333
+ "mean_abs_difference": 2.761945949725231e-10,
1334
+ "selected_tp_rank": 0
1335
+ },
1336
+ {
1337
+ "pp_rank": 1,
1338
+ "parameter": "decoder.layers.22.self_attention.linear_attn.out_proj.weight",
1339
+ "different_elements": 1198,
1340
+ "numel": 31457280,
1341
+ "max_abs_difference": 3.0517578125e-05,
1342
+ "mean_abs_difference": 1.337261967826464e-10,
1343
+ "selected_tp_rank": 0
1344
+ },
1345
+ {
1346
+ "pp_rank": 1,
1347
+ "parameter": "decoder.layers.24.self_attention.linear_attn.conv1d.weight",
1348
+ "different_elements": 7,
1349
+ "numel": 40960,
1350
+ "max_abs_difference": 1.52587890625e-05,
1351
+ "mean_abs_difference": 8.898496384190935e-10,
1352
+ "selected_tp_rank": 0
1353
+ },
1354
+ {
1355
+ "pp_rank": 1,
1356
+ "parameter": "decoder.layers.24.self_attention.linear_attn.in_proj_qkv.weight",
1357
+ "different_elements": 10669,
1358
+ "numel": 52428800,
1359
+ "max_abs_difference": 6.103515625e-05,
1360
+ "mean_abs_difference": 9.136215117777624e-10,
1361
+ "selected_tp_rank": 0
1362
+ },
1363
+ {
1364
+ "pp_rank": 1,
1365
+ "parameter": "decoder.layers.24.self_attention.linear_attn.in_proj_z.weight",
1366
+ "different_elements": 4606,
1367
+ "numel": 31457280,
1368
+ "max_abs_difference": 6.103515625e-05,
1369
+ "mean_abs_difference": 5.804748903770474e-10,
1370
+ "selected_tp_rank": 0
1371
+ },
1372
+ {
1373
+ "pp_rank": 1,
1374
+ "parameter": "decoder.layers.24.self_attention.linear_attn.in_proj_b.weight",
1375
+ "different_elements": 138,
1376
+ "numel": 245760,
1377
+ "max_abs_difference": 3.0517578125e-05,
1378
+ "mean_abs_difference": 2.1675616856953184e-09,
1379
+ "selected_tp_rank": 0
1380
+ },
1381
+ {
1382
+ "pp_rank": 1,
1383
+ "parameter": "decoder.layers.24.self_attention.linear_attn.in_proj_a.weight",
1384
+ "different_elements": 84,
1385
+ "numel": 245760,
1386
+ "max_abs_difference": 3.0517578125e-05,
1387
+ "mean_abs_difference": 1.714393738083686e-09,
1388
+ "selected_tp_rank": 0
1389
+ },
1390
+ {
1391
+ "pp_rank": 1,
1392
+ "parameter": "decoder.layers.24.self_attention.linear_attn.out_proj.weight",
1393
+ "different_elements": 4876,
1394
+ "numel": 31457280,
1395
+ "max_abs_difference": 6.103515625e-05,
1396
+ "mean_abs_difference": 6.426523757596669e-10,
1397
+ "selected_tp_rank": 0
1398
+ },
1399
+ {
1400
+ "pp_rank": 1,
1401
+ "parameter": "decoder.layers.25.self_attention.linear_attn.conv1d.weight",
1402
+ "different_elements": 1,
1403
+ "numel": 40960,
1404
+ "max_abs_difference": 3.725290298461914e-09,
1405
+ "mean_abs_difference": 9.094947153254554e-14,
1406
+ "selected_tp_rank": 0
1407
+ },
1408
+ {
1409
+ "pp_rank": 1,
1410
+ "parameter": "decoder.layers.25.self_attention.linear_attn.in_proj_qkv.weight",
1411
+ "different_elements": 1653,
1412
+ "numel": 52428800,
1413
+ "max_abs_difference": 6.103515625e-05,
1414
+ "mean_abs_difference": 1.0847993336948747e-10,
1415
+ "selected_tp_rank": 0
1416
+ },
1417
+ {
1418
+ "pp_rank": 1,
1419
+ "parameter": "decoder.layers.25.self_attention.linear_attn.in_proj_z.weight",
1420
+ "different_elements": 463,
1421
+ "numel": 31457280,
1422
+ "max_abs_difference": 3.0517578125e-05,
1423
+ "mean_abs_difference": 4.799040331793236e-11,
1424
+ "selected_tp_rank": 0
1425
+ },
1426
+ {
1427
+ "pp_rank": 1,
1428
+ "parameter": "decoder.layers.25.self_attention.linear_attn.in_proj_b.weight",
1429
+ "different_elements": 22,
1430
+ "numel": 245760,
1431
+ "max_abs_difference": 1.52587890625e-05,
1432
+ "mean_abs_difference": 1.6037422778669708e-10,
1433
+ "selected_tp_rank": 0
1434
+ },
1435
+ {
1436
+ "pp_rank": 1,
1437
+ "parameter": "decoder.layers.25.self_attention.linear_attn.in_proj_a.weight",
1438
+ "different_elements": 16,
1439
+ "numel": 245760,
1440
+ "max_abs_difference": 3.0517578125e-05,
1441
+ "mean_abs_difference": 2.5112664725490674e-10,
1442
+ "selected_tp_rank": 0
1443
+ },
1444
+ {
1445
+ "pp_rank": 1,
1446
+ "parameter": "decoder.layers.25.self_attention.linear_attn.out_proj.weight",
1447
+ "different_elements": 527,
1448
+ "numel": 31457280,
1449
+ "max_abs_difference": 3.0517578125e-05,
1450
+ "mean_abs_difference": 5.016221119036324e-11,
1451
+ "selected_tp_rank": 0
1452
+ },
1453
+ {
1454
+ "pp_rank": 1,
1455
+ "parameter": "decoder.layers.26.self_attention.linear_attn.conv1d.weight",
1456
+ "different_elements": 8,
1457
+ "numel": 40960,
1458
+ "max_abs_difference": 7.62939453125e-06,
1459
+ "mean_abs_difference": 2.3937901660886496e-10,
1460
+ "selected_tp_rank": 0
1461
+ },
1462
+ {
1463
+ "pp_rank": 1,
1464
+ "parameter": "decoder.layers.26.self_attention.linear_attn.in_proj_qkv.weight",
1465
+ "different_elements": 4895,
1466
+ "numel": 52428800,
1467
+ "max_abs_difference": 6.103515625e-05,
1468
+ "mean_abs_difference": 3.8790440304303786e-10,
1469
+ "selected_tp_rank": 0
1470
+ },
1471
+ {
1472
+ "pp_rank": 1,
1473
+ "parameter": "decoder.layers.26.self_attention.linear_attn.in_proj_z.weight",
1474
+ "different_elements": 2237,
1475
+ "numel": 31457280,
1476
+ "max_abs_difference": 3.0517578125e-05,
1477
+ "mean_abs_difference": 2.649653274566788e-10,
1478
+ "selected_tp_rank": 0
1479
+ },
1480
+ {
1481
+ "pp_rank": 1,
1482
+ "parameter": "decoder.layers.26.self_attention.linear_attn.in_proj_b.weight",
1483
+ "different_elements": 73,
1484
+ "numel": 245760,
1485
+ "max_abs_difference": 3.0517578125e-05,
1486
+ "mean_abs_difference": 9.704839154522915e-10,
1487
+ "selected_tp_rank": 0
1488
+ },
1489
+ {
1490
+ "pp_rank": 1,
1491
+ "parameter": "decoder.layers.26.self_attention.linear_attn.in_proj_a.weight",
1492
+ "different_elements": 42,
1493
+ "numel": 245760,
1494
+ "max_abs_difference": 3.0517578125e-05,
1495
+ "mean_abs_difference": 5.797507474092356e-10,
1496
+ "selected_tp_rank": 0
1497
+ },
1498
+ {
1499
+ "pp_rank": 1,
1500
+ "parameter": "decoder.layers.26.self_attention.linear_attn.out_proj.weight",
1501
+ "different_elements": 2269,
1502
+ "numel": 31457280,
1503
+ "max_abs_difference": 6.103515625e-05,
1504
+ "mean_abs_difference": 3.2059588317423504e-10,
1505
+ "selected_tp_rank": 0
1506
+ },
1507
+ {
1508
+ "pp_rank": 1,
1509
+ "parameter": "decoder.layers.28.self_attention.linear_attn.conv1d.weight",
1510
+ "different_elements": 2,
1511
+ "numel": 40960,
1512
+ "max_abs_difference": 7.62939453125e-06,
1513
+ "mean_abs_difference": 2.3283064365386963e-10,
1514
+ "selected_tp_rank": 0
1515
+ },
1516
+ {
1517
+ "pp_rank": 1,
1518
+ "parameter": "decoder.layers.28.self_attention.linear_attn.in_proj_qkv.weight",
1519
+ "different_elements": 1106,
1520
+ "numel": 52428800,
1521
+ "max_abs_difference": 3.0517578125e-05,
1522
+ "mean_abs_difference": 8.712219140560862e-11,
1523
+ "selected_tp_rank": 0
1524
+ },
1525
+ {
1526
+ "pp_rank": 1,
1527
+ "parameter": "decoder.layers.28.self_attention.linear_attn.in_proj_z.weight",
1528
+ "different_elements": 285,
1529
+ "numel": 31457280,
1530
+ "max_abs_difference": 3.0517578125e-05,
1531
+ "mean_abs_difference": 2.5129877345708707e-11,
1532
+ "selected_tp_rank": 0
1533
+ },
1534
+ {
1535
+ "pp_rank": 1,
1536
+ "parameter": "decoder.layers.28.self_attention.linear_attn.in_proj_b.weight",
1537
+ "different_elements": 19,
1538
+ "numel": 245760,
1539
+ "max_abs_difference": 3.0517578125e-05,
1540
+ "mean_abs_difference": 3.697285533288408e-10,
1541
+ "selected_tp_rank": 0
1542
+ },
1543
+ {
1544
+ "pp_rank": 1,
1545
+ "parameter": "decoder.layers.28.self_attention.linear_attn.in_proj_a.weight",
1546
+ "different_elements": 12,
1547
+ "numel": 245760,
1548
+ "max_abs_difference": 1.52587890625e-05,
1549
+ "mean_abs_difference": 2.2749493955309674e-10,
1550
+ "selected_tp_rank": 0
1551
+ },
1552
+ {
1553
+ "pp_rank": 1,
1554
+ "parameter": "decoder.layers.28.self_attention.linear_attn.out_proj.weight",
1555
+ "different_elements": 333,
1556
+ "numel": 31457280,
1557
+ "max_abs_difference": 3.0517578125e-05,
1558
+ "mean_abs_difference": 3.7015664838824236e-11,
1559
+ "selected_tp_rank": 0
1560
+ },
1561
+ {
1562
+ "pp_rank": 1,
1563
+ "parameter": "decoder.layers.30.self_attention.linear_attn.conv1d.weight",
1564
+ "different_elements": 4,
1565
+ "numel": 40960,
1566
+ "max_abs_difference": 7.62939453125e-06,
1567
+ "mean_abs_difference": 3.78531705980123e-10,
1568
+ "selected_tp_rank": 0
1569
+ },
1570
+ {
1571
+ "pp_rank": 1,
1572
+ "parameter": "decoder.layers.30.self_attention.linear_attn.in_proj_qkv.weight",
1573
+ "different_elements": 3408,
1574
+ "numel": 52428800,
1575
+ "max_abs_difference": 6.103515625e-05,
1576
+ "mean_abs_difference": 2.462122727919791e-10,
1577
+ "selected_tp_rank": 0
1578
+ },
1579
+ {
1580
+ "pp_rank": 1,
1581
+ "parameter": "decoder.layers.30.self_attention.linear_attn.in_proj_z.weight",
1582
+ "different_elements": 1602,
1583
+ "numel": 31457280,
1584
+ "max_abs_difference": 6.103515625e-05,
1585
+ "mean_abs_difference": 1.9702034448343397e-10,
1586
+ "selected_tp_rank": 0
1587
+ },
1588
+ {
1589
+ "pp_rank": 1,
1590
+ "parameter": "decoder.layers.30.self_attention.linear_attn.in_proj_b.weight",
1591
+ "different_elements": 63,
1592
+ "numel": 245760,
1593
+ "max_abs_difference": 3.0517578125e-05,
1594
+ "mean_abs_difference": 8.512794358317421e-10,
1595
+ "selected_tp_rank": 0
1596
+ },
1597
+ {
1598
+ "pp_rank": 1,
1599
+ "parameter": "decoder.layers.30.self_attention.linear_attn.in_proj_a.weight",
1600
+ "different_elements": 16,
1601
+ "numel": 245760,
1602
+ "max_abs_difference": 3.0517578125e-05,
1603
+ "mean_abs_difference": 3.135331438919309e-10,
1604
+ "selected_tp_rank": 0
1605
+ },
1606
+ {
1607
+ "pp_rank": 1,
1608
+ "parameter": "decoder.layers.30.self_attention.linear_attn.out_proj.weight",
1609
+ "different_elements": 1583,
1610
+ "numel": 31457280,
1611
+ "max_abs_difference": 3.0517578125e-05,
1612
+ "mean_abs_difference": 2.1037689645897473e-10,
1613
+ "selected_tp_rank": 0
1614
+ }
1615
+ ],
1616
+ "replica_selection": "TP rank 0 for unsharded/duplicated tensors, matching upstream HF export convention; checkpoint replicas are not all identical",
1617
+ "output_shards": [
1618
+ {
1619
+ "file": "model-00001-of-00018.safetensors",
1620
+ "bytes": 3966730552,
1621
+ "sha256": "b9dcb3e15c00823d46d92f645f8a7324cdd1163c39ff6c04de48fa6e478952a0"
1622
+ },
1623
+ {
1624
+ "file": "model-00002-of-00018.safetensors",
1625
+ "bytes": 3043080328,
1626
+ "sha256": "ecf7312565a089592217b58d761e714f3b3d85d4ec578a15d901210a9ba14773"
1627
+ },
1628
+ {
1629
+ "file": "model-00003-of-00018.safetensors",
1630
+ "bytes": 2542796952,
1631
+ "sha256": "cec21a53517089f1b9708a77936465420f266482a45da26fd4c0dd596d609d08"
1632
+ },
1633
+ {
1634
+ "file": "model-00004-of-00018.safetensors",
1635
+ "bytes": 3988973152,
1636
+ "sha256": "3a1a2bfcfd177e62fbe78f08b2b1d3edac59c495397aa64d0d1c41b727a94c73"
1637
+ },
1638
+ {
1639
+ "file": "model-00005-of-00018.safetensors",
1640
+ "bytes": 2099339864,
1641
+ "sha256": "389931270466dcdbc9125bda7572cc7712f164e1fe774e7377524abca899b0c2"
1642
+ },
1643
+ {
1644
+ "file": "model-00006-of-00018.safetensors",
1645
+ "bytes": 3979553696,
1646
+ "sha256": "da6f16eabe6be8552af22ed5ad171ae97469fa5bb5d5e613982460e43265debf"
1647
+ },
1648
+ {
1649
+ "file": "model-00007-of-00018.safetensors",
1650
+ "bytes": 2108759344,
1651
+ "sha256": "b4b9f91766d367f0260882d97b5c0862863c393ef381f54d48dfda3d362072a9"
1652
+ },
1653
+ {
1654
+ "file": "model-00008-of-00018.safetensors",
1655
+ "bytes": 3979553696,
1656
+ "sha256": "df58405d10cb8a199e3e5a01346c6b6ebf2a1bed05d6b11e14f856543b28e55c"
1657
+ },
1658
+ {
1659
+ "file": "model-00009-of-00018.safetensors",
1660
+ "bytes": 2108759344,
1661
+ "sha256": "84ba24193b7d0885ef504ce5806d3d4ebbe5506fdfd5919f4837ebc9479611e6"
1662
+ },
1663
+ {
1664
+ "file": "model-00010-of-00018.safetensors",
1665
+ "bytes": 3979553696,
1666
+ "sha256": "aacd71e34312a13bf487a06c9c54205ad31f72a8788b0a550154a04565820a97"
1667
+ },
1668
+ {
1669
+ "file": "model-00011-of-00018.safetensors",
1670
+ "bytes": 2108759344,
1671
+ "sha256": "6d967f431bdfde43c38839188023c5507d5d0dce6a1daa6bbbf8605ecb3da9aa"
1672
+ },
1673
+ {
1674
+ "file": "model-00012-of-00018.safetensors",
1675
+ "bytes": 3979553696,
1676
+ "sha256": "c36593694bc3f250449ae2196366e6cf788c122dc9a645fbe6ef116746fc5d73"
1677
+ },
1678
+ {
1679
+ "file": "model-00013-of-00018.safetensors",
1680
+ "bytes": 2108759344,
1681
+ "sha256": "b50b55cffe8be40b6eb90c43b1d3cec7672d7f3a86b2efd436db5a83619686d2"
1682
+ },
1683
+ {
1684
+ "file": "model-00014-of-00018.safetensors",
1685
+ "bytes": 3979553696,
1686
+ "sha256": "27b627a940af6ba6f26d240767ca75a4774d02a91255c3a2274dd316e364d1d1"
1687
+ },
1688
+ {
1689
+ "file": "model-00015-of-00018.safetensors",
1690
+ "bytes": 2108759344,
1691
+ "sha256": "f9203192d4a759f6fc58c32eebda8a2f00f4a15dc183f06ddc263de49595ade5"
1692
+ },
1693
+ {
1694
+ "file": "model-00016-of-00018.safetensors",
1695
+ "bytes": 3979564040,
1696
+ "sha256": "cf3333f84d3783148a58b24612998ba24b8c60dc69f77e14d86e66ef958c23de"
1697
+ },
1698
+ {
1699
+ "file": "model-00017-of-00018.safetensors",
1700
+ "bytes": 2108759344,
1701
+ "sha256": "e446dd9c79fc0255d9b2181a9a85debb2b5b51b2eb31177c4c1a7daa198221f7"
1702
+ },
1703
+ {
1704
+ "file": "model-00018-of-00018.safetensors",
1705
+ "bytes": 3392197344,
1706
+ "sha256": "2898c2335e6ce74567af2de8373d7468c403fb5e69f4b00f5f7b83fb815d71be"
1707
+ }
1708
+ ],
1709
+ "asset_sha256": {
1710
+ "chat_template.jinja": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041",
1711
+ "generation_config.json": "e70c136c1b78ddc1fb0905bac8e733a4dc448d4f852a5dd75143fffc70be550e",
1712
+ "config.json": "191e0af232104ed8b65258cf3fb2b842e288008baca7633c11b82a1ac7203aab",
1713
+ "merges.txt": "a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d",
1714
+ "preprocessor_config.json": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516",
1715
+ "tokenizer_config.json": "b11349aafa7cdc6a320767cf7ceb29ed82f7eda5d65e8e0819e76f0ce947bf27",
1716
+ "video_preprocessor_config.json": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13",
1717
+ "vocab.json": "ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003",
1718
+ "tokenizer.json": "0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3"
1719
+ },
1720
+ "parent_comparison": {
1721
+ "sampled_tensors": 851,
1722
+ "sampled_tensors_different_from_parent": 666
1723
+ },
1724
+ "all_saved_tensors_equal_to_export": true,
1725
+ "all_text_tensors_finite": true,
1726
+ "gated_qkv_roundtrip": true,
1727
+ "total_tensor_bytes": 55562855904,
1728
+ "finished_at": "2026-09-16T08:48:35.548578+00:00",
1729
+ "elapsed_seconds": 180.26648061099695
1730
+ }
compatibility/architecture-compatibility-report.md ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Swift 1.5 MLX compatibility: validated complete checkpoint
2
+
3
+ Source: `ukisai/Swift-1.5-Qwen3.8-27b` at `00ccd14e006897d28cb0ed5bf26390e60d274251`.
4
+ All 18 BF16 shards and original runtime assets passed full SHA256 verification.
5
+ Source: 55563006776 weight bytes, 1199 BF16 tensors.
6
+
7
+ The official Qwen loader dropped 333 vision and 15 MTP tensors. The isolated patch
8
+ adds `mlx_lm/models/qwen3_5_full.py` with a real vision encoder and explicit MTP
9
+ module, plus strict dispatch/index checks and config/asset preservation in
10
+ `mlx_lm/utils.py`. `tests/test_qwen3_5_full.py` exercises mapping failures,
11
+ Transformers numerical agreement, cache behavior, native nonquantized conversion,
12
+ and exact-target quantized saving/reloading.
13
+
14
+ All 16 nonquantized architecture tests and the additional fixed affine test passed
15
+ on Linux CPU. The real source loaded strictly with 851 text, 333 vision and 15 MTP
16
+ tensors. There are zero ignored or unexplained tensors. The native quant has
17
+ 2379 saved tensors because quantized weights have scales and biases.
18
+ All 609 BF16 remainder tensors equal the source values.
19
+
20
+ Text generation passed. Vision encoder execution and an MTP step using real text
21
+ hidden states passed. Image/video text integration and speculative generation are
22
+ not implemented. The original tokenizer, chat template, context, processor,
23
+ untied output head/shared MTP embeddings, norms and gating configuration are retained.
24
+
25
+ CPU runtime checks promote only in-memory floating values to FP32. The original
26
+ Linux BF16 QMM kernel produced 256 when summing 8192 exact ones; FP32 returned
27
+ 8192. This reproducible backend issue and the runtime workaround are recorded in
28
+ cpu-quantized-matmul-diagnostic.json and ../USAGE.md. Stored weights were not changed.
29
+
30
+ Apple Silicon Metal checks passed for all 2379 native parameter headers and
31
+ real packed Q4 samples from text, vision and MTP. Both native BF16 Metal and
32
+ FP32 Metal executions matched their references. Only small actual samples were
33
+ evaluated on the 16 GiB Mac; no full 27B Mac generation is claimed.
34
+
35
+ The original source was read without modification. The source tensor layout
36
+ transposes are recorded individually in `quant-tensor-mapping-manifest.json`.
37
+ No custom quantization algorithm or alternate quantization configuration was used.
38
+
39
+ Patch SHA256: `f6f1d0bdafa45863bfbf93dac0398c481c993ea04fdf38b9bae98c643f89eaec`. Apply to official MLX-LM commit
40
+ `c69d1288440a0dc4e6401fc417098b07598dccd5`; see `../USAGE.md`.
compatibility/aws-actual-source-structural-results.json ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "PASS_ACTUAL_SOURCE_LAZY_STRUCTURAL_LOAD",
3
+ "recorded_at": "2026-09-21T17:51:07.992995+00:00",
4
+ "platform": "Linux x86_64",
5
+ "source": "<SOURCE_MODEL_DIR>",
6
+ "source_repo": "ukisai/Swift-1.5-Qwen3.8-27b",
7
+ "source_revision": "00ccd14e006897d28cb0ed5bf26390e60d274251",
8
+ "source_verification": "compatibility/aws-source-verification.json",
9
+ "source_verification_sha256": "ad944ae15bedaaec80abf9cd5e9e945f478397328cf3145f19517ff96295ad9c",
10
+ "source_tensors": 1199,
11
+ "mapped_tensors": 1199,
12
+ "categories": {
13
+ "text": 851,
14
+ "MTP": 15,
15
+ "vision": 333
16
+ },
17
+ "ignored_tensors": 0,
18
+ "unexplained_tensors": 0,
19
+ "weight_backing": "actual verified source safetensors; lazy loading",
20
+ "full_parameter_evaluation": false,
21
+ "tokenizer": "Qwen2Tokenizer",
22
+ "chat_templates": [
23
+ {
24
+ "options": {
25
+ "enable_thinking": false
26
+ },
27
+ "tokens": 15,
28
+ "rendered": "<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\n"
29
+ },
30
+ {
31
+ "options": {
32
+ "reasoning_effort": "low"
33
+ },
34
+ "tokens": 43,
35
+ "rendered": "<|im_start|>system\nReasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.<|im_end|>\n<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n"
36
+ },
37
+ {
38
+ "options": {
39
+ "reasoning_effort": "xhigh"
40
+ },
41
+ "tokens": 55,
42
+ "rendered": "<|im_start|>system\nReasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.<|im_end|>\n<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n"
43
+ }
44
+ ],
45
+ "processor": "Qwen3VLProcessor",
46
+ "generation": "NOT_RUN",
47
+ "mtp_runtime": "component-only; no integrated speculative decoding",
48
+ "vision_runtime": "encoder-only; no integrated multimodal generation",
49
+ "quantization_executed": false,
50
+ "elapsed_seconds": 0.9277446469999973,
51
+ "mlx_active_memory_bytes": 8,
52
+ "process_peak_rss_bytes": 797855744
53
+ }
compatibility/aws-source-verification.json ADDED
The diff for this file is too large to render. See raw diff
 
compatibility/compatibility-tests-linux.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ ................ [100%]
2
+ 16 passed in 6.44s
compatibility/conversion-command.json ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "command": [
3
+ "mlx_lm.convert",
4
+ "--hf-path",
5
+ "<SOURCE_MODEL_DIR>",
6
+ "--mlx-path",
7
+ "<OUTPUT_MODEL_DIR>",
8
+ "--quantize",
9
+ "--q-mode",
10
+ "affine",
11
+ "--q-bits",
12
+ "4",
13
+ "--q-group-size",
14
+ "64"
15
+ ],
16
+ "started_at": "2026-09-21T17:53:29.797067+00:00",
17
+ "source_repo": "ukisai/Swift-1.5-Qwen3.8-27b",
18
+ "source_revision": "00ccd14e006897d28cb0ed5bf26390e60d274251",
19
+ "source_manifest_sha256": "0a00065b88ab003281853a7fb9bd5ce0086bc3781b36136d8c39da19933923ae",
20
+ "quantization": {
21
+ "mode": "affine",
22
+ "bits": 4,
23
+ "group_size": 64
24
+ }
25
+ }
compatibility/conversion-result.json ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "command": [
3
+ "mlx_lm.convert",
4
+ "--hf-path",
5
+ "<SOURCE_MODEL_DIR>",
6
+ "--mlx-path",
7
+ "<OUTPUT_MODEL_DIR>",
8
+ "--quantize",
9
+ "--q-mode",
10
+ "affine",
11
+ "--q-bits",
12
+ "4",
13
+ "--q-group-size",
14
+ "64"
15
+ ],
16
+ "started_at": "2026-09-21T17:53:29.797067+00:00",
17
+ "source_repo": "ukisai/Swift-1.5-Qwen3.8-27b",
18
+ "source_revision": "00ccd14e006897d28cb0ed5bf26390e60d274251",
19
+ "source_manifest_sha256": "0a00065b88ab003281853a7fb9bd5ce0086bc3781b36136d8c39da19933923ae",
20
+ "quantization": {
21
+ "mode": "affine",
22
+ "bits": 4,
23
+ "group_size": 64
24
+ },
25
+ "returncode": 0,
26
+ "elapsed_seconds": 120.018799242,
27
+ "finished_at": "2026-09-21T17:55:29.816005+00:00"
28
+ }
compatibility/conversion.log ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ [INFO] Loading
2
+ [INFO] Using dtype: bfloat16
3
+ [INFO] Quantizing
4
+ [INFO] Quantized model with 4.557 bits per weight.
compatibility/cpu-quantized-matmul-diagnostic.json ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "test": "sum of 8192 exact ones with official affine/4-bit/group-64 quantized matmul",
3
+ "bf16_cpu_result": [
4
+ [
5
+ 256.0
6
+ ]
7
+ ],
8
+ "float32_cpu_result": [
9
+ [
10
+ 8192.0
11
+ ]
12
+ ],
13
+ "reference": [
14
+ [
15
+ 8192.0
16
+ ]
17
+ ],
18
+ "stored_packed_weights_unchanged": true
19
+ }
compatibility/environment-linux.json ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "platform": "Linux x86_64",
3
+ "python": "3.12.3",
4
+ "device": "Device(cpu, 0)",
5
+ "cpu_test": [
6
+ 1,
7
+ 4,
8
+ 9
9
+ ],
10
+ "packages": {
11
+ "mlx": "0.32.2",
12
+ "mlx-cpu": "0.32.2",
13
+ "mlx-lm": "0.32.0",
14
+ "transformers": "5.14.1",
15
+ "huggingface_hub": "1.31.0",
16
+ "torch": "2.11.0+cpu",
17
+ "torchvision": "0.26.0+cpu",
18
+ "safetensors": "0.8.0"
19
+ }
20
+ }
compatibility/fixed-affine-test.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ . [100%]
2
+ 1 passed, 16 deselected in 2.48s
compatibility/mac-check/checkpoint-headers.json ADDED
The diff for this file is too large to render. See raw diff
 
compatibility/mac-check/config.json ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForConditionalGeneration"
4
+ ],
5
+ "eos_token_id": [
6
+ 248046,
7
+ 248044
8
+ ],
9
+ "image_token_id": 248056,
10
+ "language_model_only": false,
11
+ "model_type": "qwen3_5",
12
+ "quantization": {
13
+ "group_size": 64,
14
+ "bits": 4,
15
+ "mode": "affine"
16
+ },
17
+ "quantization_config": {
18
+ "group_size": 64,
19
+ "bits": 4,
20
+ "mode": "affine"
21
+ },
22
+ "text_config": {
23
+ "attention_bias": false,
24
+ "attention_dropout": 0.0,
25
+ "attn_output_gate": true,
26
+ "bos_token_id": 248044,
27
+ "dtype": "bfloat16",
28
+ "eos_token_id": 248044,
29
+ "full_attention_interval": 4,
30
+ "head_dim": 256,
31
+ "hidden_act": "silu",
32
+ "hidden_size": 5120,
33
+ "initializer_range": 0.02,
34
+ "intermediate_size": 17408,
35
+ "layer_types": [
36
+ "linear_attention",
37
+ "linear_attention",
38
+ "linear_attention",
39
+ "full_attention",
40
+ "linear_attention",
41
+ "linear_attention",
42
+ "linear_attention",
43
+ "full_attention",
44
+ "linear_attention",
45
+ "linear_attention",
46
+ "linear_attention",
47
+ "full_attention",
48
+ "linear_attention",
49
+ "linear_attention",
50
+ "linear_attention",
51
+ "full_attention",
52
+ "linear_attention",
53
+ "linear_attention",
54
+ "linear_attention",
55
+ "full_attention",
56
+ "linear_attention",
57
+ "linear_attention",
58
+ "linear_attention",
59
+ "full_attention",
60
+ "linear_attention",
61
+ "linear_attention",
62
+ "linear_attention",
63
+ "full_attention",
64
+ "linear_attention",
65
+ "linear_attention",
66
+ "linear_attention",
67
+ "full_attention",
68
+ "linear_attention",
69
+ "linear_attention",
70
+ "linear_attention",
71
+ "full_attention",
72
+ "linear_attention",
73
+ "linear_attention",
74
+ "linear_attention",
75
+ "full_attention",
76
+ "linear_attention",
77
+ "linear_attention",
78
+ "linear_attention",
79
+ "full_attention",
80
+ "linear_attention",
81
+ "linear_attention",
82
+ "linear_attention",
83
+ "full_attention",
84
+ "linear_attention",
85
+ "linear_attention",
86
+ "linear_attention",
87
+ "full_attention",
88
+ "linear_attention",
89
+ "linear_attention",
90
+ "linear_attention",
91
+ "full_attention",
92
+ "linear_attention",
93
+ "linear_attention",
94
+ "linear_attention",
95
+ "full_attention",
96
+ "linear_attention",
97
+ "linear_attention",
98
+ "linear_attention",
99
+ "full_attention"
100
+ ],
101
+ "linear_conv_kernel_dim": 4,
102
+ "linear_key_head_dim": 128,
103
+ "linear_num_key_heads": 16,
104
+ "linear_num_value_heads": 48,
105
+ "linear_value_head_dim": 128,
106
+ "mamba_ssm_dtype": "float32",
107
+ "max_position_embeddings": 262144,
108
+ "model_type": "qwen3_5_text",
109
+ "mtp_num_hidden_layers": 1,
110
+ "mtp_use_dedicated_embeddings": false,
111
+ "num_attention_heads": 24,
112
+ "num_hidden_layers": 64,
113
+ "num_key_value_heads": 4,
114
+ "output_gate_type": "swish",
115
+ "pad_token_id": null,
116
+ "partial_rotary_factor": 0.25,
117
+ "rms_norm_eps": 1e-06,
118
+ "rope_parameters": {
119
+ "mrope_interleaved": true,
120
+ "mrope_section": [
121
+ 11,
122
+ 11,
123
+ 10
124
+ ],
125
+ "partial_rotary_factor": 0.25,
126
+ "rope_theta": 10000000,
127
+ "rope_type": "default"
128
+ },
129
+ "tie_word_embeddings": false,
130
+ "use_cache": true,
131
+ "vocab_size": 248320
132
+ },
133
+ "tie_word_embeddings": false,
134
+ "transformers_version": "5.8.0.dev0",
135
+ "video_token_id": 248057,
136
+ "vision_config": {
137
+ "deepstack_visual_indexes": [],
138
+ "depth": 27,
139
+ "hidden_act": "gelu_pytorch_tanh",
140
+ "hidden_size": 1152,
141
+ "in_channels": 3,
142
+ "initializer_range": 0.02,
143
+ "intermediate_size": 4304,
144
+ "model_type": "qwen3_5",
145
+ "num_heads": 16,
146
+ "num_position_embeddings": 2304,
147
+ "out_hidden_size": 5120,
148
+ "patch_size": 16,
149
+ "spatial_merge_size": 2,
150
+ "temporal_patch_size": 2
151
+ },
152
+ "vision_end_token_id": 248054,
153
+ "vision_start_token_id": 248053
154
+ }
compatibility/mac-check/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
compatibility/mac-check/real-checkpoint-samples.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c0e3f757947e7ffdaf96b90eb8e0f0ae72e83b9685d3462b99cd6bf193a2cd5b
3
+ size 3866825
compatibility/mac-check/samples.json ADDED
@@ -0,0 +1,45 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "source": "Actual completed Swift-1.5-4bit-MLX checkpoint; no substitute weights",
3
+ "samples": [
4
+ {
5
+ "sample": 0,
6
+ "category": "text",
7
+ "checkpoint_weight": "language_model.model.layers.0.linear_attn.in_proj_a.weight",
8
+ "original_shape": [
9
+ 48,
10
+ 5120
11
+ ],
12
+ "packed_shape": [
13
+ 48,
14
+ 640
15
+ ]
16
+ },
17
+ {
18
+ "sample": 1,
19
+ "category": "vision",
20
+ "checkpoint_weight": "visual.blocks.0.attn.proj.weight",
21
+ "original_shape": [
22
+ 1152,
23
+ 1152
24
+ ],
25
+ "packed_shape": [
26
+ 1152,
27
+ 144
28
+ ]
29
+ },
30
+ {
31
+ "sample": 2,
32
+ "category": "MTP",
33
+ "checkpoint_weight": "mtp.layers.0.self_attn.k_proj.weight",
34
+ "original_shape": [
35
+ 1024,
36
+ 5120
37
+ ],
38
+ "packed_shape": [
39
+ 1024,
40
+ 640
41
+ ]
42
+ }
43
+ ],
44
+ "coverage": "Small real packed tensor samples plus all saved tensor headers; not a full-model Mac generation test"
45
+ }
compatibility/mac-compatibility-results.json ADDED
@@ -0,0 +1,72 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "PASS_MAC_NATIVE_MLX_FORMAT_AND_REAL_METAL_SAMPLES",
3
+ "platform": "macOS-26.6-arm64-arm-64bit",
4
+ "machine": "arm64",
5
+ "mlx": "0.32.2",
6
+ "mlx_lm": "0.32.0",
7
+ "device": "Device(gpu, 0)",
8
+ "quantization": {
9
+ "group_size": 64,
10
+ "bits": 4,
11
+ "mode": "affine"
12
+ },
13
+ "all_saved_tensor_headers_validated": 2379,
14
+ "all_source_parameters_accounted_for": 1199,
15
+ "strict_complete_parameter_tree": "PASS using unevaluated header fixtures; no fabricated weights saved",
16
+ "actual_checkpoint_samples": [
17
+ {
18
+ "sample": 0,
19
+ "category": "text",
20
+ "checkpoint_weight": "language_model.model.layers.0.linear_attn.in_proj_a.weight",
21
+ "original_shape": [
22
+ 48,
23
+ 5120
24
+ ],
25
+ "packed_shape": [
26
+ 48,
27
+ 640
28
+ ],
29
+ "native_metal_bf16": "PASS",
30
+ "metal_fp32": "PASS",
31
+ "bf16_max_absolute_error": 0.0004401206970214844,
32
+ "fp32_max_absolute_error": 0.0
33
+ },
34
+ {
35
+ "sample": 1,
36
+ "category": "vision",
37
+ "checkpoint_weight": "visual.blocks.0.attn.proj.weight",
38
+ "original_shape": [
39
+ 1152,
40
+ 1152
41
+ ],
42
+ "packed_shape": [
43
+ 1152,
44
+ 144
45
+ ],
46
+ "native_metal_bf16": "PASS",
47
+ "metal_fp32": "PASS",
48
+ "bf16_max_absolute_error": 0.0002315044403076172,
49
+ "fp32_max_absolute_error": 0.0
50
+ },
51
+ {
52
+ "sample": 2,
53
+ "category": "MTP",
54
+ "checkpoint_weight": "mtp.layers.0.self_attn.k_proj.weight",
55
+ "original_shape": [
56
+ 1024,
57
+ 5120
58
+ ],
59
+ "packed_shape": [
60
+ 1024,
61
+ 640
62
+ ],
63
+ "native_metal_bf16": "PASS",
64
+ "metal_fp32": "PASS",
65
+ "bf16_max_absolute_error": 0.0009589195251464844,
66
+ "fp32_max_absolute_error": 0.0
67
+ }
68
+ ],
69
+ "full_model_mac_generation": "NOT_RUN: targeted native Metal format and real-weight component checks only",
70
+ "peak_mlx_bytes": 4597960,
71
+ "peak_process_rss_bytes": 360693760
72
+ }
compatibility/missing-file-recovery-report.md ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Original Swift source recovery completed
2
+
3
+ The pinned Hub snapshot lacked these original export files:
4
+
5
+ - `model-00015-of-00018.safetensors`
6
+ - `model-00016-of-00018.safetensors`
7
+ - `model-00017-of-00018.safetensors`
8
+ - `model-00018-of-00018.safetensors`
9
+ - `preprocessor_config.json`
10
+ - `tokenizer_config.json`
11
+ - `video_preprocessor_config.json`
12
+ - `vocab.json`
13
+ - `tokenizer.json`
14
+ - `model.safetensors.index.json`
15
+
16
+ All were recovered from the verified original project BF16 export. All 18 shards and runtime assets passed the original export-manifest SHA-256 checks. The source copy was read without modification. No shard was reconstructed or borrowed from base Qwen or a derived quant. The completed source has 55,563,006,776 weight bytes and 1,199 BF16 tensors. Full per-file hashes are in `source-file-manifest.csv` and `aws-source-verification.json`. The complete Hub source is available at revision [`5ad04445d2686f525e9fbe5c077e6fa0c7df4200`](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b/tree/5ad04445d2686f525e9fbe5c077e6fa0c7df4200).
compatibility/quant-tensor-mapping-manifest.json ADDED
The diff for this file is too large to render. See raw diff
 
compatibility/quant-validation-results.json ADDED
@@ -0,0 +1,103 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "PASS",
3
+ "recorded_at": "2026-09-21T18:14:21.887850+00:00",
4
+ "source_repo": "ukisai/Swift-1.5-Qwen3.8-27b",
5
+ "source_revision": "00ccd14e006897d28cb0ed5bf26390e60d274251",
6
+ "source_shards": 18,
7
+ "source_shard_bytes": 55563006776,
8
+ "source_tensors": 1199,
9
+ "mapped_source_tensors": 1199,
10
+ "saved_tensors": 2379,
11
+ "categories": {
12
+ "text": 851,
13
+ "MTP": 15,
14
+ "vision": 333
15
+ },
16
+ "ignored_tensors": 0,
17
+ "unexplained_tensors": 0,
18
+ "exact_unquantized_tensors": 609,
19
+ "quantization": {
20
+ "group_size": 64,
21
+ "bits": 4,
22
+ "mode": "affine"
23
+ },
24
+ "all_floating_tensors_finite": true,
25
+ "tokenizer": "Qwen2Tokenizer",
26
+ "processor": "Qwen3VLProcessor",
27
+ "assets_sha256": {
28
+ "generation_config.json": "e70c136c1b78ddc1fb0905bac8e733a4dc448d4f852a5dd75143fffc70be550e",
29
+ "preprocessor_config.json": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516",
30
+ "video_preprocessor_config.json": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13",
31
+ "tokenizer.json": "0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3",
32
+ "tokenizer_config.json": "b11349aafa7cdc6a320767cf7ceb29ed82f7eda5d65e8e0819e76f0ce947bf27",
33
+ "vocab.json": "ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003",
34
+ "merges.txt": "a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d",
35
+ "chat_template.jinja": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041"
36
+ },
37
+ "chat_templates": [
38
+ {
39
+ "options": {
40
+ "enable_thinking": false
41
+ },
42
+ "rendered": "<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\n"
43
+ },
44
+ {
45
+ "options": {
46
+ "reasoning_effort": "low"
47
+ },
48
+ "rendered": "<|im_start|>system\nReasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.<|im_end|>\n<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n"
49
+ },
50
+ {
51
+ "options": {
52
+ "reasoning_effort": "xhigh"
53
+ },
54
+ "rendered": "<|im_start|>system\nReasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.<|im_end|>\n<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n"
55
+ }
56
+ ],
57
+ "load_seconds": 3.351265648000208,
58
+ "load_memory_bytes": 15826466152,
59
+ "process_peak_rss_bytes": 21142360064,
60
+ "inference_floating_dtype": "float32, CPU runtime only; stored floating tensors remain BF16",
61
+ "native_bf16_cpu_inference": "Aborted after reproducing incorrect accumulation in the official Linux BF16 quantized matmul. See cpu-quantized-matmul-diagnostic.json.",
62
+ "generation": {
63
+ "prompt": "<|im_start|>user\nReply with exactly: Hello from Swift.<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\n",
64
+ "text": "Hello from Swift.",
65
+ "token_ids": [
66
+ 9419,
67
+ 494,
68
+ 22929,
69
+ 13,
70
+ 248046
71
+ ],
72
+ "tokens": 5,
73
+ "tokens_per_second": 0.0797672292904007,
74
+ "prompt_tokens_per_second": 0.06443886297275572,
75
+ "elapsed_seconds": 374.00736365800003,
76
+ "finish_reason": "stop"
77
+ },
78
+ "mtp": {
79
+ "status": "PASS",
80
+ "shape": [
81
+ 1,
82
+ 1,
83
+ 248320
84
+ ],
85
+ "path": "Explicit MTP step with real text hidden states and shared LM head; speculative generation is not integrated"
86
+ },
87
+ "vision": {
88
+ "status": "PASS",
89
+ "shape": [
90
+ 64,
91
+ 5120
92
+ ],
93
+ "grid": [
94
+ [
95
+ 1,
96
+ 16,
97
+ 16
98
+ ]
99
+ ],
100
+ "path": "Vision encoder only; image/video insertion and multimodal text generation are not implemented"
101
+ },
102
+ "total_validation_seconds": 467.15810263900016
103
+ }
compatibility/quant-validation.log ADDED
@@ -0,0 +1,110 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Loaded 2379 saved tensors; all 1199 source tensors accounted for.
2
+ All 609 unquantized tensors equal the original BF16 values.
3
+ CPU inference uses FP32 floating values; saved 4-bit weights are unchanged.
4
+ Hello from Swift.
5
+ Text generation passed.
6
+ Real-weight MTP step passed.
7
+ Real-weight vision encoder passed.
8
+ {
9
+ "status": "PASS",
10
+ "recorded_at": "2026-09-21T18:14:21.887850+00:00",
11
+ "source_repo": "ukisai/Swift-1.5-Qwen3.8-27b",
12
+ "source_revision": "00ccd14e006897d28cb0ed5bf26390e60d274251",
13
+ "source_shards": 18,
14
+ "source_shard_bytes": 55563006776,
15
+ "source_tensors": 1199,
16
+ "mapped_source_tensors": 1199,
17
+ "saved_tensors": 2379,
18
+ "categories": {
19
+ "text": 851,
20
+ "MTP": 15,
21
+ "vision": 333
22
+ },
23
+ "ignored_tensors": 0,
24
+ "unexplained_tensors": 0,
25
+ "exact_unquantized_tensors": 609,
26
+ "quantization": {
27
+ "group_size": 64,
28
+ "bits": 4,
29
+ "mode": "affine"
30
+ },
31
+ "all_floating_tensors_finite": true,
32
+ "tokenizer": "Qwen2Tokenizer",
33
+ "processor": "Qwen3VLProcessor",
34
+ "assets_sha256": {
35
+ "generation_config.json": "e70c136c1b78ddc1fb0905bac8e733a4dc448d4f852a5dd75143fffc70be550e",
36
+ "preprocessor_config.json": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516",
37
+ "video_preprocessor_config.json": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13",
38
+ "tokenizer.json": "0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3",
39
+ "tokenizer_config.json": "b11349aafa7cdc6a320767cf7ceb29ed82f7eda5d65e8e0819e76f0ce947bf27",
40
+ "vocab.json": "ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003",
41
+ "merges.txt": "a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d",
42
+ "chat_template.jinja": "c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041"
43
+ },
44
+ "chat_templates": [
45
+ {
46
+ "options": {
47
+ "enable_thinking": false
48
+ },
49
+ "rendered": "<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\n"
50
+ },
51
+ {
52
+ "options": {
53
+ "reasoning_effort": "low"
54
+ },
55
+ "rendered": "<|im_start|>system\nReasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.<|im_end|>\n<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n"
56
+ },
57
+ {
58
+ "options": {
59
+ "reasoning_effort": "xhigh"
60
+ },
61
+ "rendered": "<|im_start|>system\nReasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.<|im_end|>\n<|im_start|>user\nSay hello.<|im_end|>\n<|im_start|>assistant\n<think>\n"
62
+ }
63
+ ],
64
+ "load_seconds": 3.351265648000208,
65
+ "load_memory_bytes": 15826466152,
66
+ "process_peak_rss_bytes": 21142360064,
67
+ "inference_floating_dtype": "float32, CPU runtime only; stored floating tensors remain BF16",
68
+ "native_bf16_cpu_inference": "Aborted after reproducing incorrect accumulation in the official Linux BF16 quantized matmul. See cpu-quantized-matmul-diagnostic.json.",
69
+ "generation": {
70
+ "prompt": "<|im_start|>user\nReply with exactly: Hello from Swift.<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\n",
71
+ "text": "Hello from Swift.",
72
+ "token_ids": [
73
+ 9419,
74
+ 494,
75
+ 22929,
76
+ 13,
77
+ 248046
78
+ ],
79
+ "tokens": 5,
80
+ "tokens_per_second": 0.0797672292904007,
81
+ "prompt_tokens_per_second": 0.06443886297275572,
82
+ "elapsed_seconds": 374.00736365800003,
83
+ "finish_reason": "stop"
84
+ },
85
+ "mtp": {
86
+ "status": "PASS",
87
+ "shape": [
88
+ 1,
89
+ 1,
90
+ 248320
91
+ ],
92
+ "path": "Explicit MTP step with real text hidden states and shared LM head; speculative generation is not integrated"
93
+ },
94
+ "vision": {
95
+ "status": "PASS",
96
+ "shape": [
97
+ 64,
98
+ 5120
99
+ ],
100
+ "grid": [
101
+ [
102
+ 1,
103
+ 16,
104
+ 16
105
+ ]
106
+ ],
107
+ "path": "Vision encoder only; image/video insertion and multimodal text generation are not implemented"
108
+ },
109
+ "total_validation_seconds": 467.15810263900016
110
+ }
compatibility/requirements-linux.txt ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ annotated-doc==0.0.5
2
+ anyio==4.15.1
3
+ certifi==2026.7.22
4
+ click==8.5.0
5
+ filelock==4.0.1
6
+ fsspec==2026.9.0
7
+ h11==0.16.0
8
+ hf-xet==1.6.0
9
+ httpcore==1.0.9
10
+ httpx==0.28.1
11
+ huggingface_hub==1.31.0
12
+ idna==3.20
13
+ iniconfig==2.3.0
14
+ Jinja2==3.1.6
15
+ markdown-it-py==4.2.0
16
+ MarkupSafe==3.0.3
17
+ mdurl==0.1.2
18
+ mlx==0.32.2
19
+ mlx-cpu==0.32.2
20
+ -e git+https://github.com/ml-explore/mlx-lm.git@c69d1288440a0dc4e6401fc417098b07598dccd5#egg=mlx_lm
21
+ mpmath==1.3.0
22
+ networkx==3.6.1
23
+ numpy==2.5.3
24
+ packaging==26.3
25
+ pillow==12.3.0
26
+ pluggy==1.6.0
27
+ protobuf==7.36.2
28
+ Pygments==2.21.0
29
+ pytest==9.1.1
30
+ PyYAML==6.0.3
31
+ regex==2026.9.10
32
+ rich==15.0.0
33
+ safetensors==0.8.0
34
+ sentencepiece==0.2.2
35
+ setuptools==78.1.0
36
+ shellingham==1.5.4
37
+ sympy==1.14.0
38
+ tokenizers==0.22.2
39
+ torch==2.11.0+cpu
40
+ torchvision==0.26.0+cpu
41
+ tqdm==4.70.1
42
+ transformers==5.14.1
43
+ typer==0.27.2
44
+ typing_extensions==4.16.0
compatibility/run-compatibility.sh ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ set -euo pipefail
3
+ repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
4
+ source_dir="${SWIFT_SOURCE_DIR:?Set SWIFT_SOURCE_DIR to the verified Swift 1.5 BF16 export}"
5
+ log_dir="${SWIFT_VALIDATION_DIR:-$repo_root/validation-output}"
6
+ mkdir -p "$log_dir"
7
+ cd "$repo_root"
8
+ export HF_HUB_OFFLINE=1 TRANSFORMERS_OFFLINE=1 PYTHONUNBUFFERED=1
9
+ export OMP_NUM_THREADS=8 OPENBLAS_NUM_THREADS=8
10
+ python3 compatibility/verify_source_readonly.py \
11
+ --root "$source_dir" \
12
+ --manifest-sha256 0a00065b88ab003281853a7fb9bd5ce0086bc3781b36136d8c39da19933923ae \
13
+ > "$log_dir/source-verification.json" 2> "$log_dir/source-verification.log"
14
+ .venv/bin/python -m pytest -q mlx-lm/tests/test_qwen3_5_full.py \
15
+ --junitxml="$log_dir/compatibility-tests-linux.xml" \
16
+ > "$log_dir/compatibility-tests-linux.log" 2>&1
17
+ .venv/bin/python compatibility/validate_aws_source_linux.py \
18
+ --source "$source_dir" \
19
+ --verification "$log_dir/source-verification.json" \
20
+ --output "$log_dir/source-structural-results.json" \
21
+ > "$log_dir/source-structural.log" 2>&1
compatibility/run-conversion.py ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Run the official fixed MLX conversion and record resource usage."""
2
+ import json
3
+ import os
4
+ import shutil
5
+ import subprocess
6
+ import time
7
+ from datetime import datetime, timezone
8
+ from pathlib import Path
9
+
10
+ root = Path(__file__).resolve().parents[1]
11
+ os.chdir(root)
12
+ source = Path(os.environ['SWIFT_SOURCE_DIR']).resolve(strict=True)
13
+ output = Path(os.environ.get('SWIFT_MLX_OUTPUT', root / 'Swift-1.5-4bit-MLX')).resolve()
14
+ logs = Path(os.environ.get('SWIFT_VALIDATION_DIR', root / 'validation-output')).resolve()
15
+ logs.mkdir(parents=True, exist_ok=True)
16
+ assert not output.exists(), 'Never overwrite an existing artifact'
17
+ command = [str(root / '.venv/bin/mlx_lm.convert'), '--hf-path', str(source), '--mlx-path', str(output), '--quantize', '--q-mode', 'affine', '--q-bits', '4', '--q-group-size', '64']
18
+ record = {'command': command, 'started_at': datetime.now(timezone.utc).isoformat(), 'source_repo': 'ukisai/Swift-1.5-Qwen3.8-27b', 'source_revision': '00ccd14e006897d28cb0ed5bf26390e60d274251', 'source_manifest_sha256': '0a00065b88ab003281853a7fb9bd5ce0086bc3781b36136d8c39da19933923ae', 'quantization': {'mode': 'affine', 'bits': 4, 'group_size': 64}}
19
+ (logs / 'conversion-command.json').write_text(json.dumps(record, indent=2)+'\n')
20
+ env = dict(os.environ, HF_HUB_OFFLINE='1', TRANSFORMERS_OFFLINE='1', PYTHONUNBUFFERED='1', OMP_NUM_THREADS='8', OPENBLAS_NUM_THREADS='8')
21
+ start = time.monotonic()
22
+ with (logs / 'conversion.log').open('x') as log, (logs / 'conversion-resources.jsonl').open('x') as monitor:
23
+ process = subprocess.Popen(command, stdout=log, stderr=subprocess.STDOUT, env=env)
24
+ record['pid'] = process.pid
25
+ (logs / 'conversion-pid').write_text(str(process.pid)+'\n')
26
+ while process.poll() is None:
27
+ memory = dict(line.split(':', 1) for line in Path('/proc/meminfo').read_text().splitlines())
28
+ status = Path(f'/proc/{process.pid}/status')
29
+ stats = dict(line.split(':', 1) for line in status.read_text().splitlines()) if status.exists() else {}
30
+ disk = shutil.disk_usage(root)
31
+ sample = {'elapsed_seconds': time.monotonic()-start, 'rss': stats.get('VmRSS', '').strip(), 'peak_rss': stats.get('VmHWM', '').strip(), 'process_swap': stats.get('VmSwap', '').strip(), 'memory_available': memory['MemAvailable'].strip(), 'swap_free': memory['SwapFree'].strip(), 'disk_free_bytes': disk.free}
32
+ monitor.write(json.dumps(sample)+'\n'); monitor.flush()
33
+ if disk.free < 1024**3:
34
+ record['critical_stop_reason'] = 'Less than 1 GiB free disk space'
35
+ process.terminate()
36
+ time.sleep(5)
37
+ record.update(returncode=process.returncode, elapsed_seconds=time.monotonic()-start, finished_at=datetime.now(timezone.utc).isoformat())
38
+ (logs / 'conversion-result.json').write_text(json.dumps(record,indent=2)+'\n')
39
+ print(json.dumps(record,indent=2))
40
+ raise SystemExit(process.returncode)
compatibility/setup-linux.sh ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ set -euo pipefail
3
+ repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
4
+ log_dir="${SWIFT_VALIDATION_DIR:-$repo_root/validation-output}"
5
+ mkdir -p "$log_dir"
6
+ export SWIFT_VALIDATION_DIR="$log_dir"
7
+ cd "$repo_root"
8
+ python3 -m venv .venv
9
+ .venv/bin/python -m pip install --upgrade pip
10
+ .venv/bin/python -m pip install 'mlx[cpu]==0.32.2' 'transformers==5.14.1' 'huggingface_hub==1.31.0' 'safetensors==0.8.0' 'pytest==9.1.1' pillow sentencepiece protobuf
11
+ .venv/bin/python -m pip install 'torch==2.11.0' 'torchvision==0.26.0' --index-url https://download.pytorch.org/whl/cpu
12
+ if [ ! -d mlx-lm/.git ]; then
13
+ git clone https://github.com/ml-explore/mlx-lm.git mlx-lm
14
+ git -C mlx-lm checkout -b swift15-preserve-components c69d1288440a0dc4e6401fc417098b07598dccd5
15
+ fi
16
+ if [ ! -f mlx-lm/mlx_lm/models/qwen3_5_full.py ]; then
17
+ git -C mlx-lm apply --check "$repo_root/compatibility/swift15-mlx-lm.patch"
18
+ git -C mlx-lm apply "$repo_root/compatibility/swift15-mlx-lm.patch"
19
+ fi
20
+ .venv/bin/python -m pip install --no-deps -e ./mlx-lm
21
+ .venv/bin/python -m pip check
22
+ .venv/bin/python -m pip freeze > "$log_dir/requirements-linux.txt"
23
+ .venv/bin/python - <<'PY'
24
+ import importlib.metadata as m,json,os,platform
25
+ from pathlib import Path
26
+ import mlx.core as mx
27
+ mx.set_default_device(mx.cpu)
28
+ x=mx.array([1,2,3]); mx.eval(x*x)
29
+ r={'platform':platform.platform(),'python':platform.python_version(),'device':str(mx.default_device()),'cpu_test':(x*x).tolist(),'packages':{n:m.version(n) for n in ('mlx','mlx-cpu','mlx-lm','transformers','huggingface_hub','torch','torchvision','safetensors')}}
30
+ Path(os.environ['SWIFT_VALIDATION_DIR'],'environment-linux.json').write_text(json.dumps(r,indent=2)+'\n')
31
+ print(json.dumps(r,indent=2))
32
+ PY
compatibility/source-file-manifest.csv ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ FILE,EXPECTED,PRESENT,SIZE_BYTES,SHA256,SOURCE_REVISION,PRESENT_IN_PINNED_HUB
2
+ model-00001-of-00018.safetensors,True,True,3966730552,b9dcb3e15c00823d46d92f645f8a7324cdd1163c39ff6c04de48fa6e478952a0,00ccd14e006897d28cb0ed5bf26390e60d274251,True
3
+ model-00002-of-00018.safetensors,True,True,3043080328,ecf7312565a089592217b58d761e714f3b3d85d4ec578a15d901210a9ba14773,00ccd14e006897d28cb0ed5bf26390e60d274251,True
4
+ model-00003-of-00018.safetensors,True,True,2542796952,cec21a53517089f1b9708a77936465420f266482a45da26fd4c0dd596d609d08,00ccd14e006897d28cb0ed5bf26390e60d274251,True
5
+ model-00004-of-00018.safetensors,True,True,3988973152,3a1a2bfcfd177e62fbe78f08b2b1d3edac59c495397aa64d0d1c41b727a94c73,00ccd14e006897d28cb0ed5bf26390e60d274251,True
6
+ model-00005-of-00018.safetensors,True,True,2099339864,389931270466dcdbc9125bda7572cc7712f164e1fe774e7377524abca899b0c2,00ccd14e006897d28cb0ed5bf26390e60d274251,True
7
+ model-00006-of-00018.safetensors,True,True,3979553696,da6f16eabe6be8552af22ed5ad171ae97469fa5bb5d5e613982460e43265debf,00ccd14e006897d28cb0ed5bf26390e60d274251,True
8
+ model-00007-of-00018.safetensors,True,True,2108759344,b4b9f91766d367f0260882d97b5c0862863c393ef381f54d48dfda3d362072a9,00ccd14e006897d28cb0ed5bf26390e60d274251,True
9
+ model-00008-of-00018.safetensors,True,True,3979553696,df58405d10cb8a199e3e5a01346c6b6ebf2a1bed05d6b11e14f856543b28e55c,00ccd14e006897d28cb0ed5bf26390e60d274251,True
10
+ model-00009-of-00018.safetensors,True,True,2108759344,84ba24193b7d0885ef504ce5806d3d4ebbe5506fdfd5919f4837ebc9479611e6,00ccd14e006897d28cb0ed5bf26390e60d274251,True
11
+ model-00010-of-00018.safetensors,True,True,3979553696,aacd71e34312a13bf487a06c9c54205ad31f72a8788b0a550154a04565820a97,00ccd14e006897d28cb0ed5bf26390e60d274251,True
12
+ model-00011-of-00018.safetensors,True,True,2108759344,6d967f431bdfde43c38839188023c5507d5d0dce6a1daa6bbbf8605ecb3da9aa,00ccd14e006897d28cb0ed5bf26390e60d274251,True
13
+ model-00012-of-00018.safetensors,True,True,3979553696,c36593694bc3f250449ae2196366e6cf788c122dc9a645fbe6ef116746fc5d73,00ccd14e006897d28cb0ed5bf26390e60d274251,True
14
+ model-00013-of-00018.safetensors,True,True,2108759344,b50b55cffe8be40b6eb90c43b1d3cec7672d7f3a86b2efd436db5a83619686d2,00ccd14e006897d28cb0ed5bf26390e60d274251,True
15
+ model-00014-of-00018.safetensors,True,True,3979553696,27b627a940af6ba6f26d240767ca75a4774d02a91255c3a2274dd316e364d1d1,00ccd14e006897d28cb0ed5bf26390e60d274251,True
16
+ model-00015-of-00018.safetensors,True,True,2108759344,f9203192d4a759f6fc58c32eebda8a2f00f4a15dc183f06ddc263de49595ade5,00ccd14e006897d28cb0ed5bf26390e60d274251,False
17
+ model-00016-of-00018.safetensors,True,True,3979564040,cf3333f84d3783148a58b24612998ba24b8c60dc69f77e14d86e66ef958c23de,00ccd14e006897d28cb0ed5bf26390e60d274251,False
18
+ model-00017-of-00018.safetensors,True,True,2108759344,e446dd9c79fc0255d9b2181a9a85debb2b5b51b2eb31177c4c1a7daa198221f7,00ccd14e006897d28cb0ed5bf26390e60d274251,False
19
+ model-00018-of-00018.safetensors,True,True,3392197344,2898c2335e6ce74567af2de8373d7468c403fb5e69f4b00f5f7b83fb815d71be,00ccd14e006897d28cb0ed5bf26390e60d274251,False
20
+ chat_template.jinja,True,True,8952,c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041,00ccd14e006897d28cb0ed5bf26390e60d274251,True
21
+ generation_config.json,True,True,202,e70c136c1b78ddc1fb0905bac8e733a4dc448d4f852a5dd75143fffc70be550e,00ccd14e006897d28cb0ed5bf26390e60d274251,True
22
+ config.json,True,True,4312,191e0af232104ed8b65258cf3fb2b842e288008baca7633c11b82a1ac7203aab,00ccd14e006897d28cb0ed5bf26390e60d274251,True
23
+ merges.txt,True,True,3353259,a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d,00ccd14e006897d28cb0ed5bf26390e60d274251,True
24
+ preprocessor_config.json,True,True,390,27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516,00ccd14e006897d28cb0ed5bf26390e60d274251,False
25
+ tokenizer_config.json,True,True,17928,b11349aafa7cdc6a320767cf7ceb29ed82f7eda5d65e8e0819e76f0ce947bf27,00ccd14e006897d28cb0ed5bf26390e60d274251,False
26
+ video_preprocessor_config.json,True,True,385,7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13,00ccd14e006897d28cb0ed5bf26390e60d274251,False
27
+ vocab.json,True,True,6722759,ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003,00ccd14e006897d28cb0ed5bf26390e60d274251,False
28
+ tokenizer.json,True,True,12809320,0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3,00ccd14e006897d28cb0ed5bf26390e60d274251,False
29
+ EXPORT_MANIFEST.json,True,True,63469,0a00065b88ab003281853a7fb9bd5ce0086bc3781b36136d8c39da19933923ae,00ccd14e006897d28cb0ed5bf26390e60d274251,True
30
+ README.md,True,True,2715,56ccec79db878d30b3a39ad7a1ac6d196fdf2e5dfc6c12f6c4c5da29e5561d7e,00ccd14e006897d28cb0ed5bf26390e60d274251,True
31
+ model.safetensors.index.json,True,True,112214,bd9f76c08ed50dccdb8a3de2c4e03a321bf6f506b99f838be11b584f36091461,00ccd14e006897d28cb0ed5bf26390e60d274251,False
compatibility/swift15-mlx-lm.patch ADDED
@@ -0,0 +1,962 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ diff --git a/mlx_lm/models/qwen3_5_full.py b/mlx_lm/models/qwen3_5_full.py
2
+ new file mode 100644
3
+ index 0000000..eb3d09b
4
+ --- /dev/null
5
+ +++ b/mlx_lm/models/qwen3_5_full.py
6
+ @@ -0,0 +1,485 @@
7
+ +# Copyright © 2026 Apple Inc.
8
+ +
9
+ +"""Complete Qwen3.5 parameter model, including the vision encoder and MTP.
10
+ +
11
+ +Text generation, the vision encoder, and explicit MTP steps are separate APIs.
12
+ +Image/video token insertion, multimodal text positions, and speculative decoding
13
+ +are not implemented here. Unsupported multimodal calls raise an error.
14
+ +"""
15
+ +
16
+ +import copy
17
+ +from dataclasses import dataclass
18
+ +from typing import Optional
19
+ +
20
+ +import mlx.core as mx
21
+ +import mlx.nn as nn
22
+ +import numpy as np
23
+ +from mlx.utils import tree_flatten, tree_unflatten
24
+ +
25
+ +from . import qwen3_5
26
+ +from .base import BaseModelArgs, create_attention_mask
27
+ +from .cache import KVCache
28
+ +
29
+ +
30
+ +class OffsetRMSNorm(nn.Module):
31
+ + """Keep HF's zero-centered weights without rounding weight + 1 to BF16."""
32
+ +
33
+ + def __init__(self, dims, eps=1e-6):
34
+ + super().__init__()
35
+ + self.weight = mx.zeros((dims,))
36
+ + self.eps = eps
37
+ +
38
+ + def __call__(self, x):
39
+ + y = x.astype(mx.float32)
40
+ + y = y * mx.rsqrt(mx.mean(y * y, axis=-1, keepdims=True) + self.eps)
41
+ + return (y * (1 + self.weight.astype(mx.float32))).astype(x.dtype)
42
+ +
43
+ +
44
+ +def _use_offset_norms(module):
45
+ + replacements = [
46
+ + (name, OffsetRMSNorm(norm.weight.shape[0], norm.eps))
47
+ + for name, norm in module.named_modules()
48
+ + if isinstance(norm, nn.RMSNorm)
49
+ + ]
50
+ + module.update_modules(tree_unflatten(replacements))
51
+ +
52
+ +
53
+ +@dataclass
54
+ +class VisionArgs(BaseModelArgs):
55
+ + depth: int
56
+ + hidden_size: int
57
+ + intermediate_size: int
58
+ + num_heads: int
59
+ + out_hidden_size: int
60
+ + num_position_embeddings: int
61
+ + patch_size: int = 16
62
+ + temporal_patch_size: int = 2
63
+ + spatial_merge_size: int = 2
64
+ + in_channels: int = 3
65
+ + hidden_act: str = "gelu_pytorch_tanh"
66
+ + deepstack_visual_indexes: Optional[list] = None
67
+ +
68
+ + def __post_init__(self):
69
+ + if self.deepstack_visual_indexes:
70
+ + raise ValueError("DeepStack vision features are not supported")
71
+ + if self.hidden_act != "gelu_pytorch_tanh":
72
+ + raise ValueError(f"Unsupported vision activation: {self.hidden_act}")
73
+ + if self.hidden_size % self.num_heads or self.hidden_size // self.num_heads % 4:
74
+ + raise ValueError("Vision head dimension must be divisible by four")
75
+ + if int(self.num_position_embeddings**0.5) ** 2 != self.num_position_embeddings:
76
+ + raise ValueError("Vision position table must be square")
77
+ +
78
+ +
79
+ +class VisionPatchEmbed(nn.Module):
80
+ + def __init__(self, args):
81
+ + super().__init__()
82
+ + self.args = args
83
+ + kernel = (args.temporal_patch_size, args.patch_size, args.patch_size)
84
+ + self.proj = nn.Conv3d(
85
+ + args.in_channels, args.hidden_size, kernel, stride=kernel, bias=True
86
+ + )
87
+ +
88
+ + def __call__(self, pixels):
89
+ + a = self.args
90
+ + x = pixels.reshape(
91
+ + -1, a.in_channels, a.temporal_patch_size, a.patch_size, a.patch_size
92
+ + )
93
+ + x = x.transpose(0, 2, 3, 4, 1).astype(self.proj.weight.dtype)
94
+ + return self.proj(x).reshape(-1, a.hidden_size)
95
+ +
96
+ +
97
+ +class VisionAttention(nn.Module):
98
+ + def __init__(self, args):
99
+ + super().__init__()
100
+ + self.num_heads = args.num_heads
101
+ + self.head_dim = args.hidden_size // args.num_heads
102
+ + self.qkv = nn.Linear(args.hidden_size, 3 * args.hidden_size)
103
+ + self.proj = nn.Linear(args.hidden_size, args.hidden_size)
104
+ +
105
+ + def __call__(self, x, cos, sin, boundaries):
106
+ + qkv = self.qkv(x).reshape(x.shape[0], 3, self.num_heads, self.head_dim)
107
+ + q, k, v = qkv.transpose(1, 0, 2, 3)
108
+ +
109
+ + def rotate(y):
110
+ + z = y.astype(mx.float32)
111
+ + half = self.head_dim // 2
112
+ + rotated = mx.concatenate([-z[..., half:], z[..., :half]], axis=-1)
113
+ + return (z * cos[:, None] + rotated * sin[:, None]).astype(y.dtype)
114
+ +
115
+ + q, k = rotate(q), rotate(k)
116
+ + outputs = []
117
+ + for start, end in zip(boundaries, boundaries[1:]):
118
+ + out = mx.fast.scaled_dot_product_attention(
119
+ + q[start:end].transpose(1, 0, 2)[None],
120
+ + k[start:end].transpose(1, 0, 2)[None],
121
+ + v[start:end].transpose(1, 0, 2)[None],
122
+ + scale=self.head_dim**-0.5,
123
+ + )
124
+ + outputs.append(out[0].transpose(1, 0, 2).reshape(end - start, -1))
125
+ + return self.proj(mx.concatenate(outputs, axis=0))
126
+ +
127
+ +
128
+ +class VisionMLP(nn.Module):
129
+ + def __init__(self, args):
130
+ + super().__init__()
131
+ + self.linear_fc1 = nn.Linear(args.hidden_size, args.intermediate_size)
132
+ + self.linear_fc2 = nn.Linear(args.intermediate_size, args.hidden_size)
133
+ +
134
+ + def __call__(self, x):
135
+ + return self.linear_fc2(nn.gelu_approx(self.linear_fc1(x)))
136
+ +
137
+ +
138
+ +class VisionBlock(nn.Module):
139
+ + def __init__(self, args):
140
+ + super().__init__()
141
+ + self.norm1 = nn.LayerNorm(args.hidden_size, eps=1e-6)
142
+ + self.norm2 = nn.LayerNorm(args.hidden_size, eps=1e-6)
143
+ + self.attn = VisionAttention(args)
144
+ + self.mlp = VisionMLP(args)
145
+ +
146
+ + def __call__(self, x, cos, sin, boundaries):
147
+ + x = x + self.attn(self.norm1(x), cos, sin, boundaries)
148
+ + return x + self.mlp(self.norm2(x))
149
+ +
150
+ +
151
+ +class VisionMerger(nn.Module):
152
+ + def __init__(self, args):
153
+ + super().__init__()
154
+ + self.hidden_size = args.hidden_size * args.spatial_merge_size**2
155
+ + self.norm = nn.LayerNorm(args.hidden_size, eps=1e-6)
156
+ + self.linear_fc1 = nn.Linear(self.hidden_size, self.hidden_size)
157
+ + self.linear_fc2 = nn.Linear(self.hidden_size, args.out_hidden_size)
158
+ +
159
+ + def __call__(self, x):
160
+ + x = self.norm(x).reshape(-1, self.hidden_size)
161
+ + return self.linear_fc2(nn.gelu(self.linear_fc1(x)))
162
+ +
163
+ +
164
+ +class VisionModel(nn.Module):
165
+ + def __init__(self, args):
166
+ + super().__init__()
167
+ + self.args = args
168
+ + self.patch_embed = VisionPatchEmbed(args)
169
+ + self.pos_embed = nn.Embedding(args.num_position_embeddings, args.hidden_size)
170
+ + self.blocks = [VisionBlock(args) for _ in range(args.depth)]
171
+ + self.merger = VisionMerger(args)
172
+ +
173
+ + def _positions(self, grid_thw):
174
+ + grid = np.asarray(
175
+ + grid_thw.tolist() if hasattr(grid_thw, "tolist") else grid_thw
176
+ + )
177
+ + if (
178
+ + grid.ndim != 2
179
+ + or grid.shape[1] != 3
180
+ + or not np.issubdtype(grid.dtype, np.integer)
181
+ + ):
182
+ + raise ValueError("grid_thw must be an integer array with shape (N, 3)")
183
+ + merge = self.args.spatial_merge_size
184
+ + side = int(self.args.num_position_embeddings**0.5)
185
+ + positions, indices, weights, boundaries = [], [], [], [0]
186
+ + for t, h, w in grid.tolist():
187
+ + if min(t, h, w) <= 0 or h % merge or w % merge:
188
+ + raise ValueError("Invalid vision grid or spatial merge dimensions")
189
+ + hp, wp = np.indices((h, w))
190
+ + block = (h // merge, merge, w // merge, merge)
191
+ + hp = hp.reshape(block).transpose(0, 2, 1, 3).reshape(-1)
192
+ + wp = wp.reshape(block).transpose(0, 2, 1, 3).reshape(-1)
193
+ + positions.append(np.tile(np.stack([hp, wp], axis=-1), (t, 1)))
194
+ + reorder = np.tile(hp * w + wp, t)
195
+ + hs = np.linspace(0, side - 1, h, dtype=np.float32)
196
+ + ws = np.linspace(0, side - 1, w, dtype=np.float32)
197
+ + hf, wf = hs.astype(np.int32), ws.astype(np.int32)
198
+ + hc, wc = np.minimum(hf + 1, side - 1), np.minimum(wf + 1, side - 1)
199
+ + dh, dw = hs - hf, ws - wf
200
+ + idx = [
201
+ + (a[:, None] * side + b[None]).reshape(-1)
202
+ + for a, b in [(hf, wf), (hf, wc), (hc, wf), (hc, wc)]
203
+ + ]
204
+ + coeff = [
205
+ + (a[:, None] * b[None]).reshape(-1)
206
+ + for a, b in [(1 - dh, 1 - dw), (1 - dh, dw), (dh, 1 - dw), (dh, dw)]
207
+ + ]
208
+ + indices.append(np.stack(idx)[:, reorder])
209
+ + weights.append(np.stack(coeff)[:, reorder])
210
+ + offset = boundaries[-1]
211
+ + boundaries.extend(offset + (i + 1) * h * w for i in range(t))
212
+ + if not positions:
213
+ + raise ValueError("At least one vision grid is required")
214
+ + return (
215
+ + mx.array(np.concatenate(positions), dtype=mx.float32),
216
+ + mx.array(np.concatenate(indices, axis=1), dtype=mx.int32),
217
+ + mx.array(np.concatenate(weights, axis=1), dtype=mx.float32),
218
+ + boundaries,
219
+ + )
220
+ +
221
+ + def __call__(self, pixels, grid_thw, return_hidden_states=False):
222
+ + positions, indices, weights, boundaries = self._positions(grid_thw)
223
+ + x = self.patch_embed(pixels)
224
+ + if x.shape[0] != boundaries[-1]:
225
+ + raise ValueError("Pixel patch count does not match grid_thw")
226
+ + pos = (self.pos_embed(indices) * weights[..., None]).sum(axis=0)
227
+ + x = x + pos.astype(x.dtype)
228
+ + dim = self.args.hidden_size // self.args.num_heads // 2
229
+ + inv_freq = 1.0 / (10000 ** (mx.arange(0, dim, 2, dtype=mx.float32) / dim))
230
+ + angles = (positions[..., None] * inv_freq).reshape(x.shape[0], -1)
231
+ + angles = mx.concatenate([angles, angles], axis=-1)
232
+ + cos, sin = mx.cos(angles), mx.sin(angles)
233
+ + for block in self.blocks:
234
+ + x = block(x, cos, sin, boundaries)
235
+ + merged = self.merger(x)
236
+ + return (x, merged) if return_hidden_states else merged
237
+ +
238
+ +
239
+ +class MultiTokenPredictor(nn.Module):
240
+ + """An explicit MTP step; embeddings and the output head belong to the LM."""
241
+ +
242
+ + def __init__(self, args, num_layers):
243
+ + super().__init__()
244
+ + self.fc = nn.Linear(2 * args.hidden_size, args.hidden_size, bias=False)
245
+ + self.pre_fc_norm_embedding = OffsetRMSNorm(args.hidden_size, args.rms_norm_eps)
246
+ + self.pre_fc_norm_hidden = OffsetRMSNorm(args.hidden_size, args.rms_norm_eps)
247
+ + self.layers = [
248
+ + qwen3_5.DecoderLayer(args, args.full_attention_interval - 1)
249
+ + for _ in range(num_layers)
250
+ + ]
251
+ + self.norm = OffsetRMSNorm(args.hidden_size, args.rms_norm_eps)
252
+ + _use_offset_norms(self)
253
+ +
254
+ + def __call__(self, hidden_states, next_token_embeddings, cache=None, step=0):
255
+ + if hidden_states.shape != next_token_embeddings.shape:
256
+ + raise ValueError("MTP hidden states and next-token embeddings must align")
257
+ + if step < 0:
258
+ + raise ValueError("MTP step must be non-negative")
259
+ + x = mx.concatenate(
260
+ + [
261
+ + self.pre_fc_norm_embedding(next_token_embeddings),
262
+ + self.pre_fc_norm_hidden(hidden_states),
263
+ + ],
264
+ + axis=-1,
265
+ + )
266
+ + x = self.fc(x)
267
+ + layer = step % len(self.layers)
268
+ + if cache is not None and len(cache) != len(self.layers):
269
+ + raise ValueError("MTP requires one KV cache per MTP layer")
270
+ + c = cache[layer] if cache is not None else None
271
+ + return self.norm(self.layers[layer](x, create_attention_mask(x, c), c))
272
+ +
273
+ + def make_cache(self):
274
+ + return [KVCache() for _ in self.layers]
275
+ +
276
+ +
277
+ +@dataclass
278
+ +class ModelArgs(BaseModelArgs):
279
+ + model_type: str
280
+ + text_config: dict
281
+ + vision_config: dict
282
+ + language_model_only: bool = False
283
+ + image_token_id: Optional[int] = None
284
+ + video_token_id: Optional[int] = None
285
+ + vision_start_token_id: Optional[int] = None
286
+ + tie_word_embeddings: bool = False
287
+ + quantization: Optional[dict] = None
288
+ +
289
+ + @classmethod
290
+ + def from_dict(cls, params):
291
+ + return super().from_dict(copy.deepcopy(params))
292
+ +
293
+ +
294
+ +class Model(nn.Module):
295
+ + extra_save_files = (
296
+ + "preprocessor_config.json",
297
+ + "video_preprocessor_config.json",
298
+ + "processor_config.json",
299
+ + "tokenizer.json",
300
+ + "tokenizer_config.json",
301
+ + "special_tokens_map.json",
302
+ + "vocab.json",
303
+ + "merges.txt",
304
+ + "chat_template.jinja",
305
+ + )
306
+ +
307
+ + def __init__(self, args):
308
+ + super().__init__()
309
+ + self.args = args
310
+ + self.model_type = args.model_type
311
+ + text = args.text_config
312
+ + if args.language_model_only:
313
+ + raise ValueError("The complete model requires language_model_only=false")
314
+ + if text.get("hidden_act", "silu") not in ("silu", "swish"):
315
+ + raise ValueError("Unsupported text MLP activation")
316
+ + if text.get("attn_output_gate", True) is not True:
317
+ + raise ValueError("Ungated attention is not supported")
318
+ + if text.get("output_gate_type", "swish") not in ("swish", "sigmoid"):
319
+ + raise ValueError("Unknown attention gate declaration")
320
+ + if text.get("mtp_use_dedicated_embeddings", False):
321
+ + raise ValueError("Dedicated MTP embeddings are not supported")
322
+ + if args.tie_word_embeddings != text.get("tie_word_embeddings", False):
323
+ + raise ValueError("Conflicting text and top-level weight tying settings")
324
+ + self._text_args = qwen3_5.TextModelArgs.from_dict(copy.deepcopy(text))
325
+ + expected_layers = [
326
+ + (
327
+ + "full_attention"
328
+ + if (i + 1) % self._text_args.full_attention_interval == 0
329
+ + else "linear_attention"
330
+ + )
331
+ + for i in range(self._text_args.num_hidden_layers)
332
+ + ]
333
+ + if text.get("layer_types", expected_layers) != expected_layers:
334
+ + raise ValueError("Layer schedule differs from the supported architecture")
335
+ + if self._text_args.num_experts:
336
+ + raise ValueError("This complete model supports the dense architecture")
337
+ + self.language_model = qwen3_5.TextModel(self._text_args)
338
+ + _use_offset_norms(self.language_model)
339
+ + self.visual = VisionModel(VisionArgs.from_dict(args.vision_config))
340
+ + if self.visual.args.out_hidden_size != self._text_args.hidden_size:
341
+ + raise ValueError("Vision output size does not match text hidden size")
342
+ + num_mtp = text.get("mtp_num_hidden_layers", 0)
343
+ + if num_mtp < 0:
344
+ + raise ValueError("Invalid MTP layer count")
345
+ + if num_mtp:
346
+ + self.mtp = MultiTokenPredictor(self._text_args, num_mtp)
347
+ + self._parameter_shapes = {
348
+ + k: tuple(v.shape) for k, v in tree_flatten(self.parameters())
349
+ + }
350
+ +
351
+ + @property
352
+ + def model(self):
353
+ + return self.language_model.model
354
+ +
355
+ + @property
356
+ + def layers(self):
357
+ + return self.language_model.layers
358
+ +
359
+ + def make_cache(self):
360
+ + return self.language_model.make_cache()
361
+ +
362
+ + def __call__(self, inputs, cache=None, input_embeddings=None, **kwargs):
363
+ + if kwargs:
364
+ + raise NotImplementedError(
365
+ + "Multimodal text integration is not implemented; use visual() for encoder features"
366
+ + )
367
+ + for token in (
368
+ + self.args.image_token_id,
369
+ + self.args.video_token_id,
370
+ + self.args.vision_start_token_id,
371
+ + ):
372
+ + if (
373
+ + token is not None
374
+ + and inputs is not None
375
+ + and bool(mx.any(inputs == token))
376
+ + ):
377
+ + raise NotImplementedError(
378
+ + "Multimodal token positions require a multimodal text runtime"
379
+ + )
380
+ + return self.language_model(inputs, cache, input_embeddings)
381
+ +
382
+ + def mtp_logits(self, next_token_ids, previous_hidden_states, cache=None, step=0):
383
+ + if not hasattr(self, "mtp"):
384
+ + raise ValueError("This checkpoint has no MTP layers")
385
+ + embeddings = self.model.embed_tokens(next_token_ids)
386
+ + hidden = self.mtp(previous_hidden_states, embeddings, cache, step)
387
+ + if self._text_args.tie_word_embeddings:
388
+ + return self.model.embed_tokens.as_linear(hidden)
389
+ + return self.language_model.lm_head(hidden)
390
+ +
391
+ + def weight_mapping(self):
392
+ + rows = []
393
+ + for name, shape in sorted(self._parameter_shapes.items()):
394
+ + source, source_shape, transform = name, shape, "identity"
395
+ + if name.startswith("language_model.model."):
396
+ + source = "model.language_model." + name[len("language_model.model.") :]
397
+ + elif name.startswith("language_model.lm_head."):
398
+ + source = name[len("language_model.") :]
399
+ + elif name.startswith("visual."):
400
+ + source = "model." + name
401
+ + if name.endswith(".conv1d.weight"):
402
+ + source_shape = (shape[0], shape[2], shape[1])
403
+ + transform = "transpose(0,2,1)"
404
+ + elif name == "visual.patch_embed.proj.weight":
405
+ + source_shape = (shape[0], shape[4], shape[1], shape[2], shape[3])
406
+ + transform = "transpose(0,2,3,4,1)"
407
+ + category = (
408
+ + "vision"
409
+ + if name.startswith("visual.")
410
+ + else "MTP" if name.startswith("mtp.") else "text"
411
+ + )
412
+ + rows.append(
413
+ + dict(
414
+ + source=source,
415
+ + source_shape=source_shape,
416
+ + destination=name,
417
+ + destination_shape=shape,
418
+ + category=category,
419
+ + transform=transform,
420
+ + )
421
+ + )
422
+ + return rows
423
+ +
424
+ + def sanitize(self, weights):
425
+ + rows = self.weight_mapping()
426
+ + hf_names = {r["source"] for r in rows}
427
+ + native_names = set(self._parameter_shapes)
428
+ + is_hf = any(k.startswith("model.language_model.") for k in weights)
429
+ + expected = hf_names if is_hf else native_names
430
+ + if self.args.quantization:
431
+ + if is_hf:
432
+ + raise ValueError("Only original unquantized HF weights can be mapped")
433
+ + return self._check_native_quantized(weights)
434
+ + missing, unexpected = expected - set(weights), set(weights) - expected
435
+ + if missing or unexpected:
436
+ + raise ValueError(
437
+ + f"Incomplete weight mapping: missing={sorted(missing)}, unexpected={sorted(unexpected)}"
438
+ + )
439
+ + mapped = {}
440
+ + for row in rows:
441
+ + key = row["source"] if is_hf else row["destination"]
442
+ + value = weights[key]
443
+ + shape = row["source_shape"] if is_hf else row["destination_shape"]
444
+ + if tuple(value.shape) != shape:
445
+ + raise ValueError(f"Shape mismatch for {key}: {value.shape} != {shape}")
446
+ + if is_hf and row["transform"] == "transpose(0,2,1)":
447
+ + value = value.transpose(0, 2, 1)
448
+ + elif is_hf and row["transform"] == "transpose(0,2,3,4,1)":
449
+ + value = value.transpose(0, 2, 3, 4, 1)
450
+ + mapped[row["destination"]] = value
451
+ + return mapped
452
+ +
453
+ + def _check_native_quantized(self, weights):
454
+ + shapes = dict(self._parameter_shapes)
455
+ + q = self.args.quantization
456
+ + for path, module in self.named_modules():
457
+ + if not hasattr(module, "to_quantized"):
458
+ + continue
459
+ + settings = q.get(path, q)
460
+ + if settings is False:
461
+ + continue
462
+ + if not isinstance(settings, dict):
463
+ + raise ValueError(f"Invalid native quantization metadata for {path}")
464
+ + bits, group, mode = (
465
+ + settings.get("bits"),
466
+ + settings.get("group_size"),
467
+ + settings.get("mode", "affine"),
468
+ + )
469
+ + if (bits, group, mode) != (4, 64, "affine"):
470
+ + raise ValueError(
471
+ + "This extension only prepares affine/4-bit/group-64 native checkpoints"
472
+ + )
473
+ + original = shapes[f"{path}.weight"]
474
+ + if original[-1] % group:
475
+ + continue
476
+ + shapes[f"{path}.weight"] = (*original[:-1], original[-1] * bits // 32)
477
+ + shapes[f"{path}.scales"] = (*original[:-1], original[-1] // group)
478
+ + shapes[f"{path}.biases"] = shapes[f"{path}.scales"]
479
+ + missing, unexpected = set(shapes) - set(weights), set(weights) - set(shapes)
480
+ + if missing or unexpected:
481
+ + raise ValueError(
482
+ + f"Incomplete native checkpoint: missing={sorted(missing)}, unexpected={sorted(unexpected)}"
483
+ + )
484
+ + for name, shape in shapes.items():
485
+ + if tuple(weights[name].shape) != shape:
486
+ + raise ValueError(f"Native checkpoint shape mismatch: {name}")
487
+ + return weights
488
+ +
489
+ + @property
490
+ + def cast_predicate(self):
491
+ + return self.language_model.cast_predicate
492
+ diff --git a/mlx_lm/utils.py b/mlx_lm/utils.py
493
+ index a00fed8..6a751a2 100644
494
+ --- a/mlx_lm/utils.py
495
+ +++ b/mlx_lm/utils.py
496
+ @@ -195,6 +195,13 @@ def _transform_awq_weights(
497
+ return new_weights, mlx_quantization
498
+
499
+
500
+ +def _is_complete_qwen3_5(config: dict) -> bool:
501
+ + return (
502
+ + config.get("model_type") == "qwen3_5"
503
+ + and config.get("language_model_only") is False
504
+ + )
505
+ +
506
+ +
507
+ def _get_classes(config: dict):
508
+ """
509
+ Retrieve the model and model args classes based on the configuration.
510
+ @@ -216,6 +223,8 @@ def _get_classes(config: dict):
511
+ break
512
+ else:
513
+ model_type = MODEL_REMAPPING.get(model_type, model_type)
514
+ + if _is_complete_qwen3_5(config):
515
+ + model_type = "qwen3_5_full"
516
+ try:
517
+ arch = importlib.import_module(f"mlx_lm.models.{model_type}")
518
+ except ImportError as e:
519
+ @@ -446,12 +455,35 @@ def load_model(
520
+
521
+ weight_files = glob.glob(str(model_path / "model*.safetensors"))
522
+
523
+ + complete_qwen = _is_complete_qwen3_5(config)
524
+ + if complete_qwen:
525
+ + with open(model_path / "model.safetensors.index.json") as stream:
526
+ + index = json.load(stream)
527
+ + expected_files = set(index["weight_map"].values())
528
+ + actual_files = {Path(file).name for file in weight_files}
529
+ + if actual_files != expected_files:
530
+ + raise ValueError(
531
+ + "Incomplete full-model checkpoint: "
532
+ + f"missing shards={sorted(expected_files - actual_files)}, "
533
+ + f"unexpected shards={sorted(actual_files - expected_files)}"
534
+ + )
535
+ +
536
+ if not weight_files and strict:
537
+ raise FileNotFoundError(f"No safetensors found in {model_path}")
538
+
539
+ weights = {}
540
+ for wf in weight_files:
541
+ - weights.update(mx.load(wf))
542
+ + shard = mx.load(wf)
543
+ + if complete_qwen:
544
+ + duplicates = weights.keys() & shard.keys()
545
+ + if duplicates:
546
+ + raise ValueError(f"Duplicate checkpoint tensors: {sorted(duplicates)}")
547
+ + for name in shard:
548
+ + if index["weight_map"].get(name) != Path(wf).name:
549
+ + raise ValueError(f"Checkpoint index mismatch: {name}")
550
+ + weights.update(shard)
551
+ + if complete_qwen and weights.keys() != index["weight_map"].keys():
552
+ + raise ValueError("Checkpoint tensor set does not match its index")
553
+
554
+ if (model_file := config.get("model_file")) is not None:
555
+ if not trust_remote_code:
556
+ @@ -1064,9 +1096,11 @@ def save_config(
557
+ config (dict): The model configuration.
558
+ config_path (Union[str, Path]): Model configuration file path.
559
+ """
560
+ - # Clean unused keys
561
+ + config = copy.deepcopy(config)
562
+ + # Complete multimodal checkpoints need the vision architecture on reload.
563
+ config.pop("_name_or_path", None)
564
+ - config.pop("vision_config", None)
565
+ + if not _is_complete_qwen3_5(config):
566
+ + config.pop("vision_config", None)
567
+ if "quantization" in config:
568
+ config["quantization_config"] = config["quantization"]
569
+
570
+ @@ -1099,7 +1133,8 @@ def save(
571
+ save_config(config, config_path=dst_path / "config.json")
572
+ tokenizer.save_pretrained(dst_path)
573
+
574
+ - for p in ["*.py", "generation_config.json"]:
575
+ + extra_files = getattr(model, "extra_save_files", ())
576
+ + for p in ["*.py", "generation_config.json", *extra_files]:
577
+ for file in glob.glob(str(src_path / p)):
578
+ shutil.copy(file, dst_path)
579
+
580
+ diff --git a/tests/test_qwen3_5_full.py b/tests/test_qwen3_5_full.py
581
+ new file mode 100644
582
+ index 0000000..2b3e5ae
583
+ --- /dev/null
584
+ +++ b/tests/test_qwen3_5_full.py
585
+ @@ -0,0 +1,377 @@
586
+ +# Copyright © 2026 Apple Inc.
587
+ +
588
+ +"""Small synthetic fixtures test architecture code, not Swift model quality."""
589
+ +
590
+ +import copy
591
+ +import json
592
+ +from pathlib import Path
593
+ +from unittest.mock import patch
594
+ +
595
+ +import mlx.core as mx
596
+ +import numpy as np
597
+ +import pytest
598
+ +import torch
599
+ +from mlx.utils import tree_flatten
600
+ +from mlx_lm.convert import convert
601
+ +from mlx_lm.models.qwen3_5_full import Model, ModelArgs, OffsetRMSNorm
602
+ +from mlx_lm.utils import _get_classes, load_model, save_config
603
+ +from transformers import Qwen3_5Config, Qwen3_5ForConditionalGeneration
604
+ +from transformers.models.qwen3_5.modeling_qwen3_5 import (
605
+ + Qwen3_5DecoderLayer,
606
+ + Qwen3_5RMSNorm,
607
+ + Qwen3_5TextRotaryEmbedding,
608
+ +)
609
+ +
610
+ +
611
+ +def small_config():
612
+ + return {
613
+ + "model_type": "qwen3_5",
614
+ + "architectures": ["Qwen3_5ForConditionalGeneration"],
615
+ + "language_model_only": False,
616
+ + "tie_word_embeddings": False,
617
+ + "image_token_id": 125,
618
+ + "video_token_id": 126,
619
+ + "vision_start_token_id": 127,
620
+ + "text_config": {
621
+ + "model_type": "qwen3_5_text",
622
+ + "hidden_size": 128,
623
+ + "intermediate_size": 256,
624
+ + "num_hidden_layers": 4,
625
+ + "num_attention_heads": 4,
626
+ + "num_key_value_heads": 2,
627
+ + "head_dim": 32,
628
+ + "vocab_size": 128,
629
+ + "full_attention_interval": 4,
630
+ + "layer_types": ["linear_attention"] * 3 + ["full_attention"],
631
+ + "linear_num_key_heads": 2,
632
+ + "linear_num_value_heads": 4,
633
+ + "linear_key_head_dim": 32,
634
+ + "linear_value_head_dim": 32,
635
+ + "linear_conv_kernel_dim": 4,
636
+ + "hidden_act": "silu",
637
+ + "attn_output_gate": True,
638
+ + "output_gate_type": "swish",
639
+ + "mamba_ssm_dtype": "float32",
640
+ + "rms_norm_eps": 1e-6,
641
+ + "max_position_embeddings": 256,
642
+ + "tie_word_embeddings": False,
643
+ + "attention_bias": False,
644
+ + "attention_dropout": 0.0,
645
+ + "mtp_num_hidden_layers": 1,
646
+ + "mtp_use_dedicated_embeddings": False,
647
+ + "rope_parameters": {
648
+ + "rope_type": "default",
649
+ + "rope_theta": 10000000,
650
+ + "partial_rotary_factor": 0.5,
651
+ + "mrope_interleaved": True,
652
+ + "mrope_section": [3, 3, 2],
653
+ + },
654
+ + },
655
+ + "vision_config": {
656
+ + "model_type": "qwen3_5",
657
+ + "depth": 2,
658
+ + "hidden_size": 32,
659
+ + "intermediate_size": 48,
660
+ + "num_heads": 4,
661
+ + "out_hidden_size": 128,
662
+ + "num_position_embeddings": 16,
663
+ + "patch_size": 2,
664
+ + "temporal_patch_size": 2,
665
+ + "spatial_merge_size": 2,
666
+ + "in_channels": 3,
667
+ + "hidden_act": "gelu_pytorch_tanh",
668
+ + "deepstack_visual_indexes": [],
669
+ + },
670
+ + }
671
+ +
672
+ +
673
+ +class ReferenceMTP(torch.nn.Module):
674
+ + """Single-step composition used by the source vLLM MTP implementation."""
675
+ +
676
+ + def __init__(self, args):
677
+ + super().__init__()
678
+ + h = args.hidden_size
679
+ + self.fc = torch.nn.Linear(2 * h, h, bias=False)
680
+ + self.pre_fc_norm_embedding = Qwen3_5RMSNorm(h, args.rms_norm_eps)
681
+ + self.pre_fc_norm_hidden = Qwen3_5RMSNorm(h, args.rms_norm_eps)
682
+ + self.layers = torch.nn.ModuleList([Qwen3_5DecoderLayer(args, 3)])
683
+ + self.norm = Qwen3_5RMSNorm(h, args.rms_norm_eps)
684
+ + self.rotary = Qwen3_5TextRotaryEmbedding(args)
685
+ +
686
+ + def forward(self, hidden, embeds):
687
+ + x = self.fc(
688
+ + torch.cat(
689
+ + [self.pre_fc_norm_embedding(embeds), self.pre_fc_norm_hidden(hidden)],
690
+ + dim=-1,
691
+ + )
692
+ + )
693
+ + positions = torch.arange(x.shape[1])[None, None].expand(3, x.shape[0], -1)
694
+ + rotary = self.rotary(x, positions)
695
+ + mask = torch.triu(
696
+ + torch.full((x.shape[0], 1, x.shape[1], x.shape[1]), float("-inf")),
697
+ + diagonal=1,
698
+ + )
699
+ + return self.norm(
700
+ + self.layers[0](x, position_embeddings=rotary, attention_mask=mask)
701
+ + )
702
+ +
703
+ +
704
+ +@pytest.fixture(scope="module")
705
+ +def reference():
706
+ + torch.set_num_threads(2)
707
+ + torch.manual_seed(71)
708
+ + config = small_config()
709
+ + hf_config = Qwen3_5Config(**copy.deepcopy(config))
710
+ + hf_config._attn_implementation = "eager"
711
+ + hf_config.text_config._attn_implementation = "eager"
712
+ + hf_config.vision_config._attn_implementation = "eager"
713
+ + hf = Qwen3_5ForConditionalGeneration(hf_config).eval()
714
+ + mtp = ReferenceMTP(hf_config.text_config).eval()
715
+ + with torch.no_grad():
716
+ + for module in (hf, mtp):
717
+ + for name, value in module.named_parameters():
718
+ + if "norm" in name and value.ndim == 1 and "model.visual" not in name:
719
+ + value.uniform_(-0.15, 0.15)
720
+ + state = {k: mx.array(v.detach().numpy()) for k, v in hf.state_dict().items()}
721
+ + state.update(
722
+ + {"mtp." + k: mx.array(v.detach().numpy()) for k, v in mtp.state_dict().items()}
723
+ + )
724
+ + model = Model(ModelArgs.from_dict(config))
725
+ + model.load_weights(list(model.sanitize(state).items()), strict=True)
726
+ + model.eval()
727
+ + return config, hf, mtp, state, model
728
+ +
729
+ +
730
+ +def assert_close(mlx_value, torch_value, atol=3e-5, rtol=3e-5):
731
+ + np.testing.assert_allclose(
732
+ + np.array(mlx_value), torch_value.detach().numpy(), atol=atol, rtol=rtol
733
+ + )
734
+ +
735
+ +
736
+ +def test_dispatch_and_mapping_preserve_every_component(reference):
737
+ + config, _, _, state, model = reference
738
+ + before = copy.deepcopy(config)
739
+ + cls, args = _get_classes(config)
740
+ + fresh = cls(args.from_dict(config))
741
+ + assert cls is Model
742
+ + assert config == before
743
+ + rows = fresh.weight_mapping()
744
+ + assert {r["source"] for r in rows} == set(state)
745
+ + assert len(rows) == len(dict(tree_flatten(model.parameters())))
746
+ + assert sum(r["category"] == "MTP" for r in rows) == 15
747
+ + assert sum(r["category"] == "vision" for r in rows) == 33
748
+ + assert model.language_model.lm_head is not model.model.embed_tokens
749
+ + assert not any("mtp.embed" in r["destination"] for r in rows)
750
+ +
751
+ +
752
+ +@pytest.mark.parametrize("prefix", ["model.language_model.", "model.visual.", "mtp."])
753
+ +def test_missing_critical_tensor_is_rejected(reference, prefix):
754
+ + _, _, _, state, model = reference
755
+ + missing = dict(state)
756
+ + missing.pop(next(k for k in state if k.startswith(prefix)))
757
+ + with pytest.raises(ValueError, match="missing="):
758
+ + model.sanitize(missing)
759
+ +
760
+ +
761
+ +def test_unexpected_and_misshaped_tensors_are_rejected(reference):
762
+ + _, _, _, state, model = reference
763
+ + with pytest.raises(ValueError, match="unexpected="):
764
+ + model.sanitize(dict(state, **{"mtp.unknown.weight": mx.zeros((1,))}))
765
+ + wrong = dict(state)
766
+ + wrong["mtp.fc.weight"] = mx.zeros((1,))
767
+ + with pytest.raises(ValueError, match="Shape mismatch"):
768
+ + model.sanitize(wrong)
769
+ +
770
+ +
771
+ +def test_native_mapping_is_idempotent_and_norm_values_are_exact(reference):
772
+ + _, _, _, state, model = reference
773
+ + once = model.sanitize(state)
774
+ + twice = model.sanitize(once)
775
+ + for key in once:
776
+ + np.testing.assert_array_equal(np.array(once[key]), np.array(twice[key]))
777
+ + for row in model.weight_mapping():
778
+ + if "norm" in row["source"]:
779
+ + np.testing.assert_array_equal(
780
+ + np.array(state[row["source"]]), np.array(once[row["destination"]])
781
+ + )
782
+ +
783
+ +
784
+ +def test_offset_norm_preserves_small_bf16_deltas():
785
+ + layer = OffsetRMSNorm(4)
786
+ + raw = mx.array([0.0001, -0.0002, 0.0003, -0.0004], dtype=mx.bfloat16)
787
+ + layer.weight = raw
788
+ + x = mx.array([[0.8, -1.2, 0.4, 2.0]], dtype=mx.bfloat16)
789
+ + torch_layer = Qwen3_5RMSNorm(4)
790
+ + torch_layer.weight.data.copy_(torch.tensor(np.array(raw.astype(mx.float32))))
791
+ + expected = torch_layer(
792
+ + torch.tensor(np.array(x.astype(mx.float32))).to(torch.bfloat16)
793
+ + ).float()
794
+ + assert_close(layer(x).astype(mx.float32), expected, atol=0, rtol=0)
795
+ + np.testing.assert_array_equal(
796
+ + np.array(layer.weight.astype(mx.float32)), np.array(raw.astype(mx.float32))
797
+ + )
798
+ +
799
+ +
800
+ +def test_text_forward_matches_transformers(reference):
801
+ + _, hf, _, _, model = reference
802
+ + ids = torch.tensor([[3, 19, 8, 21, 11]])
803
+ + with torch.no_grad():
804
+ + expected = hf(input_ids=ids, use_cache=False).logits
805
+ + actual = model(mx.array(ids.numpy()))
806
+ + assert_close(actual, expected)
807
+ +
808
+ +
809
+ +def test_text_decode_cache_matches_full_forward(reference):
810
+ + _, _, _, _, model = reference
811
+ + ids = mx.array([[3, 19, 8, 21, 11]])
812
+ + full = model(ids)
813
+ + cache = model.make_cache()
814
+ + parts = [model(ids[:, :3], cache)]
815
+ + parts.extend(model(ids[:, i : i + 1], cache) for i in range(3, 5))
816
+ + np.testing.assert_allclose(
817
+ + np.array(mx.concatenate(parts, axis=1)), np.array(full), atol=3e-5, rtol=3e-5
818
+ + )
819
+ +
820
+ +
821
+ +@pytest.mark.parametrize("grid", [[[1, 4, 4]], [[2, 2, 4], [1, 4, 2]], [[1, 6, 4]]])
822
+ +def test_vision_encoder_matches_transformers(reference, grid):
823
+ + _, hf, _, _, model = reference
824
+ + torch.manual_seed(11)
825
+ + count = sum(t * h * w for t, h, w in grid)
826
+ + pixels = torch.randn(count, 3 * 2 * 2 * 2)
827
+ + with torch.no_grad():
828
+ + expected = hf.model.visual(pixels, torch.tensor(grid))
829
+ + hidden, pooled = model.visual(
830
+ + mx.array(pixels.numpy()), grid, return_hidden_states=True
831
+ + )
832
+ + assert_close(hidden, expected.last_hidden_state)
833
+ + assert_close(pooled, expected.pooler_output)
834
+ +
835
+ +
836
+ +def test_mtp_forward_and_shared_head_match_reference(reference):
837
+ + _, hf, mtp, _, model = reference
838
+ + torch.manual_seed(9)
839
+ + hidden = torch.randn(1, 4, 128)
840
+ + ids = torch.tensor([[9, 8, 7, 6]])
841
+ + with torch.no_grad():
842
+ + embeds = hf.model.language_model.embed_tokens(ids)
843
+ + expected = mtp(hidden, embeds)
844
+ + expected_logits = hf.lm_head(expected)
845
+ + assert_close(
846
+ + model.mtp(mx.array(hidden.numpy()), mx.array(embeds.numpy())), expected
847
+ + )
848
+ + assert_close(
849
+ + model.mtp_logits(mx.array(ids.numpy()), mx.array(hidden.numpy())),
850
+ + expected_logits,
851
+ + )
852
+ + cache = model.mtp.make_cache()
853
+ + cached = mx.concatenate(
854
+ + [
855
+ + model.mtp(
856
+ + mx.array(hidden[:, i : i + 1].numpy()),
857
+ + mx.array(embeds[:, i : i + 1].numpy()),
858
+ + cache,
859
+ + )
860
+ + for i in range(4)
861
+ + ],
862
+ + axis=1,
863
+ + )
864
+ + assert_close(cached, expected)
865
+ +
866
+ +
867
+ +def test_unsupported_multimodal_generation_fails_explicitly(reference):
868
+ + _, _, _, _, model = reference
869
+ + with pytest.raises(NotImplementedError, match="Multimodal"):
870
+ + model(mx.array([[1, 2]]), pixel_values=mx.zeros((1, 24)))
871
+ + with pytest.raises(NotImplementedError, match="Multimodal"):
872
+ + model(mx.array([[1, 125]]))
873
+ +
874
+ +
875
+ +def test_save_config_keeps_vision_and_does_not_mutate_input(tmp_path):
876
+ + config = small_config()
877
+ + before = copy.deepcopy(config)
878
+ + save_config(config, tmp_path / "config.json")
879
+ + assert config == before
880
+ + assert json.loads((tmp_path / "config.json").read_text()) == before
881
+ +
882
+ +
883
+ +def test_official_nonquantized_convert_and_reload_preserve_all_tensors(
884
+ + reference, tmp_path
885
+ +):
886
+ + config, _, _, state, model = reference
887
+ + source, output = tmp_path / "hf", tmp_path / "mlx"
888
+ + source.mkdir()
889
+ + (source / "config.json").write_text(json.dumps(config))
890
+ + mx.save_safetensors(str(source / "model.safetensors"), state)
891
+ + (source / "model.safetensors.index.json").write_text(
892
+ + json.dumps({"weight_map": {k: "model.safetensors" for k in state}})
893
+ + )
894
+ + assets = ["generation_config.json", *model.extra_save_files]
895
+ + for name in assets:
896
+ + (source / name).write_text("original synthetic asset: " + name)
897
+ + (source / "generation_config.json").write_text('{"eos_token_id": 2}')
898
+ +
899
+ + class FixtureTokenizer:
900
+ + def save_pretrained(self, target):
901
+ + Path(target, "tokenizer_config.json").write_text("rewritten")
902
+ +
903
+ + def fixture_load(path, **kwargs):
904
+ + loaded, loaded_config = load_model(Path(path), lazy=True, strict=True)
905
+ + return loaded, FixtureTokenizer(), loaded_config
906
+ +
907
+ + with patch("mlx_lm.convert.load", side_effect=fixture_load):
908
+ + convert(str(source), str(output), quantize=False)
909
+ + loaded, saved_config = load_model(output, lazy=False, strict=True)
910
+ + expected = model.sanitize(state)
911
+ + actual = dict(tree_flatten(loaded.parameters()))
912
+ + assert set(actual) == set(expected)
913
+ + for name in actual:
914
+ + np.testing.assert_array_equal(np.array(actual[name]), np.array(expected[name]))
915
+ + assert saved_config["vision_config"] == config["vision_config"]
916
+ + assert saved_config["text_config"] == config["text_config"]
917
+ + assert "quantization" not in saved_config
918
+ + for name in assets:
919
+ + assert (source / name).read_bytes() == (output / name).read_bytes()
920
+ + ids = mx.array([[3, 19, 8]])
921
+ + np.testing.assert_array_equal(np.array(model(ids)), np.array(loaded(ids)))
922
+ + (output / "model.safetensors.index.json").write_text(
923
+ + json.dumps(
924
+ + {
925
+ + "weight_map": {
926
+ + **{k: "model.safetensors" for k in actual},
927
+ + "mtp.missing.weight": "missing.safetensors",
928
+ + }
929
+ + }
930
+ + )
931
+ + )
932
+ + with pytest.raises(ValueError, match="missing shards"):
933
+ + load_model(output, lazy=True)
934
+ +
935
+ +
936
+ +def test_fixed_affine_quantized_roundtrip_preserves_component_tree(reference, tmp_path):
937
+ + from mlx_lm.utils import quantize_model, save_model
938
+ +
939
+ + config, _, _, state, _ = reference
940
+ + model = Model(ModelArgs.from_dict(config))
941
+ + model.load_weights(list(model.sanitize(state).items()), strict=True)
942
+ + original_names = set(dict(tree_flatten(model.parameters())))
943
+ + model, quantized_config = quantize_model(model, config, 64, 4, mode="affine")
944
+ + save_model(tmp_path, model)
945
+ + save_config(quantized_config, tmp_path / "config.json")
946
+ + loaded, saved_config = load_model(tmp_path, lazy=False, strict=True)
947
+ + actual = dict(tree_flatten(loaded.parameters()))
948
+ + expected = dict(tree_flatten(model.parameters()))
949
+ + assert set(actual) == set(expected)
950
+ + assert original_names <= set(actual)
951
+ + assert saved_config["quantization"] == {
952
+ + "bits": 4,
953
+ + "group_size": 64,
954
+ + "mode": "affine",
955
+ + }
956
+ + for name in actual:
957
+ + np.testing.assert_array_equal(np.array(actual[name]), np.array(expected[name]))
958
+ + assert bool(mx.all(mx.isfinite(loaded(mx.array([[3, 19, 8]])))))
959
+ + missing = dict(actual)
960
+ + missing.pop("mtp.fc.scales")
961
+ + with pytest.raises(ValueError, match="Incomplete native checkpoint"):
962
+ + loaded.sanitize(missing)
compatibility/unmatched-tensor-analysis.csv ADDED
The diff for this file is too large to render. See raw diff
 
compatibility/validate-mac-format.py ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Validate complete native parameter headers and real Q4 samples on Metal.
2
+
3
+ This is not a full 27B Mac generation test. Header fixtures are never saved.
4
+ """
5
+ import importlib.metadata
6
+ import argparse
7
+ import json
8
+ import platform
9
+ import resource
10
+ from pathlib import Path
11
+ import mlx.core as mx
12
+ import mlx.nn as nn
13
+ import numpy as np
14
+ from mlx.utils import tree_flatten
15
+ from mlx_lm.utils import _get_classes
16
+
17
+ parser=argparse.ArgumentParser(description=__doc__)
18
+ parser.add_argument('--folder',type=Path,default=Path(__file__).resolve().parent/'mac-check')
19
+ parser.add_argument('--output',type=Path,default=Path('mac-compatibility-results.json'))
20
+ options=parser.parse_args()
21
+ folder=options.folder
22
+ assert platform.system()=='Darwin' and platform.machine()=='arm64'
23
+ assert mx.metal.is_available()
24
+ mx.set_default_device(mx.gpu)
25
+ config=json.loads((folder/'config.json').read_text())
26
+ assert config['quantization']=={'group_size':64,'bits':4,'mode':'affine'}
27
+ headers=json.loads((folder/'checkpoint-headers.json').read_text())
28
+ index=json.loads((folder/'model.safetensors.index.json').read_text())
29
+ assert set(headers)==set(index['weight_map'])
30
+ cls,args=_get_classes(config)
31
+ model=cls(args.from_dict(config))
32
+ dtype={'BF16':mx.bfloat16,'F16':mx.float16,'F32':mx.float32,'U32':mx.uint32}
33
+ fixtures={k:mx.broadcast_to(mx.array(0,dtype=dtype[v['dtype']]),v['shape']) for k,v in headers.items()}
34
+ fixtures=model.sanitize(fixtures)
35
+ nn.quantize(model,group_size=64,bits=4,mode='affine',class_predicate=lambda path,module: path+'.scales' in fixtures)
36
+ model.load_weights(list(fixtures.items()),strict=True)
37
+ assert set(dict(tree_flatten(model.parameters())))==set(headers)
38
+ assert len(model.weight_mapping())==1199
39
+ del fixtures,model
40
+
41
+ samples=mx.load(str(folder/'real-checkpoint-samples.safetensors'))
42
+ description=json.loads((folder/'samples.json').read_text())
43
+ results=[]
44
+ for item in description['samples']:
45
+ prefix=f"sample_{item['sample']}."
46
+ x,w,s,b,ref=[samples[prefix+k] for k in ('input','weight','scales','biases','fp32_reference')]
47
+ assert w.dtype==mx.uint32 and s.dtype==mx.bfloat16 and b.dtype==mx.bfloat16
48
+ native=mx.quantized_matmul(x,w,s,b,transpose=True,group_size=64,bits=4,mode='affine')
49
+ fp32=mx.quantized_matmul(x.astype(mx.float32),w,s.astype(mx.float32),b.astype(mx.float32),transpose=True,group_size=64,bits=4,mode='affine')
50
+ mx.eval(native,fp32,ref)
51
+ a,r,f=[np.array(value.astype(mx.float32)) for value in (native,ref,fp32)]
52
+ np.testing.assert_allclose(a,r,atol=0.01,rtol=0.02)
53
+ np.testing.assert_allclose(f,r,atol=1e-4,rtol=5e-4)
54
+ results.append(dict(item,native_metal_bf16='PASS',metal_fp32='PASS',bf16_max_absolute_error=float(np.max(np.abs(a-r))),fp32_max_absolute_error=float(np.max(np.abs(f-r)))))
55
+ result={'status':'PASS_MAC_NATIVE_MLX_FORMAT_AND_REAL_METAL_SAMPLES','platform':platform.platform(),'machine':platform.machine(),'mlx':importlib.metadata.version('mlx'),'mlx_lm':importlib.metadata.version('mlx-lm'),'device':str(mx.default_device()),'quantization':config['quantization'],'all_saved_tensor_headers_validated':len(headers),'all_source_parameters_accounted_for':1199,'strict_complete_parameter_tree':'PASS using unevaluated header fixtures; no fabricated weights saved','actual_checkpoint_samples':results,'full_model_mac_generation':'NOT_RUN: targeted native Metal format and real-weight component checks only','peak_mlx_bytes':mx.get_peak_memory(),'peak_process_rss_bytes':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss}
56
+ with options.output.open('x') as stream:json.dump(result,stream,indent=2)
57
+ print(json.dumps(result,indent=2))
compatibility/validate-quant.py ADDED
@@ -0,0 +1,134 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Validate the saved real Swift MLX artifact, including all components."""
2
+ import hashlib
3
+ import json
4
+ import platform
5
+ import resource
6
+ import time
7
+ from collections import Counter
8
+ from datetime import datetime, timezone
9
+ from pathlib import Path
10
+ import os
11
+
12
+ import mlx.core as mx
13
+ from mlx.utils import tree_flatten
14
+ from mlx_lm import load, stream_generate
15
+ from mlx_lm.sample_utils import make_sampler
16
+ from transformers import AutoProcessor, AutoTokenizer
17
+ from PIL import Image
18
+
19
+ root = Path(__file__).resolve().parents[1]
20
+ source = Path(os.environ['SWIFT_SOURCE_DIR']).resolve(strict=True)
21
+ output = Path(os.environ.get('SWIFT_MLX_OUTPUT', root / 'Swift-1.5-4bit-MLX')).resolve(strict=True)
22
+ logs = Path(os.environ.get('SWIFT_VALIDATION_DIR', root / 'validation-output')).resolve(strict=True)
23
+ verified = json.loads((logs / 'source-verification.json').read_text())
24
+ conversion = json.loads((logs / 'conversion-result.json').read_text())
25
+ assert conversion['returncode'] == 0
26
+ assert not (logs / 'quant-validation-results.json').exists()
27
+ mx.set_default_device(mx.cpu)
28
+ start = time.monotonic()
29
+ model, tokenizer, config = load(str(output), lazy=False, return_config=True)
30
+ load_seconds = time.monotonic() - start
31
+ load_memory = mx.get_active_memory()
32
+ parameters = dict(tree_flatten(model.parameters()))
33
+ assert type(model).__module__ == 'mlx_lm.models.qwen3_5_full'
34
+ assert config['quantization'] == {'mode': 'affine', 'bits': 4, 'group_size': 64}
35
+ assert config['vision_config'] == verified['config']['vision_config']
36
+ assert config['text_config'] == verified['config']['text_config']
37
+ assert config['tie_word_embeddings'] == verified['config']['tie_word_embeddings']
38
+ rows = model.weight_mapping()
39
+ assert {r['source'] for r in rows} == set(verified['tensors'])
40
+ assert len(rows) == 1199
41
+ accounted = set()
42
+ for row in rows:
43
+ assert list(row['source_shape']) == verified['tensors'][row['source']]['shape']
44
+ name = row['destination']
45
+ assert name in parameters, name
46
+ native = [name]
47
+ if name.endswith('.weight') and name[:-7]+'.scales' in parameters:
48
+ native += [name[:-7]+'.scales', name[:-7]+'.biases']
49
+ assert parameters[name].dtype == mx.uint32
50
+ row['storage'] = 'affine/4-bit/group-size-64'
51
+ else:
52
+ assert parameters[name].dtype == mx.bfloat16, name
53
+ row['storage'] = 'original BF16, with the documented layout mapping'
54
+ row['saved_tensors'] = native
55
+ accounted.update(native)
56
+ assert accounted == set(parameters)
57
+ categories = dict(Counter(r['category'] for r in rows))
58
+ assert categories == {'text': 851, 'MTP': 15, 'vision': 333}
59
+ for name, value in parameters.items():
60
+ if mx.issubdtype(value.dtype, mx.floating):
61
+ assert bool(mx.all(mx.isfinite(value))), f'Nonfinite values in {name}'
62
+ print(f'Loaded {len(parameters)} saved tensors; all 1199 source tensors accounted for.', flush=True)
63
+
64
+ # Compare every unquantized source tensor bit-for-bit after its required layout change.
65
+ unchanged = [r for r in rows if r['storage'].startswith('original BF16')]
66
+ for shard in sorted({verified['tensors'][r['source']]['shard'] for r in unchanged}):
67
+ raw = mx.load(str(source / shard))
68
+ for row in unchanged:
69
+ if verified['tensors'][row['source']]['shard'] != shard:
70
+ continue
71
+ value = raw[row['source']]
72
+ if row['transform'] == 'transpose(0,2,1)':
73
+ value = value.transpose(0, 2, 1)
74
+ elif row['transform'] == 'transpose(0,2,3,4,1)':
75
+ value = value.transpose(0, 2, 3, 4, 1)
76
+ assert bool(mx.all(value == parameters[row['destination']])), row['source']
77
+ del raw
78
+ print(f'All {len(unchanged)} unquantized tensors equal the original BF16 values.', flush=True)
79
+
80
+ assets = {}
81
+ for name in ['generation_config.json', *model.extra_save_files]:
82
+ if (source / name).is_file():
83
+ assert (source / name).read_bytes() == (output / name).read_bytes(), name
84
+ assets[name] = hashlib.sha256((output / name).read_bytes()).hexdigest()
85
+ hf_tokenizer = AutoTokenizer.from_pretrained(output, local_files_only=True, trust_remote_code=False)
86
+ processor = AutoProcessor.from_pretrained(output, local_files_only=True, trust_remote_code=False)
87
+ chats = []
88
+ for options in ({'enable_thinking': False}, {'reasoning_effort': 'low'}, {'reasoning_effort': 'xhigh'}):
89
+ prompt = hf_tokenizer.apply_chat_template([{'role': 'user', 'content': 'Say hello.'}], tokenize=False, add_generation_prompt=True, **options)
90
+ assert prompt and hf_tokenizer.encode(prompt, add_special_tokens=False)
91
+ chats.append({'options': options, 'rendered': prompt})
92
+ mapping = {'source_tensors':1199, 'mapped_source_tensors':1199, 'native_tensors':len(parameters), 'ignored':0, 'unexplained':0, 'rows':rows}
93
+ mapping_path = logs / 'quant-tensor-mapping-manifest.json'
94
+ if mapping_path.exists():
95
+ assert json.loads(mapping_path.read_text()) == json.loads(json.dumps(mapping))
96
+ else:
97
+ with mapping_path.open('x') as f:
98
+ json.dump(mapping, f, indent=2)
99
+
100
+ # The official Linux scalar BF16 QMM accumulates in BF16 (8192 ones -> 256).
101
+ # Promote only in-memory floating values; packed 4-bit tensors/files stay unchanged.
102
+ model.apply(lambda value: value.astype(mx.float32) if mx.issubdtype(value.dtype, mx.floating) else value)
103
+ mx.eval(model.parameters())
104
+ print('CPU inference uses FP32 floating values; saved 4-bit weights are unchanged.', flush=True)
105
+
106
+ prompt = tokenizer.apply_chat_template([{'role':'user','content':'Reply with exactly: Hello from Swift.'}], tokenize=False, add_generation_prompt=True, enable_thinking=False)
107
+ pieces, tokens, last = [], [], None
108
+ generation_start = time.monotonic()
109
+ for response in stream_generate(model, tokenizer, prompt=prompt, max_tokens=24, sampler=make_sampler(temp=0.0), prefill_step_size=64):
110
+ assert bool(mx.all(mx.isfinite(response.logprobs))), 'Nonfinite generation probabilities'
111
+ pieces.append(response.text); tokens.append(response.token); last = response
112
+ print(response.text, end='', flush=True)
113
+ generated = ''.join(pieces)
114
+ assert generated.strip(), 'Empty text generation'
115
+ print('\nText generation passed.', flush=True)
116
+ generation = {'prompt':prompt, 'text':generated, 'token_ids':tokens, 'tokens':last.generation_tokens, 'tokens_per_second':last.generation_tps, 'prompt_tokens_per_second':last.prompt_tps, 'elapsed_seconds':time.monotonic()-generation_start, 'finish_reason':last.finish_reason}
117
+
118
+ ids = mx.array([hf_tokenizer.encode('Hello', add_special_tokens=False)[:2]], dtype=mx.int32)
119
+ hidden = model.model(ids)
120
+ mtp = model.mtp_logits(ids, hidden)
121
+ mx.eval(mtp)
122
+ assert bool(mx.all(mx.isfinite(mtp)))
123
+ mtp_result = {'status':'PASS', 'shape':list(mtp.shape), 'path':'Explicit MTP step with real text hidden states and shared LM head; speculative generation is not integrated'}
124
+ print('Real-weight MTP step passed.', flush=True)
125
+ pixels = processor.image_processor(images=[Image.new('RGB', (256,256), (64,128,192))], return_tensors='np')
126
+ features = model.visual(mx.array(pixels['pixel_values']), pixels['image_grid_thw'])
127
+ mx.eval(features)
128
+ assert bool(mx.all(mx.isfinite(features)))
129
+ vision_result = {'status':'PASS', 'shape':list(features.shape), 'grid':pixels['image_grid_thw'].tolist(), 'path':'Vision encoder only; image/video insertion and multimodal text generation are not implemented'}
130
+ print('Real-weight vision encoder passed.', flush=True)
131
+
132
+ result = {'status':'PASS', 'recorded_at':datetime.now(timezone.utc).isoformat(), 'source_repo':conversion['source_repo'], 'source_revision':conversion['source_revision'], 'source_shards':18, 'source_shard_bytes':verified['shard_bytes'], 'source_tensors':1199, 'mapped_source_tensors':1199, 'saved_tensors':len(parameters), 'categories':categories, 'ignored_tensors':0, 'unexplained_tensors':0, 'exact_unquantized_tensors':len(unchanged), 'quantization':config['quantization'], 'all_floating_tensors_finite':True, 'tokenizer':type(hf_tokenizer).__name__, 'processor':type(processor).__name__, 'assets_sha256':assets, 'chat_templates':chats, 'load_seconds':load_seconds, 'load_memory_bytes':load_memory, 'process_peak_rss_bytes':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss * (1024 if platform.system()=='Linux' else 1), 'inference_floating_dtype':'float32, CPU runtime only; stored floating tensors remain BF16', 'native_bf16_cpu_inference':'Aborted after reproducing incorrect accumulation in the official Linux BF16 quantized matmul. See cpu-quantized-matmul-diagnostic.json.', 'generation':generation, 'mtp':mtp_result, 'vision':vision_result, 'total_validation_seconds':time.monotonic()-start}
133
+ with (logs / 'quant-validation-results.json').open('x') as f:json.dump(result,f,indent=2)
134
+ print(json.dumps(result,indent=2),flush=True)
compatibility/validate_aws_source_linux.py ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Load the actual verified Swift BF16 source lazily, without quantization."""
2
+
3
+ import argparse
4
+ import hashlib
5
+ import json
6
+ import platform
7
+ import resource
8
+ import time
9
+ from collections import Counter
10
+ from datetime import datetime, timezone
11
+ from pathlib import Path
12
+
13
+ import mlx.core as mx
14
+ from mlx.utils import tree_flatten
15
+ from transformers import AutoProcessor, AutoTokenizer
16
+
17
+ from mlx_lm.utils import load_model
18
+
19
+
20
+ def main():
21
+ parser = argparse.ArgumentParser(description=__doc__)
22
+ parser.add_argument("--source", required=True, type=Path)
23
+ parser.add_argument("--verification", required=True, type=Path)
24
+ parser.add_argument("--output", required=True, type=Path)
25
+ args = parser.parse_args()
26
+ if args.output.exists():
27
+ raise FileExistsError(args.output)
28
+ source = args.source.resolve(strict=True)
29
+ verified = json.loads(args.verification.read_text())
30
+ assert verified["status"] == "PASS"
31
+ assert Path(verified["source_root"]).resolve() == source
32
+ manifest_hash = "0a00065b88ab003281853a7fb9bd5ce0086bc3781b36136d8c39da19933923ae"
33
+ assert verified["manifest_sha256"] == manifest_hash
34
+ assert hashlib.sha256((source / "EXPORT_MANIFEST.json").read_bytes()).hexdigest() == manifest_hash
35
+ assert verified["shard_count"] == 18
36
+ assert verified["shard_bytes"] == 55563006776
37
+ assert verified["tensor_count"] == 1199
38
+ for item in verified["files"]:
39
+ assert (source / item["name"]).stat().st_size == item["bytes"]
40
+
41
+ started = time.monotonic()
42
+ model, config = load_model(source, lazy=True, strict=True)
43
+ assert type(model).__module__ == "mlx_lm.models.qwen3_5_full"
44
+ assert not config.get("quantization") and not config.get("quantization_config")
45
+ rows = model.weight_mapping()
46
+ parameters = dict(tree_flatten(model.parameters()))
47
+ assert {row["source"] for row in rows} == set(verified["tensors"])
48
+ assert {row["destination"] for row in rows} == set(parameters)
49
+ for row in rows:
50
+ header = verified["tensors"][row["source"]]
51
+ value = parameters[row["destination"]]
52
+ assert list(row["source_shape"]) == header["shape"]
53
+ assert list(value.shape) == list(row["destination_shape"])
54
+ assert value.dtype == mx.bfloat16 and header["dtype"] == "BF16"
55
+ categories = dict(Counter(row["category"] for row in rows))
56
+ assert categories == {"text": 851, "vision": 333, "MTP": 15}
57
+
58
+ tokenizer = AutoTokenizer.from_pretrained(source, local_files_only=True, trust_remote_code=False)
59
+ assert tokenizer.chat_template
60
+ chats = []
61
+ for options in [{"enable_thinking": False}, {"reasoning_effort": "low"}, {"reasoning_effort": "xhigh"}]:
62
+ prompt = tokenizer.apply_chat_template(
63
+ [{"role": "user", "content": "Say hello."}],
64
+ tokenize=False, add_generation_prompt=True, **options,
65
+ )
66
+ ids = tokenizer.encode(prompt, add_special_tokens=False)
67
+ assert ids and all(0 <= token < config["text_config"]["vocab_size"] for token in ids)
68
+ chats.append({"options": options, "tokens": len(ids), "rendered": prompt})
69
+ processor = AutoProcessor.from_pretrained(source, local_files_only=True, trust_remote_code=False)
70
+ result = {
71
+ "status": "PASS_ACTUAL_SOURCE_LAZY_STRUCTURAL_LOAD",
72
+ "recorded_at": datetime.now(timezone.utc).isoformat(),
73
+ "platform": platform.platform(),
74
+ "source": str(source),
75
+ "source_repo": "ukisai/Swift-1.5-Qwen3.8-27b",
76
+ "source_revision": "00ccd14e006897d28cb0ed5bf26390e60d274251",
77
+ "source_verification": str(args.verification.resolve()),
78
+ "source_verification_sha256": hashlib.sha256(args.verification.read_bytes()).hexdigest(),
79
+ "source_tensors": len(verified["tensors"]),
80
+ "mapped_tensors": len(parameters),
81
+ "categories": categories,
82
+ "ignored_tensors": 0,
83
+ "unexplained_tensors": 0,
84
+ "weight_backing": "actual verified source safetensors; lazy loading",
85
+ "full_parameter_evaluation": False,
86
+ "tokenizer": type(tokenizer).__name__,
87
+ "chat_templates": chats,
88
+ "processor": type(processor).__name__,
89
+ "generation": "NOT_RUN",
90
+ "mtp_runtime": "component-only; no integrated speculative decoding",
91
+ "vision_runtime": "encoder-only; no integrated multimodal generation",
92
+ "quantization_executed": False,
93
+ "elapsed_seconds": time.monotonic() - started,
94
+ "mlx_active_memory_bytes": mx.get_active_memory(),
95
+ "process_peak_rss_bytes": resource.getrusage(resource.RUSAGE_SELF).ru_maxrss * (1024 if platform.system() == "Linux" else 1),
96
+ }
97
+ with args.output.open("x") as stream:
98
+ json.dump(result, stream, indent=2)
99
+ stream.write("\n")
100
+ print(json.dumps(result, indent=2))
101
+
102
+
103
+ if __name__ == "__main__":
104
+ main()
compatibility/verify_source_readonly.py ADDED
@@ -0,0 +1,92 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Verify the original Swift export without changing or loading its weights."""
2
+ import argparse
3
+ import hashlib
4
+ import json
5
+ import math
6
+ import struct
7
+ import sys
8
+ from datetime import datetime, timezone
9
+ from pathlib import Path
10
+
11
+ parser = argparse.ArgumentParser()
12
+ parser.add_argument("--root", required=True)
13
+ parser.add_argument("--manifest-sha256", required=True)
14
+ args = parser.parse_args()
15
+ root = Path(args.root)
16
+
17
+
18
+ def digest(path):
19
+ result = hashlib.sha256()
20
+ with path.open("rb") as stream:
21
+ for block in iter(lambda: stream.read(16 * 1024 * 1024), b""):
22
+ result.update(block)
23
+ return result.hexdigest()
24
+
25
+
26
+ manifest_path = root / "EXPORT_MANIFEST.json"
27
+ assert digest(manifest_path) == args.manifest_sha256, "Export provenance mismatch"
28
+ manifest = json.loads(manifest_path.read_text())
29
+ assert manifest["dtype"] == "bfloat16"
30
+ shards = manifest["output_shards"]
31
+ assert len(shards) == 18
32
+ index = json.loads((root / "model.safetensors.index.json").read_text())
33
+ expected_shards = {item["file"] for item in shards}
34
+ assert set(index["weight_map"].values()) == expected_shards
35
+ assert {f.name for f in root.glob("*.safetensors")} == expected_shards
36
+ files, tensors = [], {}
37
+ for item in shards:
38
+ name = item["file"]
39
+ assert Path(name).name == name
40
+ path = root / name
41
+ size = path.stat().st_size
42
+ assert size == item["bytes"], f"Size mismatch: {name}"
43
+ sha = digest(path)
44
+ assert sha == item["sha256"], f"SHA256 mismatch: {name}"
45
+ with path.open("rb") as stream:
46
+ header_size = struct.unpack("<Q", stream.read(8))[0]
47
+ assert 0 < header_size < 16 * 1024 * 1024
48
+ header = json.loads(stream.read(header_size))
49
+ ranges = []
50
+ for tensor_name, metadata in header.items():
51
+ if tensor_name == "__metadata__":
52
+ continue
53
+ assert tensor_name not in tensors, f"Duplicate tensor: {tensor_name}"
54
+ assert index["weight_map"][tensor_name] == name
55
+ assert metadata["dtype"] == "BF16", tensor_name
56
+ start, end = metadata["data_offsets"]
57
+ count = math.prod(metadata["shape"])
58
+ assert end - start == count * 2
59
+ ranges.append((start, end))
60
+ tensors[tensor_name] = dict(metadata, shard=name, parameters=count, bytes=end-start)
61
+ ranges.sort()
62
+ assert ranges[0][0] == 0
63
+ assert all(left[1] == right[0] for left, right in zip(ranges, ranges[1:]))
64
+ assert ranges[-1][1] + 8 + header_size == size
65
+ files.append({"name": name, "bytes": size, "sha256": sha, "kind": "bf16_shard"})
66
+ print(f"VERIFIED {name} {size} bytes", file=sys.stderr, flush=True)
67
+ assert set(tensors) == set(index["weight_map"])
68
+ assert sum(v["bytes"] for v in tensors.values()) == manifest["total_tensor_bytes"]
69
+ assert index["metadata"]["total_size"] == manifest["total_tensor_bytes"]
70
+ for name, sha in manifest["asset_sha256"].items():
71
+ path = root / name
72
+ assert path.is_file(), f"Missing asset: {name}"
73
+ actual = digest(path)
74
+ assert actual == sha, f"Asset hash mismatch: {name}"
75
+ files.append({"name": name, "bytes": path.stat().st_size, "sha256": actual, "kind": "source_asset"})
76
+ recorded = {item["name"] for item in files}
77
+ for path in sorted(root.iterdir()):
78
+ if path.is_file() and path.name not in recorded:
79
+ files.append({"name": path.name, "bytes": path.stat().st_size, "sha256": digest(path), "kind": "additional_source_file"})
80
+ result = {
81
+ "status": "PASS", "recorded_at": datetime.now(timezone.utc).isoformat(),
82
+ "source_root": str(root), "manifest_sha256": args.manifest_sha256,
83
+ "shard_count": len(shards), "shard_bytes": sum(item["bytes"] for item in shards),
84
+ "source_file_bytes": sum(item["bytes"] for item in files),
85
+ "tensor_count": len(tensors), "parameter_count": sum(v["parameters"] for v in tensors.values()),
86
+ "files": files, "tensors": tensors,
87
+ "config": json.loads((root / "config.json").read_text()),
88
+ "custom_modeling_files": [path.name for path in root.glob("*.py")],
89
+ "special_tokens_map_present": (root / "special_tokens_map.json").is_file(),
90
+ "original_unchanged": True,
91
+ }
92
+ print(json.dumps(result, indent=2))
config.json ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForConditionalGeneration"
4
+ ],
5
+ "eos_token_id": [
6
+ 248046,
7
+ 248044
8
+ ],
9
+ "image_token_id": 248056,
10
+ "language_model_only": false,
11
+ "model_type": "qwen3_5",
12
+ "quantization": {
13
+ "group_size": 64,
14
+ "bits": 4,
15
+ "mode": "affine"
16
+ },
17
+ "quantization_config": {
18
+ "group_size": 64,
19
+ "bits": 4,
20
+ "mode": "affine"
21
+ },
22
+ "text_config": {
23
+ "attention_bias": false,
24
+ "attention_dropout": 0.0,
25
+ "attn_output_gate": true,
26
+ "bos_token_id": 248044,
27
+ "dtype": "bfloat16",
28
+ "eos_token_id": 248044,
29
+ "full_attention_interval": 4,
30
+ "head_dim": 256,
31
+ "hidden_act": "silu",
32
+ "hidden_size": 5120,
33
+ "initializer_range": 0.02,
34
+ "intermediate_size": 17408,
35
+ "layer_types": [
36
+ "linear_attention",
37
+ "linear_attention",
38
+ "linear_attention",
39
+ "full_attention",
40
+ "linear_attention",
41
+ "linear_attention",
42
+ "linear_attention",
43
+ "full_attention",
44
+ "linear_attention",
45
+ "linear_attention",
46
+ "linear_attention",
47
+ "full_attention",
48
+ "linear_attention",
49
+ "linear_attention",
50
+ "linear_attention",
51
+ "full_attention",
52
+ "linear_attention",
53
+ "linear_attention",
54
+ "linear_attention",
55
+ "full_attention",
56
+ "linear_attention",
57
+ "linear_attention",
58
+ "linear_attention",
59
+ "full_attention",
60
+ "linear_attention",
61
+ "linear_attention",
62
+ "linear_attention",
63
+ "full_attention",
64
+ "linear_attention",
65
+ "linear_attention",
66
+ "linear_attention",
67
+ "full_attention",
68
+ "linear_attention",
69
+ "linear_attention",
70
+ "linear_attention",
71
+ "full_attention",
72
+ "linear_attention",
73
+ "linear_attention",
74
+ "linear_attention",
75
+ "full_attention",
76
+ "linear_attention",
77
+ "linear_attention",
78
+ "linear_attention",
79
+ "full_attention",
80
+ "linear_attention",
81
+ "linear_attention",
82
+ "linear_attention",
83
+ "full_attention",
84
+ "linear_attention",
85
+ "linear_attention",
86
+ "linear_attention",
87
+ "full_attention",
88
+ "linear_attention",
89
+ "linear_attention",
90
+ "linear_attention",
91
+ "full_attention",
92
+ "linear_attention",
93
+ "linear_attention",
94
+ "linear_attention",
95
+ "full_attention",
96
+ "linear_attention",
97
+ "linear_attention",
98
+ "linear_attention",
99
+ "full_attention"
100
+ ],
101
+ "linear_conv_kernel_dim": 4,
102
+ "linear_key_head_dim": 128,
103
+ "linear_num_key_heads": 16,
104
+ "linear_num_value_heads": 48,
105
+ "linear_value_head_dim": 128,
106
+ "mamba_ssm_dtype": "float32",
107
+ "max_position_embeddings": 262144,
108
+ "model_type": "qwen3_5_text",
109
+ "mtp_num_hidden_layers": 1,
110
+ "mtp_use_dedicated_embeddings": false,
111
+ "num_attention_heads": 24,
112
+ "num_hidden_layers": 64,
113
+ "num_key_value_heads": 4,
114
+ "output_gate_type": "swish",
115
+ "pad_token_id": null,
116
+ "partial_rotary_factor": 0.25,
117
+ "rms_norm_eps": 1e-06,
118
+ "rope_parameters": {
119
+ "mrope_interleaved": true,
120
+ "mrope_section": [
121
+ 11,
122
+ 11,
123
+ 10
124
+ ],
125
+ "partial_rotary_factor": 0.25,
126
+ "rope_theta": 10000000,
127
+ "rope_type": "default"
128
+ },
129
+ "tie_word_embeddings": false,
130
+ "use_cache": true,
131
+ "vocab_size": 248320
132
+ },
133
+ "tie_word_embeddings": false,
134
+ "transformers_version": "5.8.0.dev0",
135
+ "video_token_id": 248057,
136
+ "vision_config": {
137
+ "deepstack_visual_indexes": [],
138
+ "depth": 27,
139
+ "hidden_act": "gelu_pytorch_tanh",
140
+ "hidden_size": 1152,
141
+ "in_channels": 3,
142
+ "initializer_range": 0.02,
143
+ "intermediate_size": 4304,
144
+ "model_type": "qwen3_5",
145
+ "num_heads": 16,
146
+ "num_position_embeddings": 2304,
147
+ "out_hidden_size": 5120,
148
+ "patch_size": 16,
149
+ "spatial_merge_size": 2,
150
+ "temporal_patch_size": 2
151
+ },
152
+ "vision_end_token_id": 248054,
153
+ "vision_start_token_id": 248053
154
+ }
generation_config.json ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 248044,
3
+ "do_sample": true,
4
+ "eos_token_id": [
5
+ 248046,
6
+ 248044
7
+ ],
8
+ "pad_token_id": 248044,
9
+ "temperature": 1.0,
10
+ "top_k": 20,
11
+ "top_p": 0.95
12
+ }
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model-00001-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d91b70ca84dff315addeeb8a599dce2beccfec33c7483686ede2e8ff3cc459dc
3
+ size 5328325554
model-00002-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eddcff1a6ef0971990f7cd01dba77ab9a8c20d59821bb4cd7e25234fe3809346
3
+ size 5354185158
model-00003-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:644cc5322ff706c9608b42ea1ff83e0beb12b828d49b36c468a366f44593b8eb
3
+ size 5144253923
model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
preprocessor_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "size": {
3
+ "longest_edge": 16777216,
4
+ "shortest_edge": 65536
5
+ },
6
+ "patch_size": 16,
7
+ "temporal_patch_size": 2,
8
+ "merge_size": 2,
9
+ "image_mean": [
10
+ 0.5,
11
+ 0.5,
12
+ 0.5
13
+ ],
14
+ "image_std": [
15
+ 0.5,
16
+ 0.5,
17
+ 0.5
18
+ ],
19
+ "processor_class": "Qwen3VLProcessor",
20
+ "image_processor_type": "Qwen2VLImageProcessorFast"
21
+ }