ixim commited on
Commit
4f03424
·
verified ·
1 Parent(s): ba3cf8d

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +1 -0
  2. CHANGES.md +5 -0
  3. LICENSE +55 -0
  4. MANIFEST.json +417 -0
  5. Notice +4 -0
  6. README.md +132 -0
  7. benchmarks/cases.json +9 -0
  8. conversion.json +512 -0
  9. evaluation/bf16/composition-s42.png +3 -0
  10. evaluation/cache-parity-8bit.json +21 -0
  11. evaluation/cache-parity-bf16.json +21 -0
  12. evaluation/offload-parity.json +47 -0
  13. evaluation/precision-probe.json +483 -0
  14. evaluation/report.md +37 -0
  15. evaluation/runtime-manifest.json +18 -0
  16. evaluation/source-files.json +164 -0
  17. evaluation/summary.json +101 -0
  18. evaluation/system-context.json +14 -0
  19. evaluation/visual-review.json +68 -0
  20. model_index.json +24 -0
  21. processor/added_tokens.json +28 -0
  22. processor/chat_template.jinja +120 -0
  23. processor/merges.txt +0 -0
  24. processor/preprocessor_config.json +39 -0
  25. processor/special_tokens_map.json +31 -0
  26. processor/tokenizer_config.json +240 -0
  27. processor/video_preprocessor_config.json +41 -0
  28. processor/vocab.json +0 -0
  29. requirements.lock.txt +59 -0
  30. requirements.txt +3 -0
  31. scheduler/scheduler_config.json +18 -0
  32. scripts/MLX_VLM_LICENSE.txt +21 -0
  33. scripts/__init__.py +0 -0
  34. scripts/audit.py +44 -0
  35. scripts/benchmark.py +85 -0
  36. scripts/cards.py +166 -0
  37. scripts/common.py +60 -0
  38. scripts/convert.py +106 -0
  39. scripts/infer.py +42 -0
  40. scripts/login.py +23 -0
  41. scripts/mlx_pipeline.py +351 -0
  42. scripts/probe.py +55 -0
  43. scripts/release.py +91 -0
  44. scripts/report.py +82 -0
  45. scripts/runtime.py +48 -0
  46. scripts/upload.py +67 -0
  47. scripts/verify_cache.py +42 -0
  48. text_encoder/config.json +71 -0
  49. text_encoder/model.safetensors.index.json +0 -0
  50. transformer/config.json +26 -0
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ evaluation/bf16/composition-s42.png filter=lfs diff=lfs merge=lfs -text
CHANGES.md ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ # Modifications
2
+
3
+ Modified by ixim / iximbox for Image21-MLX: converted from the pinned BF16 source to MLX layout; eligible linear weights use groupwise affine quantization. See conversion.json for precision and exceptions. Built with Qwen.
4
+
5
+ VAE convolution layout is transposed without reducing FP32 precision. The entire vision tower, token embeddings, language head, norms, transformer input/output projections, timestep embedding and modulation remain floating point.
LICENSE ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Qwen RESEARCH LICENSE AGREEMENT
2
+
3
+ Qwen RESEARCH LICENSE AGREEMENT Release Date: September 20, 2026
4
+
5
+ By clicking to agree or by using or distributing any portion or element of the Qwen Materials, you will be deemed to have recognized and accepted the content of this Agreement, which is effective immediately.
6
+
7
+ 1. Definitions
8
+ a. This Qwen RESEARCH LICENSE AGREEMENT (this "Agreement") shall mean the terms and conditions for use, reproduction, distribution and modification of the Materials as defined by this Agreement.
9
+ b. "We" (or "Us") shall mean Hangzhou Tongyi Laboratory Technology Co., Ltd.
10
+ c. "You" (or "Your") shall mean a natural person or legal entity exercising the rights granted by this Agreement and/or using the Materials for any purpose and in any field of use.
11
+ d. "Third Parties" shall mean individuals or legal entities that are not under common control with us or you.
12
+ e. "Qwen" shall mean the large language models, diffusion models, and software and algorithms, consisting of trained model weights, parameters (including optimizer states), machine-learning model code, inference-enabling code, training-enabling code, fine-tuning enabling code and other elements of the foregoing distributed by us.
13
+ f. "Materials" shall mean, collectively, our proprietary Qwen and Documentation (and any portion thereof) made available under this Agreement.
14
+ g. "Source" form shall mean the preferred form for making modifications, including but not limited to model source code, documentation source, and configuration files.
15
+ h. "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
16
+ i. "Non-Commercial" shall mean for research or evaluation purposes only.
17
+
18
+ 2. Grant of Rights
19
+ a. You are granted a non-exclusive, worldwide, non-transferable and royalty-free limited license under our intellectual property or other rights owned by us embodied in the Materials to use, reproduce, distribute, copy, create derivative works of, and make modifications to the Materials FOR NON-COMMERCIAL PURPOSES ONLY.
20
+ b. You shall not use the Materials for any commercial purpose without obtaining a separate commercial license from us. If you wish to use the Materials commercially, you shall request a license from us at model-business@notice.qwencloud.com.
21
+
22
+ 3. Redistribution
23
+ Subject to Section 2 (Grant of Rights), you may distribute copies or make the Materials, or derivative works thereof, available as part of a product or service that contains any of them, with or without modifications, and in Source or Object form, provided that you meet the following conditions:
24
+ a. You shall give any other recipients of the Materials or derivative works a copy of this Agreement;
25
+ b. You shall cause any modified files to carry prominent notices stating that you changed the files;
26
+ c. You shall retain in all copies of the Materials that you distribute the following attribution notices within a "Notice" text file distributed as a part of such copies: "Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved."; and
27
+ d. You may add your own copyright statement to your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of your modifications, or for any such derivative works as a whole, provided your use, reproduction, and distribution of the work otherwise complies with the terms and conditions of this Agreement.
28
+
29
+ 4. Rules of use
30
+ a. The Materials may be subject to export controls or restrictions in China, the United States or other countries or regions. You shall comply with applicable laws and regulations in your use of the Materials.
31
+ b. If you use the Materials or any outputs or results therefrom to create, train, fine-tune, or improve an AI model that is distributed or made available, you shall prominently display “Built with Qwen” or “Improved using Qwen” in the related product documentation.
32
+ c. You shall not use "Qwen" as the primary name or identifier of any derivative works or products; reasonable descriptive use (e.g., "fine-tuned from Qwen Image") is permitted.
33
+
34
+ 5. Intellectual Property
35
+ a. We retain ownership of all intellectual property rights in and to the Materials and derivatives made by or for us. Conditioned upon compliance with the terms and conditions of this Agreement, with respect to any derivative works and modifications of the Materials that are made by you, you are and will be the owner of such derivative works and modifications.
36
+ b. No trademark license is granted to use the trade names, trademarks, service marks, or product names of us, except as required to fulfill notice requirements under this Agreement or as required for reasonable and customary use in describing and redistributing the Materials.
37
+ c. If you commence a lawsuit or other proceedings (including a cross-claim or counterclaim in a lawsuit) against us or any entity alleging that the Materials or any output therefrom, or any part of the foregoing, infringe any intellectual property or other right owned or licensable by you, then all licenses granted to you under this Agreement shall terminate as of the date such lawsuit or other proceeding is commenced or brought.
38
+
39
+ 6. Disclaimer of Warranty and Limitation of Liability
40
+ a. We are not obligated to support, update, provide training for, or develop any further version of the Qwen Materials or to grant any license thereto.
41
+ b. THE MATERIALS ARE PROVIDED "AS IS" WITHOUT ANY EXPRESS OR IMPLIED WARRANTY OF ANY KIND INCLUDING WARRANTIES OF MERCHANTABILITY, NONINFRINGEMENT, OR FITNESS FOR A PARTICULAR PURPOSE. WE MAKE NO WARRANTY AND ASSUME NO RESPONSIBILITY FOR THE SAFETY OR STABILITY OF THE MATERIALS AND ANY OUTPUT THEREFROM.
42
+ c. IN NO EVENT SHALL WE BE LIABLE TO YOU FOR ANY DAMAGES, INCLUDING, BUT NOT LIMITED TO ANY DIRECT, OR INDIRECT, SPECIAL OR CONSEQUENTIAL DAMAGES ARISING FROM YOUR USE OR INABILITY TO USE THE MATERIALS OR ANY OUTPUT OF IT, NO MATTER HOW IT’S CAUSED.
43
+ d. You will defend, indemnify and hold harmless us from and against any claim by any third party arising out of or related to your use or distribution of the Materials.
44
+
45
+ 7. Survival and Termination.
46
+ a. The term of this Agreement shall commence upon your acceptance of this Agreement or access to the Materials and will continue in full force and effect until terminated in accordance with the terms and conditions herein.
47
+ b. We may terminate this Agreement if you breach any of the terms or conditions of this Agreement. Upon termination of this Agreement, you must delete and cease use of the Materials. Sections 6 and 8 shall survive the termination of this Agreement.
48
+
49
+ 8. Governing Law and Jurisdiction.
50
+ a. This Agreement and any dispute arising out of or relating to it will be governed by the laws of China, without regard to conflict of law principles, and the UN Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
51
+ b. The People's Courts in Hangzhou City shall have exclusive jurisdiction over any dispute arising out of this Agreement.
52
+
53
+ 9. Other Terms and Conditions.
54
+ a. Any arrangements, understandings, or agreements regarding the Material not stated herein are separate from and independent of the terms and conditions of this Agreement. You shall request a separate license from us, if you use the Materials in ways not expressly agreed to in this Agreement.
55
+ b. We shall not be bound by any additional or different terms or conditions communicated by you unless expressly agreed.
MANIFEST.json ADDED
@@ -0,0 +1,417 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "path": "CHANGES.md",
4
+ "size": 471,
5
+ "sha256": "01c72c43bec801321010cccff2049a39940a7fa2ddc21e0a448793bb288dc35c"
6
+ },
7
+ {
8
+ "path": "LICENSE",
9
+ "size": 7831,
10
+ "sha256": "8dc973f024ff95966bea25866efa443fd16776dcb1001e681e3d467ea572b28d"
11
+ },
12
+ {
13
+ "path": "Notice",
14
+ "size": 388,
15
+ "sha256": "6e5545e72d7e21a0b1b5e1e4254bcf444fbf69c8639409b14248bcff94e63f7d"
16
+ },
17
+ {
18
+ "path": "README.md",
19
+ "size": 6850,
20
+ "sha256": "90ea8c96bc0c3b842fed9d78467d7adb6b60bf115875d205ec4c38e49dc8addf"
21
+ },
22
+ {
23
+ "path": "benchmarks/cases.json",
24
+ "size": 1919,
25
+ "sha256": "2663ee3d489e1eb136e02b083963b8622146fc65ac301d6acc5a6aab21e63356"
26
+ },
27
+ {
28
+ "path": "conversion.json",
29
+ "size": 25509,
30
+ "sha256": "5a0185391cd88f5eec7101c52b7fa4e0c876f542903ba1ba66ff208327f185dd"
31
+ },
32
+ {
33
+ "path": "evaluation/8bit/COMPLETE.json",
34
+ "size": 73,
35
+ "sha256": "cd40aa24c15f7458da2a63b571fb0afee8c0c1a2eee9d8339872e0aface2ba8d"
36
+ },
37
+ {
38
+ "path": "evaluation/8bit/chinese_text-s42.png",
39
+ "size": 1098691,
40
+ "sha256": "a004a65048217fc709f110d5d00bd4aeb284ef76d42d0df4fc8a5d0df4779e1b"
41
+ },
42
+ {
43
+ "path": "evaluation/8bit/composition-s42.png",
44
+ "size": 1121051,
45
+ "sha256": "0add44cd7e9f92f7b4b9fe2d2a1c831a4ab43f8109e0eee2f9191f61b937aee2"
46
+ },
47
+ {
48
+ "path": "evaluation/8bit/edit-s1000042.png",
49
+ "size": 1726278,
50
+ "sha256": "152b1dcfd216713dd9d7e79b1de22dc1618692e6e5f1aa35459b3a3ce88d21cb"
51
+ },
52
+ {
53
+ "path": "evaluation/8bit/english_text-s42.png",
54
+ "size": 968820,
55
+ "sha256": "283a5b4147a9552701ffc16da7a8763fec928499d8678e205e74afae6f475a52"
56
+ },
57
+ {
58
+ "path": "evaluation/8bit/environment.json",
59
+ "size": 27251,
60
+ "sha256": "b84d6aec6456999d088c55578a720add96a90dcb424de6bfba46e9f9381bfd71"
61
+ },
62
+ {
63
+ "path": "evaluation/8bit/portrait-s42.png",
64
+ "size": 1828666,
65
+ "sha256": "aaa56d1bd3329428332b527279999472c00ed4bbdd1e6bb13a84d365a13db00e"
66
+ },
67
+ {
68
+ "path": "evaluation/8bit/results.jsonl",
69
+ "size": 6597,
70
+ "sha256": "742a088a5e0d30a04a5f8b2ce6eb5ae9fa3613167fed0072227eb1bb991437ff"
71
+ },
72
+ {
73
+ "path": "evaluation/8bit/rgba-s42.png",
74
+ "size": 689083,
75
+ "sha256": "722a3d2015ceb8cbfa994ac0b38a3b4d6fbb4c68a80a8198d24ac01de6af121e"
76
+ },
77
+ {
78
+ "path": "evaluation/8bit/texture-s42.png",
79
+ "size": 1764465,
80
+ "sha256": "23435675875678ebe9550d697d6b798fcbac9c461e0d2b9c686535182cb42795"
81
+ },
82
+ {
83
+ "path": "evaluation/bf16/COMPLETE.json",
84
+ "size": 73,
85
+ "sha256": "cd40aa24c15f7458da2a63b571fb0afee8c0c1a2eee9d8339872e0aface2ba8d"
86
+ },
87
+ {
88
+ "path": "evaluation/bf16/chinese_text-s42.png",
89
+ "size": 1100420,
90
+ "sha256": "d2e903aa478d0ea500372b800e5276669f07340ea27227544fe560a900d766ac"
91
+ },
92
+ {
93
+ "path": "evaluation/bf16/composition-s42.png",
94
+ "size": 1114487,
95
+ "sha256": "b361dd45e536a8d4d2baa45d412966c1820eb21e0d096829387cd0afa736aaef"
96
+ },
97
+ {
98
+ "path": "evaluation/bf16/edit-s1000042.png",
99
+ "size": 1726291,
100
+ "sha256": "51d46aaf58fa9488f5094a553daac0d5ef921b045c619014d75a5256331170d0"
101
+ },
102
+ {
103
+ "path": "evaluation/bf16/english_text-s42.png",
104
+ "size": 959474,
105
+ "sha256": "2ce4e2a35c0023832bcea130d1955ded09c390f3571ad43c6aaa44581362e55c"
106
+ },
107
+ {
108
+ "path": "evaluation/bf16/environment.json",
109
+ "size": 1747,
110
+ "sha256": "3ac177a51fd089875326d4c7fdb3f78fe7c77cb77fd8870e7da5f18ee6fa5c0e"
111
+ },
112
+ {
113
+ "path": "evaluation/bf16/portrait-s42.png",
114
+ "size": 1829384,
115
+ "sha256": "72493104ac6662f1de9e7857e8ec03f4f3971c31c4ea77207d33ae029a3a0c41"
116
+ },
117
+ {
118
+ "path": "evaluation/bf16/results.jsonl",
119
+ "size": 6601,
120
+ "sha256": "2c9856ece4b83473cf884aa37c7a55be9cec3a52df31920e6a41b6e07418657d"
121
+ },
122
+ {
123
+ "path": "evaluation/bf16/rgba-s42.png",
124
+ "size": 687815,
125
+ "sha256": "c7adbb4da64425795a221e780cf872ee642dc6d8d47c7efdefadc75fdcc0396b"
126
+ },
127
+ {
128
+ "path": "evaluation/bf16/texture-s42.png",
129
+ "size": 1757770,
130
+ "sha256": "6399da2b528016420f934dd33613ace33d30cfed338f7815a14645e15b01b1f2"
131
+ },
132
+ {
133
+ "path": "evaluation/cache-parity-8bit.json",
134
+ "size": 550,
135
+ "sha256": "125b9eb038bc4a7aa74281e0d819f216dab0e50edbc7b24f82f097d1b257edba"
136
+ },
137
+ {
138
+ "path": "evaluation/cache-parity-bf16.json",
139
+ "size": 548,
140
+ "sha256": "47d3ca6081acfa67a1aa0d5575fef11b42caab4ac55b010cf6f0d6d13b558ab2"
141
+ },
142
+ {
143
+ "path": "evaluation/comparison.png",
144
+ "size": 4156154,
145
+ "sha256": "19083714350e5b2cdd596df81794ab62b724898b0f63f1245df5a191f1506d8e"
146
+ },
147
+ {
148
+ "path": "evaluation/offload-parity.json",
149
+ "size": 1340,
150
+ "sha256": "36343d5931725290cedb150d98f97da3ec9f98d8d4ef43ab2ca7b87fefb94344"
151
+ },
152
+ {
153
+ "path": "evaluation/precision-probe.json",
154
+ "size": 13644,
155
+ "sha256": "950a98b43817f51642d59a794712962129133bbc04715831e7aebcba095d70f3"
156
+ },
157
+ {
158
+ "path": "evaluation/report.md",
159
+ "size": 4544,
160
+ "sha256": "982d457ebc32c455c58afebf78c0e2dba8d6c0b7ee0858adc87209514603fafc"
161
+ },
162
+ {
163
+ "path": "evaluation/runtime-manifest.json",
164
+ "size": 509,
165
+ "sha256": "ee348c543f136b162f3c285f15a66cd59fa51f8d889aad07e6e26e8e02bf39bb"
166
+ },
167
+ {
168
+ "path": "evaluation/source-files.json",
169
+ "size": 5163,
170
+ "sha256": "0c0ddea15b77f9a8093144d083ac887b5f8977fc92db52f1a168008deba8b4bc"
171
+ },
172
+ {
173
+ "path": "evaluation/summary.json",
174
+ "size": 3280,
175
+ "sha256": "e2c1820cea379a4ac66bd2f1779c56645666ffaafa757855fe3cf0f2943613cc"
176
+ },
177
+ {
178
+ "path": "evaluation/system-context.json",
179
+ "size": 1188,
180
+ "sha256": "82fd648e290d6b5f10ee583a827ab08761edc00b9bca448c7793d65c48225a05"
181
+ },
182
+ {
183
+ "path": "evaluation/visual-review.json",
184
+ "size": 6277,
185
+ "sha256": "8b41b30803ae40972945ea5777f592c1681fad02de7368a20d4e9e3c68e3696d"
186
+ },
187
+ {
188
+ "path": "model_index.json",
189
+ "size": 447,
190
+ "sha256": "cf1ecd104ea090855d60d8cec0895c1e9d6ee41b2f87e231b678beecd1cf7809"
191
+ },
192
+ {
193
+ "path": "processor/added_tokens.json",
194
+ "size": 707,
195
+ "sha256": "c0284b582e14987fbd3d5a2cb2bd139084371ed9acbae488829a1c900833c680"
196
+ },
197
+ {
198
+ "path": "processor/chat_template.jinja",
199
+ "size": 5292,
200
+ "sha256": "3636d0f0bd6bef02654cdffdc447b79cb2cef8ab02cc75267345946291a489e4"
201
+ },
202
+ {
203
+ "path": "processor/merges.txt",
204
+ "size": 1671853,
205
+ "sha256": "8831e4f1a044471340f7c0a83d7bd71306a5b867e95fd870f74d0c5308a904d5"
206
+ },
207
+ {
208
+ "path": "processor/preprocessor_config.json",
209
+ "size": 782,
210
+ "sha256": "93585062a80db5e8ca038efc7726a3e6411d9db948472d81d63c6303993be8c5"
211
+ },
212
+ {
213
+ "path": "processor/special_tokens_map.json",
214
+ "size": 613,
215
+ "sha256": "76862e765266b85aa9459767e33cbaf13970f327a0e88d1c65846c2ddd3a1ecd"
216
+ },
217
+ {
218
+ "path": "processor/tokenizer.json",
219
+ "size": 11422654,
220
+ "sha256": "aeb13307a71acd8fe81861d94ad54ab689df773318809eed3cbe794b4492dae4"
221
+ },
222
+ {
223
+ "path": "processor/tokenizer_config.json",
224
+ "size": 5445,
225
+ "sha256": "81ec7bb9530159b326c0bef1d0b6c33d392090524014ea3f0123a3c1eb9c2af5"
226
+ },
227
+ {
228
+ "path": "processor/video_preprocessor_config.json",
229
+ "size": 817,
230
+ "sha256": "59c5c9eb52182eb14c06ffb10ca9effd29adce5f238a95de23ca14a38dbd2cb1"
231
+ },
232
+ {
233
+ "path": "processor/vocab.json",
234
+ "size": 2776833,
235
+ "sha256": "ca10d7e9fb3ed18575dd1e277a2579c16d108e32f27439684afa0e10b1440910"
236
+ },
237
+ {
238
+ "path": "requirements.lock.txt",
239
+ "size": 1110,
240
+ "sha256": "ec2e0402146f97b80165946f96e39374bd68aa7d54c32163e7ade178a06511c5"
241
+ },
242
+ {
243
+ "path": "requirements.txt",
244
+ "size": 125,
245
+ "sha256": "4501caf47b8d383e90c5e5ee8e649c7e0dd71de29c508bf2edb34b02e0494f90"
246
+ },
247
+ {
248
+ "path": "scheduler/scheduler_config.json",
249
+ "size": 485,
250
+ "sha256": "5895f3a167c14a967fe9ac70c64924ae5acc79799e0679fd12907e594a713cd1"
251
+ },
252
+ {
253
+ "path": "scripts/MLX_VLM_LICENSE.txt",
254
+ "size": 1069,
255
+ "sha256": "f74c448746be3376e27dee43e2be9ec9d55f4820cd7466c603a6c7d1fd2d25b8"
256
+ },
257
+ {
258
+ "path": "scripts/__init__.py",
259
+ "size": 0,
260
+ "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
261
+ },
262
+ {
263
+ "path": "scripts/audit.py",
264
+ "size": 2255,
265
+ "sha256": "7e6183ffbd1861fc701f247603fc6be44b889a379ecada04e2e1d15529b2fbed"
266
+ },
267
+ {
268
+ "path": "scripts/benchmark.py",
269
+ "size": 5088,
270
+ "sha256": "318e2bde9fc68fdd547c8fd32a0fb9fbe69f3cf99e62b5049240ee00754b1566"
271
+ },
272
+ {
273
+ "path": "scripts/cards.py",
274
+ "size": 9369,
275
+ "sha256": "740ab304501946e7b8ecce32afd10c2879715a979c2e3b7afdecd42d9560f56a"
276
+ },
277
+ {
278
+ "path": "scripts/common.py",
279
+ "size": 2712,
280
+ "sha256": "77ad9a6a8875ed0a1ef5d920058d3406b75cf0b8d038d3f8e9eed39c2a1d6914"
281
+ },
282
+ {
283
+ "path": "scripts/convert.py",
284
+ "size": 5970,
285
+ "sha256": "f177576669880b7501924fab5cec8119fcbb7e4c8007665f40ae5c486fdf0b36"
286
+ },
287
+ {
288
+ "path": "scripts/infer.py",
289
+ "size": 2175,
290
+ "sha256": "708a48f9dd6ca8a332875799637904bc63656f2bc8b158b81e9e46c5aba7e7e3"
291
+ },
292
+ {
293
+ "path": "scripts/login.py",
294
+ "size": 898,
295
+ "sha256": "b39d663babb2b685993433b662b42125ed357332d1d4271e35455c2fc62d259e"
296
+ },
297
+ {
298
+ "path": "scripts/mlx_pipeline.py",
299
+ "size": 13583,
300
+ "sha256": "ac001b3cf8b79304fc02efd3127576a67d7b03551a6295a6fc1c2a13d032614b"
301
+ },
302
+ {
303
+ "path": "scripts/probe.py",
304
+ "size": 2888,
305
+ "sha256": "3e450e1b077f153ed0c6762e3f3c0b5d0c455c0f11ee8e5578bc05226589a154"
306
+ },
307
+ {
308
+ "path": "scripts/release.py",
309
+ "size": 5206,
310
+ "sha256": "d1abeb50b719aa0aa943b6abc0a7596898ea97484ab3b6ba8930e2166eca06be"
311
+ },
312
+ {
313
+ "path": "scripts/report.py",
314
+ "size": 5600,
315
+ "sha256": "87634473beb077f9b2617160510d52c1b181bcafdd77de7d458dc27477696563"
316
+ },
317
+ {
318
+ "path": "scripts/runtime.py",
319
+ "size": 2490,
320
+ "sha256": "ad32ba856fddf5168d50e930204603fd2caeb016bc70fe9972d769aade6a79f4"
321
+ },
322
+ {
323
+ "path": "scripts/upload.py",
324
+ "size": 4004,
325
+ "sha256": "fe50fdf1d2c4eb8e88539324797ca878667a3f7ce03b2a2572b6bbb597b6c54c"
326
+ },
327
+ {
328
+ "path": "scripts/verify_cache.py",
329
+ "size": 2297,
330
+ "sha256": "97a2ff08dd26d9f1e085843cbaa9d1a40196f5239c5e521d3f3b6c144db0d612"
331
+ },
332
+ {
333
+ "path": "text_encoder/config.json",
334
+ "size": 1874,
335
+ "sha256": "ce16d11ac7f4f9383bce66fcae7932665992ed29b48d5d8ffc83e1585044e2b5"
336
+ },
337
+ {
338
+ "path": "text_encoder/model-00001-of-00006.safetensors",
339
+ "size": 1152816452,
340
+ "sha256": "c990998d46948fe255332cfe5f75b0d3d55aabc88717e173f18aa38604d1d797"
341
+ },
342
+ {
343
+ "path": "text_encoder/model-00002-of-00006.safetensors",
344
+ "size": 1957771354,
345
+ "sha256": "06c8e0698ce37b7ab750a33a56332b23b2147620482f36ec28c98b3f61094e9a"
346
+ },
347
+ {
348
+ "path": "text_encoder/model-00003-of-00006.safetensors",
349
+ "size": 1996689654,
350
+ "sha256": "781e557722909cf1b5db5e13582a4b35be3d49615ba8920a4f7541ac0d54d564"
351
+ },
352
+ {
353
+ "path": "text_encoder/model-00004-of-00006.safetensors",
354
+ "size": 1952106414,
355
+ "sha256": "51cff2fe048456449a8131cc0642e091cce43d0116414599fd3ebae04b4fa6fa"
356
+ },
357
+ {
358
+ "path": "text_encoder/model-00005-of-00006.safetensors",
359
+ "size": 1996689818,
360
+ "sha256": "023b173e9208a5f406acd516382817c7983b58ef0f57154c55627ba96589f0d6"
361
+ },
362
+ {
363
+ "path": "text_encoder/model-00006-of-00006.safetensors",
364
+ "size": 1966674243,
365
+ "sha256": "cb57f09b7e78a82fbede723c680b9d0231357dfec772856f6db4fe8322890898"
366
+ },
367
+ {
368
+ "path": "text_encoder/model.safetensors.index.json",
369
+ "size": 116487,
370
+ "sha256": "2caeee17a1f27f641e70bfd07e7f7b7527814d1f1dbab3e1875ea8c8a6f3f59e"
371
+ },
372
+ {
373
+ "path": "transformer/config.json",
374
+ "size": 727,
375
+ "sha256": "ae72e14e73a31cb130f1a5a7b5e712c574d3a314c4ed5d8d4f73321b19ec3ced"
376
+ },
377
+ {
378
+ "path": "transformer/model-00001-of-00004.safetensors",
379
+ "size": 1984463956,
380
+ "sha256": "2aa12f9c914fe8d152d2a932d485b475a2c30398d8ee70ee15278383e98586c0"
381
+ },
382
+ {
383
+ "path": "transformer/model-00002-of-00004.safetensors",
384
+ "size": 1996515918,
385
+ "sha256": "74fae4addedd82000aa70b84c98a7bed253c8ba4d006198841725fae3a3d2ac3"
386
+ },
387
+ {
388
+ "path": "transformer/model-00003-of-00004.safetensors",
389
+ "size": 1996516732,
390
+ "sha256": "fafec3d438a2c8d0e6766ff9e8c6d91eccdc7474af13a1a59b78bb3574fcffaa"
391
+ },
392
+ {
393
+ "path": "transformer/model-00004-of-00004.safetensors",
394
+ "size": 1709726423,
395
+ "sha256": "a612b703a280a936dc9e10828832e1c030113b0cdeaf628eb0ac5da88c339e80"
396
+ },
397
+ {
398
+ "path": "transformer/model.safetensors.index.json",
399
+ "size": 62901,
400
+ "sha256": "83dceafc8c4c4808997e4551f96d160d492e784ac11fc2d2f373768adc7578f8"
401
+ },
402
+ {
403
+ "path": "vae/config.json",
404
+ "size": 2353,
405
+ "sha256": "e672be7b41be0d4016dd78e4104fb6f2c246f9f63acb91e4e44e0862a16cbbf8"
406
+ },
407
+ {
408
+ "path": "vae/model-00001-of-00001.safetensors",
409
+ "size": 1350989238,
410
+ "sha256": "44e83fe1529df6753a00a0d8981cb3da0332826518f040937777cae4123225b9"
411
+ },
412
+ {
413
+ "path": "vae/model.safetensors.index.json",
414
+ "size": 20646,
415
+ "sha256": "86eee223fbf1df185b791785fc5b3775abfb59104a3ed91f19ec819eafe11b37"
416
+ }
417
+ ]
Notice ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved.
2
+
3
+ Built with Qwen
4
+ Modified by ixim / iximbox for Image21-MLX: converted from the pinned BF16 source to MLX layout; eligible linear weights use groupwise affine quantization. See conversion.json for precision and exceptions. Built with Qwen.
README.md ADDED
@@ -0,0 +1,132 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: qwen-research
4
+ license_link: LICENSE
5
+ base_model: Qwen/Qwen-Image-2.1
6
+ library_name: mlx
7
+ pipeline_tag: text-to-image
8
+ tags:
9
+ - mlx
10
+ - mlx-vlm
11
+ - apple-silicon
12
+ - quantized
13
+ - image-to-image
14
+ - rgba
15
+ ---
16
+ # Image21-MLX-8bit
17
+
18
+ **Built with Qwen.** An independent native MLX quantization of Qwen-Image-2.1 for
19
+ Apple Silicon, by ixim / iximbox. **Non-commercial research and evaluation only**
20
+ under the original Qwen Research License. This is not an official Qwen release.
21
+
22
+ The complete checkpoint is approximately **18.68 GiB**. DiT attention/MLP and
23
+ language-encoder attention/MLP linears use **8-bit affine weights, group size 64**,
24
+ with BF16 activations. The full vision tower, token embeddings, language head,
25
+ norms and DiT input/output/timestep/modulation layers retain floating-point precision.
26
+ The VAE retains its original **FP32** weights. See the exact 476 quantized modules in conversion.json.
27
+
28
+ Supports text-to-image, native reference-image editing and RGBA transparency through
29
+ the included scripts. The wrapper preserves the sampler's alpha channel, which the
30
+ pinned upstream text-to-image convenience method otherwise slices away.
31
+ The runtime also enables the upstream fixed-prefix KV cache for text-to-image.
32
+ Components are loaded and released by phase to reduce unified-memory use.
33
+ See scripts/mlx_pipeline.py and its retained MIT attribution.
34
+
35
+ With the included phase-loading runtime, plan for **32GB unified memory as a starting
36
+ budget; 48GB or more gives more room for other applications** at 1024px with one reference.
37
+ These are capacity estimates, not verified minimums: only the 128GB M4 Max was tested.
38
+ The measured Q8 MLX allocation peak is about **16.20 GiB**, excluding OS/driver overhead.
39
+ 16/24GB Macs are not the target for this default 1024px configuration. 2048px and
40
+ 10-reference editing have not been benchmarked. Allow roughly **50GB free disk space**
41
+ for ordinary inference; rebuilding also needs the original source and baseline checkpoints.
42
+
43
+ ## Visual findings / 视觉检查
44
+
45
+ Assistant inspection of all seven paired cases found no obvious severe quality regression in this small sample. English and Chinese text are correct in both precisions; cup count/order and sweater recoloring succeed; real RGBA transparency is retained. The English poster changes visibly in typeface and lamp structure, and the dragon changes expression/details. Other pairs are closer. Both portraits are cropped more tightly than requested, and both transparency samples contain slight near-zero background alpha. This is an unblinded, single-seed visual check, not evidence of lossless quantization or statistical equivalence.
46
+
47
+ ## Download / 下载
48
+
49
+ On an Apple Silicon Mac with [uv](https://docs.astral.sh/uv/) installed:
50
+
51
+ ```bash
52
+ uvx --from huggingface-hub==1.33.0 hf download ixim/Image21-MLX-8bit --local-dir Image21-MLX-8bit
53
+ cd Image21-MLX-8bit
54
+ ```
55
+
56
+ ## Reproducible inference
57
+
58
+ ```bash
59
+ uv venv --python 3.13 .venv
60
+ uv pip install --python .venv/bin/python -r requirements.lock.txt
61
+ .venv/bin/python -m scripts.infer --model . \
62
+ --prompt 'A natural portrait in soft window light' --output outputs/portrait.png
63
+ ```
64
+
65
+ ```bash
66
+ .venv/bin/python -m scripts.infer --model . --input input.png \
67
+ --prompt 'Change only the blue sweater to a red sweater. Preserve the person and background.' \
68
+ --source-seed 42 --seed 1000042 --output outputs/edit.png
69
+ ```
70
+
71
+ Transparent generation prompt:
72
+ `This is an RGBA image with transparency. A cute cartoon dragon sticker. The image has alpha channel and the background is transparent.`
73
+
74
+ ## Measurements / 实测
75
+
76
+ Apple M4 Max, 40 GPU cores, 128 GiB unified memory, macOS 26.6.2, MLX 0.32.2.
77
+ Both models use the same pinned MLX runtime and internal SSD; 1024×1024, 40 steps,
78
+ CFG=1, no VAE tiling, one full warm-up per process, seven cases, one seed per case.
79
+ Sequential component loading is enabled in both models. Per-image time includes
80
+ component loading, prompt encoding, denoising and VAE decoding; PNG writing is excluded.
81
+ This is a desktop session with other applications open. BF16 ran before Q8;
82
+ there were no repeated or interleaved trials to control order and thermal effects.
83
+
84
+ | Case | BF16 seconds | 8-bit seconds | BF16 peak GiB | 8-bit peak GiB |
85
+ |---|---:|---:|---:|---:|
86
+ | portrait | 440.91 | 568.86 | 16.41 | 16.19 |
87
+ | english_text | 464.32 | 566.88 | 16.43 | 16.20 |
88
+ | chinese_text | 474.14 | 567.43 | 16.44 | 16.20 |
89
+ | composition | 480.91 | 562.52 | 16.44 | 16.20 |
90
+ | texture | 487.64 | 445.07 | 16.43 | 16.20 |
91
+ | rgba | 485.45 | 557.30 | 16.42 | 16.20 |
92
+ | edit | 606.29 | 665.00 | 18.68 | 16.20 |
93
+
94
+ Six-case text-to-image mean / 六类文生图平均:BF16 **7.87 min**, Q8 **9.08 min**. Q8 generation time relative to BF16 / Q8 相对耗时:**+15.3%** in this run.
95
+
96
+ Peak figures measure MLX allocations, not minimum physical RAM. Full raw records,
97
+ original RGBA samples and the visual review are in [evaluation](evaluation/report.md).
98
+ These are informal measurements, not an official benchmark. The BF16 baseline is
99
+ the same MLX implementation; cross-runtime CUDA equivalence is not claimed.
100
+ One seed per case is insufficient to establish statistical quality equivalence.
101
+ There is no claim of lossless quantization or a guaranteed speedup.
102
+
103
+ ![All seven paired samples](evaluation/comparison.png)
104
+
105
+ ## Provenance
106
+
107
+ - Source: `Qwen/Qwen-Image-2.1@b3179ad355be050328e483a9dfdd9e60cd62adfa`.
108
+ - Runtime: `Blaizzy/mlx-vlm@95b01ccad2d9f65a9e87f6a87bd1c5df69626261`.
109
+ - Native MLX affine packed weights; not CUDA bitsandbytes INT8, FP8, GGUF or an MFLUX checkpoint.
110
+ - All three components passed exact tensor round-trip verification.
111
+ - [conversion.json](conversion.json), [modifications](CHANGES.md), [Notice](Notice), [license](LICENSE), [file hashes](MANIFEST.json).
112
+
113
+ To rebuild, first download the original source snapshot at the revision above to
114
+ `source-bf16` (requires additional disk space), then run from this repository:
115
+
116
+ ```bash
117
+ .venv/bin/python -m scripts.audit --source source-bf16 --manifest evaluation/source-files.json
118
+ .venv/bin/python -m scripts.convert --source source-bf16 --bits 8 --output rebuilt-8bit
119
+ ```
120
+ The audit verifies the original source file hashes before conversion. Rebuilding
121
+ uses the original floating-point weights, never a dequantized CUDA INT8 checkpoint.
122
+
123
+ To repeat the paired evaluation after the source audit, create a native baseline
124
+ and run each precision in its own process. This is a lengthy research workflow
125
+ and needs extra memory and disk space beyond ordinary inference:
126
+
127
+ ```bash
128
+ .venv/bin/python -m scripts.convert --source source-bf16 --bits 16 --output rebuilt-bf16
129
+ .venv/bin/python -m scripts.benchmark --model rebuilt-bf16 --output artifacts/eval/bf16
130
+ .venv/bin/python -m scripts.benchmark --model . --output artifacts/eval/8bit --edit-input artifacts/eval/bf16/portrait-s42.png
131
+ .venv/bin/python -m scripts.report
132
+ ```
benchmarks/cases.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {"id":"portrait","category":"portrait","prompt":"A natural documentary portrait of an elderly woman with silver hair and freckles, wearing a blue wool sweater, standing beside a window in soft morning light. Realistic skin texture, gentle expression, waist-up composition."},
3
+ {"id":"english_text","category":"typography_en","prompt":"A clean editorial poster with a deep blue background. Large exact headline at the top: \"CREATE WITH LIGHT\". Smaller exact subtitle: \"September 2026\". A realistic orange desk lamp occupies the lower half. Elegant balanced typography, no additional text."},
4
+ {"id":"chinese_text","category":"typography_zh","prompt":"设计一张现代咖啡馆海报,米白背景,中央一杯拉花拿铁。顶部准确写上中文大标题“慢下来,喝杯咖啡”,下方准确写上“小店今日营业”。文字清晰,留白充分,暖色摄影,不添加其他文字。"},
5
+ {"id":"composition","category":"spatial_counting","prompt":"A studio photograph on a light gray tabletop: exactly three ceramic cups in a row, a red cup on the left, a blue cup in the middle, and a yellow cup on the right. A single green apple sits in front of the blue cup. Soft shadows, no text."},
6
+ {"id":"texture","category":"texture","prompt":"Macro photography of a small kingfisher perched on a mossy branch beside clear water, detailed blue feathers, droplets, natural sunlight, softly blurred forest background, realistic textures."},
7
+ {"id":"rgba","category":"transparency","prompt":"This is an RGBA image with transparency. A cute cartoon dragon sticker, full body, green scales and small orange wings. The image has alpha channel and the background is transparent."},
8
+ {"id":"edit","category":"image_editing","prompt":"Change only the blue sweater to a red sweater. Preserve the same person, face, pose, lighting and background.","input":"artifacts/eval/bf16/portrait-s42.png"}
9
+ ]
conversion.json ADDED
@@ -0,0 +1,512 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "Image21-MLX",
3
+ "source_model": "Qwen/Qwen-Image-2.1",
4
+ "source_revision": "b3179ad355be050328e483a9dfdd9e60cd62adfa",
5
+ "runtime_revision": "95b01ccad2d9f65a9e87f6a87bd1c5df69626261",
6
+ "method": "MLX affine weight-only",
7
+ "bits": 8,
8
+ "group_size": 64,
9
+ "activation_dtype": "bfloat16",
10
+ "vae_dtype": "float32",
11
+ "components": {
12
+ "transformer": {
13
+ "quantized_modules": [
14
+ "transformer_blocks.0.attn.to_q",
15
+ "transformer_blocks.0.attn.to_k",
16
+ "transformer_blocks.0.attn.to_v",
17
+ "transformer_blocks.0.attn.to_out.0",
18
+ "transformer_blocks.0.img_mlp.proj",
19
+ "transformer_blocks.0.img_mlp.out",
20
+ "transformer_blocks.0.img_mlp.gate_layer",
21
+ "transformer_blocks.1.attn.to_q",
22
+ "transformer_blocks.1.attn.to_k",
23
+ "transformer_blocks.1.attn.to_v",
24
+ "transformer_blocks.1.attn.to_out.0",
25
+ "transformer_blocks.1.img_mlp.proj",
26
+ "transformer_blocks.1.img_mlp.out",
27
+ "transformer_blocks.1.img_mlp.gate_layer",
28
+ "transformer_blocks.2.attn.to_q",
29
+ "transformer_blocks.2.attn.to_k",
30
+ "transformer_blocks.2.attn.to_v",
31
+ "transformer_blocks.2.attn.to_out.0",
32
+ "transformer_blocks.2.img_mlp.proj",
33
+ "transformer_blocks.2.img_mlp.out",
34
+ "transformer_blocks.2.img_mlp.gate_layer",
35
+ "transformer_blocks.3.attn.to_q",
36
+ "transformer_blocks.3.attn.to_k",
37
+ "transformer_blocks.3.attn.to_v",
38
+ "transformer_blocks.3.attn.to_out.0",
39
+ "transformer_blocks.3.img_mlp.proj",
40
+ "transformer_blocks.3.img_mlp.out",
41
+ "transformer_blocks.3.img_mlp.gate_layer",
42
+ "transformer_blocks.4.attn.to_q",
43
+ "transformer_blocks.4.attn.to_k",
44
+ "transformer_blocks.4.attn.to_v",
45
+ "transformer_blocks.4.attn.to_out.0",
46
+ "transformer_blocks.4.img_mlp.proj",
47
+ "transformer_blocks.4.img_mlp.out",
48
+ "transformer_blocks.4.img_mlp.gate_layer",
49
+ "transformer_blocks.5.attn.to_q",
50
+ "transformer_blocks.5.attn.to_k",
51
+ "transformer_blocks.5.attn.to_v",
52
+ "transformer_blocks.5.attn.to_out.0",
53
+ "transformer_blocks.5.img_mlp.proj",
54
+ "transformer_blocks.5.img_mlp.out",
55
+ "transformer_blocks.5.img_mlp.gate_layer",
56
+ "transformer_blocks.6.attn.to_q",
57
+ "transformer_blocks.6.attn.to_k",
58
+ "transformer_blocks.6.attn.to_v",
59
+ "transformer_blocks.6.attn.to_out.0",
60
+ "transformer_blocks.6.img_mlp.proj",
61
+ "transformer_blocks.6.img_mlp.out",
62
+ "transformer_blocks.6.img_mlp.gate_layer",
63
+ "transformer_blocks.7.attn.to_q",
64
+ "transformer_blocks.7.attn.to_k",
65
+ "transformer_blocks.7.attn.to_v",
66
+ "transformer_blocks.7.attn.to_out.0",
67
+ "transformer_blocks.7.img_mlp.proj",
68
+ "transformer_blocks.7.img_mlp.out",
69
+ "transformer_blocks.7.img_mlp.gate_layer",
70
+ "transformer_blocks.8.attn.to_q",
71
+ "transformer_blocks.8.attn.to_k",
72
+ "transformer_blocks.8.attn.to_v",
73
+ "transformer_blocks.8.attn.to_out.0",
74
+ "transformer_blocks.8.img_mlp.proj",
75
+ "transformer_blocks.8.img_mlp.out",
76
+ "transformer_blocks.8.img_mlp.gate_layer",
77
+ "transformer_blocks.9.attn.to_q",
78
+ "transformer_blocks.9.attn.to_k",
79
+ "transformer_blocks.9.attn.to_v",
80
+ "transformer_blocks.9.attn.to_out.0",
81
+ "transformer_blocks.9.img_mlp.proj",
82
+ "transformer_blocks.9.img_mlp.out",
83
+ "transformer_blocks.9.img_mlp.gate_layer",
84
+ "transformer_blocks.10.attn.to_q",
85
+ "transformer_blocks.10.attn.to_k",
86
+ "transformer_blocks.10.attn.to_v",
87
+ "transformer_blocks.10.attn.to_out.0",
88
+ "transformer_blocks.10.img_mlp.proj",
89
+ "transformer_blocks.10.img_mlp.out",
90
+ "transformer_blocks.10.img_mlp.gate_layer",
91
+ "transformer_blocks.11.attn.to_q",
92
+ "transformer_blocks.11.attn.to_k",
93
+ "transformer_blocks.11.attn.to_v",
94
+ "transformer_blocks.11.attn.to_out.0",
95
+ "transformer_blocks.11.img_mlp.proj",
96
+ "transformer_blocks.11.img_mlp.out",
97
+ "transformer_blocks.11.img_mlp.gate_layer",
98
+ "transformer_blocks.12.attn.to_q",
99
+ "transformer_blocks.12.attn.to_k",
100
+ "transformer_blocks.12.attn.to_v",
101
+ "transformer_blocks.12.attn.to_out.0",
102
+ "transformer_blocks.12.img_mlp.proj",
103
+ "transformer_blocks.12.img_mlp.out",
104
+ "transformer_blocks.12.img_mlp.gate_layer",
105
+ "transformer_blocks.13.attn.to_q",
106
+ "transformer_blocks.13.attn.to_k",
107
+ "transformer_blocks.13.attn.to_v",
108
+ "transformer_blocks.13.attn.to_out.0",
109
+ "transformer_blocks.13.img_mlp.proj",
110
+ "transformer_blocks.13.img_mlp.out",
111
+ "transformer_blocks.13.img_mlp.gate_layer",
112
+ "transformer_blocks.14.attn.to_q",
113
+ "transformer_blocks.14.attn.to_k",
114
+ "transformer_blocks.14.attn.to_v",
115
+ "transformer_blocks.14.attn.to_out.0",
116
+ "transformer_blocks.14.img_mlp.proj",
117
+ "transformer_blocks.14.img_mlp.out",
118
+ "transformer_blocks.14.img_mlp.gate_layer",
119
+ "transformer_blocks.15.attn.to_q",
120
+ "transformer_blocks.15.attn.to_k",
121
+ "transformer_blocks.15.attn.to_v",
122
+ "transformer_blocks.15.attn.to_out.0",
123
+ "transformer_blocks.15.img_mlp.proj",
124
+ "transformer_blocks.15.img_mlp.out",
125
+ "transformer_blocks.15.img_mlp.gate_layer",
126
+ "transformer_blocks.16.attn.to_q",
127
+ "transformer_blocks.16.attn.to_k",
128
+ "transformer_blocks.16.attn.to_v",
129
+ "transformer_blocks.16.attn.to_out.0",
130
+ "transformer_blocks.16.img_mlp.proj",
131
+ "transformer_blocks.16.img_mlp.out",
132
+ "transformer_blocks.16.img_mlp.gate_layer",
133
+ "transformer_blocks.17.attn.to_q",
134
+ "transformer_blocks.17.attn.to_k",
135
+ "transformer_blocks.17.attn.to_v",
136
+ "transformer_blocks.17.attn.to_out.0",
137
+ "transformer_blocks.17.img_mlp.proj",
138
+ "transformer_blocks.17.img_mlp.out",
139
+ "transformer_blocks.17.img_mlp.gate_layer",
140
+ "transformer_blocks.18.attn.to_q",
141
+ "transformer_blocks.18.attn.to_k",
142
+ "transformer_blocks.18.attn.to_v",
143
+ "transformer_blocks.18.attn.to_out.0",
144
+ "transformer_blocks.18.img_mlp.proj",
145
+ "transformer_blocks.18.img_mlp.out",
146
+ "transformer_blocks.18.img_mlp.gate_layer",
147
+ "transformer_blocks.19.attn.to_q",
148
+ "transformer_blocks.19.attn.to_k",
149
+ "transformer_blocks.19.attn.to_v",
150
+ "transformer_blocks.19.attn.to_out.0",
151
+ "transformer_blocks.19.img_mlp.proj",
152
+ "transformer_blocks.19.img_mlp.out",
153
+ "transformer_blocks.19.img_mlp.gate_layer",
154
+ "transformer_blocks.20.attn.to_q",
155
+ "transformer_blocks.20.attn.to_k",
156
+ "transformer_blocks.20.attn.to_v",
157
+ "transformer_blocks.20.attn.to_out.0",
158
+ "transformer_blocks.20.img_mlp.proj",
159
+ "transformer_blocks.20.img_mlp.out",
160
+ "transformer_blocks.20.img_mlp.gate_layer",
161
+ "transformer_blocks.21.attn.to_q",
162
+ "transformer_blocks.21.attn.to_k",
163
+ "transformer_blocks.21.attn.to_v",
164
+ "transformer_blocks.21.attn.to_out.0",
165
+ "transformer_blocks.21.img_mlp.proj",
166
+ "transformer_blocks.21.img_mlp.out",
167
+ "transformer_blocks.21.img_mlp.gate_layer",
168
+ "transformer_blocks.22.attn.to_q",
169
+ "transformer_blocks.22.attn.to_k",
170
+ "transformer_blocks.22.attn.to_v",
171
+ "transformer_blocks.22.attn.to_out.0",
172
+ "transformer_blocks.22.img_mlp.proj",
173
+ "transformer_blocks.22.img_mlp.out",
174
+ "transformer_blocks.22.img_mlp.gate_layer",
175
+ "transformer_blocks.23.attn.to_q",
176
+ "transformer_blocks.23.attn.to_k",
177
+ "transformer_blocks.23.attn.to_v",
178
+ "transformer_blocks.23.attn.to_out.0",
179
+ "transformer_blocks.23.img_mlp.proj",
180
+ "transformer_blocks.23.img_mlp.out",
181
+ "transformer_blocks.23.img_mlp.gate_layer",
182
+ "transformer_blocks.24.attn.to_q",
183
+ "transformer_blocks.24.attn.to_k",
184
+ "transformer_blocks.24.attn.to_v",
185
+ "transformer_blocks.24.attn.to_out.0",
186
+ "transformer_blocks.24.img_mlp.proj",
187
+ "transformer_blocks.24.img_mlp.out",
188
+ "transformer_blocks.24.img_mlp.gate_layer",
189
+ "transformer_blocks.25.attn.to_q",
190
+ "transformer_blocks.25.attn.to_k",
191
+ "transformer_blocks.25.attn.to_v",
192
+ "transformer_blocks.25.attn.to_out.0",
193
+ "transformer_blocks.25.img_mlp.proj",
194
+ "transformer_blocks.25.img_mlp.out",
195
+ "transformer_blocks.25.img_mlp.gate_layer",
196
+ "transformer_blocks.26.attn.to_q",
197
+ "transformer_blocks.26.attn.to_k",
198
+ "transformer_blocks.26.attn.to_v",
199
+ "transformer_blocks.26.attn.to_out.0",
200
+ "transformer_blocks.26.img_mlp.proj",
201
+ "transformer_blocks.26.img_mlp.out",
202
+ "transformer_blocks.26.img_mlp.gate_layer",
203
+ "transformer_blocks.27.attn.to_q",
204
+ "transformer_blocks.27.attn.to_k",
205
+ "transformer_blocks.27.attn.to_v",
206
+ "transformer_blocks.27.attn.to_out.0",
207
+ "transformer_blocks.27.img_mlp.proj",
208
+ "transformer_blocks.27.img_mlp.out",
209
+ "transformer_blocks.27.img_mlp.gate_layer",
210
+ "transformer_blocks.28.attn.to_q",
211
+ "transformer_blocks.28.attn.to_k",
212
+ "transformer_blocks.28.attn.to_v",
213
+ "transformer_blocks.28.attn.to_out.0",
214
+ "transformer_blocks.28.img_mlp.proj",
215
+ "transformer_blocks.28.img_mlp.out",
216
+ "transformer_blocks.28.img_mlp.gate_layer",
217
+ "transformer_blocks.29.attn.to_q",
218
+ "transformer_blocks.29.attn.to_k",
219
+ "transformer_blocks.29.attn.to_v",
220
+ "transformer_blocks.29.attn.to_out.0",
221
+ "transformer_blocks.29.img_mlp.proj",
222
+ "transformer_blocks.29.img_mlp.out",
223
+ "transformer_blocks.29.img_mlp.gate_layer",
224
+ "transformer_blocks.30.attn.to_q",
225
+ "transformer_blocks.30.attn.to_k",
226
+ "transformer_blocks.30.attn.to_v",
227
+ "transformer_blocks.30.attn.to_out.0",
228
+ "transformer_blocks.30.img_mlp.proj",
229
+ "transformer_blocks.30.img_mlp.out",
230
+ "transformer_blocks.30.img_mlp.gate_layer",
231
+ "transformer_blocks.31.attn.to_q",
232
+ "transformer_blocks.31.attn.to_k",
233
+ "transformer_blocks.31.attn.to_v",
234
+ "transformer_blocks.31.attn.to_out.0",
235
+ "transformer_blocks.31.img_mlp.proj",
236
+ "transformer_blocks.31.img_mlp.out",
237
+ "transformer_blocks.31.img_mlp.gate_layer"
238
+ ],
239
+ "tensor_bytes": 7687135232,
240
+ "tensors": 745,
241
+ "exact_roundtrip": true
242
+ },
243
+ "text_encoder": {
244
+ "quantized_modules": [
245
+ "language_model.model.layers.0.self_attn.q_proj",
246
+ "language_model.model.layers.0.self_attn.k_proj",
247
+ "language_model.model.layers.0.self_attn.v_proj",
248
+ "language_model.model.layers.0.self_attn.o_proj",
249
+ "language_model.model.layers.0.mlp.gate_proj",
250
+ "language_model.model.layers.0.mlp.up_proj",
251
+ "language_model.model.layers.0.mlp.down_proj",
252
+ "language_model.model.layers.1.self_attn.q_proj",
253
+ "language_model.model.layers.1.self_attn.k_proj",
254
+ "language_model.model.layers.1.self_attn.v_proj",
255
+ "language_model.model.layers.1.self_attn.o_proj",
256
+ "language_model.model.layers.1.mlp.gate_proj",
257
+ "language_model.model.layers.1.mlp.up_proj",
258
+ "language_model.model.layers.1.mlp.down_proj",
259
+ "language_model.model.layers.2.self_attn.q_proj",
260
+ "language_model.model.layers.2.self_attn.k_proj",
261
+ "language_model.model.layers.2.self_attn.v_proj",
262
+ "language_model.model.layers.2.self_attn.o_proj",
263
+ "language_model.model.layers.2.mlp.gate_proj",
264
+ "language_model.model.layers.2.mlp.up_proj",
265
+ "language_model.model.layers.2.mlp.down_proj",
266
+ "language_model.model.layers.3.self_attn.q_proj",
267
+ "language_model.model.layers.3.self_attn.k_proj",
268
+ "language_model.model.layers.3.self_attn.v_proj",
269
+ "language_model.model.layers.3.self_attn.o_proj",
270
+ "language_model.model.layers.3.mlp.gate_proj",
271
+ "language_model.model.layers.3.mlp.up_proj",
272
+ "language_model.model.layers.3.mlp.down_proj",
273
+ "language_model.model.layers.4.self_attn.q_proj",
274
+ "language_model.model.layers.4.self_attn.k_proj",
275
+ "language_model.model.layers.4.self_attn.v_proj",
276
+ "language_model.model.layers.4.self_attn.o_proj",
277
+ "language_model.model.layers.4.mlp.gate_proj",
278
+ "language_model.model.layers.4.mlp.up_proj",
279
+ "language_model.model.layers.4.mlp.down_proj",
280
+ "language_model.model.layers.5.self_attn.q_proj",
281
+ "language_model.model.layers.5.self_attn.k_proj",
282
+ "language_model.model.layers.5.self_attn.v_proj",
283
+ "language_model.model.layers.5.self_attn.o_proj",
284
+ "language_model.model.layers.5.mlp.gate_proj",
285
+ "language_model.model.layers.5.mlp.up_proj",
286
+ "language_model.model.layers.5.mlp.down_proj",
287
+ "language_model.model.layers.6.self_attn.q_proj",
288
+ "language_model.model.layers.6.self_attn.k_proj",
289
+ "language_model.model.layers.6.self_attn.v_proj",
290
+ "language_model.model.layers.6.self_attn.o_proj",
291
+ "language_model.model.layers.6.mlp.gate_proj",
292
+ "language_model.model.layers.6.mlp.up_proj",
293
+ "language_model.model.layers.6.mlp.down_proj",
294
+ "language_model.model.layers.7.self_attn.q_proj",
295
+ "language_model.model.layers.7.self_attn.k_proj",
296
+ "language_model.model.layers.7.self_attn.v_proj",
297
+ "language_model.model.layers.7.self_attn.o_proj",
298
+ "language_model.model.layers.7.mlp.gate_proj",
299
+ "language_model.model.layers.7.mlp.up_proj",
300
+ "language_model.model.layers.7.mlp.down_proj",
301
+ "language_model.model.layers.8.self_attn.q_proj",
302
+ "language_model.model.layers.8.self_attn.k_proj",
303
+ "language_model.model.layers.8.self_attn.v_proj",
304
+ "language_model.model.layers.8.self_attn.o_proj",
305
+ "language_model.model.layers.8.mlp.gate_proj",
306
+ "language_model.model.layers.8.mlp.up_proj",
307
+ "language_model.model.layers.8.mlp.down_proj",
308
+ "language_model.model.layers.9.self_attn.q_proj",
309
+ "language_model.model.layers.9.self_attn.k_proj",
310
+ "language_model.model.layers.9.self_attn.v_proj",
311
+ "language_model.model.layers.9.self_attn.o_proj",
312
+ "language_model.model.layers.9.mlp.gate_proj",
313
+ "language_model.model.layers.9.mlp.up_proj",
314
+ "language_model.model.layers.9.mlp.down_proj",
315
+ "language_model.model.layers.10.self_attn.q_proj",
316
+ "language_model.model.layers.10.self_attn.k_proj",
317
+ "language_model.model.layers.10.self_attn.v_proj",
318
+ "language_model.model.layers.10.self_attn.o_proj",
319
+ "language_model.model.layers.10.mlp.gate_proj",
320
+ "language_model.model.layers.10.mlp.up_proj",
321
+ "language_model.model.layers.10.mlp.down_proj",
322
+ "language_model.model.layers.11.self_attn.q_proj",
323
+ "language_model.model.layers.11.self_attn.k_proj",
324
+ "language_model.model.layers.11.self_attn.v_proj",
325
+ "language_model.model.layers.11.self_attn.o_proj",
326
+ "language_model.model.layers.11.mlp.gate_proj",
327
+ "language_model.model.layers.11.mlp.up_proj",
328
+ "language_model.model.layers.11.mlp.down_proj",
329
+ "language_model.model.layers.12.self_attn.q_proj",
330
+ "language_model.model.layers.12.self_attn.k_proj",
331
+ "language_model.model.layers.12.self_attn.v_proj",
332
+ "language_model.model.layers.12.self_attn.o_proj",
333
+ "language_model.model.layers.12.mlp.gate_proj",
334
+ "language_model.model.layers.12.mlp.up_proj",
335
+ "language_model.model.layers.12.mlp.down_proj",
336
+ "language_model.model.layers.13.self_attn.q_proj",
337
+ "language_model.model.layers.13.self_attn.k_proj",
338
+ "language_model.model.layers.13.self_attn.v_proj",
339
+ "language_model.model.layers.13.self_attn.o_proj",
340
+ "language_model.model.layers.13.mlp.gate_proj",
341
+ "language_model.model.layers.13.mlp.up_proj",
342
+ "language_model.model.layers.13.mlp.down_proj",
343
+ "language_model.model.layers.14.self_attn.q_proj",
344
+ "language_model.model.layers.14.self_attn.k_proj",
345
+ "language_model.model.layers.14.self_attn.v_proj",
346
+ "language_model.model.layers.14.self_attn.o_proj",
347
+ "language_model.model.layers.14.mlp.gate_proj",
348
+ "language_model.model.layers.14.mlp.up_proj",
349
+ "language_model.model.layers.14.mlp.down_proj",
350
+ "language_model.model.layers.15.self_attn.q_proj",
351
+ "language_model.model.layers.15.self_attn.k_proj",
352
+ "language_model.model.layers.15.self_attn.v_proj",
353
+ "language_model.model.layers.15.self_attn.o_proj",
354
+ "language_model.model.layers.15.mlp.gate_proj",
355
+ "language_model.model.layers.15.mlp.up_proj",
356
+ "language_model.model.layers.15.mlp.down_proj",
357
+ "language_model.model.layers.16.self_attn.q_proj",
358
+ "language_model.model.layers.16.self_attn.k_proj",
359
+ "language_model.model.layers.16.self_attn.v_proj",
360
+ "language_model.model.layers.16.self_attn.o_proj",
361
+ "language_model.model.layers.16.mlp.gate_proj",
362
+ "language_model.model.layers.16.mlp.up_proj",
363
+ "language_model.model.layers.16.mlp.down_proj",
364
+ "language_model.model.layers.17.self_attn.q_proj",
365
+ "language_model.model.layers.17.self_attn.k_proj",
366
+ "language_model.model.layers.17.self_attn.v_proj",
367
+ "language_model.model.layers.17.self_attn.o_proj",
368
+ "language_model.model.layers.17.mlp.gate_proj",
369
+ "language_model.model.layers.17.mlp.up_proj",
370
+ "language_model.model.layers.17.mlp.down_proj",
371
+ "language_model.model.layers.18.self_attn.q_proj",
372
+ "language_model.model.layers.18.self_attn.k_proj",
373
+ "language_model.model.layers.18.self_attn.v_proj",
374
+ "language_model.model.layers.18.self_attn.o_proj",
375
+ "language_model.model.layers.18.mlp.gate_proj",
376
+ "language_model.model.layers.18.mlp.up_proj",
377
+ "language_model.model.layers.18.mlp.down_proj",
378
+ "language_model.model.layers.19.self_attn.q_proj",
379
+ "language_model.model.layers.19.self_attn.k_proj",
380
+ "language_model.model.layers.19.self_attn.v_proj",
381
+ "language_model.model.layers.19.self_attn.o_proj",
382
+ "language_model.model.layers.19.mlp.gate_proj",
383
+ "language_model.model.layers.19.mlp.up_proj",
384
+ "language_model.model.layers.19.mlp.down_proj",
385
+ "language_model.model.layers.20.self_attn.q_proj",
386
+ "language_model.model.layers.20.self_attn.k_proj",
387
+ "language_model.model.layers.20.self_attn.v_proj",
388
+ "language_model.model.layers.20.self_attn.o_proj",
389
+ "language_model.model.layers.20.mlp.gate_proj",
390
+ "language_model.model.layers.20.mlp.up_proj",
391
+ "language_model.model.layers.20.mlp.down_proj",
392
+ "language_model.model.layers.21.self_attn.q_proj",
393
+ "language_model.model.layers.21.self_attn.k_proj",
394
+ "language_model.model.layers.21.self_attn.v_proj",
395
+ "language_model.model.layers.21.self_attn.o_proj",
396
+ "language_model.model.layers.21.mlp.gate_proj",
397
+ "language_model.model.layers.21.mlp.up_proj",
398
+ "language_model.model.layers.21.mlp.down_proj",
399
+ "language_model.model.layers.22.self_attn.q_proj",
400
+ "language_model.model.layers.22.self_attn.k_proj",
401
+ "language_model.model.layers.22.self_attn.v_proj",
402
+ "language_model.model.layers.22.self_attn.o_proj",
403
+ "language_model.model.layers.22.mlp.gate_proj",
404
+ "language_model.model.layers.22.mlp.up_proj",
405
+ "language_model.model.layers.22.mlp.down_proj",
406
+ "language_model.model.layers.23.self_attn.q_proj",
407
+ "language_model.model.layers.23.self_attn.k_proj",
408
+ "language_model.model.layers.23.self_attn.v_proj",
409
+ "language_model.model.layers.23.self_attn.o_proj",
410
+ "language_model.model.layers.23.mlp.gate_proj",
411
+ "language_model.model.layers.23.mlp.up_proj",
412
+ "language_model.model.layers.23.mlp.down_proj",
413
+ "language_model.model.layers.24.self_attn.q_proj",
414
+ "language_model.model.layers.24.self_attn.k_proj",
415
+ "language_model.model.layers.24.self_attn.v_proj",
416
+ "language_model.model.layers.24.self_attn.o_proj",
417
+ "language_model.model.layers.24.mlp.gate_proj",
418
+ "language_model.model.layers.24.mlp.up_proj",
419
+ "language_model.model.layers.24.mlp.down_proj",
420
+ "language_model.model.layers.25.self_attn.q_proj",
421
+ "language_model.model.layers.25.self_attn.k_proj",
422
+ "language_model.model.layers.25.self_attn.v_proj",
423
+ "language_model.model.layers.25.self_attn.o_proj",
424
+ "language_model.model.layers.25.mlp.gate_proj",
425
+ "language_model.model.layers.25.mlp.up_proj",
426
+ "language_model.model.layers.25.mlp.down_proj",
427
+ "language_model.model.layers.26.self_attn.q_proj",
428
+ "language_model.model.layers.26.self_attn.k_proj",
429
+ "language_model.model.layers.26.self_attn.v_proj",
430
+ "language_model.model.layers.26.self_attn.o_proj",
431
+ "language_model.model.layers.26.mlp.gate_proj",
432
+ "language_model.model.layers.26.mlp.up_proj",
433
+ "language_model.model.layers.26.mlp.down_proj",
434
+ "language_model.model.layers.27.self_attn.q_proj",
435
+ "language_model.model.layers.27.self_attn.k_proj",
436
+ "language_model.model.layers.27.self_attn.v_proj",
437
+ "language_model.model.layers.27.self_attn.o_proj",
438
+ "language_model.model.layers.27.mlp.gate_proj",
439
+ "language_model.model.layers.27.mlp.up_proj",
440
+ "language_model.model.layers.27.mlp.down_proj",
441
+ "language_model.model.layers.28.self_attn.q_proj",
442
+ "language_model.model.layers.28.self_attn.k_proj",
443
+ "language_model.model.layers.28.self_attn.v_proj",
444
+ "language_model.model.layers.28.self_attn.o_proj",
445
+ "language_model.model.layers.28.mlp.gate_proj",
446
+ "language_model.model.layers.28.mlp.up_proj",
447
+ "language_model.model.layers.28.mlp.down_proj",
448
+ "language_model.model.layers.29.self_attn.q_proj",
449
+ "language_model.model.layers.29.self_attn.k_proj",
450
+ "language_model.model.layers.29.self_attn.v_proj",
451
+ "language_model.model.layers.29.self_attn.o_proj",
452
+ "language_model.model.layers.29.mlp.gate_proj",
453
+ "language_model.model.layers.29.mlp.up_proj",
454
+ "language_model.model.layers.29.mlp.down_proj",
455
+ "language_model.model.layers.30.self_attn.q_proj",
456
+ "language_model.model.layers.30.self_attn.k_proj",
457
+ "language_model.model.layers.30.self_attn.v_proj",
458
+ "language_model.model.layers.30.self_attn.o_proj",
459
+ "language_model.model.layers.30.mlp.gate_proj",
460
+ "language_model.model.layers.30.mlp.up_proj",
461
+ "language_model.model.layers.30.mlp.down_proj",
462
+ "language_model.model.layers.31.self_attn.q_proj",
463
+ "language_model.model.layers.31.self_attn.k_proj",
464
+ "language_model.model.layers.31.self_attn.v_proj",
465
+ "language_model.model.layers.31.self_attn.o_proj",
466
+ "language_model.model.layers.31.mlp.gate_proj",
467
+ "language_model.model.layers.31.mlp.up_proj",
468
+ "language_model.model.layers.31.mlp.down_proj",
469
+ "language_model.model.layers.32.self_attn.q_proj",
470
+ "language_model.model.layers.32.self_attn.k_proj",
471
+ "language_model.model.layers.32.self_attn.v_proj",
472
+ "language_model.model.layers.32.self_attn.o_proj",
473
+ "language_model.model.layers.32.mlp.gate_proj",
474
+ "language_model.model.layers.32.mlp.up_proj",
475
+ "language_model.model.layers.32.mlp.down_proj",
476
+ "language_model.model.layers.33.self_attn.q_proj",
477
+ "language_model.model.layers.33.self_attn.k_proj",
478
+ "language_model.model.layers.33.self_attn.v_proj",
479
+ "language_model.model.layers.33.self_attn.o_proj",
480
+ "language_model.model.layers.33.mlp.gate_proj",
481
+ "language_model.model.layers.33.mlp.up_proj",
482
+ "language_model.model.layers.33.mlp.down_proj",
483
+ "language_model.model.layers.34.self_attn.q_proj",
484
+ "language_model.model.layers.34.self_attn.k_proj",
485
+ "language_model.model.layers.34.self_attn.v_proj",
486
+ "language_model.model.layers.34.self_attn.o_proj",
487
+ "language_model.model.layers.34.mlp.gate_proj",
488
+ "language_model.model.layers.34.mlp.up_proj",
489
+ "language_model.model.layers.34.mlp.down_proj",
490
+ "language_model.model.layers.35.self_attn.q_proj",
491
+ "language_model.model.layers.35.self_attn.k_proj",
492
+ "language_model.model.layers.35.self_attn.v_proj",
493
+ "language_model.model.layers.35.self_attn.o_proj",
494
+ "language_model.model.layers.35.mlp.gate_proj",
495
+ "language_model.model.layers.35.mlp.up_proj",
496
+ "language_model.model.layers.35.mlp.down_proj"
497
+ ],
498
+ "tensor_bytes": 11022590432,
499
+ "tensors": 1254,
500
+ "exact_roundtrip": true
501
+ },
502
+ "vae": {
503
+ "quantized_modules": [],
504
+ "tensor_bytes": 1350961616,
505
+ "tensors": 238,
506
+ "exact_roundtrip": true
507
+ }
508
+ },
509
+ "source_audit_sha256": "149b336e13a8dc556e1b8992fc333e0e844eb3dbe58fb7346c6f58aea95023f3",
510
+ "status": "converted_and_roundtrip_verified",
511
+ "seconds": 28.819035249995068
512
+ }
evaluation/bf16/composition-s42.png ADDED

Git LFS Details

  • SHA256: b361dd45e536a8d4d2baa45d412966c1820eb21e0d096829387cd0afa736aaef
  • Pointer size: 132 Bytes
  • Size of remote file: 1.11 MB
evaluation/cache-parity-8bit.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "models/Image21-MLX-8bit",
3
+ "passed": true,
4
+ "rows": [
5
+ {
6
+ "timestep": 0.7,
7
+ "max_abs_error": 0.0,
8
+ "relative_rmse": 0.0,
9
+ "uncached_seconds": 2.1813871250487864,
10
+ "cached_seconds": 2.0793312090681866
11
+ },
12
+ {
13
+ "timestep": 0.3,
14
+ "max_abs_error": 0.0,
15
+ "relative_rmse": 0.0,
16
+ "uncached_seconds": 2.1828467500163242,
17
+ "cached_seconds": 1.9954760000109673
18
+ }
19
+ ],
20
+ "note": "Real transformer with synthetic conditioning/latents; BF16 numerical tolerance, not pixel equality."
21
+ }
evaluation/cache-parity-bf16.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "models/Image21-MLX-bf16",
3
+ "passed": true,
4
+ "rows": [
5
+ {
6
+ "timestep": 0.7,
7
+ "max_abs_error": 0.0,
8
+ "relative_rmse": 0.0,
9
+ "uncached_seconds": 2.253576749935746,
10
+ "cached_seconds": 1.9403239579405636
11
+ },
12
+ {
13
+ "timestep": 0.3,
14
+ "max_abs_error": 0.0,
15
+ "relative_rmse": 0.0,
16
+ "uncached_seconds": 2.003779707942158,
17
+ "cached_seconds": 1.9097302500158548
18
+ }
19
+ ],
20
+ "note": "Real transformer with synthetic conditioning/latents; BF16 numerical tolerance, not pixel equality."
21
+ }
evaluation/offload-parity.json ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "passed": true,
3
+ "size": 512,
4
+ "steps": 2,
5
+ "rows": [
6
+ {
7
+ "case": "portrait",
8
+ "identical_png": true,
9
+ "resident_peak_gib": 23.64565789140761,
10
+ "phased_peak_gib": 10.36156104132533,
11
+ "precision": "8bit"
12
+ },
13
+ {
14
+ "case": "rgba",
15
+ "identical_png": true,
16
+ "resident_peak_gib": 23.64657341875136,
17
+ "phased_peak_gib": 10.389103142544627,
18
+ "precision": "8bit"
19
+ },
20
+ {
21
+ "case": "edit",
22
+ "identical_png": true,
23
+ "resident_peak_gib": 23.646100413054228,
24
+ "phased_peak_gib": 11.152842032723129,
25
+ "precision": "8bit"
26
+ },
27
+ {
28
+ "case": "portrait",
29
+ "precision": "bf16",
30
+ "identical_png": true,
31
+ "resident_peak_gib": 35.80386101640761,
32
+ "phased_peak_gib": 16.414859991520643
33
+ }
34
+ ],
35
+ "note": "PNG SHA256 identical in all four resident-versus-phase checks. Short smoke checks are not quality benchmarks.",
36
+ "full_resolution_check": {
37
+ "case": "portrait",
38
+ "precision": "bf16",
39
+ "size": 1024,
40
+ "steps": 40,
41
+ "identical_png": true,
42
+ "png_sha256": "72493104ac6662f1de9e7857e8ec03f4f3971c31c4ea77207d33ae029a3a0c41",
43
+ "resident_peak_gib": 45.77874500118196,
44
+ "phased_peak_gib": 16.41486000828445,
45
+ "note": "Resident timing excluded because it belonged to the earlier memory-pressure diagnostic."
46
+ }
47
+ }
evaluation/precision-probe.json ADDED
@@ -0,0 +1,483 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "group_size": 64,
3
+ "mode": "affine",
4
+ "device": {
5
+ "device_name": "Apple M4 Max",
6
+ "max_recommended_working_set_size": 115448725504,
7
+ "memory_size": 137438953472,
8
+ "architecture": "applegpu_g16s",
9
+ "max_buffer_length": 86586540032,
10
+ "resource_limit": 499000
11
+ },
12
+ "results": [
13
+ {
14
+ "component": "transformer",
15
+ "tensor": "transformer_blocks.0.attn.to_q.weight",
16
+ "source_shape": [
17
+ 4096,
18
+ 4096
19
+ ],
20
+ "sample_rows": 256,
21
+ "bits": 4,
22
+ "weight_relative_rmse": 0.09608591347932816,
23
+ "synthetic_matmul_relative_rmse": 0.09557931870222092,
24
+ "matmul_ms": 0.5911250016652048
25
+ },
26
+ {
27
+ "component": "transformer",
28
+ "tensor": "transformer_blocks.0.attn.to_q.weight",
29
+ "source_shape": [
30
+ 4096,
31
+ 4096
32
+ ],
33
+ "sample_rows": 256,
34
+ "bits": 6,
35
+ "weight_relative_rmse": 0.02344648726284504,
36
+ "synthetic_matmul_relative_rmse": 0.023693682625889778,
37
+ "matmul_ms": 0.4109000088647008
38
+ },
39
+ {
40
+ "component": "transformer",
41
+ "tensor": "transformer_blocks.0.attn.to_q.weight",
42
+ "source_shape": [
43
+ 4096,
44
+ 4096
45
+ ],
46
+ "sample_rows": 256,
47
+ "bits": 8,
48
+ "weight_relative_rmse": 0.007525453343987465,
49
+ "synthetic_matmul_relative_rmse": 0.008602947928011417,
50
+ "matmul_ms": 0.4843916976824403
51
+ },
52
+ {
53
+ "component": "transformer",
54
+ "tensor": "transformer_blocks.0.img_mlp.out.weight",
55
+ "source_shape": [
56
+ 4096,
57
+ 12288
58
+ ],
59
+ "sample_rows": 256,
60
+ "bits": 4,
61
+ "weight_relative_rmse": 0.09441296756267548,
62
+ "synthetic_matmul_relative_rmse": 0.09449099004268646,
63
+ "matmul_ms": 0.4845582996495068
64
+ },
65
+ {
66
+ "component": "transformer",
67
+ "tensor": "transformer_blocks.0.img_mlp.out.weight",
68
+ "source_shape": [
69
+ 4096,
70
+ 12288
71
+ ],
72
+ "sample_rows": 256,
73
+ "bits": 6,
74
+ "weight_relative_rmse": 0.02303435653448105,
75
+ "synthetic_matmul_relative_rmse": 0.02340426668524742,
76
+ "matmul_ms": 0.46681660460308194
77
+ },
78
+ {
79
+ "component": "transformer",
80
+ "tensor": "transformer_blocks.0.img_mlp.out.weight",
81
+ "source_shape": [
82
+ 4096,
83
+ 12288
84
+ ],
85
+ "sample_rows": 256,
86
+ "bits": 8,
87
+ "weight_relative_rmse": 0.007468077354133129,
88
+ "synthetic_matmul_relative_rmse": 0.008637943305075169,
89
+ "matmul_ms": 0.36163330078125
90
+ },
91
+ {
92
+ "component": "transformer",
93
+ "tensor": "transformer_blocks.15.attn.to_q.weight",
94
+ "source_shape": [
95
+ 4096,
96
+ 4096
97
+ ],
98
+ "sample_rows": 256,
99
+ "bits": 4,
100
+ "weight_relative_rmse": 0.09535665810108185,
101
+ "synthetic_matmul_relative_rmse": 0.09527087211608887,
102
+ "matmul_ms": 0.57477499358356
103
+ },
104
+ {
105
+ "component": "transformer",
106
+ "tensor": "transformer_blocks.15.attn.to_q.weight",
107
+ "source_shape": [
108
+ 4096,
109
+ 4096
110
+ ],
111
+ "sample_rows": 256,
112
+ "bits": 6,
113
+ "weight_relative_rmse": 0.02322564087808132,
114
+ "synthetic_matmul_relative_rmse": 0.023668956011533737,
115
+ "matmul_ms": 0.4584625014103949
116
+ },
117
+ {
118
+ "component": "transformer",
119
+ "tensor": "transformer_blocks.15.attn.to_q.weight",
120
+ "source_shape": [
121
+ 4096,
122
+ 4096
123
+ ],
124
+ "sample_rows": 256,
125
+ "bits": 8,
126
+ "weight_relative_rmse": 0.0074961441569030285,
127
+ "synthetic_matmul_relative_rmse": 0.008681540377438068,
128
+ "matmul_ms": 0.533016596455127
129
+ },
130
+ {
131
+ "component": "transformer",
132
+ "tensor": "transformer_blocks.15.img_mlp.out.weight",
133
+ "source_shape": [
134
+ 4096,
135
+ 12288
136
+ ],
137
+ "sample_rows": 256,
138
+ "bits": 4,
139
+ "weight_relative_rmse": 0.09448622912168503,
140
+ "synthetic_matmul_relative_rmse": 0.09489456564188004,
141
+ "matmul_ms": 0.9572708979249
142
+ },
143
+ {
144
+ "component": "transformer",
145
+ "tensor": "transformer_blocks.15.img_mlp.out.weight",
146
+ "source_shape": [
147
+ 4096,
148
+ 12288
149
+ ],
150
+ "sample_rows": 256,
151
+ "bits": 6,
152
+ "weight_relative_rmse": 0.02305353432893753,
153
+ "synthetic_matmul_relative_rmse": 0.02346830628812313,
154
+ "matmul_ms": 0.645641703158617
155
+ },
156
+ {
157
+ "component": "transformer",
158
+ "tensor": "transformer_blocks.15.img_mlp.out.weight",
159
+ "source_shape": [
160
+ 4096,
161
+ 12288
162
+ ],
163
+ "sample_rows": 256,
164
+ "bits": 8,
165
+ "weight_relative_rmse": 0.007503869011998177,
166
+ "synthetic_matmul_relative_rmse": 0.00874588917940855,
167
+ "matmul_ms": 0.6554292049258947
168
+ },
169
+ {
170
+ "component": "transformer",
171
+ "tensor": "transformer_blocks.31.attn.to_q.weight",
172
+ "source_shape": [
173
+ 4096,
174
+ 4096
175
+ ],
176
+ "sample_rows": 256,
177
+ "bits": 4,
178
+ "weight_relative_rmse": 0.09847480803728104,
179
+ "synthetic_matmul_relative_rmse": 0.0985177680850029,
180
+ "matmul_ms": 0.5378333968110383
181
+ },
182
+ {
183
+ "component": "transformer",
184
+ "tensor": "transformer_blocks.31.attn.to_q.weight",
185
+ "source_shape": [
186
+ 4096,
187
+ 4096
188
+ ],
189
+ "sample_rows": 256,
190
+ "bits": 6,
191
+ "weight_relative_rmse": 0.0240206029266119,
192
+ "synthetic_matmul_relative_rmse": 0.024259869009256363,
193
+ "matmul_ms": 0.5431666970252991
194
+ },
195
+ {
196
+ "component": "transformer",
197
+ "tensor": "transformer_blocks.31.attn.to_q.weight",
198
+ "source_shape": [
199
+ 4096,
200
+ 4096
201
+ ],
202
+ "sample_rows": 256,
203
+ "bits": 8,
204
+ "weight_relative_rmse": 0.007802714128047228,
205
+ "synthetic_matmul_relative_rmse": 0.008978405967354774,
206
+ "matmul_ms": 0.5484999972395599
207
+ },
208
+ {
209
+ "component": "transformer",
210
+ "tensor": "transformer_blocks.31.img_mlp.out.weight",
211
+ "source_shape": [
212
+ 4096,
213
+ 12288
214
+ ],
215
+ "sample_rows": 256,
216
+ "bits": 4,
217
+ "weight_relative_rmse": 0.09683173894882202,
218
+ "synthetic_matmul_relative_rmse": 0.09696819633245468,
219
+ "matmul_ms": 0.9709125035442412
220
+ },
221
+ {
222
+ "component": "transformer",
223
+ "tensor": "transformer_blocks.31.img_mlp.out.weight",
224
+ "source_shape": [
225
+ 4096,
226
+ 12288
227
+ ],
228
+ "sample_rows": 256,
229
+ "bits": 6,
230
+ "weight_relative_rmse": 0.023639416322112083,
231
+ "synthetic_matmul_relative_rmse": 0.0238750372081995,
232
+ "matmul_ms": 0.3942042007111013
233
+ },
234
+ {
235
+ "component": "transformer",
236
+ "tensor": "transformer_blocks.31.img_mlp.out.weight",
237
+ "source_shape": [
238
+ 4096,
239
+ 12288
240
+ ],
241
+ "sample_rows": 256,
242
+ "bits": 8,
243
+ "weight_relative_rmse": 0.007727540098130703,
244
+ "synthetic_matmul_relative_rmse": 0.008899149484932423,
245
+ "matmul_ms": 0.4157708026468754
246
+ },
247
+ {
248
+ "component": "text_encoder",
249
+ "tensor": "model.language_model.layers.0.mlp.down_proj.weight",
250
+ "source_shape": [
251
+ 4096,
252
+ 12288
253
+ ],
254
+ "sample_rows": 256,
255
+ "bits": 4,
256
+ "weight_relative_rmse": 0.09275667369365692,
257
+ "synthetic_matmul_relative_rmse": 0.09330281615257263,
258
+ "matmul_ms": 0.4116749973036349
259
+ },
260
+ {
261
+ "component": "text_encoder",
262
+ "tensor": "model.language_model.layers.0.mlp.down_proj.weight",
263
+ "source_shape": [
264
+ 4096,
265
+ 12288
266
+ ],
267
+ "sample_rows": 256,
268
+ "bits": 6,
269
+ "weight_relative_rmse": 0.022620145231485367,
270
+ "synthetic_matmul_relative_rmse": 0.022967973724007607,
271
+ "matmul_ms": 0.3953375038690865
272
+ },
273
+ {
274
+ "component": "text_encoder",
275
+ "tensor": "model.language_model.layers.0.mlp.down_proj.weight",
276
+ "source_shape": [
277
+ 4096,
278
+ 12288
279
+ ],
280
+ "sample_rows": 256,
281
+ "bits": 8,
282
+ "weight_relative_rmse": 0.0073314751498401165,
283
+ "synthetic_matmul_relative_rmse": 0.008502925746142864,
284
+ "matmul_ms": 0.40935829747468233
285
+ },
286
+ {
287
+ "component": "text_encoder",
288
+ "tensor": "model.language_model.layers.0.self_attn.q_proj.weight",
289
+ "source_shape": [
290
+ 4096,
291
+ 4096
292
+ ],
293
+ "sample_rows": 256,
294
+ "bits": 4,
295
+ "weight_relative_rmse": 0.09217474609613419,
296
+ "synthetic_matmul_relative_rmse": 0.09356217831373215,
297
+ "matmul_ms": 0.28917910531163216
298
+ },
299
+ {
300
+ "component": "text_encoder",
301
+ "tensor": "model.language_model.layers.0.self_attn.q_proj.weight",
302
+ "source_shape": [
303
+ 4096,
304
+ 4096
305
+ ],
306
+ "sample_rows": 256,
307
+ "bits": 6,
308
+ "weight_relative_rmse": 0.022505931556224823,
309
+ "synthetic_matmul_relative_rmse": 0.02307399921119213,
310
+ "matmul_ms": 0.2340042032301426
311
+ },
312
+ {
313
+ "component": "text_encoder",
314
+ "tensor": "model.language_model.layers.0.self_attn.q_proj.weight",
315
+ "source_shape": [
316
+ 4096,
317
+ 4096
318
+ ],
319
+ "sample_rows": 256,
320
+ "bits": 8,
321
+ "weight_relative_rmse": 0.007364622782915831,
322
+ "synthetic_matmul_relative_rmse": 0.008642347529530525,
323
+ "matmul_ms": 0.23023750400170684
324
+ },
325
+ {
326
+ "component": "text_encoder",
327
+ "tensor": "model.language_model.layers.17.mlp.down_proj.weight",
328
+ "source_shape": [
329
+ 4096,
330
+ 12288
331
+ ],
332
+ "sample_rows": 256,
333
+ "bits": 4,
334
+ "weight_relative_rmse": 0.0940268263220787,
335
+ "synthetic_matmul_relative_rmse": 0.09416663646697998,
336
+ "matmul_ms": 0.7600749959237874
337
+ },
338
+ {
339
+ "component": "text_encoder",
340
+ "tensor": "model.language_model.layers.17.mlp.down_proj.weight",
341
+ "source_shape": [
342
+ 4096,
343
+ 12288
344
+ ],
345
+ "sample_rows": 256,
346
+ "bits": 6,
347
+ "weight_relative_rmse": 0.02295832335948944,
348
+ "synthetic_matmul_relative_rmse": 0.02339865453541279,
349
+ "matmul_ms": 0.3463750006631017
350
+ },
351
+ {
352
+ "component": "text_encoder",
353
+ "tensor": "model.language_model.layers.17.mlp.down_proj.weight",
354
+ "source_shape": [
355
+ 4096,
356
+ 12288
357
+ ],
358
+ "sample_rows": 256,
359
+ "bits": 8,
360
+ "weight_relative_rmse": 0.007490983698517084,
361
+ "synthetic_matmul_relative_rmse": 0.00866878591477871,
362
+ "matmul_ms": 0.40897090220823884
363
+ },
364
+ {
365
+ "component": "text_encoder",
366
+ "tensor": "model.language_model.layers.17.self_attn.q_proj.weight",
367
+ "source_shape": [
368
+ 4096,
369
+ 4096
370
+ ],
371
+ "sample_rows": 256,
372
+ "bits": 4,
373
+ "weight_relative_rmse": 0.09371878206729889,
374
+ "synthetic_matmul_relative_rmse": 0.09475100785493851,
375
+ "matmul_ms": 0.22580409422516823
376
+ },
377
+ {
378
+ "component": "text_encoder",
379
+ "tensor": "model.language_model.layers.17.self_attn.q_proj.weight",
380
+ "source_shape": [
381
+ 4096,
382
+ 4096
383
+ ],
384
+ "sample_rows": 256,
385
+ "bits": 6,
386
+ "weight_relative_rmse": 0.022860271856188774,
387
+ "synthetic_matmul_relative_rmse": 0.02319853939116001,
388
+ "matmul_ms": 0.2775333006866276
389
+ },
390
+ {
391
+ "component": "text_encoder",
392
+ "tensor": "model.language_model.layers.17.self_attn.q_proj.weight",
393
+ "source_shape": [
394
+ 4096,
395
+ 4096
396
+ ],
397
+ "sample_rows": 256,
398
+ "bits": 8,
399
+ "weight_relative_rmse": 0.007405295968055725,
400
+ "synthetic_matmul_relative_rmse": 0.008685958571732044,
401
+ "matmul_ms": 0.21041249856352806
402
+ },
403
+ {
404
+ "component": "text_encoder",
405
+ "tensor": "model.language_model.layers.35.mlp.down_proj.weight",
406
+ "source_shape": [
407
+ 4096,
408
+ 12288
409
+ ],
410
+ "sample_rows": 256,
411
+ "bits": 4,
412
+ "weight_relative_rmse": 0.09651479870080948,
413
+ "synthetic_matmul_relative_rmse": 0.09677258133888245,
414
+ "matmul_ms": 0.8071957970969379
415
+ },
416
+ {
417
+ "component": "text_encoder",
418
+ "tensor": "model.language_model.layers.35.mlp.down_proj.weight",
419
+ "source_shape": [
420
+ 4096,
421
+ 12288
422
+ ],
423
+ "sample_rows": 256,
424
+ "bits": 6,
425
+ "weight_relative_rmse": 0.02353922463953495,
426
+ "synthetic_matmul_relative_rmse": 0.02398456446826458,
427
+ "matmul_ms": 0.5642374977469444
428
+ },
429
+ {
430
+ "component": "text_encoder",
431
+ "tensor": "model.language_model.layers.35.mlp.down_proj.weight",
432
+ "source_shape": [
433
+ 4096,
434
+ 12288
435
+ ],
436
+ "sample_rows": 256,
437
+ "bits": 8,
438
+ "weight_relative_rmse": 0.0075887893326580524,
439
+ "synthetic_matmul_relative_rmse": 0.008758915588259697,
440
+ "matmul_ms": 0.5837332922965288
441
+ },
442
+ {
443
+ "component": "text_encoder",
444
+ "tensor": "model.language_model.layers.35.self_attn.q_proj.weight",
445
+ "source_shape": [
446
+ 4096,
447
+ 4096
448
+ ],
449
+ "sample_rows": 256,
450
+ "bits": 4,
451
+ "weight_relative_rmse": 0.0928456261754036,
452
+ "synthetic_matmul_relative_rmse": 0.09324980527162552,
453
+ "matmul_ms": 0.344270805362612
454
+ },
455
+ {
456
+ "component": "text_encoder",
457
+ "tensor": "model.language_model.layers.35.self_attn.q_proj.weight",
458
+ "source_shape": [
459
+ 4096,
460
+ 4096
461
+ ],
462
+ "sample_rows": 256,
463
+ "bits": 6,
464
+ "weight_relative_rmse": 0.022607337683439255,
465
+ "synthetic_matmul_relative_rmse": 0.02301245927810669,
466
+ "matmul_ms": 0.24173329584300518
467
+ },
468
+ {
469
+ "component": "text_encoder",
470
+ "tensor": "model.language_model.layers.35.self_attn.q_proj.weight",
471
+ "source_shape": [
472
+ 4096,
473
+ 4096
474
+ ],
475
+ "sample_rows": 256,
476
+ "bits": 8,
477
+ "weight_relative_rmse": 0.007241594605147839,
478
+ "synthetic_matmul_relative_rmse": 0.008508691564202309,
479
+ "matmul_ms": 0.20297080045565963
480
+ }
481
+ ],
482
+ "limitations": "Sampled rows and synthetic inputs; not calibrated activations, image quality or end-to-end speed."
483
+ }
evaluation/report.md ADDED
@@ -0,0 +1,37 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Image21-MLX informal evaluation
2
+
3
+ One seed per case, one machine, same pinned MLX runtime. Sequential BF16-then-Q8 desktop run without repeated or interleaved trials; order, thermal state and other applications may affect timing. Pixel similarity is not a perceptual quality score; no cross-runtime parity claim.
4
+
5
+ | Case | BF16 s | Q8 s | BF16 peak GiB | Q8 peak GiB | RGB PSNR dB |
6
+ |---|---:|---:|---:|---:|---:|
7
+ | portrait | 440.91 | 568.86 | 16.41 | 16.19 | 42.03 |
8
+ | english_text | 464.32 | 566.88 | 16.43 | 16.20 | 19.89 |
9
+ | chinese_text | 474.14 | 567.43 | 16.44 | 16.20 | 36.97 |
10
+ | composition | 480.91 | 562.52 | 16.44 | 16.20 | 37.17 |
11
+ | texture | 487.64 | 445.07 | 16.43 | 16.20 | 38.41 |
12
+ | rgba | 485.45 | 557.30 | 16.42 | 16.20 | 29.81 |
13
+ | edit | 606.29 | 665.00 | 18.68 | 16.20 | 51.09 |
14
+
15
+ MLX peak allocated memory excludes the OS, other applications and some process allocations. It is not minimum machine RAM.
16
+
17
+ Timing includes prompt encoding, phase-by-phase component loading, denoising and VAE decoding; PNG writing is excluded. The after-image RSS field is a snapshot after component release, not a peak.
18
+
19
+ Existing system swap is recorded separately from per-image swap change. This desktop-session run leaves other applications open. See system-context.json and the per-model environment.json files.
20
+
21
+ All editing pairs use the same BF16 portrait input. Visual findings and release limitations are in visual-review.json.
22
+
23
+ ![All evaluated pairs](comparison.png)
24
+
25
+ ## Visual review / 视觉检查
26
+
27
+ 逐对检查七类样本,未发现明显严重的量化退化:中英文文字均正确,杯子数量与顺序、毛衣改色及真实透明通道均保留。英文海报的字体和台灯结构有可见变化,龙贴纸的表情与细节也有变化,其余样本更接近。两种精度的人物图都比提示词要求裁得更紧,透明图均有少量接近零的背景 alpha 残留。这是助手进行的非盲法、每类单种子视觉检查,不代表无损或统计等价。
28
+
29
+ Assistant inspection of all seven paired cases found no obvious severe quality regression in this small sample. English and Chinese text are correct in both precisions; cup count/order and sweater recoloring succeed; real RGBA transparency is retained. The English poster changes visibly in typeface and lamp structure, and the dragon changes expression/details. Other pairs are closer. Both portraits are cropped more tightly than requested, and both transparency samples contain slight near-zero background alpha. This is an unblinded, single-seed visual check, not evidence of lossless quantization or statistical equivalence.
30
+
31
+ - **portrait**: Very close to BF16 in facial features, framing, silver hair, blue knit sweater and window light. Small local skin/hair/knit texture changes are visible, without an obvious severe degradation. Same chest-up rather than waist-up crop limitation.
32
+ - **english_text**: Exact, legible CREATE WITH LIGHT and September 2026; no extra text. Compared with BF16 the serif typeface/underline and lamp stem/base change visibly, but both remain coherent and satisfy the prompt. This demonstrates non-identical generation, not an obvious readability regression.
33
+ - **chinese_text**: Both Chinese lines and comma are correct and legible, matching BF16. Layout, cream backdrop and cup placement remain very close; small latte/foam texture variations are visible. No obvious text accuracy regression in this sample.
34
+ - **composition**: Exactly three cups in the requested red/blue/yellow order and one green apple in front of the blue cup. Overall layout is very close to BF16. Minor handle/shadow geometry changes are visible; no counting or spatial-order regression.
35
+ - **texture**: Very close to BF16: coherent bird anatomy, fine blue/orange feather detail, wet moss and droplets. Minor local texture/brightness changes without obvious smoothing or broken detail.
36
+ - **rgba**: Full-body green dragon with orange wings and white sticker outline. Expression and surface details change visibly from BF16 while silhouette and requested content remain coherent. 60.69% of pixels have alpha exactly zero (BF16 60.47%); both have canvas-edge alpha at most 4/255. No loss of actual transparency; neither output is a perfectly binary matte.
37
+ - **edit**: Successful red sweater recoloring with facial features, hair, pose and window background very close to the BF16 edit. Minor local texture changes; no obvious edit-consistency regression. Neither output guarantees pixel-identical preservation outside the garment.
evaluation/runtime-manifest.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "path": "scripts/runtime.py",
4
+ "sha256": "ad32ba856fddf5168d50e930204603fd2caeb016bc70fe9972d769aade6a79f4"
5
+ },
6
+ {
7
+ "path": "scripts/mlx_pipeline.py",
8
+ "sha256": "ac001b3cf8b79304fc02efd3127576a67d7b03551a6295a6fc1c2a13d032614b"
9
+ },
10
+ {
11
+ "path": "scripts/benchmark.py",
12
+ "sha256": "318e2bde9fc68fdd547c8fd32a0fb9fbe69f3cf99e62b5049240ee00754b1566"
13
+ },
14
+ {
15
+ "path": "requirements.lock.txt",
16
+ "sha256": "ec2e0402146f97b80165946f96e39374bd68aa7d54c32163e7ade178a06511c5"
17
+ }
18
+ ]
evaluation/source-files.json ADDED
@@ -0,0 +1,164 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "revision": "b3179ad355be050328e483a9dfdd9e60cd62adfa",
3
+ "files": [
4
+ {
5
+ "path": "text_encoder/model-00001-of-00004.safetensors",
6
+ "size": 4998056552,
7
+ "sha256": "dde00291b5f7fb92013895310a3da0ddba78674df9f10d505d375243dc01fc6f"
8
+ },
9
+ {
10
+ "path": "text_encoder/model-00002-of-00004.safetensors",
11
+ "size": 4915962464,
12
+ "sha256": "9047faccc0a6d98496a52d55f27be1c94a9c259d1e283fbea0128d054a948d42"
13
+ },
14
+ {
15
+ "path": "text_encoder/model-00003-of-00004.safetensors",
16
+ "size": 4915962496,
17
+ "sha256": "8c54187654c0176b73ae73785bf791dc9a14c9df7fb4310083a09d42048cb57e"
18
+ },
19
+ {
20
+ "path": "text_encoder/model-00004-of-00004.safetensors",
21
+ "size": 2704357976,
22
+ "sha256": "5311532aaaeae3259eb6a7b2c600636be1159adf7ded35f53579f7d0e7d43cdd"
23
+ },
24
+ {
25
+ "path": "transformer/diffusion_pytorch_model-00001-of-00002.safetensors",
26
+ "size": 9968332504,
27
+ "sha256": "9e6bc2d641e67bf277895ea8777141044a38f3edb7101bc469b2961dd7c36b4b"
28
+ },
29
+ {
30
+ "path": "transformer/diffusion_pytorch_model-00002-of-00002.safetensors",
31
+ "size": 4261951904,
32
+ "sha256": "3aaf234dcbe128530479735854a346b5e3e66283b7c11db56f836bbd1c13ebaa"
33
+ },
34
+ {
35
+ "path": "vae/diffusion_pytorch_model.safetensors",
36
+ "size": 1350989512,
37
+ "sha256": "a07a1b7c4ee2966a1b3bdc37de9b4f983d56937e46619f709a80b6e490675417"
38
+ },
39
+ {
40
+ "path": "LICENSE",
41
+ "sha256": "8dc973f024ff95966bea25866efa443fd16776dcb1001e681e3d467ea572b28d",
42
+ "size": 7831
43
+ },
44
+ {
45
+ "path": "model_index.json",
46
+ "sha256": "cf1ecd104ea090855d60d8cec0895c1e9d6ee41b2f87e231b678beecd1cf7809",
47
+ "size": 447
48
+ },
49
+ {
50
+ "path": "processor/added_tokens.json",
51
+ "sha256": "c0284b582e14987fbd3d5a2cb2bd139084371ed9acbae488829a1c900833c680",
52
+ "size": 707
53
+ },
54
+ {
55
+ "path": "processor/chat_template.jinja",
56
+ "sha256": "3636d0f0bd6bef02654cdffdc447b79cb2cef8ab02cc75267345946291a489e4",
57
+ "size": 5292
58
+ },
59
+ {
60
+ "path": "processor/merges.txt",
61
+ "sha256": "8831e4f1a044471340f7c0a83d7bd71306a5b867e95fd870f74d0c5308a904d5",
62
+ "size": 1671853
63
+ },
64
+ {
65
+ "path": "processor/preprocessor_config.json",
66
+ "sha256": "93585062a80db5e8ca038efc7726a3e6411d9db948472d81d63c6303993be8c5",
67
+ "size": 782
68
+ },
69
+ {
70
+ "path": "processor/special_tokens_map.json",
71
+ "sha256": "76862e765266b85aa9459767e33cbaf13970f327a0e88d1c65846c2ddd3a1ecd",
72
+ "size": 613
73
+ },
74
+ {
75
+ "path": "processor/tokenizer.json",
76
+ "sha256": "aeb13307a71acd8fe81861d94ad54ab689df773318809eed3cbe794b4492dae4",
77
+ "size": 11422654
78
+ },
79
+ {
80
+ "path": "processor/tokenizer_config.json",
81
+ "sha256": "81ec7bb9530159b326c0bef1d0b6c33d392090524014ea3f0123a3c1eb9c2af5",
82
+ "size": 5445
83
+ },
84
+ {
85
+ "path": "processor/video_preprocessor_config.json",
86
+ "sha256": "59c5c9eb52182eb14c06ffb10ca9effd29adce5f238a95de23ca14a38dbd2cb1",
87
+ "size": 817
88
+ },
89
+ {
90
+ "path": "processor/vocab.json",
91
+ "sha256": "ca10d7e9fb3ed18575dd1e277a2579c16d108e32f27439684afa0e10b1440910",
92
+ "size": 2776833
93
+ },
94
+ {
95
+ "path": "scheduler/scheduler_config.json",
96
+ "sha256": "5895f3a167c14a967fe9ac70c64924ae5acc79799e0679fd12907e594a713cd1",
97
+ "size": 485
98
+ },
99
+ {
100
+ "path": "text_encoder/config.json",
101
+ "sha256": "6001331949f7f86ab6d791a80b12246d4f8da5159fa08c3e62368a39f36d63a9",
102
+ "size": 1517
103
+ },
104
+ {
105
+ "path": "text_encoder/generation_config.json",
106
+ "sha256": "4d9818c3d27895c0058828a5f68bc7a4de80c3ae1bcdac90936ce24178063f59",
107
+ "size": 213
108
+ },
109
+ {
110
+ "path": "text_encoder/model.safetensors.index.json",
111
+ "sha256": "859146bff8db1d617fca8b457a5223edff963bb70076b3645f2189fe0540a147",
112
+ "size": 67795
113
+ },
114
+ {
115
+ "path": "transformer/config.json",
116
+ "sha256": "56ae3281c4e6c2d1aa3658252d187488071815fd79bef15808bb0205fc1a2241",
117
+ "size": 370
118
+ },
119
+ {
120
+ "path": "transformer/diffusion_pytorch_model.safetensors.index.json",
121
+ "sha256": "17987f6623b1c814d0ef55a137d99142b7b3b040eb1bf241b5575dd35af803a2",
122
+ "size": 30283
123
+ },
124
+ {
125
+ "path": "vae/config.json",
126
+ "sha256": "9785d527b278cb8b210e9f48a8028d92a4190968ce7e24c8fc85d9d1b82f6ba2",
127
+ "size": 2079
128
+ }
129
+ ],
130
+ "inventory": {
131
+ "transformer": {
132
+ "parameters": 7115124736,
133
+ "quantizable_parameters": 6979321856,
134
+ "tensor_bytes": 14230249472,
135
+ "dtypes": [
136
+ "BF16"
137
+ ]
138
+ },
139
+ "text_encoder": {
140
+ "parameters": 8767123696,
141
+ "quantizable_parameters": 6945767424,
142
+ "tensor_bytes": 17534247392,
143
+ "dtypes": [
144
+ "BF16"
145
+ ]
146
+ },
147
+ "vae": {
148
+ "parameters": 337740404,
149
+ "quantizable_parameters": 0,
150
+ "tensor_bytes": 1350961616,
151
+ "dtypes": [
152
+ "F32"
153
+ ]
154
+ }
155
+ },
156
+ "estimated_weight_bytes": {
157
+ "4": 13098142640.0,
158
+ "6": 16579414960.0,
159
+ "8": 20060687280.0
160
+ },
161
+ "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O",
162
+ "source_model": "Qwen/Qwen-Image-2.1",
163
+ "note": "Public source-file hashes and precision inventory; local source directory omitted."
164
+ }
evaluation/summary.json ADDED
@@ -0,0 +1,101 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "pairs": 7,
3
+ "cases": 7,
4
+ "size": 1024,
5
+ "steps": 40,
6
+ "warmup": true,
7
+ "phase_offload": true,
8
+ "bf16_mean_seconds": 491.3783866848597,
9
+ "q8_mean_seconds": 561.8662204464033,
10
+ "bf16_t2i_mean_seconds": 472.2269847920009,
11
+ "q8_t2i_mean_seconds": 544.6779421319758,
12
+ "bf16_max_peak_gib": 18.68414589576423,
13
+ "q8_max_peak_gib": 16.196932649239898,
14
+ "rows": [
15
+ {
16
+ "case_id": "portrait",
17
+ "seed": 42,
18
+ "bf16_seconds": 440.9093492079992,
19
+ "q8_seconds": 568.857087665936,
20
+ "bf16_peak_gib": 16.41486000828445,
21
+ "q8_peak_gib": 16.19291958771646,
22
+ "bf16_swap_delta_gib": 0.0,
23
+ "q8_swap_delta_gib": 0.0,
24
+ "rgb_psnr_db": 42.02969487919134,
25
+ "alpha_mae": 0.011185646057128906
26
+ },
27
+ {
28
+ "case_id": "english_text",
29
+ "seed": 42,
30
+ "bf16_seconds": 464.31536945910193,
31
+ "q8_seconds": 566.884881041944,
32
+ "bf16_peak_gib": 16.431492125615478,
33
+ "q8_peak_gib": 16.196902131661773,
34
+ "bf16_swap_delta_gib": -0.0078125,
35
+ "q8_swap_delta_gib": 0.0,
36
+ "rgb_psnr_db": 19.89081080658791,
37
+ "alpha_mae": 0.055230140686035156
38
+ },
39
+ {
40
+ "case_id": "chinese_text",
41
+ "seed": 42,
42
+ "bf16_seconds": 474.137312249979,
43
+ "q8_seconds": 567.4320387080079,
44
+ "bf16_peak_gib": 16.43712263740599,
45
+ "q8_peak_gib": 16.196932649239898,
46
+ "bf16_swap_delta_gib": 0.0,
47
+ "q8_swap_delta_gib": -0.0078125,
48
+ "rgb_psnr_db": 36.96905851474672,
49
+ "alpha_mae": 0.007161140441894531
50
+ },
51
+ {
52
+ "case_id": "composition",
53
+ "seed": 42,
54
+ "bf16_seconds": 480.91022845893167,
55
+ "q8_seconds": 562.5225267919013,
56
+ "bf16_peak_gib": 16.43643598817289,
57
+ "q8_peak_gib": 16.196932649239898,
58
+ "bf16_swap_delta_gib": 0.0,
59
+ "q8_swap_delta_gib": 0.0,
60
+ "rgb_psnr_db": 37.16673614061903,
61
+ "alpha_mae": 0.023729324340820312
62
+ },
63
+ {
64
+ "case_id": "texture",
65
+ "seed": 42,
66
+ "bf16_seconds": 487.6394271670142,
67
+ "q8_seconds": 445.06702766695525,
68
+ "bf16_peak_gib": 16.425296997651458,
69
+ "q8_peak_gib": 16.196780061349273,
70
+ "bf16_swap_delta_gib": 0.0,
71
+ "q8_swap_delta_gib": 0.0,
72
+ "rgb_psnr_db": 38.407761046077326,
73
+ "alpha_mae": 0.021371841430664062
74
+ },
75
+ {
76
+ "case_id": "rgba",
77
+ "seed": 42,
78
+ "bf16_seconds": 485.45022220897954,
79
+ "q8_seconds": 557.3040909171104,
80
+ "bf16_peak_gib": 16.422336785122752,
81
+ "q8_peak_gib": 16.19676480256021,
82
+ "bf16_swap_delta_gib": 0.0,
83
+ "q8_swap_delta_gib": 0.0,
84
+ "rgb_psnr_db": 29.813264394212403,
85
+ "alpha_mae": 1.0648202896118164
86
+ },
87
+ {
88
+ "case_id": "edit",
89
+ "seed": 1000042,
90
+ "bf16_seconds": 606.2867980420124,
91
+ "q8_seconds": 664.995890332968,
92
+ "bf16_peak_gib": 18.68414589576423,
93
+ "q8_peak_gib": 16.196322314441204,
94
+ "bf16_swap_delta_gib": 0.0,
95
+ "q8_swap_delta_gib": 0.0,
96
+ "rgb_psnr_db": 51.09023795494214,
97
+ "alpha_mae": 0.006680488586425781
98
+ }
99
+ ],
100
+ "limitations": "One seed per case, one machine, same pinned MLX runtime. Sequential BF16-then-Q8 desktop run without repeated or interleaved trials; order, thermal state and other applications may affect timing. Pixel similarity is not a perceptual quality score; no cross-runtime parity claim."
101
+ }
evaluation/system-context.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "power": "AC",
3
+ "low_power_mode": false,
4
+ "thermal_warning_reported": false,
5
+ "other_model_service": "llama-server present, observed 0.1% CPU, approximately 15.6 GiB physical footprint",
6
+ "caveat": "Desktop-session benchmark; other applications remain open. Not an isolated laboratory run.",
7
+ "idle_sleep": "Temporarily inhibited for benchmark process only",
8
+ "phase_offload": true,
9
+ "swap_before_formal_run_gib": 19.60015869140625,
10
+ "swap_note": "Swap allocated during earlier resident diagnostics; per-image swap deltas are recorded separately.",
11
+ "other_model_services": "Other pre-existing Python and llama-server services left running; earlier physical footprints approximately 25.5 and 15.6 GiB.",
12
+ "mid_run_observation": "At approximately 23:35 PDT both pre-existing model services were still present and around 0.1% CPU; pmset reported no thermal or performance warning. GPU contention and thermal state were not continuously instrumented.",
13
+ "timing_variability": "Q8 texture was substantially faster than its preceding Q8 cases despite unchanged runtime; do not interpret the aggregate latency difference as a controlled causal estimate of quantization speed."
14
+ }
evaluation/visual-review.json ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "approved_for_research_release": true,
3
+ "review_method": "Assistant visual inspection of original PNGs, followed by paired comparison; not a blinded human study.",
4
+ "review_date": "2026-09-26",
5
+ "summary_en": "Assistant inspection of all seven paired cases found no obvious severe quality regression in this small sample. English and Chinese text are correct in both precisions; cup count/order and sweater recoloring succeed; real RGBA transparency is retained. The English poster changes visibly in typeface and lamp structure, and the dragon changes expression/details. Other pairs are closer. Both portraits are cropped more tightly than requested, and both transparency samples contain slight near-zero background alpha. This is an unblinded, single-seed visual check, not evidence of lossless quantization or statistical equivalence.",
6
+ "summary_zh": "逐对检查七类样本,未发现明显严重的量化退化:中英文文字均正确,杯子数量与顺序、毛衣改色及真实透明通道均保留。英文海报的字体和台灯结构有可见变化,龙贴纸的表情与细节也有变化,其余样本更接近。两种精度的人物图都比提示词要求裁得更紧,透明图均有少量接近零的背景 alpha 残留。这是助手进行的非盲法、每类单种子视觉检查,不代表无损或统计等价。",
7
+ "results_sha256": {
8
+ "bf16": "2c9856ece4b83473cf884aa37c7a55be9cec3a52df31920e6a41b6e07418657d",
9
+ "8bit": "742a088a5e0d30a04a5f8b2ce6eb5ae9fa3613167fed0072227eb1bb991437ff"
10
+ },
11
+ "runtime_manifest_sha256": "ee348c543f136b162f3c285f15a66cd59fa51f8d889aad07e6e26e8e02bf39bb",
12
+ "cases": {
13
+ "portrait": {
14
+ "bf16": "Natural elderly face, detailed silver hair/skin and blue knit sweater, coherent window light. Crop is chest-up rather than the requested waist-up. No obvious broken facial features.",
15
+ "q8": "Very close to BF16 in facial features, framing, silver hair, blue knit sweater and window light. Small local skin/hair/knit texture changes are visible, without an obvious severe degradation. Same chest-up rather than waist-up crop limitation.",
16
+ "obvious_severe_quantization_regression": false
17
+ },
18
+ "english_text": {
19
+ "bf16": "Headline CREATE WITH LIGHT and subtitle September 2026 are exact and legible; no extra text. Orange desk lamp, dark blue backdrop and balanced layout match the prompt.",
20
+ "q8": "Exact, legible CREATE WITH LIGHT and September 2026; no extra text. Compared with BF16 the serif typeface/underline and lamp stem/base change visibly, but both remain coherent and satisfy the prompt. This demonstrates non-identical generation, not an obvious readability regression.",
21
+ "obvious_severe_quantization_regression": false
22
+ },
23
+ "chinese_text": {
24
+ "bf16": "Both requested lines 慢下来,喝杯咖啡 and 小店今日营业 are correctly rendered and legible, with the comma present. Cream background and central latte art match the prompt; no extra text.",
25
+ "q8": "Both Chinese lines and comma are correct and legible, matching BF16. Layout, cream backdrop and cup placement remain very close; small latte/foam texture variations are visible. No obvious text accuracy regression in this sample.",
26
+ "obvious_severe_quantization_regression": false
27
+ },
28
+ "composition": {
29
+ "bf16": "Exactly three cups, red left / blue center / yellow right, and one green apple in front of the blue cup. Soft studio shadows and gray tabletop are coherent; no text.",
30
+ "q8": "Exactly three cups in the requested red/blue/yellow order and one green apple in front of the blue cup. Overall layout is very close to BF16. Minor handle/shadow geometry changes are visible; no counting or spatial-order regression.",
31
+ "obvious_severe_quantization_regression": false
32
+ },
33
+ "texture": {
34
+ "bf16": "Coherent kingfisher with detailed blue feathers, orange breast, moist mossy branch and water below. Fine textures and droplets are visible, with a blurred natural background; no obvious structural failure.",
35
+ "q8": "Very close to BF16: coherent bird anatomy, fine blue/orange feather detail, wet moss and droplets. Minor local texture/brightness changes without obvious smoothing or broken detail.",
36
+ "obvious_severe_quantization_regression": false
37
+ },
38
+ "rgba": {
39
+ "bf16": "Full-body green cartoon dragon with small orange wings and a white sticker outline. Real transparency: 60.47% of pixels have alpha exactly zero; canvas edge alpha is at most 4/255. Minor residual near-zero background alpha is present.",
40
+ "q8": "Full-body green dragon with orange wings and white sticker outline. Expression and surface details change visibly from BF16 while silhouette and requested content remain coherent. 60.69% of pixels have alpha exactly zero (BF16 60.47%); both have canvas-edge alpha at most 4/255. No loss of actual transparency; neither output is a perfectly binary matte.",
41
+ "obvious_severe_quantization_regression": false
42
+ },
43
+ "edit": {
44
+ "bf16": "Sweater recolored red. Facial features, silver hair, pose, lighting and window background remain visually consistent with the input. Small local texture variations exist beyond the garment; this is not a pixel-locked masked edit.",
45
+ "q8": "Successful red sweater recoloring with facial features, hair, pose and window background very close to the BF16 edit. Minor local texture changes; no obvious edit-consistency regression. Neither output guarantees pixel-identical preservation outside the garment.",
46
+ "obvious_severe_quantization_regression": false
47
+ }
48
+ },
49
+ "alpha_checks": {
50
+ "bf16": {
51
+ "zero_fraction": 0.6047077178955078,
52
+ "near_zero_fraction": 0.6159038543701172,
53
+ "opaque_fraction": 0.3482656478881836,
54
+ "edge_max": 4
55
+ },
56
+ "8bit": {
57
+ "zero_fraction": 0.6069240570068359,
58
+ "near_zero_fraction": 0.6179990768432617,
59
+ "opaque_fraction": 0.3452177047729492,
60
+ "edge_max": 4
61
+ }
62
+ },
63
+ "limitations": [
64
+ "One seed per case and one machine; no blinded human panel or statistical quality benchmark.",
65
+ "No CUDA/MLX cross-runtime equivalence claim.",
66
+ "Long prompts, 2048px, multiple-reference stress tests and smaller-memory Macs were not evaluated."
67
+ ]
68
+ }
model_index.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "QwenImage21Pipeline",
3
+ "_diffusers_version": "0.37.0.dev0",
4
+ "processor": [
5
+ "transformers",
6
+ "Qwen3VLProcessor"
7
+ ],
8
+ "scheduler": [
9
+ "diffusers",
10
+ "FlowMatchEulerDiscreteScheduler"
11
+ ],
12
+ "text_encoder": [
13
+ "transformers",
14
+ "Qwen3VLForConditionalGeneration"
15
+ ],
16
+ "transformer": [
17
+ "diffusers",
18
+ "QwenImage21Transformer2DModel"
19
+ ],
20
+ "vae": [
21
+ "diffusers",
22
+ "AutoencoderKLQwenImage21"
23
+ ]
24
+ }
processor/added_tokens.json ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</think>": 151668,
3
+ "</tool_call>": 151658,
4
+ "</tool_response>": 151666,
5
+ "<think>": 151667,
6
+ "<tool_call>": 151657,
7
+ "<tool_response>": 151665,
8
+ "<|box_end|>": 151649,
9
+ "<|box_start|>": 151648,
10
+ "<|endoftext|>": 151643,
11
+ "<|file_sep|>": 151664,
12
+ "<|fim_middle|>": 151660,
13
+ "<|fim_pad|>": 151662,
14
+ "<|fim_prefix|>": 151659,
15
+ "<|fim_suffix|>": 151661,
16
+ "<|im_end|>": 151645,
17
+ "<|im_start|>": 151644,
18
+ "<|image_pad|>": 151655,
19
+ "<|object_ref_end|>": 151647,
20
+ "<|object_ref_start|>": 151646,
21
+ "<|quad_end|>": 151651,
22
+ "<|quad_start|>": 151650,
23
+ "<|repo_name|>": 151663,
24
+ "<|video_pad|>": 151656,
25
+ "<|vision_end|>": 151653,
26
+ "<|vision_pad|>": 151654,
27
+ "<|vision_start|>": 151652
28
+ }
processor/chat_template.jinja ADDED
@@ -0,0 +1,120 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0].role == 'system' %}
4
+ {%- if messages[0].content is string %}
5
+ {{- messages[0].content }}
6
+ {%- else %}
7
+ {%- for content in messages[0].content %}
8
+ {%- if 'text' in content %}
9
+ {{- content.text }}
10
+ {%- endif %}
11
+ {%- endfor %}
12
+ {%- endif %}
13
+ {{- '\n\n' }}
14
+ {%- endif %}
15
+ {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
16
+ {%- for tool in tools %}
17
+ {{- "\n" }}
18
+ {{- tool | tojson }}
19
+ {%- endfor %}
20
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
21
+ {%- else %}
22
+ {%- if messages[0].role == 'system' %}
23
+ {{- '<|im_start|>system\n' }}
24
+ {%- if messages[0].content is string %}
25
+ {{- messages[0].content }}
26
+ {%- else %}
27
+ {%- for content in messages[0].content %}
28
+ {%- if 'text' in content %}
29
+ {{- content.text }}
30
+ {%- endif %}
31
+ {%- endfor %}
32
+ {%- endif %}
33
+ {{- '<|im_end|>\n' }}
34
+ {%- endif %}
35
+ {%- endif %}
36
+ {%- set image_count = namespace(value=0) %}
37
+ {%- set video_count = namespace(value=0) %}
38
+ {%- for message in messages %}
39
+ {%- if message.role == "user" %}
40
+ {{- '<|im_start|>' + message.role + '\n' }}
41
+ {%- if message.content is string %}
42
+ {{- message.content }}
43
+ {%- else %}
44
+ {%- for content in message.content %}
45
+ {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
46
+ {%- set image_count.value = image_count.value + 1 %}
47
+ {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
48
+ <|vision_start|><|image_pad|><|vision_end|>
49
+ {%- elif content.type == 'video' or 'video' in content %}
50
+ {%- set video_count.value = video_count.value + 1 %}
51
+ {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
52
+ <|vision_start|><|video_pad|><|vision_end|>
53
+ {%- elif 'text' in content %}
54
+ {{- content.text }}
55
+ {%- endif %}
56
+ {%- endfor %}
57
+ {%- endif %}
58
+ {{- '<|im_end|>\n' }}
59
+ {%- elif message.role == "assistant" %}
60
+ {{- '<|im_start|>' + message.role + '\n' }}
61
+ {%- if message.content is string %}
62
+ {{- message.content }}
63
+ {%- else %}
64
+ {%- for content_item in message.content %}
65
+ {%- if 'text' in content_item %}
66
+ {{- content_item.text }}
67
+ {%- endif %}
68
+ {%- endfor %}
69
+ {%- endif %}
70
+ {%- if message.tool_calls %}
71
+ {%- for tool_call in message.tool_calls %}
72
+ {%- if (loop.first and message.content) or (not loop.first) %}
73
+ {{- '\n' }}
74
+ {%- endif %}
75
+ {%- if tool_call.function %}
76
+ {%- set tool_call = tool_call.function %}
77
+ {%- endif %}
78
+ {{- '<tool_call>\n{"name": "' }}
79
+ {{- tool_call.name }}
80
+ {{- '", "arguments": ' }}
81
+ {%- if tool_call.arguments is string %}
82
+ {{- tool_call.arguments }}
83
+ {%- else %}
84
+ {{- tool_call.arguments | tojson }}
85
+ {%- endif %}
86
+ {{- '}\n</tool_call>' }}
87
+ {%- endfor %}
88
+ {%- endif %}
89
+ {{- '<|im_end|>\n' }}
90
+ {%- elif message.role == "tool" %}
91
+ {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
92
+ {{- '<|im_start|>user' }}
93
+ {%- endif %}
94
+ {{- '\n<tool_response>\n' }}
95
+ {%- if message.content is string %}
96
+ {{- message.content }}
97
+ {%- else %}
98
+ {%- for content in message.content %}
99
+ {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
100
+ {%- set image_count.value = image_count.value + 1 %}
101
+ {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
102
+ <|vision_start|><|image_pad|><|vision_end|>
103
+ {%- elif content.type == 'video' or 'video' in content %}
104
+ {%- set video_count.value = video_count.value + 1 %}
105
+ {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
106
+ <|vision_start|><|video_pad|><|vision_end|>
107
+ {%- elif 'text' in content %}
108
+ {{- content.text }}
109
+ {%- endif %}
110
+ {%- endfor %}
111
+ {%- endif %}
112
+ {{- '\n</tool_response>' }}
113
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
114
+ {{- '<|im_end|>\n' }}
115
+ {%- endif %}
116
+ {%- endif %}
117
+ {%- endfor %}
118
+ {%- if add_generation_prompt %}
119
+ {{- '<|im_start|>assistant\n' }}
120
+ {%- endif %}
processor/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
processor/preprocessor_config.json ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "crop_size": null,
3
+ "data_format": "channels_first",
4
+ "default_to_square": true,
5
+ "device": null,
6
+ "disable_grouping": null,
7
+ "do_center_crop": null,
8
+ "do_convert_rgb": true,
9
+ "do_normalize": true,
10
+ "do_pad": null,
11
+ "do_rescale": true,
12
+ "do_resize": true,
13
+ "image_mean": [
14
+ 0.5,
15
+ 0.5,
16
+ 0.5
17
+ ],
18
+ "image_processor_type": "Qwen2VLImageProcessorFast",
19
+ "image_std": [
20
+ 0.5,
21
+ 0.5,
22
+ 0.5
23
+ ],
24
+ "input_data_format": null,
25
+ "max_pixels": null,
26
+ "merge_size": 2,
27
+ "min_pixels": null,
28
+ "pad_size": null,
29
+ "patch_size": 16,
30
+ "processor_class": "Qwen3VLProcessor",
31
+ "resample": 3,
32
+ "rescale_factor": 0.00392156862745098,
33
+ "return_tensors": null,
34
+ "size": {
35
+ "longest_edge": 16777216,
36
+ "shortest_edge": 65536
37
+ },
38
+ "temporal_patch_size": 2
39
+ }
processor/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|im_end|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
processor/tokenizer_config.json ADDED
@@ -0,0 +1,240 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ },
181
+ "151665": {
182
+ "content": "<tool_response>",
183
+ "lstrip": false,
184
+ "normalized": false,
185
+ "rstrip": false,
186
+ "single_word": false,
187
+ "special": false
188
+ },
189
+ "151666": {
190
+ "content": "</tool_response>",
191
+ "lstrip": false,
192
+ "normalized": false,
193
+ "rstrip": false,
194
+ "single_word": false,
195
+ "special": false
196
+ },
197
+ "151667": {
198
+ "content": "<think>",
199
+ "lstrip": false,
200
+ "normalized": false,
201
+ "rstrip": false,
202
+ "single_word": false,
203
+ "special": false
204
+ },
205
+ "151668": {
206
+ "content": "</think>",
207
+ "lstrip": false,
208
+ "normalized": false,
209
+ "rstrip": false,
210
+ "single_word": false,
211
+ "special": false
212
+ }
213
+ },
214
+ "additional_special_tokens": [
215
+ "<|im_start|>",
216
+ "<|im_end|>",
217
+ "<|object_ref_start|>",
218
+ "<|object_ref_end|>",
219
+ "<|box_start|>",
220
+ "<|box_end|>",
221
+ "<|quad_start|>",
222
+ "<|quad_end|>",
223
+ "<|vision_start|>",
224
+ "<|vision_end|>",
225
+ "<|vision_pad|>",
226
+ "<|image_pad|>",
227
+ "<|video_pad|>"
228
+ ],
229
+ "bos_token": null,
230
+ "clean_up_tokenization_spaces": false,
231
+ "eos_token": "<|im_end|>",
232
+ "errors": "replace",
233
+ "extra_special_tokens": {},
234
+ "model_max_length": 262144,
235
+ "pad_token": "<|endoftext|>",
236
+ "processor_class": "Qwen3VLProcessor",
237
+ "split_special_tokens": false,
238
+ "tokenizer_class": "Qwen2Tokenizer",
239
+ "unk_token": null
240
+ }
processor/video_preprocessor_config.json ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "crop_size": null,
3
+ "data_format": "channels_first",
4
+ "default_to_square": true,
5
+ "device": null,
6
+ "do_center_crop": null,
7
+ "do_convert_rgb": true,
8
+ "do_normalize": true,
9
+ "do_rescale": true,
10
+ "do_resize": true,
11
+ "do_sample_frames": true,
12
+ "fps": 2,
13
+ "image_mean": [
14
+ 0.5,
15
+ 0.5,
16
+ 0.5
17
+ ],
18
+ "image_std": [
19
+ 0.5,
20
+ 0.5,
21
+ 0.5
22
+ ],
23
+ "input_data_format": null,
24
+ "max_frames": 768,
25
+ "merge_size": 2,
26
+ "min_frames": 4,
27
+ "num_frames": null,
28
+ "pad_size": null,
29
+ "patch_size": 16,
30
+ "processor_class": "Qwen3VLProcessor",
31
+ "resample": 3,
32
+ "rescale_factor": 0.00392156862745098,
33
+ "return_metadata": false,
34
+ "size": {
35
+ "longest_edge": 25165824,
36
+ "shortest_edge": 4096
37
+ },
38
+ "temporal_patch_size": 2,
39
+ "video_metadata": null,
40
+ "video_processor_type": "Qwen3VLVideoProcessor"
41
+ }
processor/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
requirements.lock.txt ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ annotated-doc==0.0.5
2
+ annotated-types==0.8.0
3
+ anyio==4.15.1
4
+ certifi==2026.7.22
5
+ cffi==2.1.1
6
+ charset-normalizer==3.5.1
7
+ click==8.5.0
8
+ cryptography==50.0.1
9
+ fastapi==0.141.1
10
+ filelock==4.0.3
11
+ fsspec==2026.9.0
12
+ h11==0.16.0
13
+ hf-xet==1.6.0
14
+ httpcore==1.0.9
15
+ httpx==0.28.1
16
+ huggingface-hub==1.33.0
17
+ idna==3.20
18
+ jinja2==3.1.6
19
+ llguidance==1.8.0
20
+ markdown-it-py==4.2.0
21
+ markupsafe==3.0.3
22
+ mdurl==0.1.2
23
+ miniaudio==1.71
24
+ mlx==0.32.2
25
+ mlx-audio==0.5.6
26
+ mlx-metal==0.32.2
27
+ mlx-vlm @ https://github.com/Blaizzy/mlx-vlm/archive/95b01ccad2d9f65a9e87f6a87bd1c5df69626261.zip
28
+ modelscope==1.40.1
29
+ modelscope-hub==0.4.5
30
+ numpy==2.5.3
31
+ opencv-python==5.0.0.93
32
+ packaging==26.3
33
+ pillow==12.3.0
34
+ psutil==7.2.2
35
+ pycparser==3.0
36
+ pydantic==2.13.5
37
+ pydantic-core==2.46.5
38
+ pygments==2.21.0
39
+ python-multipart==0.0.32
40
+ pyyaml==6.0.3
41
+ regex==2026.9.10
42
+ requests==2.34.2
43
+ rich==15.0.0
44
+ safetensors==0.8.0
45
+ scipy==1.18.1
46
+ sentencepiece==0.2.2
47
+ setuptools==84.0.0
48
+ shellingham==1.5.4
49
+ sounddevice==0.5.6
50
+ starlette==1.7.0
51
+ tokenizers==0.23.2
52
+ tqdm==4.70.1
53
+ transformers==5.17.0
54
+ typer==0.27.2
55
+ typing-extensions==4.16.0
56
+ typing-inspection==0.4.4
57
+ urllib3==2.8.0
58
+ uvicorn==0.54.0
59
+ websockets==17.1
requirements.txt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ mlx-vlm @ https://github.com/Blaizzy/mlx-vlm/archive/95b01ccad2d9f65a9e87f6a87bd1c5df69626261.zip
2
+ psutil>=7
3
+ modelscope>=1.30
scheduler/scheduler_config.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "FlowMatchEulerDiscreteScheduler",
3
+ "_diffusers_version": "0.37.0.dev0",
4
+ "base_image_seq_len": 256,
5
+ "base_shift": 0.5,
6
+ "invert_sigmas": false,
7
+ "max_image_seq_len": 8192,
8
+ "max_shift": 0.9,
9
+ "num_train_timesteps": 1000,
10
+ "shift": 1.0,
11
+ "shift_terminal": 0.02,
12
+ "stochastic_sampling": false,
13
+ "time_shift_type": "exponential",
14
+ "use_beta_sigmas": false,
15
+ "use_dynamic_shifting": true,
16
+ "use_exponential_sigmas": false,
17
+ "use_karras_sigmas": false
18
+ }
scripts/MLX_VLM_LICENSE.txt ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ MIT License
2
+
3
+ Copyright © 2025 Prince Canuma
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
scripts/__init__.py ADDED
File without changes
scripts/audit.py ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Verify the complete BF16 source against the saved official revision."""
2
+ import argparse
3
+ import json
4
+ import platform
5
+ import subprocess
6
+ from pathlib import Path
7
+ from scripts.common import SOURCE, REVISION, inventory, sha256, write_json
8
+
9
+ def main():
10
+ ap = argparse.ArgumentParser()
11
+ ap.add_argument('--source', type=Path, default=SOURCE)
12
+ ap.add_argument('--output', type=Path, default=Path('artifacts/source-audit.json'))
13
+ ap.add_argument('--manifest', type=Path, help='Published evaluation/source-files.json for a portable rebuild')
14
+ args = ap.parse_args()
15
+ if args.manifest:
16
+ info=json.loads(args.manifest.read_text())
17
+ if info['revision']!=REVISION: raise ValueError('Unexpected source revision')
18
+ rows=info['files']
19
+ else:
20
+ info = json.loads(Path('artifacts/upstream/source-model-info.json').read_text())
21
+ if info['sha'] != REVISION:
22
+ raise ValueError('Unexpected source revision')
23
+ rows = [{'path': x['rfilename'], 'size': x['size'], 'sha256': x['lfs']['sha256']}
24
+ for x in info['siblings'] if x['rfilename'].endswith('.safetensors')]
25
+ rows += [r for r in json.loads(Path('artifacts/upstream/source-config-manifest.json').read_text())
26
+ if r['path'] not in ('.gitattributes', 'README.md')]
27
+ for row in rows:
28
+ file = args.source / row['path']
29
+ if file.stat().st_size != row['size'] or sha256(file) != row['sha256']:
30
+ raise ValueError(f'Source mismatch: {file}')
31
+ print('Verified', row['path'], flush=True)
32
+ inv = inventory(args.source)
33
+ estimates = {}
34
+ for bits in (4, 6, 8):
35
+ estimates[str(bits)] = sum(v['tensor_bytes'] - v['quantizable_parameters'] * 2 +
36
+ v['quantizable_parameters'] * (bits / 8 + 4 / 64)
37
+ for v in inv.values())
38
+ write_json(args.output, dict(source=str(args.source), revision=REVISION, files=rows,
39
+ inventory=inv, estimated_weight_bytes=estimates,
40
+ platform=platform.platform()))
41
+ print(json.dumps(dict(inventory=inv, estimated_weight_GiB={k:v/2**30 for k,v in estimates.items()}),indent=2))
42
+
43
+ if __name__ == '__main__':
44
+ main()
scripts/benchmark.py ADDED
@@ -0,0 +1,85 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Paired 1024px evaluation, separate processes per precision, unchanged RGBA samples."""
2
+ import argparse
3
+ import gc
4
+ import importlib.metadata
5
+ import json
6
+ import platform
7
+ import time
8
+ from pathlib import Path
9
+ import mlx.core as mx
10
+ import numpy as np
11
+ import psutil
12
+ from scripts.common import sha256, write_json
13
+ from scripts.runtime import load, generate, enable_progress
14
+
15
+ def main():
16
+ ap=argparse.ArgumentParser()
17
+ ap.add_argument('--model',type=Path,required=True)
18
+ ap.add_argument('--output',type=Path,required=True)
19
+ ap.add_argument('--cases',type=Path,default=Path('benchmarks/cases.json'))
20
+ ap.add_argument('--only',nargs='+')
21
+ ap.add_argument('--seeds',nargs='+',type=int,default=[42])
22
+ ap.add_argument('--steps',type=int,default=40)
23
+ ap.add_argument('--size',type=int,default=1024)
24
+ ap.add_argument('--edit-input',type=Path)
25
+ ap.add_argument('--no-warmup',action='store_true')
26
+ ap.add_argument('--resident',action='store_true',help='Diagnostic mode: keep all components resident')
27
+ args=ap.parse_args()
28
+ if args.output.exists(): raise FileExistsError(args.output)
29
+ cases=json.loads(args.cases.read_text())
30
+ if args.only: cases=[x for x in cases if x['id'] in args.only]
31
+ if not cases: raise ValueError('No cases selected')
32
+ args.output.mkdir(parents=True)
33
+ env=dict(model=str(args.model.resolve()),device=mx.device_info(),platform=platform.platform(),
34
+ total_memory_gib=psutil.virtual_memory().total/2**30,
35
+ packages={p:importlib.metadata.version(p) for p in ('mlx','mlx-vlm','numpy','transformers')},
36
+ cases_sha256=sha256(args.cases),warmup=not args.no_warmup,
37
+ phase_offload=not args.resident,
38
+ note='MLX allocated peak is not whole-system memory or a proven minimum RAM requirement.')
39
+ if (args.model/'conversion.json').exists(): env['conversion']=json.loads((args.model/'conversion.json').read_text())
40
+ write_json(args.output/'environment.json',env)
41
+ enable_progress()
42
+ start=time.perf_counter(); pipe=load(args.model,phase_offload=not args.resident)
43
+ if args.resident:
44
+ mx.eval(pipe.transformer.parameters(),pipe.text_encoder.model.parameters(),pipe.vae.parameters())
45
+ load_seconds=time.perf_counter()-start
46
+ print(f'Pipeline ready in {load_seconds:.2f}s; phase loads are included in image timings',flush=True)
47
+ if not args.no_warmup:
48
+ print('Full untimed warm-up',flush=True)
49
+ generate(pipe,cases[0]['prompt'],seed=20260926,steps=args.steps,width=args.size,height=args.size)
50
+ mx.synchronize(); gc.collect(); mx.clear_cache()
51
+ for case in cases:
52
+ for generation_seed in args.seeds:
53
+ is_edit=case['id']=='edit'
54
+ seed=1_000_000+generation_seed if is_edit else generation_seed
55
+ inputs=None
56
+ if is_edit:
57
+ ref=args.edit_input or args.output/f'portrait-s{generation_seed}.png'
58
+ if not ref.exists(): raise FileNotFoundError(f'Provide the shared BF16 portrait: {ref}')
59
+ inputs=[str(ref)]
60
+ gc.collect(); mx.clear_cache(); mx.reset_peak_memory()
61
+ swap_before=psutil.swap_memory().used/2**30
62
+ start=time.perf_counter()
63
+ print(f'Running {case["id"]} seed={seed}',flush=True)
64
+ img=generate(pipe,case['prompt'],seed=seed,steps=args.steps,width=args.size,height=args.size,
65
+ inputs=inputs,source_seed=generation_seed if is_edit else None,resolution=args.size)
66
+ mx.synchronize(); seconds=time.perf_counter()-start
67
+ name=f'{case["id"]}-s{seed}.png'; img.save(args.output/name)
68
+ pixels=np.asarray(img); alpha=pixels[:,:,3]
69
+ row=dict(case_id=case['id'],prompt=case['prompt'],seed=seed,width=args.size,height=args.size,
70
+ steps=args.steps,cfg=1.0,vae_tiling=False,kv_cache=True,phase_offload=not args.resident,
71
+ input_sha256=sha256(inputs[0]) if inputs else None,pipeline_init_seconds=load_seconds,
72
+ seconds=seconds,mlx_peak_gib=mx.get_peak_memory()/2**30,
73
+ rss_gib=psutil.Process().memory_info().rss/2**30,
74
+ system_available_gib=psutil.virtual_memory().available/2**30,
75
+ swap_used_gib=psutil.swap_memory().used/2**30,
76
+ swap_before_gib=swap_before,swap_delta_gib=psutil.swap_memory().used/2**30-swap_before,
77
+ output=name,output_sha256=sha256(args.output/name),mode=img.mode,
78
+ alpha_min=int(alpha.min()),alpha_max=int(alpha.max()),
79
+ alpha_fraction_below_250=float(np.mean(alpha<250)),
80
+ rgb_std=float(pixels[:,:,:3].astype(np.float32).std()),warmup=not args.no_warmup)
81
+ with (args.output/'results.jsonl').open('a') as f: f.write(json.dumps(row,ensure_ascii=False)+'\n')
82
+ print(json.dumps(row,ensure_ascii=False),flush=True)
83
+ write_json(args.output/'COMPLETE.json',dict(cases=len(cases),seeds=args.seeds,steps=args.steps,size=args.size))
84
+
85
+ if __name__=='__main__': main()
scripts/cards.py ADDED
@@ -0,0 +1,166 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Render bilingual model cards from actual evaluation records."""
2
+ import json
3
+ from pathlib import Path
4
+
5
+ def main():
6
+ summary=json.loads(Path('artifacts/eval/summary.json').read_text())
7
+ review=json.loads(Path('artifacts/eval/visual-review.json').read_text())
8
+ if review.get('approved_for_research_release') is not True:
9
+ raise ValueError('Complete the paired visual review before rendering release cards')
10
+ conversion=json.loads(Path('models/Image21-MLX-8bit/conversion.json').read_text())
11
+ gib=sum(x['tensor_bytes'] for x in conversion['components'].values())/2**30
12
+ header='''---
13
+ license: other
14
+ license_name: qwen-research
15
+ license_link: LICENSE
16
+ base_model: Qwen/Qwen-Image-2.1
17
+ library_name: mlx
18
+ pipeline_tag: text-to-image
19
+ tags:
20
+ - mlx
21
+ - mlx-vlm
22
+ - apple-silicon
23
+ - quantized
24
+ - image-to-image
25
+ - rgba
26
+ ---
27
+ '''
28
+ common=f'''
29
+ ## Reproducible inference
30
+
31
+ ```bash
32
+ uv venv --python 3.13 .venv
33
+ uv pip install --python .venv/bin/python -r requirements.lock.txt
34
+ .venv/bin/python -m scripts.infer --model . \\
35
+ --prompt 'A natural portrait in soft window light' --output outputs/portrait.png
36
+ ```
37
+
38
+ ```bash
39
+ .venv/bin/python -m scripts.infer --model . --input input.png \\
40
+ --prompt 'Change only the blue sweater to a red sweater. Preserve the person and background.' \\
41
+ --source-seed 42 --seed 1000042 --output outputs/edit.png
42
+ ```
43
+
44
+ Transparent generation prompt:
45
+ `This is an RGBA image with transparency. A cute cartoon dragon sticker. The image has alpha channel and the background is transparent.`
46
+
47
+ ## Measurements / 实测
48
+
49
+ Apple M4 Max, 40 GPU cores, 128 GiB unified memory, macOS 26.6.2, MLX 0.32.2.
50
+ Both models use the same pinned MLX runtime and internal SSD; 1024×1024, 40 steps,
51
+ CFG=1, no VAE tiling, one full warm-up per process, seven cases, one seed per case.
52
+ Sequential component loading is enabled in both models. Per-image time includes
53
+ component loading, prompt encoding, denoising and VAE decoding; PNG writing is excluded.
54
+ This is a desktop session with other applications open. BF16 ran before Q8;
55
+ there were no repeated or interleaved trials to control order and thermal effects.
56
+
57
+ | Case | BF16 seconds | 8-bit seconds | BF16 peak GiB | 8-bit peak GiB |
58
+ |---|---:|---:|---:|---:|
59
+ '''
60
+ for r in summary['rows']:
61
+ common+=f'| {r["case_id"]} | {r["bf16_seconds"]:.2f} | {r["q8_seconds"]:.2f} | {r["bf16_peak_gib"]:.2f} | {r["q8_peak_gib"]:.2f} |\n'
62
+ relative=100*(summary['q8_t2i_mean_seconds']/summary['bf16_t2i_mean_seconds']-1)
63
+ common+=f'\nSix-case text-to-image mean / 六类文生图平均:BF16 **{summary["bf16_t2i_mean_seconds"]/60:.2f} min**, Q8 **{summary["q8_t2i_mean_seconds"]/60:.2f} min**. Q8 generation time relative to BF16 / Q8 相对耗时:**{relative:+.1f}%** in this run.\n'
64
+ common+='''
65
+ Peak figures measure MLX allocations, not minimum physical RAM. Full raw records,
66
+ original RGBA samples and the visual review are in [evaluation](evaluation/report.md).
67
+ These are informal measurements, not an official benchmark. The BF16 baseline is
68
+ the same MLX implementation; cross-runtime CUDA equivalence is not claimed.
69
+ One seed per case is insufficient to establish statistical quality equivalence.
70
+ There is no claim of lossless quantization or a guaranteed speedup.
71
+
72
+ ![All seven paired samples](evaluation/comparison.png)
73
+
74
+ ## Provenance
75
+
76
+ - Source: `Qwen/Qwen-Image-2.1@b3179ad355be050328e483a9dfdd9e60cd62adfa`.
77
+ - Runtime: `Blaizzy/mlx-vlm@95b01ccad2d9f65a9e87f6a87bd1c5df69626261`.
78
+ - Native MLX affine packed weights; not CUDA bitsandbytes INT8, FP8, GGUF or an MFLUX checkpoint.
79
+ - All three components passed exact tensor round-trip verification.
80
+ - [conversion.json](conversion.json), [modifications](CHANGES.md), [Notice](Notice), [license](LICENSE), [file hashes](MANIFEST.json).
81
+
82
+ To rebuild, first download the original source snapshot at the revision above to
83
+ `source-bf16` (requires additional disk space), then run from this repository:
84
+
85
+ ```bash
86
+ .venv/bin/python -m scripts.audit --source source-bf16 --manifest evaluation/source-files.json
87
+ .venv/bin/python -m scripts.convert --source source-bf16 --bits 8 --output rebuilt-8bit
88
+ ```
89
+ The audit verifies the original source file hashes before conversion. Rebuilding
90
+ uses the original floating-point weights, never a dequantized CUDA INT8 checkpoint.
91
+
92
+ To repeat the paired evaluation after the source audit, create a native baseline
93
+ and run each precision in its own process. This is a lengthy research workflow
94
+ and needs extra memory and disk space beyond ordinary inference:
95
+
96
+ ```bash
97
+ .venv/bin/python -m scripts.convert --source source-bf16 --bits 16 --output rebuilt-bf16
98
+ .venv/bin/python -m scripts.benchmark --model rebuilt-bf16 --output artifacts/eval/bf16
99
+ .venv/bin/python -m scripts.benchmark --model . --output artifacts/eval/8bit --edit-input artifacts/eval/bf16/portrait-s42.png
100
+ .venv/bin/python -m scripts.report
101
+ ```
102
+ '''
103
+ en=f'''# Image21-MLX-8bit
104
+
105
+ **Built with Qwen.** An independent native MLX quantization of Qwen-Image-2.1 for
106
+ Apple Silicon, by ixim / iximbox. **Non-commercial research and evaluation only**
107
+ under the original Qwen Research License. This is not an official Qwen release.
108
+
109
+ The complete checkpoint is approximately **{gib:.2f} GiB**. DiT attention/MLP and
110
+ language-encoder attention/MLP linears use **8-bit affine weights, group size 64**,
111
+ with BF16 activations. The full vision tower, token embeddings, language head,
112
+ norms and DiT input/output/timestep/modulation layers retain floating-point precision.
113
+ The VAE retains its original **FP32** weights. See the exact 476 quantized modules in conversion.json.
114
+
115
+ Supports text-to-image, native reference-image editing and RGBA transparency through
116
+ the included scripts. The wrapper preserves the sampler's alpha channel, which the
117
+ pinned upstream text-to-image convenience method otherwise slices away.
118
+ The runtime also enables the upstream fixed-prefix KV cache for text-to-image.
119
+ Components are loaded and released by phase to reduce unified-memory use.
120
+ See scripts/mlx_pipeline.py and its retained MIT attribution.
121
+
122
+ With the included phase-loading runtime, plan for **32GB unified memory as a starting
123
+ budget; 48GB or more gives more room for other applications** at 1024px with one reference.
124
+ These are capacity estimates, not verified minimums: only the 128GB M4 Max was tested.
125
+ The measured Q8 MLX allocation peak is about **16.20 GiB**, excluding OS/driver overhead.
126
+ 16/24GB Macs are not the target for this default 1024px configuration. 2048px and
127
+ 10-reference editing have not been benchmarked. Allow roughly **50GB free disk space**
128
+ for ordinary inference; rebuilding also needs the original source and baseline checkpoints.
129
+ '''
130
+ zh=f'''# Image21-MLX-8bit
131
+
132
+ **Built with Qwen。** ixim / iximbox 基于 Qwen-Image-2.1 原始 BF16 权重制作的
133
+ Apple Silicon 原生 MLX 量化版。**仅限非商业研究与评估**,沿用 Qwen Research License;
134
+ 本项目为独立衍生模型,不是官方发布。
135
+
136
+ 完整权重约 **{gib:.2f} GiB**。DiT 和文本编码器中的注意力/MLP 线性层采用
137
+ **8-bit affine、group size 64、BF16 激活**。整个视觉编码器、词嵌入、语言输出头、
138
+ 归一化及 DiT 输入/输出、时间嵌入和 modulation 保留浮点精度;VAE 保留原始 **FP32**。
139
+ 全部 476 个量化模块详见 conversion.json。本模型从原始权重直接转换,没有从 CUDA INT8 二次量化。
140
+
141
+ 通过配套脚本支持文生图、参考图指令编辑和 RGBA 透明图;适配器保留原生 sampler 的 alpha,
142
+ 避免固定版本上游文生图便捷方法丢弃透明通道。
143
+ 运行时还将上游固定前缀 KV 缓存启用于文生图;完整改动与 MIT 归属保留在 scripts 中。
144
+ 配套脚本默认按阶段加载与释放组件,组件加载开销计入下表时间。
145
+
146
+ 使用配套分阶段加载运行时,1024px、单参考图建议按 **32GB 起、48GB 以上更有余量**规划。
147
+ 这是容量估算,未验证最低内存边界;本次只实测了 128GB M4 Max。量化版 MLX 分配峰值约
148
+ **16.20 GiB**,还需为操作系统、驱动和其他软件留空间。16/24GB 不是默认 1024px 配置的目标。
149
+ 2048px 与十张参考图的编辑尚未测评。普通推理建议预留约 **50GB 可用磁盘**;
150
+ 重建量化与 BF16 基线还需额外存放原始及基线权重。
151
+
152
+ 下表为非正式、同后端、单种子对照,不能证明无损,也不能与 CUDA 时间直接横向比较。
153
+ MLX 分配峰值不等于整机内存需求。参考图编辑共用同一张 BF16 原图,编辑种子与原图生成种子不同。
154
+ '''
155
+ Path('cards').mkdir(exist_ok=True)
156
+ for platform,body in [('huggingface',en),('modelscope',zh)]:
157
+ command=(
158
+ 'uvx --from huggingface-hub==1.33.0 hf download ixim/Image21-MLX-8bit --local-dir Image21-MLX-8bit'
159
+ if platform=='huggingface' else
160
+ 'uvx --from modelscope-hub==0.4.5 modelscope download iximbox/Image21-MLX-8bit --local-dir Image21-MLX-8bit'
161
+ )
162
+ download=f'\n## Download / 下载\n\nOn an Apple Silicon Mac with [uv](https://docs.astral.sh/uv/) installed:\n\n```bash\n{command}\ncd Image21-MLX-8bit\n```\n'
163
+ quality='\n## Visual findings / 视觉检查\n\n'+review['summary_en' if platform=='huggingface' else 'summary_zh']+'\n'
164
+ Path(f'cards/{platform}.md').write_text(header+body+quality+download+common)
165
+
166
+ if __name__=='__main__': main()
scripts/common.py ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Local, reproducible model identities and safetensors inspection."""
2
+ import hashlib
3
+ import json
4
+ import math
5
+ import struct
6
+ from pathlib import Path
7
+
8
+ SOURCE = Path('/Volumes/ZX6 1TB/Qwen-Image-2.1/models/bf16')
9
+ REVISION = 'b3179ad355be050328e483a9dfdd9e60cd62adfa'
10
+ RUNTIME_REVISION = '95b01ccad2d9f65a9e87f6a87bd1c5df69626261'
11
+ NOTICE = ('Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 '
12
+ 'Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved.')
13
+ MODIFICATION = ('Modified by ixim / iximbox for Image21-MLX: converted from the pinned '
14
+ 'BF16 source to MLX layout; eligible linear weights use groupwise affine '
15
+ 'quantization. See conversion.json for precision and exceptions. Built with Qwen.')
16
+
17
+ def write_json(path, value):
18
+ path = Path(path)
19
+ path.parent.mkdir(parents=True, exist_ok=True)
20
+ path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + '\n')
21
+
22
+ def sha256(path):
23
+ with Path(path).open('rb') as f:
24
+ return hashlib.file_digest(f, 'sha256').hexdigest()
25
+
26
+ def header(path):
27
+ with Path(path).open('rb') as f:
28
+ length = struct.unpack('<Q', f.read(8))[0]
29
+ if length > 100_000_000:
30
+ raise ValueError('Invalid safetensors header')
31
+ return json.loads(f.read(length))
32
+
33
+ def tensors(root):
34
+ for path in sorted(Path(root).glob('*.safetensors')):
35
+ if path.name.startswith('._'):
36
+ continue
37
+ for name, info in header(path).items():
38
+ if name != '__metadata__':
39
+ yield path, name, info
40
+
41
+ def eligible(component, name, shape):
42
+ # Preserve all visual encoder, embeddings, norms, modulation and boundary layers.
43
+ if len(shape) != 2 or shape[-1] % 64 or not name.endswith('.weight'):
44
+ return False
45
+ if component == 'transformer':
46
+ return name.startswith('transformer_blocks.') and ('.attn.to_' in name or '.img_mlp.' in name)
47
+ if component == 'text_encoder':
48
+ return 'language_model' in name and '.layers.' in name and ('.self_attn.' in name or '.mlp.' in name)
49
+ return False
50
+
51
+ def inventory(root):
52
+ result = {}
53
+ for component in ('transformer', 'text_encoder', 'vae'):
54
+ rows = list(tensors(Path(root) / component))
55
+ params = sum(math.prod(v['shape']) for _, _, v in rows)
56
+ qparams = sum(math.prod(v['shape']) for _, n, v in rows if eligible(component, n, v['shape']))
57
+ nbytes = sum(v['data_offsets'][1] - v['data_offsets'][0] for _, _, v in rows)
58
+ result[component] = dict(parameters=params, quantizable_parameters=qparams, tensor_bytes=nbytes,
59
+ dtypes=sorted({v['dtype'] for _, _, v in rows}))
60
+ return result
scripts/convert.py ADDED
@@ -0,0 +1,106 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Convert verified BF16 components to a native MLX checkpoint, with round-trip checks."""
2
+ import argparse
3
+ import gc
4
+ import json
5
+ import shutil
6
+ import time
7
+ from pathlib import Path
8
+ import mlx.core as mx
9
+ import mlx.nn as nn
10
+ from mlx.utils import tree_flatten
11
+ from mlx_vlm.models.qwen_image.weights import load_transformer, load_text_encoder, load_vae
12
+ from scripts.common import (SOURCE, REVISION, RUNTIME_REVISION, MODIFICATION, NOTICE,
13
+ eligible, sha256, write_json)
14
+
15
+ LOADERS = dict(transformer=load_transformer, text_encoder=load_text_encoder, vae=load_vae)
16
+
17
+ def save_component(model, folder, config, limit=2_000_000_000):
18
+ folder.mkdir(parents=True)
19
+ weights = dict(tree_flatten(model.parameters()))
20
+ chunks, current, size = [], {}, 0
21
+ for name, value in weights.items():
22
+ if current and size + value.nbytes > limit:
23
+ chunks.append(current); current, size = {}, 0
24
+ current[name] = value; size += value.nbytes
25
+ if current: chunks.append(current)
26
+ index = {'metadata': {'total_size': sum(v.nbytes for v in weights.values()),
27
+ 'modification_notice': MODIFICATION}, 'weight_map': {}}
28
+ for i, chunk in enumerate(chunks,1):
29
+ name = f'model-{i:05d}-of-{len(chunks):05d}.safetensors'
30
+ mx.save_safetensors(str(folder/name), chunk,
31
+ metadata={'format':'mlx', 'modification_notice':MODIFICATION,
32
+ 'source_revision':REVISION})
33
+ index['weight_map'].update({key:name for key in chunk})
34
+ config['mlx_format'] = True
35
+ config['_modification_notice'] = MODIFICATION
36
+ write_json(folder/'config.json', config)
37
+ write_json(folder/'model.safetensors.index.json', index)
38
+ return weights
39
+
40
+ def main():
41
+ ap=argparse.ArgumentParser()
42
+ ap.add_argument('--source',type=Path,default=SOURCE)
43
+ ap.add_argument('--output',type=Path,required=True)
44
+ ap.add_argument('--bits',type=int,choices=(4,6,8,16),default=8)
45
+ ap.add_argument('--audit',type=Path,default=Path('artifacts/source-audit.json'))
46
+ args=ap.parse_args()
47
+ if args.output.exists(): raise FileExistsError(f'Refusing to overwrite {args.output}')
48
+ audit=json.loads(args.audit.read_text())
49
+ if audit['revision'] != REVISION or Path(audit['source']).resolve()!=args.source.resolve():
50
+ raise ValueError('Source audit does not match')
51
+ # Audit SHA256 was computed in this workspace; reject changes since then.
52
+ for row in audit['files']:
53
+ p=args.source/row['path']
54
+ if p.stat().st_size != row['size'] or p.stat().st_mtime > args.audit.stat().st_mtime:
55
+ raise ValueError(f'Source changed since audit: {p}')
56
+ args.output.mkdir(parents=True)
57
+ report=dict(name='Image21-MLX',source_model='Qwen/Qwen-Image-2.1',source_revision=REVISION,
58
+ runtime_revision=RUNTIME_REVISION,method='MLX affine weight-only' if args.bits<16 else 'MLX BF16 layout conversion',bits=args.bits,
59
+ group_size=64,activation_dtype='bfloat16',vae_dtype='float32',components={},
60
+ source_audit_sha256=sha256(args.audit),status='in_progress')
61
+ write_json(args.output/'conversion.json',report)
62
+ started=time.perf_counter()
63
+ for component, loader in LOADERS.items():
64
+ print(f'Loading {component}',flush=True)
65
+ model=loader(args.source)
66
+ quantized=[]
67
+ if component!='vae' and args.bits<16:
68
+ def predicate(path, module):
69
+ yes=isinstance(module,nn.Linear) and eligible(component,path+'.weight',module.weight.shape)
70
+ if yes: quantized.append(path)
71
+ return yes
72
+ nn.quantize(model,bits=args.bits,group_size=64,mode='affine',class_predicate=predicate)
73
+ if not quantized: raise RuntimeError(f'No quantized modules: {component}')
74
+ mx.eval(model.parameters())
75
+ config=json.loads((args.source/component/'config.json').read_text())
76
+ if quantized: config['quantization']=dict(bits=args.bits,group_size=64,mode='affine')
77
+ expected=save_component(model,args.output/component,config)
78
+ print(f'Saved {component}; verifying fresh loader',flush=True)
79
+ restored=loader(args.output)
80
+ actual=dict(tree_flatten(restored.parameters()))
81
+ if set(expected)!=set(actual): raise ValueError(f'Reload keys differ: {component}')
82
+ for key,value in expected.items():
83
+ other=actual[key]
84
+ if value.dtype!=other.dtype or value.shape!=other.shape or not mx.array_equal(value,other).item():
85
+ raise ValueError(f'Reload mismatch: {component}/{key}')
86
+ report['components'][component]=dict(quantized_modules=quantized,
87
+ tensor_bytes=sum(v.nbytes for v in expected.values()),
88
+ tensors=len(expected),exact_roundtrip=True)
89
+ write_json(args.output/'conversion.json',report)
90
+ del model,restored,actual,expected
91
+ gc.collect(); mx.clear_cache()
92
+ for sub in ('processor','scheduler'):
93
+ shutil.copytree(args.source/sub,args.output/sub,
94
+ ignore=shutil.ignore_patterns('._*','.cache','*.lock'))
95
+ for name in ('model_index.json','LICENSE'):
96
+ shutil.copy2(args.source/name,args.output/name)
97
+ (args.output/'Notice').write_text(NOTICE+'\n\nBuilt with Qwen\n'+MODIFICATION+'\n')
98
+ (args.output/'CHANGES.md').write_text('# Modifications\n\n'+MODIFICATION+'\n\n'
99
+ 'VAE convolution layout is transposed without reducing FP32 precision. '
100
+ 'The entire vision tower, token embeddings, language head, norms, '
101
+ 'transformer input/output projections, timestep embedding and modulation remain floating point.\n')
102
+ report.update(status='converted_and_roundtrip_verified',seconds=time.perf_counter()-started)
103
+ write_json(args.output/'conversion.json',report)
104
+ print(json.dumps(report,indent=2),flush=True)
105
+
106
+ if __name__=='__main__': main()
scripts/infer.py ADDED
@@ -0,0 +1,42 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Generate an RGBA image from a local Image21-MLX or original BF16 model."""
2
+ import argparse
3
+ import time
4
+ from pathlib import Path
5
+ import mlx.core as mx
6
+ from scripts.runtime import load,generate
7
+ from scripts.common import write_json
8
+
9
+ def main():
10
+ ap=argparse.ArgumentParser()
11
+ ap.add_argument('--model',type=Path,required=True)
12
+ ap.add_argument('--prompt',required=True)
13
+ ap.add_argument('--output',type=Path,required=True)
14
+ ap.add_argument('--input',nargs='+')
15
+ ap.add_argument('--seed',type=int)
16
+ ap.add_argument('--source-seed',type=int)
17
+ ap.add_argument('--steps',type=int,default=40)
18
+ ap.add_argument('--width',type=int)
19
+ ap.add_argument('--height',type=int)
20
+ ap.add_argument('--resolution',type=int,default=1024)
21
+ args=ap.parse_args()
22
+ if args.output.exists(): raise FileExistsError(args.output)
23
+ if (args.width is None)!=(args.height is None): ap.error('Set width and height together')
24
+ if args.width is None:
25
+ if args.input:
26
+ from PIL import Image
27
+ from mlx_vlm.models.qwen_image.pipeline import _image_dimensions
28
+ with Image.open(args.input[-1]) as source:
29
+ args.width,args.height=_image_dimensions(args.resolution,source.width/source.height)
30
+ else: args.width=args.height=args.resolution
31
+ seed=args.seed if args.seed is not None else (1000042 if args.input else 42)
32
+ start=time.perf_counter(); pipe=load(args.model)
33
+ load_s=time.perf_counter()-start; mx.reset_peak_memory(); start=time.perf_counter()
34
+ img=generate(pipe,args.prompt,seed=seed,steps=args.steps,width=args.width,height=args.height,
35
+ inputs=args.input,source_seed=args.source_seed,resolution=args.resolution)
36
+ seconds=time.perf_counter()-start
37
+ args.output.parent.mkdir(parents=True,exist_ok=True); img.save(args.output)
38
+ row=dict(model=str(args.model),prompt=args.prompt,seed=seed,steps=args.steps,width=args.width,height=args.height,
39
+ mode=img.mode,pipeline_init_seconds=load_s,seconds=seconds,mlx_peak_gib=mx.get_peak_memory()/2**30)
40
+ write_json(args.output.with_suffix('.json'),row); print(row,flush=True)
41
+
42
+ if __name__=='__main__': main()
scripts/login.py ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Run interactively in your own terminal; never pass tokens on the command line."""
2
+ import argparse
3
+ from getpass import getpass
4
+
5
+ def main():
6
+ ap=argparse.ArgumentParser()
7
+ ap.add_argument('platform',choices=('huggingface','modelscope'))
8
+ args=ap.parse_args()
9
+ if args.platform=='huggingface':
10
+ from huggingface_hub import login,HfApi
11
+ login(token=getpass('Hugging Face write token (hidden): '),add_to_git_credential=False)
12
+ owner=HfApi().whoami()['name']
13
+ expected='ixim'
14
+ else:
15
+ from modelscope_hub import HubApi
16
+ api=HubApi()
17
+ user=api.login(getpass('ModelScope SDK write token (hidden): '))
18
+ owner=user.username
19
+ expected='iximbox'
20
+ if owner!=expected: raise ValueError(f'Expected {expected}; logged in as {owner}')
21
+ print(f'Authenticated {owner}. No token has been printed.')
22
+
23
+ if __name__=='__main__': main()
scripts/mlx_pipeline.py ADDED
@@ -0,0 +1,351 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Adapted from MLX-VLM 95b01ccad2d9f65a9e87f6a87bd1c5df69626261.
2
+ # Copyright (c) 2025 Prince Canuma. MIT: see MLX_VLM_LICENSE.txt.
3
+ # Modifications: text-to-image prefix KV cache and sequential component loading.
4
+ """Generation and reference-image editing for Qwen-Image-2.1."""
5
+
6
+ from __future__ import annotations
7
+
8
+ import json
9
+ import math
10
+ import gc
11
+ from collections.abc import Sequence
12
+ from pathlib import Path
13
+
14
+ import mlx.core as mx
15
+ import numpy as np
16
+ from PIL import Image
17
+
18
+ from mlx_vlm.models.qwen_image.config import QwenImageVariant, get_variant
19
+ from mlx_vlm.models.qwen_image.download import download_model, validate_model_layout
20
+ from mlx_vlm.models.qwen_image.kv_cache import QwenImageKVCache
21
+ from mlx_vlm.models.qwen_image.scheduler import FlowMatchEulerDiscreteScheduler
22
+ from mlx_vlm.models.qwen_image.text_encoder import QwenImageTextEncoder
23
+ from mlx_vlm.models.qwen_image.weights import _read_quant, load_text_encoder, load_transformer, load_vae
24
+
25
+
26
+ def _image_dimensions(resolution: int, ratio: float) -> tuple[int, int]:
27
+ width = math.sqrt(resolution * resolution * ratio)
28
+ return max(32, round(width / 32) * 32), max(32, round(width / ratio / 32) * 32)
29
+
30
+
31
+ class QwenImagePipeline:
32
+ def __init__(
33
+ self,
34
+ *,
35
+ variant: QwenImageVariant,
36
+ model_path: str | Path,
37
+ text_encoder,
38
+ transformer,
39
+ vae,
40
+ phase_offload: bool = True,
41
+ ) -> None:
42
+ self.variant = variant
43
+ self.model_path = Path(model_path)
44
+ self.text_encoder = QwenImageTextEncoder(
45
+ model=text_encoder, model_path=self.model_path
46
+ )
47
+ self.transformer = transformer
48
+ self.vae = vae
49
+ self.phase_offload = phase_offload
50
+ cfg = json.loads((self.model_path / "vae" / "config.json").read_text())
51
+ self.z_dim = cfg["z_dim"]
52
+ self.latents_mean = mx.array(cfg["latents_mean"]).reshape(
53
+ 1, self.z_dim, 1, 1, 1
54
+ )
55
+ self.latents_std = mx.array(cfg["latents_std"]).reshape(1, self.z_dim, 1, 1, 1)
56
+ self.quantization_config = _read_quant(self.model_path / "transformer")
57
+ scheduler_path = self.model_path / "scheduler" / "scheduler_config.json"
58
+ scheduler_config = (
59
+ json.loads(scheduler_path.read_text()) if scheduler_path.exists() else {}
60
+ )
61
+ self.scheduler_config = {
62
+ key: scheduler_config[key]
63
+ for key in (
64
+ "base_shift",
65
+ "max_shift",
66
+ "base_image_seq_len",
67
+ "max_image_seq_len",
68
+ "num_train_timesteps",
69
+ "shift_terminal",
70
+ )
71
+ if key in scheduler_config
72
+ }
73
+
74
+ @classmethod
75
+ def from_pretrained(
76
+ cls,
77
+ variant: str | QwenImageVariant = "qwen-image-2.1",
78
+ *,
79
+ model_path: str | Path | None = None,
80
+ download: bool = True,
81
+ token: str | None = None,
82
+ revision: str | None = None,
83
+ force_download: bool = False,
84
+ phase_offload: bool = True,
85
+ ) -> "QwenImagePipeline":
86
+ variant = get_variant(variant)
87
+ if model_path is None:
88
+ if not download:
89
+ raise ValueError("model_path is required when download=False")
90
+ model_path = download_model(
91
+ variant, token=token, revision=revision, force_download=force_download
92
+ )
93
+ model_path = validate_model_layout(model_path)
94
+ return cls(
95
+ variant=variant,
96
+ model_path=model_path,
97
+ text_encoder=None if phase_offload else load_text_encoder(model_path),
98
+ transformer=None if phase_offload else load_transformer(model_path, variant),
99
+ vae=None if phase_offload else load_vae(model_path),
100
+ phase_offload=phase_offload,
101
+ )
102
+
103
+ def activate(self, component: str) -> None:
104
+ """Keep only the currently needed component resident in unified memory.
105
+
106
+ Callers materialize activations before switching, so no lazy graph keeps
107
+ the preceding model alive. Loading happens before resetting sample RNG.
108
+ """
109
+ if self.phase_offload:
110
+ if component != 'text_encoder': self.text_encoder.model = None
111
+ if component != 'transformer': self.transformer = None
112
+ if component != 'vae': self.vae = None
113
+ gc.collect()
114
+ mx.clear_cache()
115
+ if component == 'text_encoder':
116
+ if self.text_encoder.model is None:
117
+ self.text_encoder.model = load_text_encoder(self.model_path)
118
+ model = self.text_encoder.model
119
+ elif component == 'transformer':
120
+ if self.transformer is None:
121
+ self.transformer = load_transformer(self.model_path, self.variant)
122
+ model = self.transformer
123
+ elif component == 'vae':
124
+ if self.vae is None: self.vae = load_vae(self.model_path)
125
+ model = self.vae
126
+ else: raise ValueError(component)
127
+ mx.eval(model.parameters())
128
+
129
+ def release(self) -> None:
130
+ if self.phase_offload:
131
+ self.text_encoder.model = self.transformer = self.vae = None
132
+ gc.collect()
133
+ mx.clear_cache()
134
+
135
+ def count_prompt_tokens(self, prompt: str) -> int:
136
+ return len(self.text_encoder.tokenizer(prompt)["input_ids"])
137
+
138
+ def generate_array(
139
+ self,
140
+ prompt: str,
141
+ *,
142
+ seed: int = 0,
143
+ steps: int = 30,
144
+ width: int = 512,
145
+ height: int = 512,
146
+ guidance: float = 1.0,
147
+ negative_prompt: str = " ",
148
+ num_images: int = 1,
149
+ ) -> mx.array:
150
+ """Generate one ``[H, W, 3]`` image, or ``[N, H, W, 3]`` when num_images > 1."""
151
+ self.activate('text_encoder')
152
+ emb = self.text_encoder.encode(prompt).astype(mx.bfloat16)
153
+ do_cfg = guidance is not None and guidance > 1.0
154
+ neg = (
155
+ self.text_encoder.encode(negative_prompt).astype(mx.bfloat16)
156
+ if do_cfg
157
+ else None
158
+ )
159
+
160
+ return self._sample(
161
+ emb,
162
+ neg,
163
+ seed=seed,
164
+ steps=steps,
165
+ width=width,
166
+ height=height,
167
+ guidance=guidance,
168
+ num_images=num_images,
169
+ )[..., :3]
170
+
171
+ def edit_array(
172
+ self,
173
+ prompt: str,
174
+ image_paths: Sequence[str | Path],
175
+ *,
176
+ seed: int = 0,
177
+ steps: int = 40,
178
+ width: int | None = None,
179
+ height: int | None = None,
180
+ guidance: float = 1.0,
181
+ negative_prompt: str = " ",
182
+ output_resolution: int = 1024,
183
+ use_kv_cache: bool = True,
184
+ ) -> mx.array:
185
+ """Edit one or more references; return an [H, W, 4] uint8 RGBA image."""
186
+ if not image_paths:
187
+ raise ValueError("At least one reference image is required")
188
+ if output_resolution < 256:
189
+ raise ValueError("output_resolution must be at least 256")
190
+ self.activate('vae')
191
+ references, reference_latents, reference_shapes = [], [], []
192
+ for path in image_paths:
193
+ with Image.open(Path(path).expanduser()) as source:
194
+ image = source.convert("RGBA")
195
+ size = _image_dimensions(output_resolution, image.width / image.height)
196
+ image = image.resize(size, Image.Resampling.LANCZOS)
197
+ references.append(image)
198
+ pixels = mx.array(np.asarray(image).astype(np.float32) / 127.5 - 1.0)
199
+ pixels = pixels.transpose(2, 0, 1)[None, :, None].astype(mx.bfloat16)
200
+ mean, variance = self.vae.encode(pixels)
201
+ mx.eval(mean, variance)
202
+ normalized = (
203
+ mean - self.latents_mean.astype(mean.dtype)
204
+ ) / self.latents_std.astype(mean.dtype)
205
+ h, w = normalized.shape[-2:]
206
+ reference_shapes.append((1, h, w))
207
+ reference_latents.append(
208
+ normalized.reshape(1, self.z_dim, h * w).transpose(0, 2, 1)
209
+ )
210
+ mx.eval(reference_latents[-1])
211
+ default_width, default_height = references[-1].size
212
+ width = (default_width if width is None else width) // 32 * 32
213
+ height = (default_height if height is None else height) // 32 * 32
214
+ if width < 32 or height < 32:
215
+ raise ValueError("Output width and height must be at least 32")
216
+ self.activate('text_encoder')
217
+ emb, image_pad_mask = self.text_encoder.encode_edit(prompt, references)
218
+ neg = negative_image_pad_mask = None
219
+ if guidance is not None and guidance > 1.0:
220
+ neg, negative_image_pad_mask = self.text_encoder.encode_edit(
221
+ negative_prompt, references
222
+ )
223
+ neg = neg.astype(mx.bfloat16)
224
+ return self._sample(
225
+ emb.astype(mx.bfloat16),
226
+ neg,
227
+ seed=seed,
228
+ steps=steps,
229
+ width=width,
230
+ height=height,
231
+ guidance=guidance,
232
+ reference_latents=mx.concatenate(reference_latents, axis=1).astype(
233
+ mx.bfloat16
234
+ ),
235
+ reference_image_shapes=reference_shapes,
236
+ image_pad_mask=image_pad_mask,
237
+ negative_image_pad_mask=negative_image_pad_mask,
238
+ use_kv_cache=use_kv_cache,
239
+ )
240
+
241
+ def _sample(
242
+ self,
243
+ emb,
244
+ neg,
245
+ *,
246
+ seed,
247
+ steps,
248
+ width,
249
+ height,
250
+ guidance,
251
+ num_images=1,
252
+ reference_latents=None,
253
+ reference_image_shapes=None,
254
+ image_pad_mask=None,
255
+ negative_image_pad_mask=None,
256
+ use_kv_cache=False,
257
+ ) -> mx.array:
258
+ mx.eval(emb, [] if neg is None else neg,
259
+ [] if reference_latents is None else reference_latents)
260
+ self.activate('transformer')
261
+ z = self.z_dim
262
+ h_lat, w_lat = height // 16, width // 16
263
+ tokens = h_lat * w_lat
264
+ mx.random.seed(seed)
265
+ latents = mx.random.normal((num_images, 1, z, h_lat, w_lat)).astype(mx.bfloat16)
266
+ latents = latents.reshape(num_images, z, tokens).transpose(0, 2, 1)
267
+ if num_images > 1:
268
+ emb = mx.broadcast_to(emb, (num_images, *emb.shape[1:]))
269
+ if neg is not None:
270
+ neg = mx.broadcast_to(neg, (num_images, *neg.shape[1:]))
271
+ scheduler = FlowMatchEulerDiscreteScheduler(
272
+ image_seq_len=tokens, num_inference_steps=steps, **self.scheduler_config
273
+ )
274
+ cache = negative_cache = None
275
+ if (
276
+ use_kv_cache
277
+ and steps > 1
278
+ and self.transformer.causal_condition
279
+ ):
280
+ num_layers = len(self.transformer.transformer_blocks)
281
+ cache = QwenImageKVCache(num_layers)
282
+ if neg is not None:
283
+ negative_cache = QwenImageKVCache(num_layers)
284
+ for i in range(steps):
285
+ t = scheduler.timesteps[i : i + 1].astype(latents.dtype) / 1000
286
+ model_input = latents
287
+ edit_kwargs = {}
288
+ if reference_latents is not None:
289
+ if cache is None or i == 0:
290
+ model_input = mx.concatenate([reference_latents, latents], axis=1)
291
+ edit_kwargs = dict(
292
+ reference_image_shapes=reference_image_shapes,
293
+ image_pad_mask=image_pad_mask,
294
+ )
295
+ if cache is not None:
296
+ edit_kwargs.update(
297
+ kv_cache=cache, kv_cache_mode="extract" if i == 0 else "cached"
298
+ )
299
+ pred = self.transformer(
300
+ hidden_states=model_input,
301
+ encoder_hidden_states=emb,
302
+ timestep=t,
303
+ img_shape=(1, h_lat, w_lat),
304
+ **edit_kwargs,
305
+ )
306
+ if neg is not None:
307
+ if reference_latents is not None:
308
+ edit_kwargs["image_pad_mask"] = negative_image_pad_mask
309
+ if negative_cache is not None:
310
+ edit_kwargs["kv_cache"] = negative_cache
311
+ neg_pred = self.transformer(
312
+ hidden_states=model_input,
313
+ encoder_hidden_states=neg,
314
+ timestep=t,
315
+ img_shape=(1, h_lat, w_lat),
316
+ **edit_kwargs,
317
+ )
318
+ pred = neg_pred + guidance * (pred - neg_pred)
319
+ latents = scheduler.step(noise=pred, step_index=i, latents=latents)
320
+ if cache is not None and i == 0:
321
+ # Materialize compact prefix storage at the existing step boundary,
322
+ # releasing the extraction graph before the next denoising step.
323
+ mx.eval(
324
+ latents,
325
+ cache.arrays(),
326
+ [] if negative_cache is None else negative_cache.arrays(),
327
+ )
328
+ else:
329
+ mx.eval(latents)
330
+
331
+ # The VAE decoder does not need the per-layer prefix buffers.
332
+ if cache is not None:
333
+ cache.clear()
334
+ if negative_cache is not None:
335
+ negative_cache.clear()
336
+
337
+ z_lat = (
338
+ latents.transpose(0, 2, 1)
339
+ .reshape(num_images, z, 1, h_lat, w_lat)
340
+ .astype(mx.float32)
341
+ )
342
+ z_lat = z_lat * self.latents_std + self.latents_mean
343
+ mx.eval(z_lat)
344
+ self.activate('vae')
345
+ image = self.vae.decode(z_lat)[:, :, 0]
346
+ image = ((mx.clip(image, -1.0, 1.0) + 1.0) / 2.0 * 255).astype(mx.uint8)
347
+ images = image.transpose(0, 2, 3, 1)
348
+ return images[0] if num_images == 1 else images
349
+
350
+
351
+ __all__ = ["QwenImagePipeline"]
scripts/probe.py ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stratified weight reconstruction and GPU matmul probe, not an image-quality score."""
2
+ import argparse
3
+ import gc
4
+ import json
5
+ import time
6
+ from pathlib import Path
7
+ import mlx.core as mx
8
+ from scripts.common import SOURCE, eligible, tensors, write_json
9
+
10
+ def main():
11
+ ap = argparse.ArgumentParser()
12
+ ap.add_argument('--source', type=Path, default=SOURCE)
13
+ ap.add_argument('--output', default='artifacts/precision-probe.json')
14
+ args = ap.parse_args()
15
+ results = []
16
+ for component, layers in [('transformer', ('0', '15', '31')), ('text_encoder', ('0','17','35'))]:
17
+ for path, name, info in tensors(args.source / component):
18
+ if not eligible(component, name, info['shape']):
19
+ continue
20
+ if not any(f'.{i}.' in name for i in layers):
21
+ continue
22
+ if not any(x in name for x in ('to_q.weight', 'img_mlp.out.weight', 'q_proj.weight', 'down_proj.weight')):
23
+ continue
24
+ all_weights = mx.load(str(path))
25
+ # 256 evenly spaced rows, all columns/groups: bounded memory and reproducible.
26
+ full = all_weights[name]
27
+ weight = full[mx.linspace(0, full.shape[0]-1, 256).astype(mx.int32)]
28
+ mx.eval(weight)
29
+ del full, all_weights
30
+ wf = weight.astype(mx.float32)
31
+ mx.random.seed(2026)
32
+ x = mx.random.normal((1,256,weight.shape[-1])).astype(mx.bfloat16)
33
+ expected = x @ weight.T
34
+ mx.eval(expected)
35
+ for bits in (4,6,8):
36
+ q,s,b = mx.quantize(weight, bits=bits, group_size=64)
37
+ rec = mx.dequantize(q,s,b,bits=bits,group_size=64).astype(mx.float32)
38
+ err = float(mx.sqrt(mx.mean((rec-wf)**2) / mx.mean(wf**2)).item())
39
+ fn = lambda: mx.quantized_matmul(x,q,s,b,bits=bits,group_size=64,transpose=True)
40
+ for _ in range(3): mx.eval(fn())
41
+ start=time.perf_counter()
42
+ for _ in range(10): mx.eval(fn())
43
+ ms=(time.perf_counter()-start)*100
44
+ y=fn().astype(mx.float32); e=expected.astype(mx.float32)
45
+ out_err=float(mx.sqrt(mx.mean((y-e)**2)/mx.mean(e**2)).item())
46
+ row=dict(component=component,tensor=name,source_shape=info['shape'],sample_rows=256,
47
+ bits=bits,weight_relative_rmse=err,synthetic_matmul_relative_rmse=out_err,matmul_ms=ms)
48
+ results.append(row)
49
+ print(json.dumps(row),flush=True)
50
+ del weight,wf,x,expected,q,s,b,rec,y,e
51
+ gc.collect(); mx.clear_cache()
52
+ write_json(args.output,dict(group_size=64,mode='affine',device=mx.metal.device_info(),results=results,
53
+ limitations='Sampled rows and synthetic inputs; not calibrated activations, image quality or end-to-end speed.'))
54
+
55
+ if __name__ == '__main__': main()
scripts/release.py ADDED
@@ -0,0 +1,91 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage only verified weights, reproducible tools and paired evaluation evidence."""
2
+ import argparse
3
+ import json
4
+ import shutil
5
+ from pathlib import Path
6
+ from scripts.common import REVISION,RUNTIME_REVISION,sha256,header,write_json
7
+ from scripts.report import REQUIRED_CASES,pair_records,records
8
+
9
+ def manifest(root):
10
+ root=Path(root)
11
+ return [dict(path=p.relative_to(root).as_posix(),size=p.stat().st_size,sha256=sha256(p))
12
+ for p in sorted(root.rglob('*')) if p.is_file() and p.name!='MANIFEST.json'
13
+ and not any(x.startswith('.') or x=='__pycache__' for x in p.relative_to(root).parts)]
14
+
15
+ def validate_manifest(root):
16
+ root=Path(root).resolve()
17
+ rows=json.loads((root/'MANIFEST.json').read_text())
18
+ names=set()
19
+ for row in rows:
20
+ name=row['path']; original=root/name; path=original.resolve()
21
+ if original.is_symlink() or not path.is_relative_to(root) or name in names:
22
+ raise ValueError(f'Invalid manifest path: {name}')
23
+ names.add(name)
24
+ if path.stat().st_size!=row['size'] or sha256(path)!=row['sha256']:
25
+ raise ValueError(f'Manifest mismatch: {name}')
26
+ actual={p.relative_to(root).as_posix() for p in root.rglob('*') if p.is_file()
27
+ and not any(x.startswith('.') or x=='__pycache__' for x in p.relative_to(root).parts)}
28
+ if actual!=names|{'MANIFEST.json'}: raise ValueError('Unexpected/missing release files')
29
+ return rows
30
+
31
+ def validate_evaluation(root):
32
+ root=Path(root)
33
+ pairs=list(pair_records(records(root/'bf16'),records(root/'8bit')))
34
+ if {a['case_id'] for a,b in pairs}!=REQUIRED_CASES: raise ValueError('Incomplete seven-case suite')
35
+ for a,b in pairs:
36
+ if a['steps']!=40 or a['width']!=1024 or a['height']!=1024 or not a['warmup']:
37
+ raise ValueError('Release requires the full 1024px/40-step warmed protocol')
38
+ if b['mode']!='RGBA' or b['rgb_std']<1: raise ValueError('Broken output')
39
+ review=json.loads((root/'visual-review.json').read_text())
40
+ if review.get('approved_for_research_release') is not True:
41
+ raise ValueError('Visual review incomplete or failed')
42
+ if set(review.get('cases',{}))!=REQUIRED_CASES: raise ValueError('Missing visual case reviews')
43
+ for name in ('bf16','8bit'):
44
+ if review['results_sha256'][name]!=sha256(root/name/'results.jsonl'):
45
+ raise ValueError('Visual review does not match current results')
46
+
47
+ def main():
48
+ ap=argparse.ArgumentParser()
49
+ ap.add_argument('--model',type=Path,default=Path('models/Image21-MLX-8bit'))
50
+ ap.add_argument('--evaluation',type=Path,default=Path('artifacts/eval'))
51
+ ap.add_argument('--platform',choices=('huggingface','modelscope'),required=True)
52
+ ap.add_argument('--output',type=Path)
53
+ args=ap.parse_args(); out=args.output or Path('release')/args.platform
54
+ if out.exists(): raise FileExistsError(out)
55
+ validate_evaluation(args.evaluation)
56
+ conversion=json.loads((args.model/'conversion.json').read_text())
57
+ if conversion['status']!='converted_and_roundtrip_verified' or conversion['source_revision']!=REVISION or conversion['runtime_revision']!=RUNTIME_REVISION:
58
+ raise ValueError('Unverified conversion')
59
+ for comp in conversion['components'].values():
60
+ if not comp['exact_roundtrip']: raise ValueError('Roundtrip not verified')
61
+ for p in args.model.glob('*/*.safetensors'):
62
+ if 'modification_notice' not in header(p).get('__metadata__',{}): raise ValueError('Missing modification notice')
63
+ shutil.copytree(args.model,out,copy_function=shutil.copy2)
64
+ shutil.copytree(args.evaluation,out/'evaluation',ignore=shutil.ignore_patterns('._*','*.tmp'))
65
+ # Publish useful machine measurements without the workstation's absolute paths.
66
+ for precision in ('bf16','8bit'):
67
+ path=out/'evaluation'/precision/'environment.json'
68
+ env=json.loads(path.read_text())
69
+ env['model']=f'Image21-MLX-{precision}'
70
+ write_json(path,env)
71
+ audit=json.loads(Path('artifacts/source-audit.json').read_text())
72
+ audit.pop('source')
73
+ audit['source_model']='Qwen/Qwen-Image-2.1'
74
+ audit['note']='Public source-file hashes and precision inventory; local source directory omitted.'
75
+ write_json(out/'evaluation'/'source-files.json',audit)
76
+ shutil.copytree('scripts',out/'scripts',ignore=shutil.ignore_patterns('__pycache__'))
77
+ shutil.copytree('benchmarks',out/'benchmarks')
78
+ for name in ('requirements.txt','requirements.lock.txt'):
79
+ shutil.copy2(name,out/name)
80
+ shutil.copy2(f'cards/{args.platform}.md',out/'README.md')
81
+ shutil.copy2('artifacts/precision-probe.json',out/'evaluation'/'precision-probe.json')
82
+ for name in ('cache-parity-bf16.json','cache-parity-8bit.json','offload-parity.json'):
83
+ shutil.copy2(Path('artifacts')/name,out/'evaluation'/name)
84
+ for row in json.loads((args.evaluation/'runtime-manifest.json').read_text()):
85
+ if sha256(out/row['path'])!=row['sha256']:
86
+ raise ValueError(f'Runtime changed after evaluation: {row["path"]}')
87
+ write_json(out/'MANIFEST.json',manifest(out))
88
+ rows=validate_manifest(out)
89
+ print(f'Staged and verified {len(rows)} files in {out}')
90
+
91
+ if __name__=='__main__': main()
scripts/report.py ADDED
@@ -0,0 +1,82 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Pair raw records and generate inspectable image comparisons (no invented quality score)."""
2
+ import argparse
3
+ import json
4
+ import math
5
+ import statistics
6
+ from pathlib import Path
7
+ import numpy as np
8
+ from PIL import Image,ImageDraw
9
+ from scripts.common import sha256,write_json
10
+
11
+ PAIR_KEYS=('case_id','prompt','seed','width','height','steps','cfg','vae_tiling','kv_cache','phase_offload','input_sha256','warmup')
12
+ REQUIRED_CASES={'portrait','english_text','chinese_text','composition','texture','rgba','edit'}
13
+
14
+ def records(root):
15
+ root=Path(root)
16
+ if not (root/'COMPLETE.json').exists(): raise ValueError(f'Incomplete evaluation: {root}')
17
+ data=[json.loads(s) for s in (root/'results.jsonl').read_text().splitlines()]
18
+ out={}
19
+ for row in data:
20
+ key=(row['case_id'],row['seed'])
21
+ if key in out: raise ValueError('Duplicate case/seed')
22
+ if sha256(root/row['output'])!=row['output_sha256']: raise ValueError('Image hash mismatch')
23
+ out[key]=row
24
+ return out
25
+
26
+ def pair_records(baseline,candidate):
27
+ if set(baseline)!=set(candidate): raise ValueError('Different evaluation cases')
28
+ for key,a in baseline.items():
29
+ b=candidate[key]
30
+ for k in PAIR_KEYS:
31
+ if a.get(k)!=b.get(k): raise ValueError(f'Incomparable pair: {key}/{k}')
32
+ yield a,b
33
+
34
+ def main():
35
+ ap=argparse.ArgumentParser()
36
+ ap.add_argument('--baseline',type=Path,default=Path('artifacts/eval/bf16'))
37
+ ap.add_argument('--candidate',type=Path,default=Path('artifacts/eval/8bit'))
38
+ ap.add_argument('--output',type=Path,default=Path('artifacts/eval'))
39
+ args=ap.parse_args(); args.output.mkdir(parents=True,exist_ok=True)
40
+ pairs=list(pair_records(records(args.baseline),records(args.candidate)))
41
+ rows=[]; sheet=Image.new('RGB',(1024,len(pairs)*550),'#eeeeee'); draw=ImageDraw.Draw(sheet)
42
+ for i,(a,b) in enumerate(pairs):
43
+ images=[Image.open(root/row['output']).convert('RGBA') for root,row in [(args.baseline,a),(args.candidate,b)]]
44
+ av,bv=[np.asarray(img).astype(np.float32) for img in images]
45
+ mse=float(np.mean((av[:,:,:3]-bv[:,:,:3])**2))
46
+ row=dict(case_id=a['case_id'],seed=a['seed'],bf16_seconds=a['seconds'],q8_seconds=b['seconds'],
47
+ bf16_peak_gib=a['mlx_peak_gib'],q8_peak_gib=b['mlx_peak_gib'],
48
+ bf16_swap_delta_gib=a.get('swap_delta_gib'),q8_swap_delta_gib=b.get('swap_delta_gib'),
49
+ rgb_psnr_db=10*math.log10(255**2/max(mse,1e-12)),
50
+ alpha_mae=float(np.mean(np.abs(av[:,:,3]-bv[:,:,3]))))
51
+ rows.append(row)
52
+ for j,(img,label) in enumerate(zip(images,('BF16','Q8'))):
53
+ checker=Image.new('RGBA',img.size,'white'); cd=ImageDraw.Draw(checker)
54
+ for y in range(0,img.height,32):
55
+ for x in range(0,img.width,32):
56
+ if (x//32+y//32)%2: cd.rectangle((x,y,x+31,y+31),fill='#d7d7d7')
57
+ thumb=Image.alpha_composite(checker,img).convert('RGB').resize((512,512),Image.Resampling.LANCZOS)
58
+ sheet.paste(thumb,(j*512,i*550+30)); draw.text((j*512+10,i*550+8),f'{a["case_id"]} / {label} / seed {a["seed"]}',fill='black')
59
+ sheet.save(args.output/'comparison.png')
60
+ summary=dict(pairs=len(pairs),cases=len({x['case_id'] for x in rows}),size=pairs[0][0]['width'],steps=pairs[0][0]['steps'],
61
+ warmup=pairs[0][0]['warmup'],phase_offload=pairs[0][0].get('phase_offload',False),
62
+ bf16_mean_seconds=statistics.mean(x['bf16_seconds'] for x in rows),
63
+ q8_mean_seconds=statistics.mean(x['q8_seconds'] for x in rows),
64
+ bf16_t2i_mean_seconds=statistics.mean(x['bf16_seconds'] for x in rows if x['case_id']!='edit'),
65
+ q8_t2i_mean_seconds=statistics.mean(x['q8_seconds'] for x in rows if x['case_id']!='edit'),
66
+ bf16_max_peak_gib=max(x['bf16_peak_gib'] for x in rows),
67
+ q8_max_peak_gib=max(x['q8_peak_gib'] for x in rows),
68
+ rows=rows,limitations='One seed per case, one machine, same pinned MLX runtime. Sequential BF16-then-Q8 desktop run without repeated or interleaved trials; order, thermal state and other applications may affect timing. Pixel similarity is not a perceptual quality score; no cross-runtime parity claim.')
69
+ write_json(args.output/'summary.json',summary)
70
+ text=['# Image21-MLX informal evaluation','',summary['limitations'],'',
71
+ '| Case | BF16 s | Q8 s | BF16 peak GiB | Q8 peak GiB | RGB PSNR dB |',
72
+ '|---|---:|---:|---:|---:|---:|']
73
+ for r in rows: text.append(f'| {r["case_id"]} | {r["bf16_seconds"]:.2f} | {r["q8_seconds"]:.2f} | {r["bf16_peak_gib"]:.2f} | {r["q8_peak_gib"]:.2f} | {r["rgb_psnr_db"]:.2f} |')
74
+ text+=['','MLX peak allocated memory excludes the OS, other applications and some process allocations. It is not minimum machine RAM.',
75
+ '', 'Timing includes prompt encoding, phase-by-phase component loading, denoising and VAE decoding; PNG writing is excluded. The after-image RSS field is a snapshot after component release, not a peak.',
76
+ '', 'Existing system swap is recorded separately from per-image swap change. This desktop-session run leaves other applications open. See system-context.json and the per-model environment.json files.',
77
+ '', 'All editing pairs use the same BF16 portrait input. Visual findings and release limitations are in visual-review.json.',
78
+ '', '![All evaluated pairs](comparison.png)','']
79
+ (args.output/'report.md').write_text('\n'.join(text))
80
+ print(json.dumps(summary,indent=2))
81
+
82
+ if __name__=='__main__': main()
scripts/runtime.py ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Pinned MLX runtime adapter preserving the source model's RGBA output."""
2
+ from pathlib import Path
3
+ import mlx.core as mx
4
+ import numpy as np
5
+ from PIL import Image
6
+ from scripts.mlx_pipeline import QwenImagePipeline
7
+
8
+ def enable_progress():
9
+ """Materialize at the existing Euler boundary and show progress every ten steps."""
10
+ from scripts import mlx_pipeline as module
11
+ from mlx_vlm.models.qwen_image.scheduler import FlowMatchEulerDiscreteScheduler
12
+ import time
13
+ class ProgressScheduler(FlowMatchEulerDiscreteScheduler):
14
+ def step(self,*,noise,step_index,latents):
15
+ value=super().step(noise=noise,step_index=step_index,latents=latents)
16
+ mx.eval(value)
17
+ if step_index==0 or (step_index+1)%10==0:
18
+ print(f' denoise step {step_index+1} at {time.strftime("%H:%M:%S")}',flush=True)
19
+ return value
20
+ module.FlowMatchEulerDiscreteScheduler=ProgressScheduler
21
+
22
+ def load(model,phase_offload=True):
23
+ return QwenImagePipeline.from_pretrained(model_path=Path(model),download=False,phase_offload=phase_offload)
24
+
25
+ def generate(pipe,prompt,*,seed=42,steps=40,width=1024,height=1024,inputs=None,
26
+ source_seed=None,resolution=1024):
27
+ if steps<1 or width<256 or height<256 or width%32 or height%32:
28
+ raise ValueError('Use positive steps and dimensions >=256 divisible by 32')
29
+ if inputs:
30
+ if source_seed is not None and seed==source_seed:
31
+ raise ValueError('Editing must use a different seed from the source image')
32
+ if len(inputs)>10: raise ValueError('At most ten references are supported')
33
+ result=pipe.edit_array(prompt,inputs,seed=seed,steps=steps,width=width,height=height,
34
+ guidance=1.0,output_resolution=resolution,use_kv_cache=True)
35
+ else:
36
+ # The pinned upstream generate_array slices RGB. Call its unchanged RGBA
37
+ # sampler so transparent generation keeps all four native output channels.
38
+ print('Encoding prompt',flush=True)
39
+ pipe.activate('text_encoder')
40
+ emb=pipe.text_encoder.encode(prompt).astype(mx.bfloat16)
41
+ mx.eval(emb)
42
+ print('Denoising',flush=True)
43
+ result=pipe._sample(emb,None,seed=seed,steps=steps,width=width,height=height,guidance=1.0,use_kv_cache=True)
44
+ mx.eval(result)
45
+ pixels=np.asarray(result)
46
+ pipe.release()
47
+ if pixels.shape!=(height,width,4): raise ValueError(f'Unexpected output: {pixels.shape}')
48
+ return Image.fromarray(pixels)
scripts/upload.py ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Publish the explicit release inventory; never upload workspace caches or credentials."""
2
+ import argparse
3
+ import hashlib
4
+ import json
5
+ import os
6
+ from pathlib import Path
7
+ from scripts.common import write_json,sha256
8
+ from scripts.release import validate_manifest,validate_evaluation
9
+
10
+ def main():
11
+ ap=argparse.ArgumentParser()
12
+ ap.add_argument('platform',choices=('huggingface','modelscope'))
13
+ ap.add_argument('--folder',type=Path)
14
+ ap.add_argument('--dry-run',action='store_true')
15
+ args=ap.parse_args(); folder=args.folder or Path('release')/args.platform
16
+ owner='ixim' if args.platform=='huggingface' else 'iximbox'
17
+ repo=f'{owner}/Image21-MLX-8bit'
18
+ validate_evaluation(folder/'evaluation')
19
+ rows=validate_manifest(folder); names=[x['path'] for x in rows]+['MANIFEST.json']
20
+ remote_rows=rows+[dict(path='MANIFEST.json',size=(folder/'MANIFEST.json').stat().st_size,
21
+ sha256=sha256(folder/'MANIFEST.json'))]
22
+ print(f'Validated {len(rows)} files: {repo}',flush=True)
23
+ if args.dry_run: return
24
+ if args.platform=='huggingface':
25
+ from huggingface_hub import HfApi,get_token
26
+ if not get_token(): raise RuntimeError('Run python -m scripts.login huggingface locally')
27
+ api=HfApi()
28
+ if api.whoami()['name']!=owner: raise ValueError('Wrong HF account')
29
+ api.create_repo(repo_id=repo,repo_type='model',private=False,exist_ok=True)
30
+ result=api.upload_folder(repo_id=repo,repo_type='model',folder_path=folder,allow_patterns=names,
31
+ commit_message='Release Image21-MLX 8-bit with reproducible evaluation')
32
+ info=api.model_info(repo,files_metadata=True)
33
+ remote={x.rfilename:x for x in info.siblings}
34
+ for row in remote_rows:
35
+ if row['path'] not in remote or remote[row['path']].size!=row['size']:
36
+ raise ValueError(f'Remote size mismatch: {row["path"]}')
37
+ lfs=remote[row['path']].lfs
38
+ if lfs is not None and lfs.sha256!=row['sha256']: raise ValueError('Remote LFS hash mismatch')
39
+ if lfs is None:
40
+ data=(folder/row['path']).read_bytes()
41
+ blob=hashlib.sha1(f'blob {len(data)}\0'.encode()+data).hexdigest()
42
+ if remote[row['path']].blob_id!=blob: raise ValueError('Remote Git blob hash mismatch')
43
+ record=dict(repo_id=repo,url=f'https://huggingface.co/{repo}',revision=info.sha,
44
+ commit_url=str(result.commit_url),verified_files=len(remote_rows))
45
+ else:
46
+ from modelscope_hub import HubApi
47
+ from modelscope_hub.errors import NotExistError
48
+ api=HubApi()
49
+ if api.whoami().username!=owner: raise ValueError('Wrong ModelScope account')
50
+ # create only on an explicit not-found response; never hide authentication failures.
51
+ try: api.get_repo(repo,repo_type='model')
52
+ except NotExistError:
53
+ api.create_repo(repo,repo_type='model',visibility='public',license='other',
54
+ description='Image21-MLX 8-bit. Built with Qwen. Research/evaluation only.')
55
+ result=api.upload_folder(repo_id=repo,repo_type='model',folder_path=folder,allow_patterns=names,
56
+ max_workers=4,commit_message='Release verified Image21-MLX 8-bit')
57
+ remote={x.path:x for x in api.list_repo_files(repo,repo_type='model')}
58
+ for row in remote_rows:
59
+ item=remote.get(row['path'])
60
+ if item is None or item.size!=row['size']: raise ValueError(f'Remote size mismatch: {row["path"]}')
61
+ if item.sha256 and item.sha256!=row['sha256']: raise ValueError('Remote SHA256 mismatch')
62
+ record=dict(repo_id=repo,url=f'https://modelscope.cn/models/{repo}',
63
+ verified_files=len(remote_rows),remote_verification='file sizes and available SHA256')
64
+ write_json(Path('artifacts')/f'publication-{args.platform}.json',record)
65
+ print(json.dumps(record,indent=2))
66
+
67
+ if __name__=='__main__': main()
scripts/verify_cache.py ADDED
@@ -0,0 +1,42 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Validate native text-prefix caching against the exact uncached transformer."""
2
+ import argparse
3
+ import json
4
+ import time
5
+ from pathlib import Path
6
+ import mlx.core as mx
7
+ from mlx_vlm.models.qwen_image.weights import load_transformer
8
+ from mlx_vlm.models.qwen_image.kv_cache import QwenImageKVCache
9
+ from scripts.common import write_json
10
+
11
+ def main():
12
+ ap=argparse.ArgumentParser()
13
+ ap.add_argument('--model',type=Path,required=True)
14
+ ap.add_argument('--output',type=Path,required=True)
15
+ args=ap.parse_args()
16
+ model=load_transformer(args.model); mx.eval(model.parameters())
17
+ mx.random.seed(20260926)
18
+ emb=mx.random.normal((1,48,4096)).astype(mx.bfloat16)
19
+ latent=mx.random.normal((1,1024,64)).astype(mx.bfloat16)
20
+ cache=QwenImageKVCache(len(model.transformer_blocks))
21
+ kw=dict(encoder_hidden_states=emb,img_shape=(1,32,32))
22
+ init=model(hidden_states=latent,timestep=mx.array([0.9],dtype=mx.bfloat16),kv_cache=cache,kv_cache_mode='extract',**kw)
23
+ mx.eval(init,cache.arrays())
24
+ rows=[]
25
+ for t in (0.7,0.3):
26
+ latent=mx.random.normal((1,1024,64)).astype(mx.bfloat16)
27
+ timestep=mx.array([t],dtype=mx.bfloat16)
28
+ start=time.perf_counter(); expected=model(hidden_states=latent,timestep=timestep,**kw); mx.eval(expected)
29
+ uncached=time.perf_counter()-start
30
+ start=time.perf_counter(); actual=model(hidden_states=latent,timestep=timestep,kv_cache=cache,kv_cache_mode='cached',**kw); mx.eval(actual)
31
+ cached=time.perf_counter()-start
32
+ delta=actual.astype(mx.float32)-expected.astype(mx.float32)
33
+ rel=float(mx.sqrt(mx.mean(delta**2)/mx.mean(expected.astype(mx.float32)**2)).item())
34
+ row=dict(timestep=t,max_abs_error=float(mx.max(mx.abs(delta)).item()),relative_rmse=rel,
35
+ uncached_seconds=uncached,cached_seconds=cached)
36
+ rows.append(row); print(row,flush=True)
37
+ # Kernel splitting can change BF16 rounding; guard against material drift.
38
+ if rel>0.01: raise ValueError(f'KV cache changes predictions materially: {rel}')
39
+ write_json(args.output,dict(model=str(args.model),passed=True,rows=rows,
40
+ note='Real transformer with synthetic conditioning/latents; BF16 numerical tolerance, not pixel equality.'))
41
+
42
+ if __name__=='__main__': main()
text_encoder/config.json ADDED
@@ -0,0 +1,71 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3VLForConditionalGeneration"
4
+ ],
5
+ "dtype": "bfloat16",
6
+ "image_token_id": 151655,
7
+ "model_type": "qwen3_vl",
8
+ "text_config": {
9
+ "attention_bias": false,
10
+ "attention_dropout": 0.0,
11
+ "bos_token_id": 151643,
12
+ "dtype": "bfloat16",
13
+ "eos_token_id": 151645,
14
+ "head_dim": 128,
15
+ "hidden_act": "silu",
16
+ "hidden_size": 4096,
17
+ "initializer_range": 0.02,
18
+ "intermediate_size": 12288,
19
+ "max_position_embeddings": 262144,
20
+ "model_type": "qwen3_vl_text",
21
+ "num_attention_heads": 32,
22
+ "num_hidden_layers": 36,
23
+ "num_key_value_heads": 8,
24
+ "rms_norm_eps": 1e-06,
25
+ "rope_scaling": {
26
+ "mrope_interleaved": true,
27
+ "mrope_section": [
28
+ 24,
29
+ 20,
30
+ 20
31
+ ],
32
+ "rope_type": "default"
33
+ },
34
+ "rope_theta": 5000000,
35
+ "use_cache": true,
36
+ "vocab_size": 151936
37
+ },
38
+ "tie_word_embeddings": false,
39
+ "transformers_version": "4.57.1",
40
+ "video_token_id": 151656,
41
+ "vision_config": {
42
+ "deepstack_visual_indexes": [
43
+ 8,
44
+ 16,
45
+ 24
46
+ ],
47
+ "depth": 27,
48
+ "dtype": "bfloat16",
49
+ "hidden_act": "gelu_pytorch_tanh",
50
+ "hidden_size": 1152,
51
+ "in_channels": 3,
52
+ "initializer_range": 0.02,
53
+ "intermediate_size": 4304,
54
+ "model_type": "qwen3_vl",
55
+ "num_heads": 16,
56
+ "num_position_embeddings": 2304,
57
+ "out_hidden_size": 4096,
58
+ "patch_size": 16,
59
+ "spatial_merge_size": 2,
60
+ "temporal_patch_size": 2
61
+ },
62
+ "vision_end_token_id": 151653,
63
+ "vision_start_token_id": 151652,
64
+ "quantization": {
65
+ "bits": 8,
66
+ "group_size": 64,
67
+ "mode": "affine"
68
+ },
69
+ "mlx_format": true,
70
+ "_modification_notice": "Modified by ixim / iximbox for Image21-MLX: converted from the pinned BF16 source to MLX layout; eligible linear weights use groupwise affine quantization. See conversion.json for precision and exceptions. Built with Qwen."
71
+ }
text_encoder/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
transformer/config.json ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "QwenImage21Transformer2DModel",
3
+ "_diffusers_version": "0.37.0.dev0",
4
+ "attention_head_dim": 128,
5
+ "axes_dims_rope": [
6
+ 16,
7
+ 56,
8
+ 56
9
+ ],
10
+ "context_in_dim": 4096,
11
+ "in_channels": 64,
12
+ "num_attention_heads": 32,
13
+ "num_layers": 32,
14
+ "out_channels": 64,
15
+ "patch_size": 1,
16
+ "mlp_ratio": 3,
17
+ "eps": 1e-06,
18
+ "causal_condition": true,
19
+ "quantization": {
20
+ "bits": 8,
21
+ "group_size": 64,
22
+ "mode": "affine"
23
+ },
24
+ "mlx_format": true,
25
+ "_modification_notice": "Modified by ixim / iximbox for Image21-MLX: converted from the pinned BF16 source to MLX layout; eligible linear weights use groupwise affine quantization. See conversion.json for precision and exceptions. Built with Qwen."
26
+ }