mlboydaisuke commited on
Commit
a9b5c94
·
verified ·
1 Parent(s): f37625e

RGBA-Image-2.1: Core AI port of Qwen-Image-2.1 (macOS 27, bf16 DiT + w16a32 encoder + fp32 RGBA VAE)

Browse files
.gitattributes CHANGED
@@ -33,3 +33,15 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ host/rope_axis0_cos.f32 filter=lfs diff=lfs merge=lfs -text
37
+ host/rope_axis0_sin.f32 filter=lfs diff=lfs merge=lfs -text
38
+ host/rope_axis1_cos.f32 filter=lfs diff=lfs merge=lfs -text
39
+ host/rope_axis1_sin.f32 filter=lfs diff=lfs merge=lfs -text
40
+ host/rope_axis2_cos.f32 filter=lfs diff=lfs merge=lfs -text
41
+ host/rope_axis2_sin.f32 filter=lfs diff=lfs merge=lfs -text
42
+ qi21_dit_full_bf16_dyn_iofp32.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
43
+ qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
44
+ qi21_vae_1024_fp32.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
45
+ qi21_vae_256_fp32.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
46
+ qi21_vae_512_fp32.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
47
+ tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Qwen RESEARCH LICENSE AGREEMENT
2
+
3
+ Qwen RESEARCH LICENSE AGREEMENT Release Date: September 20, 2026
4
+
5
+ By clicking to agree or by using or distributing any portion or element of the Qwen Materials, you will be deemed to have recognized and accepted the content of this Agreement, which is effective immediately.
6
+
7
+ 1. Definitions
8
+ a. This Qwen RESEARCH LICENSE AGREEMENT (this "Agreement") shall mean the terms and conditions for use, reproduction, distribution and modification of the Materials as defined by this Agreement.
9
+ b. "We" (or "Us") shall mean Hangzhou Tongyi Laboratory Technology Co., Ltd.
10
+ c. "You" (or "Your") shall mean a natural person or legal entity exercising the rights granted by this Agreement and/or using the Materials for any purpose and in any field of use.
11
+ d. "Third Parties" shall mean individuals or legal entities that are not under common control with us or you.
12
+ e. "Qwen" shall mean the large language models, diffusion models, and software and algorithms, consisting of trained model weights, parameters (including optimizer states), machine-learning model code, inference-enabling code, training-enabling code, fine-tuning enabling code and other elements of the foregoing distributed by us.
13
+ f. "Materials" shall mean, collectively, our proprietary Qwen and Documentation (and any portion thereof) made available under this Agreement.
14
+ g. "Source" form shall mean the preferred form for making modifications, including but not limited to model source code, documentation source, and configuration files.
15
+ h. "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
16
+ i. "Non-Commercial" shall mean for research or evaluation purposes only.
17
+
18
+ 2. Grant of Rights
19
+ a. You are granted a non-exclusive, worldwide, non-transferable and royalty-free limited license under our intellectual property or other rights owned by us embodied in the Materials to use, reproduce, distribute, copy, create derivative works of, and make modifications to the Materials FOR NON-COMMERCIAL PURPOSES ONLY.
20
+ b. You shall not use the Materials for any commercial purpose without obtaining a separate commercial license from us. If you wish to use the Materials commercially, you shall request a license from us at model-business@notice.qwencloud.com.
21
+
22
+ 3. Redistribution
23
+ Subject to Section 2 (Grant of Rights), you may distribute copies or make the Materials, or derivative works thereof, available as part of a product or service that contains any of them, with or without modifications, and in Source or Object form, provided that you meet the following conditions:
24
+ a. You shall give any other recipients of the Materials or derivative works a copy of this Agreement;
25
+ b. You shall cause any modified files to carry prominent notices stating that you changed the files;
26
+ c. You shall retain in all copies of the Materials that you distribute the following attribution notices within a "Notice" text file distributed as a part of such copies: "Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved."; and
27
+ d. You may add your own copyright statement to your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of your modifications, or for any such derivative works as a whole, provided your use, reproduction, and distribution of the work otherwise complies with the terms and conditions of this Agreement.
28
+
29
+ 4. Rules of use
30
+ a. The Materials may be subject to export controls or restrictions in China, the United States or other countries or regions. You shall comply with applicable laws and regulations in your use of the Materials.
31
+ b. If you use the Materials or any outputs or results therefrom to create, train, fine-tune, or improve an AI model that is distributed or made available, you shall prominently display “Built with Qwen” or “Improved using Qwen” in the related product documentation.
32
+ c. You shall not use "Qwen" as the primary name or identifier of any derivative works or products; reasonable descriptive use (e.g., "fine-tuned from Qwen Image") is permitted.
33
+
34
+ 5. Intellectual Property
35
+ a. We retain ownership of all intellectual property rights in and to the Materials and derivatives made by or for us. Conditioned upon compliance with the terms and conditions of this Agreement, with respect to any derivative works and modifications of the Materials that are made by you, you are and will be the owner of such derivative works and modifications.
36
+ b. No trademark license is granted to use the trade names, trademarks, service marks, or product names of us, except as required to fulfill notice requirements under this Agreement or as required for reasonable and customary use in describing and redistributing the Materials.
37
+ c. If you commence a lawsuit or other proceedings (including a cross-claim or counterclaim in a lawsuit) against us or any entity alleging that the Materials or any output therefrom, or any part of the foregoing, infringe any intellectual property or other right owned or licensable by you, then all licenses granted to you under this Agreement shall terminate as of the date such lawsuit or other proceeding is commenced or brought.
38
+
39
+ 6. Disclaimer of Warranty and Limitation of Liability
40
+ a. We are not obligated to support, update, provide training for, or develop any further version of the Qwen Materials or to grant any license thereto.
41
+ b. THE MATERIALS ARE PROVIDED "AS IS" WITHOUT ANY EXPRESS OR IMPLIED WARRANTY OF ANY KIND INCLUDING WARRANTIES OF MERCHANTABILITY, NONINFRINGEMENT, OR FITNESS FOR A PARTICULAR PURPOSE. WE MAKE NO WARRANTY AND ASSUME NO RESPONSIBILITY FOR THE SAFETY OR STABILITY OF THE MATERIALS AND ANY OUTPUT THEREFROM.
42
+ c. IN NO EVENT SHALL WE BE LIABLE TO YOU FOR ANY DAMAGES, INCLUDING, BUT NOT LIMITED TO ANY DIRECT, OR INDIRECT, SPECIAL OR CONSEQUENTIAL DAMAGES ARISING FROM YOUR USE OR INABILITY TO USE THE MATERIALS OR ANY OUTPUT OF IT, NO MATTER HOW IT’S CAUSED.
43
+ d. You will defend, indemnify and hold harmless us from and against any claim by any third party arising out of or related to your use or distribution of the Materials.
44
+
45
+ 7. Survival and Termination.
46
+ a. The term of this Agreement shall commence upon your acceptance of this Agreement or access to the Materials and will continue in full force and effect until terminated in accordance with the terms and conditions herein.
47
+ b. We may terminate this Agreement if you breach any of the terms or conditions of this Agreement. Upon termination of this Agreement, you must delete and cease use of the Materials. Sections 6 and 8 shall survive the termination of this Agreement.
48
+
49
+ 8. Governing Law and Jurisdiction.
50
+ a. This Agreement and any dispute arising out of or relating to it will be governed by the laws of China, without regard to conflict of law principles, and the UN Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
51
+ b. The People's Courts in Hangzhou City shall have exclusive jurisdiction over any dispute arising out of this Agreement.
52
+
53
+ 9. Other Terms and Conditions.
54
+ a. Any arrangements, understandings, or agreements regarding the Material not stated herein are separate from and independent of the terms and conditions of this Agreement. You shall request a separate license from us, if you use the Materials in ways not expressly agreed to in this Agreement.
55
+ b. We shall not be bound by any additional or different terms or conditions communicated by you unless expressly agreed.
NOTICE ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved.
2
+
3
+ The model files in this repository were converted and re-authored from the Qwen-Image-2.1 weights
4
+ (Qwen/Qwen-Image-2.1, revision 790c92633540aa0cb11d9abf19eb46d861714758) into Apple Core AI
5
+ .aimodel graphs. They were changed as follows.
6
+
7
+ qi21_dit_full_bf16_dyn_iofp32.aimodel — the diffusion transformer (from transformer/)
8
+ - Re-authored in plain PyTorch with the checkpoint's parameter names, then exported as one graph.
9
+ - Block-causal attention is computed as two scaled-dot-product-attention calls per block
10
+ (text tokens attend causally to text; image tokens attend to every token).
11
+ - Rotary position embeddings are applied as real-valued cos/sin pairs that the host supplies.
12
+ - The prefix key/value cache is removed: the text prefix is recomputed at every step.
13
+ - Weights and computation are bfloat16; every graph input and output is float32.
14
+ - Both sequence axes are dynamic (text 8-512 tokens, image 64-4096 tokens). Text-to-image only:
15
+ the graph has no condition-image input.
16
+
17
+ qi21_encoder_dynL_w16a32_ids_iofp32.aimodel — the text encoder (from text_encoder/)
18
+ - Only the text path of the Qwen3-VL-8B encoder is kept. The vision tower, the language-model head
19
+ and the final RMSNorm are removed; the output is the residual stream after the last layer.
20
+ - Re-authored in plain PyTorch; the token embedding is inside the graph (int32 token ids in).
21
+ - Weights are stored in bfloat16 and all computation is float32; the output is float32.
22
+ - The sequence axis is dynamic (16-512 tokens).
23
+
24
+ qi21_vae_256_fp32.aimodel, qi21_vae_512_fp32.aimodel, qi21_vae_1024_fp32.aimodel — the VAE (from vae/)
25
+ - Decoder only, one frame; the VAE encoder is not included.
26
+ - The latent un-normalisation (latents * std + mean) and the token-to-grid unpacking are inside
27
+ the graph.
28
+ - One graph per output size (256x256, 512x512, 1024x1024), float32.
29
+
30
+ Copied without changes: LICENSE; config.json (= transformer/config.json); tokenizer/ (= processor/
31
+ tokenizer.json, tokenizer_config.json, vocab.json, merges.txt, special_tokens_map.json,
32
+ added_tokens.json, chat_template.jinja).
33
+
34
+ Added by the conversion: host/ (rotary tables and sampler constants computed from the model
35
+ definition, and host_contract.md) and README.md.
README.md ADDED
@@ -0,0 +1,248 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: qwen-research
4
+ license_link: LICENSE
5
+ base_model: Qwen/Qwen-Image-2.1
6
+ pipeline_tag: text-to-image
7
+ library_name: coreai
8
+ tags:
9
+ - coreai
10
+ - rgba
11
+ - text-to-image
12
+ - apple-silicon
13
+ ---
14
+
15
+ # RGBA-Image-2.1 — Core AI port of Qwen-Image-2.1
16
+
17
+ **Built with Qwen.**
18
+
19
+ **Non-commercial use only** — Qwen Research License (research or evaluation purposes).
20
+
21
+ **macOS 27 on Apple silicon only.** Five Core AI bundles, 32.41 GB (30.18 GiB) in total.
22
+
23
+ This is [`Qwen/Qwen-Image-2.1`](https://huggingface.co/Qwen/Qwen-Image-2.1) (revision `790c926`)
24
+ converted to Core AI `.aimodel` graphs that run on the Mac GPU. A Qwen3-VL-8B text encoder (text
25
+ path only) conditions a 7B single-stream DiT (32 blocks, block-causal attention). A 64-channel VAE
26
+ decodes to four channels, RGBA. The sampler is the pipeline default: 40 FlowMatch Euler steps, no
27
+ CFG. Ask for a transparent background in the prompt and the fourth channel is a real alpha.
28
+
29
+ Sizes: 256², 512² and 1024², square. The DiT and the text encoder have dynamic axes (see Graph
30
+ contracts); the VAE has one graph per size. The upstream default of 2048² needs 16,384 image
31
+ tokens, outside this DiT's 64–4096-token axis.
32
+
33
+ ## Bundle
34
+
35
+ [🤗 mlboydaisuke/RGBA-Image-2.1-CoreAI](https://huggingface.co/mlboydaisuke/RGBA-Image-2.1-CoreAI)
36
+
37
+ | file | what it is | size |
38
+ | --- | --- | --- |
39
+ | `qi21_dit_full_bf16_dyn_iofp32.aimodel` | the DiT: bf16 weights and compute, fp32 inputs and outputs | 14.23 GB (13.25 GiB) |
40
+ | `qi21_encoder_dynL_w16a32_ids_iofp32.aimodel` | the text encoder: bf16 weights, fp32 compute; token ids in, `embed_tokens` inside | 15.14 GB (14.10 GiB) |
41
+ | `qi21_vae_{256,512,1024}_fp32.aimodel` | the VAE decoder, fp32, one per size | 1.01 GB (0.94 GiB) each |
42
+ | `host/` | per-axis RoPE tables, `scheduler.json`, [`host_contract.md`](https://huggingface.co/mlboydaisuke/RGBA-Image-2.1-CoreAI/blob/main/host/host_contract.md) | |
43
+ | `tokenizer/` | the source's `processor/` tokenizer files, unchanged | |
44
+ | `config.json` | the source's `transformer/config.json`, unchanged | |
45
+ | `LICENSE`, `NOTICE` | the Qwen Research License and the change notice | |
46
+
47
+ No quantized variant ships. The int8 DiT does not compile on the one GPU path that works (next
48
+ paragraph). int8 weights break the text encoder's `<|im_start|>` tokens (Lessons, 2). Compiled, the int8
49
+ encoder is also 34.32 GiB; this one is 14.10 GiB.
50
+
51
+ **Compile before you run.** On macOS 27.0 (26A428) the DiT crashes in the Python runtime when its
52
+ `.aimodel` is compiled just-in-time. What runs is an ahead-of-time compile with
53
+ `--expect-frequent-reshapes` (step 2 below). The compiled copies need their own disk: 27 GB for
54
+ the DiT and 14.10 GiB for the encoder.
55
+
56
+ ## Use it
57
+
58
+ There is no Swift app for this model yet. The Python engine
59
+ [`conversion/qwenimage21/pipeline_engine.py`](https://github.com/john-rocky/coreai-model-zoo/blob/main/conversion/qwenimage21/pipeline_engine.py)
60
+ runs the whole loop on the three bundles: tokenize, encode, 40 DiT steps, decode, write a PNG.
61
+
62
+ ```
63
+ # 1. the bundles, and the tokenizer the engine reads from the source repo's processor/
64
+ hf download mlboydaisuke/RGBA-Image-2.1-CoreAI --local-dir RGBA-Image-2.1-CoreAI
65
+ hf download Qwen/Qwen-Image-2.1 --revision 790c92633540aa0cb11d9abf19eb46d861714758 --include "processor/*"
66
+
67
+ # 2. compile for the Mac GPU (once per bundle; the gates used --architecture h16c on an M4 Max)
68
+ cd RGBA-Image-2.1-CoreAI
69
+ for m in qi21_encoder_dynL_w16a32_ids_iofp32 qi21_dit_full_bf16_dyn_iofp32 qi21_vae_512_fp32; do
70
+ xcrun coreai-build compile $m.aimodel --output aot/$m --platform macOS --architecture h16c \
71
+ --preferred-compute gpu --expect-frequent-reshapes
72
+ done
73
+ A=$PWD/aot
74
+
75
+ # 3. generate, from a checkout of the zoo
76
+ # (Python with coreai-core 1.0.0b2, torch, numpy, tokenizers, pillow)
77
+ cd /path/to/coreai-model-zoo/conversion/qwenimage21
78
+ python pipeline_engine.py \
79
+ --prompt "This is an RGBA image with transparency. A cute cartoon dragon sticker. The image has alpha channel and the background is transparent." \
80
+ --size 512 --seed 42 --tag dragon \
81
+ --encoder $A/qi21_encoder_dynL_w16a32_ids_iofp32/qi21_encoder_dynL_w16a32_ids_iofp32.h16c.aimodelc \
82
+ --dit $A/qi21_dit_full_bf16_dyn_iofp32/qi21_dit_full_bf16_dyn_iofp32.h16c.aimodelc \
83
+ --vae $A/qi21_vae_512_fp32/qi21_vae_512_fp32.h16c.aimodelc
84
+ # -> _work/samples/dragon.png (RGBA) and dragon_rgb.png (composited on white)
85
+ ```
86
+
87
+ The prompt above is the upstream card's recommended form for transparent images: "This is an RGBA
88
+ image with transparency. {subject}. The image has alpha channel and the background is
89
+ transparent." The noise is `torch.randn` on a CPU generator seeded with `--seed`, the same draw the
90
+ reference pipeline makes with a CPU generator and that seed.
91
+
92
+ To check the port against the fp32 reference, record the reference once and run the engine in
93
+ oracle mode. `capture_oracle.py` runs the diffusers-main pipeline in fp32 on the CPU, about 2 min
94
+ at 256². It needs the full `Qwen/Qwen-Image-2.1` snapshot and a venv with diffusers main 4295ee3
95
+ and transformers 5.17.
96
+
97
+ ```
98
+ python capture_oracle.py --size 256 --steps 40 # -> oracle/256/
99
+ python pipeline_engine.py --oracle oracle/256 \
100
+ --encoder $A/qi21_encoder_dynL_w16a32_ids_iofp32/qi21_encoder_dynL_w16a32_ids_iofp32.h16c.aimodelc \
101
+ --dit $A/qi21_dit_full_bf16_dyn_iofp32/qi21_dit_full_bf16_dyn_iofp32.h16c.aimodelc \
102
+ --vae $A/qi21_vae_256_fp32/qi21_vae_256_fp32.h16c.aimodelc
103
+ # prints the latent corr vs the reference after every step, then the RGBA and white-composited PSNR
104
+ ```
105
+
106
+ ## Which image model should I use?
107
+
108
+ The zoo's Mac text-to-image models are not ranked; pick by trade-off. RGBA-Image-2.1 writes RGBA
109
+ natively, so it fits images that need an alpha channel.
110
+
111
+ | | params | sampler | time @1024 | precision |
112
+ | --- | --- | --- | --- | --- |
113
+ | **FLUX.2 klein** | 4B | 4 steps, guidance-distilled (no CFG) | ~17 s | int4 |
114
+ | **Z-Image-Turbo** | 6B | 8 steps + CFG (16 forwards) | ~70 s | bf16, near-lossless |
115
+ | **GLM-Image** | 16B (9B AR + 7B DiT) | AR prior + 20-step DiT | ~208 s | int8 |
116
+ | **RGBA-Image-2.1** | 7B | 40 steps, no CFG | ~190 s | bf16 |
117
+
118
+ Times are the ones each card reports, on an M4 Max. For RGBA-Image-2.1 it is 40 DiT steps at the
119
+ warm median of 4.757 s per forward. It leaves out the encoder, the VAE and loading.
120
+
121
+ ## Graph contracts
122
+
123
+ Every graph has one function, `main`, and fp32 inputs and outputs except `input_ids`.
124
+
125
+ | graph | inputs | output |
126
+ | --- | --- | --- |
127
+ | encoder | `input_ids [1,Lfull]` int32, `Lfull` 16..512 | `hidden [1,Lfull,4096]`: the last layer's residual stream, before the final norm |
128
+ | DiT | `img_tokens [1,N,64]`, `txt_feats [1,L,4096]`, `timestep [1]`, `txt_cos`/`txt_sin [1,L,64]`, `img_cos`/`img_sin [1,N,64]`; `L` 8..512, `N` 64..4096 | `vel [1,N,64]` |
129
+ | VAE | `latents_packed [1,N,64]`, the sampler's latent unchanged | `image [1,4,S,S]`, RGBA in [-1, 1] |
130
+
131
+ - **Prompt.** One tokenization of the text-to-image chat template around the prompt, no padding.
132
+ The DiT reads `hidden[:, drop_idx:]`. `drop_idx` is the token count of the template's system
133
+ part: 14 with this tokenizer. Compute it; do not hard-code it.
134
+ - **Sequence.** `[text L | image N]`, image tokens in raster order, one token per 16×16 px tile
135
+ (256² → 256 tokens, 512² → 1024, 1024² → 4096). There is no 2×2 latent packing.
136
+ - **RoPE.** Text token `i` sits at `(i, i, i)`. Image token `(y, x)` sits at `(L, y − (h − h//2),
137
+ x − (w − w//2))`. `host/` has the per-axis tables for positions −1024..8191.
138
+ - **Sampler.** Shifted sigmas: `μ` from the image-token count, exponential shift, terminal stretch
139
+ to 0.02. The DiT's `timestep` input is `timesteps[i] / 1000` in fp32 (σ to within 1 ulp), and each step is `x += (σᵢ₊₁ − σᵢ)·v`.
140
+ - **VAE.** The graph unpacks the tokens and applies `latents·std + mean` itself. Feed it the raw
141
+ sampler latent.
142
+
143
+ The full contract, with the formulas a Swift host needs:
144
+ [`host/host_contract.md`](https://huggingface.co/mlboydaisuke/RGBA-Image-2.1-CoreAI/blob/main/host/host_contract.md).
145
+
146
+ ## Measured
147
+
148
+ M4 Max, macOS 27.0 (26A428). Bundles compiled ahead of time with `--expect-frequent-reshapes`,
149
+ run with `SpecializationOptions.default()`.
150
+
151
+ **DiT speed** (`bench_dit.py`, text L = 40, GPU lock held, load average ~10):
152
+
153
+ | size | image tokens | first call | s/forward (warm median of 5) | 40 steps |
154
+ | --- | --- | --- | --- | --- |
155
+ | 256² | 256 | 2.94 s | 0.343 s | 13.7 s |
156
+ | 512² | 1024 | 1.14 s | 1.104 s | 44 s |
157
+ | 1024² | 4096 | 4.77 s | 4.757 s | 190 s |
158
+
159
+ **Text encoder:** 0.181 s per call at 32 tokens and 0.271 s at 128 (warm median). The first call
160
+ takes 0.37 s; loading takes 19.6 s.
161
+
162
+ **One 256² image end to end:** encoder 1.15 s, DiT 40 steps 45 s (the first step 32 s, then
163
+ 0.34 s per step), VAE 0.16 s.
164
+
165
+ **Memory.** A 512² run with all three compiled bundles loaded in one process (`pipeline_engine.py`,
166
+ free prompt): peak resident set 58.3 GB (`/usr/bin/time -l`), wall 108 s of which loading is 43 s,
167
+ the DiT 40 steps 63 s (first step 21 s, then 1.08 s), encoder 0.5 s, VAE 0.4 s.
168
+
169
+ **Fidelity** against the fp32 diffusers reference (prompt "a red apple on a wooden table, studio
170
+ lighting", seed 1234, the same noise, 40 steps):
171
+
172
+ | size | white-composited RGB PSNR | alpha max\|Δ\| | final latent corr |
173
+ | --- | --- | --- | --- |
174
+ | 256² | 46.72 dB (46.7–50.4 over 5 runs) | 1/255 | 0.999987 |
175
+ | 512² | 35.49 dB (23.6–43.4 over 6 runs, 4 of them ≥ 30 dB) | 2/255 | 0.999559 |
176
+
177
+ The ranges are over runs whose prompt embeddings differ at the 1e-5 level.
178
+
179
+ **The model's own bf16 band.** The official pipeline run in bf16 (diffusers main, MPS), scored
180
+ against the same fp32 reference:
181
+
182
+ | size | official pipeline in bf16 | this port |
183
+ | --- | --- | --- |
184
+ | 256² | 43.47 dB, final latent corr 0.999947 | 46.72 dB, 0.999987 |
185
+ | 512² | 33.19 dB, 0.999053 | 35.49 dB, 0.999559 |
186
+
187
+ **Gates:**
188
+
189
+ - DiT re-authored in plain PyTorch vs diffusers (fp32, 2 random layers): max|Δ| 7.2e-7. With the
190
+ real weights in fp32, teacher-forced on all 40 steps: corr 1.000000000.
191
+ - DiT bundle (bf16, GPU), teacher-forced against the fp32 reference: 40/40 steps corr ≥ 0.999854
192
+ at 256² (NaN 0) and ≥ 0.999844 at 512².
193
+ - Text encoder re-authored in fp32: bit-exact with the reference on all 32 tokens. The bundle: min
194
+ per-token corr 0.999999999 on the reference prompt, and ≥ 0.99999998 over 3 prompts × 7 lengths.
195
+ The host tokenizer gives the processor's ids on all 3 prompts, `drop_idx` 14.
196
+ - VAE bundles on all four channels: corr 1.0000000, max|Δ| 1.35e-5 at 256² and 1.29e-5 at 512²;
197
+ 2.33e-5 at 1024² against torch on a synthetic latent.
198
+ - Sampler: sigmas, timesteps and all 40 Euler steps bit-exact at 256² and 512².
199
+ - Transparency: the dragon prompt above at 512², seed 42: alpha min 0, mean 126; the background
200
+ corners average 0.96/255.
201
+
202
+ ## Lessons
203
+
204
+ 1. **A 2-layer probe does not clear a 32-layer graph.** On 26A428 the Python runtime puts a Neural
205
+ Engine region inside the 32-block bf16 DiT, and the ANE inference fails (`Code=-19`). It fails
206
+ under JIT and under a plain AOT compile; the 2-layer probe never triggered it. The GPU path that
207
+ runs is AOT with `--expect-frequent-reshapes`: 2 min 9 s to compile, 27 GB, MPSGraph delegates
208
+ only. The int8 DiT does not compile with that flag (`Pass failed: MPSMemrefAllocFusion`).
209
+ 2. **The encoder's `<|im_start|>` tokens need fp32 compute, not fp32 storage.** Token 14, the `<|im_start|>`
210
+ that opens the user turn, is the first token the DiT reads. Its residual grows to |h| ≈ 9,100 in
211
+ layers 17–34, and the last two layers cancel it to ≈ 100. bf16 cannot hold that cancellation:
212
+ per-token corr 0.968 in torch bf16, 0.976 on the engine. An fp32 residual stream alone did not
213
+ hold across prompts and lengths. Full fp32 compute over bf16-stored weights did: min token corr
214
+ 0.999999999 at the same 14.10 GiB. The DiT barely notices the bf16 error (velocity corr
215
+ ≥ 0.99998); only a per-token gate catches it.
216
+ 3. **Judge a bf16 port by the model's own bf16 band, and look at the image when PSNR drops.** At
217
+ 512², six runs whose prompt embeddings differ at the 1e-5 level span 23.6–43.4 dB. The two runs
218
+ near 24 dB show the same apple with one extra leaf on the stem: a semantic fork, not noise. The
219
+ fp32 reference stays at 82 dB under a 1e-5 perturbation, so the fork comes from the bf16 DiT's
220
+ per-step error (0.4–1.8 %). The official pipeline in bf16 scores 33.19 dB at 512²; this port's
221
+ 35.49 dB is in that band.
222
+
223
+ Port notes, every gate and the dead ends:
224
+ [`knowledge/qwenimage21-port.md`](https://github.com/john-rocky/coreai-model-zoo/blob/main/knowledge/qwenimage21-port.md).
225
+ Scripts: [`conversion/qwenimage21/`](https://github.com/john-rocky/coreai-model-zoo/tree/main/conversion/qwenimage21).
226
+
227
+ ## Licence
228
+
229
+ The weights are Qwen Materials under the **Qwen RESEARCH LICENSE AGREEMENT** (release date
230
+ 2026-09-20), included unchanged as
231
+ [`LICENSE`](https://huggingface.co/mlboydaisuke/RGBA-Image-2.1-CoreAI/blob/main/LICENSE). A summary
232
+ follows; the Agreement is what binds.
233
+
234
+ - **§2, non-commercial only.** You may use, copy, modify and redistribute the Materials for
235
+ research or evaluation purposes only. Commercial use needs a separate licence from Hangzhou
236
+ Tongyi Laboratory Technology Co., Ltd.; §2(b) gives the contact.
237
+ - **§3, redistribution,** under three conditions: every recipient gets a copy of the Agreement
238
+ (`LICENSE`); modified files carry a notice that they were changed (`NOTICE` lists every change);
239
+ and copies keep the attribution text below in a "Notice" file (`NOTICE`, first line).
240
+ - **§4(b).** An AI model created from the Materials and made available displays "Built with Qwen"
241
+ prominently in its documentation. This card does, at the top.
242
+ - **§4(c).** "Qwen" is not the primary name of a derivative. This port is named RGBA-Image-2.1;
243
+ "Core AI port of Qwen-Image-2.1" is the descriptive use the Agreement permits.
244
+
245
+ [`NOTICE`](https://huggingface.co/mlboydaisuke/RGBA-Image-2.1-CoreAI/blob/main/NOTICE) begins with
246
+ the attribution text §3(c) requires:
247
+
248
+ > Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved.
config.json ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "QwenImage21Transformer2DModel",
3
+ "_diffusers_version": "0.37.0.dev0",
4
+ "attention_head_dim": 128,
5
+ "axes_dims_rope": [
6
+ 16,
7
+ 56,
8
+ 56
9
+ ],
10
+ "context_in_dim": 4096,
11
+ "in_channels": 64,
12
+ "num_attention_heads": 32,
13
+ "num_layers": 32,
14
+ "out_channels": 64,
15
+ "patch_size": 1,
16
+ "mlp_ratio": 3,
17
+ "eps": 1e-06,
18
+ "causal_condition": true
19
+ }
host/host_contract.md ADDED
@@ -0,0 +1,153 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Host contract — RGBA-Image-2.1 (Core AI port of Qwen-Image-2.1)
2
+
3
+ Everything a host computes around the three graphs, for text-to-image, batch 1, no CFG.
4
+ The Python reference of every step below is in the zoo's
5
+ [`conversion/qwenimage21/`](https://github.com/john-rocky/coreai-model-zoo/tree/main/conversion/qwenimage21):
6
+ `qi21_tokenize.py` (tokenizer), `qi21_host.py` (RoPE, pack/unpack), `qi21_sched.py` (sampler),
7
+ `pipeline_engine.py` (the whole loop on the three bundles). The first three match the diffusers-main
8
+ pipeline (`Qwen/Qwen-Image-2.1` @ `790c926`) exactly: the same token ids, bit-identical RoPE tables,
9
+ bit-identical sigmas and Euler steps. `pipeline_engine.py` is scored against the pipeline's fp32 images.
10
+
11
+ ## 1. The graphs
12
+
13
+ Every graph has one function, `main`. All tensors crossing a graph boundary are fp32, except
14
+ `input_ids` (int32).
15
+
16
+ | bundle | inputs | output | axes |
17
+ | --- | --- | --- | --- |
18
+ | `qi21_encoder_dynL_w16a32_ids_iofp32.aimodel` | `input_ids [1,Lfull]` int32 | `hidden [1,Lfull,4096]` | `Lfull` 16..512 |
19
+ | `qi21_dit_full_bf16_dyn_iofp32.aimodel` | `img_tokens [1,N,64]`, `txt_feats [1,L,4096]`, `timestep [1]`, `txt_cos [1,L,64]`, `txt_sin [1,L,64]`, `img_cos [1,N,64]`, `img_sin [1,N,64]` | `vel [1,N,64]` | `L` 8..512, `N` 64..4096 |
20
+ | `qi21_vae_{256,512,1024}_fp32.aimodel` | `latents_packed [1,N,64]` | `image [1,4,S,S]` | fixed: `N = (S/16)²` |
21
+
22
+ - Encoder: the Qwen3-VL-8B text stack (36 layers). bf16 weights, fp32 compute. `embed_tokens` is
23
+ inside the graph. The output is the residual stream after the last layer, **before** the final
24
+ RMSNorm. Do not apply a norm on the host.
25
+ - DiT: 32 blocks, bf16 weights and compute, fp32 boundary.
26
+ - VAE: decoder only, fp32. The graph unpacks the tokens and applies `latents * std + mean` itself.
27
+
28
+ ## 2. Tokenize
29
+
30
+ Tokenizer: `tokenizer/tokenizer.json` (Qwen2 BPE, byte-level). No BOS token. An empty prompt is
31
+ replaced by a single space `" "`.
32
+
33
+ Template (text-to-image):
34
+
35
+ ```
36
+ <|im_start|>system\nComprehend and analyze the provided prompt.<|im_end|>\n<|im_start|>user\n{prompt}<|im_end|>\n<|im_start|>assistant\n
37
+ ```
38
+
39
+ `\n` is a newline character. Encode the whole string once, with no padding and no truncation, to get
40
+ `ids` (length `Lfull`).
41
+
42
+ **`drop_idx`** is the number of tokens of the system part alone,
43
+ `<|im_start|>system\nComprehend and analyze the provided prompt.<|im_end|>\n`. Compute it by encoding
44
+ that string. With this tokenizer it is 14:
45
+ `[151644, 8948, 198, 1092, 30782, 408, 323, 23643, 279, 3897, 9934, 13, 151645, 198]`.
46
+ Check that `ids[:drop_idx]` equals those tokens.
47
+
48
+ `Lfull` must be 16..512 (the encoder's axis), and the DiT needs `L = Lfull − drop_idx` in 8..512.
49
+ The template around the empty-prompt substitute `" "` is 23 tokens (`L` = 9).
50
+
51
+ ## 3. Encode
52
+
53
+ ```
54
+ hidden = encoder(input_ids = ids as int32 [1, Lfull]) # [1, Lfull, 4096]
55
+ prompt_embeds = hidden[:, drop_idx : Lfull] # [1, L, 4096], L = Lfull - drop_idx
56
+ ```
57
+
58
+ Pass exactly `Lfull` tokens; the axis is dynamic, so no padding is needed. (Attention is causal, so
59
+ pad tokens never reach the first `Lfull` outputs, but a longer input moves the GPU result by about
60
+ 4e-6 relative.) Run the encoder once per image.
61
+
62
+ ## 4. Sequence layout and RoPE
63
+
64
+ The DiT's sequence is `[text L | image N]`: the `L` prompt tokens, then the image tokens in raster
65
+ order. One image token covers one 16×16 px tile of the output: for an `S×S` image,
66
+ `h = w = S/16` and `N = h·w` (256² → 256 tokens, 512² → 1024, 1024² → 4096). There is no 2×2 packing.
67
+
68
+ Each token has a 3-axis position `(frame, height, width)`:
69
+
70
+ - text token `i` (0-based) → `(i, i, i)`;
71
+ - image token at row `y`, column `x` (0-based, token index `y·w + x`) →
72
+ `(L, y − (h − h//2), x − (w − w//2))`. The grid is centred on zero. For `h = 16`, rows go −8..7.
73
+
74
+ `host/rope_axis{a}_{cos,sin}.f32` hold one row per position −1024..8191: fp32, little-endian,
75
+ row-major `[9216, P_a]` with `P = (8, 28, 28)`. The row for position `p` is `p + 1024`. A token's
76
+ 64 values are the three rows concatenated in axis order:
77
+
78
+ ```
79
+ cos[token] = axis0_cos[f + 1024] ++ axis1_cos[hpos + 1024] ++ axis2_cos[wpos + 1024] # 8 + 28 + 28
80
+ sin[token] = the same with the _sin tables
81
+ txt_cos, txt_sin = rows of the L text tokens # [1, L, 64]
82
+ img_cos, img_sin = rows of the N image tokens # [1, N, 64]
83
+ ```
84
+
85
+ The tables are `torch.polar(1, outer(pos, 1 / 10000^(arange(0, d, 2)/d)))` for axis widths
86
+ `d = 16, 56, 56`, the same numbers as `QwenImage21Rope.freqs` of diffusers main (bit-exact). Axes 1
87
+ and 2 have the same width, so their files are byte-identical. The four tensors depend only on `L`,
88
+ `h` and `w`: build them once per image, not per step.
89
+
90
+ ## 5. Noise and pack
91
+
92
+ The initial latent is standard normal noise: 64 channels × `h` rows × `w` columns, channel-major.
93
+ The reference pipeline draws it as `randn((1, 1, 64, h, w))` and packs it into DiT tokens:
94
+
95
+ ```
96
+ x[0, y·w + x_, c] = z[c, y, x_] # = z.view(1, 64, h·w).transpose(1, 2)
97
+ ```
98
+
99
+ Any N(0, 1) noise generates an image. Matching the Python engine for a given seed needs its exact
100
+ draw: `torch.randn((1, 1, 64, h, w), generator=torch.Generator("cpu").manual_seed(seed))`.
101
+
102
+ ## 6. Sampler (FlowMatch Euler, 40 steps, no CFG)
103
+
104
+ Constants in `host/scheduler.json` (from the checkpoint's `scheduler_config.json`). All arrays fp32.
105
+
106
+ ```
107
+ sigmas = linspace(1, 1/steps, steps) # steps = 40
108
+ mu = N · m + b, m = (max_shift − base_shift) / (max_image_seq_len − base_image_seq_len),
109
+ b = base_shift − m · base_image_seq_len # N = image tokens
110
+ sigmas = e^mu / (e^mu + (1/sigmas − 1)) # time_shift_type "exponential"
111
+ sigmas = 1 − (1 − sigmas) / ((1 − sigmas[-1]) / (1 − shift_terminal))
112
+ timesteps = sigmas · num_train_timesteps # fp32
113
+ sigmas = sigmas ++ [0] # 41 values
114
+ for i in 0 ..< steps:
115
+ t = timesteps[i] / 1000 # fp32; this is the DiT `timestep` input
116
+ vel = dit(img_tokens = x, txt_feats = prompt_embeds, timestep = [t], txt_cos, txt_sin, img_cos, img_sin)
117
+ x = x + (sigmas[i+1] − sigmas[i]) · vel # fp32
118
+ ```
119
+
120
+ `mu` is 0.5 at 256² (N = 256), 0.5387… at 512², 0.6935… at 1024². Keep `t = timesteps[i] / 1000` as
121
+ written (multiply by 1000, then divide) to match the reference bit for bit. `qi21_sched.py`
122
+ reproduces the reference sigmas, timesteps and every Euler step bit-exactly at 256² and 512².
123
+
124
+ ## 7. Decode
125
+
126
+ ```
127
+ image = vae_S(latents_packed = x) # x after the last step, [1, N, 64], exactly as the sampler holds it
128
+ rgba8 = round(clip(image · 0.5 + 0.5, 0, 1) · 255) # [1, 4, S, S] -> channels R, G, B, A
129
+ ```
130
+
131
+ Feed the sampler's latent unchanged: no unpacking and no `* std + mean` on the host (the graph does
132
+ both). The four channels are what the reference pipeline saves as an RGBA PNG.
133
+
134
+ ## 8. Compile before running on the GPU
135
+
136
+ On macOS 27.0 (26A428) the Python runtime crashes on the 32-block DiT when the `.aimodel` is
137
+ compiled just-in-time: the MPSGraph delegate forms a Neural Engine region inside the graph and the
138
+ ANE inference fails (`ANERegion.mm:414 failed assertion … Code=-19`). A plain ahead-of-time compile
139
+ fails the same way. What runs is an ahead-of-time compile with `--expect-frequent-reshapes`, loaded
140
+ with `SpecializationOptions.default()`:
141
+
142
+ ```
143
+ xcrun coreai-build compile qi21_dit_full_bf16_dyn_iofp32.aimodel \
144
+ --output aot/qi21_dit_full_bf16_dyn_iofp32 \
145
+ --platform macOS --architecture h16c --preferred-compute gpu --expect-frequent-reshapes
146
+ # -> aot/qi21_dit_full_bf16_dyn_iofp32/qi21_dit_full_bf16_dyn_iofp32.h16c.aimodelc
147
+ ```
148
+
149
+ Do the same for the encoder and the VAE bundle you use; every gate of this port ran all three this
150
+ way, on an M4 Max with `--architecture h16c` (without `--architecture`, `coreai-build` compiles for
151
+ every supported architecture). The compiled DiT is about 1.9× its `.aimodel` (27 GB); the encoder's
152
+ stays at 14.10 GiB. Whether a Swift host using `GraphModel(computeUnits: .gpu)` hits the same Neural
153
+ Engine region has not been tested.
host/rope_axis0_cos.f32 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2a67b7ed9fea4c3f775dea442cf921f6cfa25be5d7da2671a87a01b154af3b5b
3
+ size 294912
host/rope_axis0_sin.f32 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a6da418893c5551a329985da7efbeb20858736f3fb21077239405f9ffb5f578f
3
+ size 294912
host/rope_axis1_cos.f32 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8152a557c6febd21a2baeded29590c6b5bf18c439da5ddbe110c1273379c6ee3
3
+ size 1032192
host/rope_axis1_sin.f32 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a9e468de85274824afd0a30ecf8ca8bb53005ed0d606b656a65703e51c1b21cc
3
+ size 1032192
host/rope_axis2_cos.f32 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8152a557c6febd21a2baeded29590c6b5bf18c439da5ddbe110c1273379c6ee3
3
+ size 1032192
host/rope_axis2_sin.f32 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a9e468de85274824afd0a30ecf8ca8bb53005ed0d606b656a65703e51c1b21cc
3
+ size 1032192
host/scheduler.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base_image_seq_len": 256,
3
+ "max_image_seq_len": 8192,
4
+ "base_shift": 0.5,
5
+ "max_shift": 0.9,
6
+ "shift_terminal": 0.02,
7
+ "time_shift_type": "exponential",
8
+ "num_train_timesteps": 1000,
9
+ "steps": 40
10
+ }
qi21_dit_full_bf16_dyn_iofp32.aimodel/main.hash ADDED
@@ -0,0 +1 @@
 
 
1
+ ���6l�A��T����?*b���coDw��bN
qi21_dit_full_bf16_dyn_iofp32.aimodel/main.mlirb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c87e0b2366cc61e4185b154dfece8d23f2a628483bc636f4477fcce624e1310
3
+ size 14231004496
qi21_dit_full_bf16_dyn_iofp32.aimodel/metadata.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "producer" : "coreai-core 1.0.0b2",
3
+ "assetVersion" : "2.0",
4
+ "license" : "qwen-research",
5
+ "description" : "Qwen-Image-2.1 DiT (full), bf16 weights+compute, fp32 I\/O, dynamic text\/image axes. Source: Qwen\/Qwen-Image-2.1@790c926.",
6
+ "creationDate" : "20260925T055156Z"
7
+ }
qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/main.hash ADDED
@@ -0,0 +1 @@
 
 
1
+ �{�dv6H�j2�:!��-�?�,8��]�d�4V2C
qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/main.mlirb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d37b8e64763648cb6a32b23a2199b42d0ee93fac2c3899835d8e64d534563243
3
+ size 15138226830
qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/metadata.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "creationDate" : "20260925T065421Z",
3
+ "assetVersion" : "2.0",
4
+ "description" : "Qwen-Image-2.1 text encoder (Qwen3-VL-8B text stack, all 36 layers), bf16 weights, fp32 compute, int32 input_ids -> fp32 hidden (last layer, before the final norm), dynamic L 16..512. Source: Qwen\/Qwen-Image-2.1@790c926.",
5
+ "license" : "qwen-research",
6
+ "producer" : "coreai-core 1.0.0b2"
7
+ }
qi21_vae_1024_fp32.aimodel/main.hash ADDED
@@ -0,0 +1 @@
 
 
1
+ �/�6�6ޡrr�N��g�}� H[�%1��9D
qi21_vae_1024_fp32.aimodel/main.mlirb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d82fc436a51e36dea17272c64e8dd667df7dcd0c1d485bf225319303ed823944
3
+ size 1012440351
qi21_vae_1024_fp32.aimodel/metadata.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "description" : "Qwen-Image-2.1 VAE decoder, 1024x1024, fp32: latents_packed [1,4096,64] (sampler output, normalised) -> image [1,4,1024,1024] RGBA in [-1,1]; unpack and latents*std+mean inside the graph. Source: Qwen\/Qwen-Image-2.1@790c926.",
3
+ "assetVersion" : "2.0",
4
+ "license" : "qwen-research",
5
+ "creationDate" : "20260925T071221Z",
6
+ "producer" : "coreai-core 1.0.0b2"
7
+ }
qi21_vae_256_fp32.aimodel/main.hash ADDED
@@ -0,0 +1 @@
 
 
1
+ �9bf�DN�k�G�ø�&+U"�w�?Rwb
qi21_vae_256_fp32.aimodel/main.mlirb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:82396266f39abd444e11fd6b17b20347eeb7c3b8d5262b5522f877b13f527762
3
+ size 1012434454
qi21_vae_256_fp32.aimodel/metadata.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "creationDate" : "20260925T071101Z",
3
+ "producer" : "coreai-core 1.0.0b2",
4
+ "assetVersion" : "2.0",
5
+ "license" : "qwen-research",
6
+ "description" : "Qwen-Image-2.1 VAE decoder, 256x256, fp32: latents_packed [1,256,64] (sampler output, normalised) -> image [1,4,256,256] RGBA in [-1,1]; unpack and latents*std+mean inside the graph. Source: Qwen\/Qwen-Image-2.1@790c926."
7
+ }
qi21_vae_512_fp32.aimodel/main.hash ADDED
@@ -0,0 +1 @@
 
 
1
+ ϋpIc��(f����$�6w�����~��(.V
qi21_vae_512_fp32.aimodel/main.mlirb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cf8b1b14704963bc912866b81482a2ac24b1367796c6f60eeed17e95e9282e56
3
+ size 1012436431
qi21_vae_512_fp32.aimodel/metadata.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "description" : "Qwen-Image-2.1 VAE decoder, 512x512, fp32: latents_packed [1,1024,64] (sampler output, normalised) -> image [1,4,512,512] RGBA in [-1,1]; unpack and latents*std+mean inside the graph. Source: Qwen\/Qwen-Image-2.1@790c926.",
3
+ "assetVersion" : "2.0",
4
+ "license" : "qwen-research",
5
+ "creationDate" : "20260925T071145Z",
6
+ "producer" : "coreai-core 1.0.0b2"
7
+ }
tokenizer/added_tokens.json ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</think>": 151668,
3
+ "</tool_call>": 151658,
4
+ "</tool_response>": 151666,
5
+ "<think>": 151667,
6
+ "<tool_call>": 151657,
7
+ "<tool_response>": 151665,
8
+ "<|box_end|>": 151649,
9
+ "<|box_start|>": 151648,
10
+ "<|endoftext|>": 151643,
11
+ "<|file_sep|>": 151664,
12
+ "<|fim_middle|>": 151660,
13
+ "<|fim_pad|>": 151662,
14
+ "<|fim_prefix|>": 151659,
15
+ "<|fim_suffix|>": 151661,
16
+ "<|im_end|>": 151645,
17
+ "<|im_start|>": 151644,
18
+ "<|image_pad|>": 151655,
19
+ "<|object_ref_end|>": 151647,
20
+ "<|object_ref_start|>": 151646,
21
+ "<|quad_end|>": 151651,
22
+ "<|quad_start|>": 151650,
23
+ "<|repo_name|>": 151663,
24
+ "<|video_pad|>": 151656,
25
+ "<|vision_end|>": 151653,
26
+ "<|vision_pad|>": 151654,
27
+ "<|vision_start|>": 151652
28
+ }
tokenizer/chat_template.jinja ADDED
@@ -0,0 +1,120 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0].role == 'system' %}
4
+ {%- if messages[0].content is string %}
5
+ {{- messages[0].content }}
6
+ {%- else %}
7
+ {%- for content in messages[0].content %}
8
+ {%- if 'text' in content %}
9
+ {{- content.text }}
10
+ {%- endif %}
11
+ {%- endfor %}
12
+ {%- endif %}
13
+ {{- '\n\n' }}
14
+ {%- endif %}
15
+ {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
16
+ {%- for tool in tools %}
17
+ {{- "\n" }}
18
+ {{- tool | tojson }}
19
+ {%- endfor %}
20
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
21
+ {%- else %}
22
+ {%- if messages[0].role == 'system' %}
23
+ {{- '<|im_start|>system\n' }}
24
+ {%- if messages[0].content is string %}
25
+ {{- messages[0].content }}
26
+ {%- else %}
27
+ {%- for content in messages[0].content %}
28
+ {%- if 'text' in content %}
29
+ {{- content.text }}
30
+ {%- endif %}
31
+ {%- endfor %}
32
+ {%- endif %}
33
+ {{- '<|im_end|>\n' }}
34
+ {%- endif %}
35
+ {%- endif %}
36
+ {%- set image_count = namespace(value=0) %}
37
+ {%- set video_count = namespace(value=0) %}
38
+ {%- for message in messages %}
39
+ {%- if message.role == "user" %}
40
+ {{- '<|im_start|>' + message.role + '\n' }}
41
+ {%- if message.content is string %}
42
+ {{- message.content }}
43
+ {%- else %}
44
+ {%- for content in message.content %}
45
+ {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
46
+ {%- set image_count.value = image_count.value + 1 %}
47
+ {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
48
+ <|vision_start|><|image_pad|><|vision_end|>
49
+ {%- elif content.type == 'video' or 'video' in content %}
50
+ {%- set video_count.value = video_count.value + 1 %}
51
+ {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
52
+ <|vision_start|><|video_pad|><|vision_end|>
53
+ {%- elif 'text' in content %}
54
+ {{- content.text }}
55
+ {%- endif %}
56
+ {%- endfor %}
57
+ {%- endif %}
58
+ {{- '<|im_end|>\n' }}
59
+ {%- elif message.role == "assistant" %}
60
+ {{- '<|im_start|>' + message.role + '\n' }}
61
+ {%- if message.content is string %}
62
+ {{- message.content }}
63
+ {%- else %}
64
+ {%- for content_item in message.content %}
65
+ {%- if 'text' in content_item %}
66
+ {{- content_item.text }}
67
+ {%- endif %}
68
+ {%- endfor %}
69
+ {%- endif %}
70
+ {%- if message.tool_calls %}
71
+ {%- for tool_call in message.tool_calls %}
72
+ {%- if (loop.first and message.content) or (not loop.first) %}
73
+ {{- '\n' }}
74
+ {%- endif %}
75
+ {%- if tool_call.function %}
76
+ {%- set tool_call = tool_call.function %}
77
+ {%- endif %}
78
+ {{- '<tool_call>\n{"name": "' }}
79
+ {{- tool_call.name }}
80
+ {{- '", "arguments": ' }}
81
+ {%- if tool_call.arguments is string %}
82
+ {{- tool_call.arguments }}
83
+ {%- else %}
84
+ {{- tool_call.arguments | tojson }}
85
+ {%- endif %}
86
+ {{- '}\n</tool_call>' }}
87
+ {%- endfor %}
88
+ {%- endif %}
89
+ {{- '<|im_end|>\n' }}
90
+ {%- elif message.role == "tool" %}
91
+ {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
92
+ {{- '<|im_start|>user' }}
93
+ {%- endif %}
94
+ {{- '\n<tool_response>\n' }}
95
+ {%- if message.content is string %}
96
+ {{- message.content }}
97
+ {%- else %}
98
+ {%- for content in message.content %}
99
+ {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
100
+ {%- set image_count.value = image_count.value + 1 %}
101
+ {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
102
+ <|vision_start|><|image_pad|><|vision_end|>
103
+ {%- elif content.type == 'video' or 'video' in content %}
104
+ {%- set video_count.value = video_count.value + 1 %}
105
+ {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
106
+ <|vision_start|><|video_pad|><|vision_end|>
107
+ {%- elif 'text' in content %}
108
+ {{- content.text }}
109
+ {%- endif %}
110
+ {%- endfor %}
111
+ {%- endif %}
112
+ {{- '\n</tool_response>' }}
113
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
114
+ {{- '<|im_end|>\n' }}
115
+ {%- endif %}
116
+ {%- endif %}
117
+ {%- endfor %}
118
+ {%- if add_generation_prompt %}
119
+ {{- '<|im_start|>assistant\n' }}
120
+ {%- endif %}
tokenizer/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|im_end|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
tokenizer/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:aeb13307a71acd8fe81861d94ad54ab689df773318809eed3cbe794b4492dae4
3
+ size 11422654
tokenizer/tokenizer_config.json ADDED
@@ -0,0 +1,240 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ },
181
+ "151665": {
182
+ "content": "<tool_response>",
183
+ "lstrip": false,
184
+ "normalized": false,
185
+ "rstrip": false,
186
+ "single_word": false,
187
+ "special": false
188
+ },
189
+ "151666": {
190
+ "content": "</tool_response>",
191
+ "lstrip": false,
192
+ "normalized": false,
193
+ "rstrip": false,
194
+ "single_word": false,
195
+ "special": false
196
+ },
197
+ "151667": {
198
+ "content": "<think>",
199
+ "lstrip": false,
200
+ "normalized": false,
201
+ "rstrip": false,
202
+ "single_word": false,
203
+ "special": false
204
+ },
205
+ "151668": {
206
+ "content": "</think>",
207
+ "lstrip": false,
208
+ "normalized": false,
209
+ "rstrip": false,
210
+ "single_word": false,
211
+ "special": false
212
+ }
213
+ },
214
+ "additional_special_tokens": [
215
+ "<|im_start|>",
216
+ "<|im_end|>",
217
+ "<|object_ref_start|>",
218
+ "<|object_ref_end|>",
219
+ "<|box_start|>",
220
+ "<|box_end|>",
221
+ "<|quad_start|>",
222
+ "<|quad_end|>",
223
+ "<|vision_start|>",
224
+ "<|vision_end|>",
225
+ "<|vision_pad|>",
226
+ "<|image_pad|>",
227
+ "<|video_pad|>"
228
+ ],
229
+ "bos_token": null,
230
+ "clean_up_tokenization_spaces": false,
231
+ "eos_token": "<|im_end|>",
232
+ "errors": "replace",
233
+ "extra_special_tokens": {},
234
+ "model_max_length": 262144,
235
+ "pad_token": "<|endoftext|>",
236
+ "processor_class": "Qwen3VLProcessor",
237
+ "split_special_tokens": false,
238
+ "tokenizer_class": "Qwen2Tokenizer",
239
+ "unk_token": null
240
+ }
tokenizer/vocab.json ADDED
The diff for this file is too large to render. See raw diff