RGBA-Image-2.1: Core AI port of Qwen-Image-2.1 (macOS 27, bf16 DiT + w16a32 encoder + fp32 RGBA VAE)
Browse files- .gitattributes +12 -0
- LICENSE +55 -0
- NOTICE +35 -0
- README.md +248 -0
- config.json +19 -0
- host/host_contract.md +153 -0
- host/rope_axis0_cos.f32 +3 -0
- host/rope_axis0_sin.f32 +3 -0
- host/rope_axis1_cos.f32 +3 -0
- host/rope_axis1_sin.f32 +3 -0
- host/rope_axis2_cos.f32 +3 -0
- host/rope_axis2_sin.f32 +3 -0
- host/scheduler.json +10 -0
- qi21_dit_full_bf16_dyn_iofp32.aimodel/main.hash +1 -0
- qi21_dit_full_bf16_dyn_iofp32.aimodel/main.mlirb +3 -0
- qi21_dit_full_bf16_dyn_iofp32.aimodel/metadata.json +7 -0
- qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/main.hash +1 -0
- qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/main.mlirb +3 -0
- qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/metadata.json +7 -0
- qi21_vae_1024_fp32.aimodel/main.hash +1 -0
- qi21_vae_1024_fp32.aimodel/main.mlirb +3 -0
- qi21_vae_1024_fp32.aimodel/metadata.json +7 -0
- qi21_vae_256_fp32.aimodel/main.hash +1 -0
- qi21_vae_256_fp32.aimodel/main.mlirb +3 -0
- qi21_vae_256_fp32.aimodel/metadata.json +7 -0
- qi21_vae_512_fp32.aimodel/main.hash +1 -0
- qi21_vae_512_fp32.aimodel/main.mlirb +3 -0
- qi21_vae_512_fp32.aimodel/metadata.json +7 -0
- tokenizer/added_tokens.json +28 -0
- tokenizer/chat_template.jinja +120 -0
- tokenizer/merges.txt +0 -0
- tokenizer/special_tokens_map.json +31 -0
- tokenizer/tokenizer.json +3 -0
- tokenizer/tokenizer_config.json +240 -0
- tokenizer/vocab.json +0 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,15 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
host/rope_axis0_cos.f32 filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
host/rope_axis0_sin.f32 filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
host/rope_axis1_cos.f32 filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
host/rope_axis1_sin.f32 filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
host/rope_axis2_cos.f32 filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
host/rope_axis2_sin.f32 filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
qi21_dit_full_bf16_dyn_iofp32.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
qi21_vae_1024_fp32.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
qi21_vae_256_fp32.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
qi21_vae_512_fp32.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
|
| 47 |
+
tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
LICENSE
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Qwen RESEARCH LICENSE AGREEMENT
|
| 2 |
+
|
| 3 |
+
Qwen RESEARCH LICENSE AGREEMENT Release Date: September 20, 2026
|
| 4 |
+
|
| 5 |
+
By clicking to agree or by using or distributing any portion or element of the Qwen Materials, you will be deemed to have recognized and accepted the content of this Agreement, which is effective immediately.
|
| 6 |
+
|
| 7 |
+
1. Definitions
|
| 8 |
+
a. This Qwen RESEARCH LICENSE AGREEMENT (this "Agreement") shall mean the terms and conditions for use, reproduction, distribution and modification of the Materials as defined by this Agreement.
|
| 9 |
+
b. "We" (or "Us") shall mean Hangzhou Tongyi Laboratory Technology Co., Ltd.
|
| 10 |
+
c. "You" (or "Your") shall mean a natural person or legal entity exercising the rights granted by this Agreement and/or using the Materials for any purpose and in any field of use.
|
| 11 |
+
d. "Third Parties" shall mean individuals or legal entities that are not under common control with us or you.
|
| 12 |
+
e. "Qwen" shall mean the large language models, diffusion models, and software and algorithms, consisting of trained model weights, parameters (including optimizer states), machine-learning model code, inference-enabling code, training-enabling code, fine-tuning enabling code and other elements of the foregoing distributed by us.
|
| 13 |
+
f. "Materials" shall mean, collectively, our proprietary Qwen and Documentation (and any portion thereof) made available under this Agreement.
|
| 14 |
+
g. "Source" form shall mean the preferred form for making modifications, including but not limited to model source code, documentation source, and configuration files.
|
| 15 |
+
h. "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
|
| 16 |
+
i. "Non-Commercial" shall mean for research or evaluation purposes only.
|
| 17 |
+
|
| 18 |
+
2. Grant of Rights
|
| 19 |
+
a. You are granted a non-exclusive, worldwide, non-transferable and royalty-free limited license under our intellectual property or other rights owned by us embodied in the Materials to use, reproduce, distribute, copy, create derivative works of, and make modifications to the Materials FOR NON-COMMERCIAL PURPOSES ONLY.
|
| 20 |
+
b. You shall not use the Materials for any commercial purpose without obtaining a separate commercial license from us. If you wish to use the Materials commercially, you shall request a license from us at model-business@notice.qwencloud.com.
|
| 21 |
+
|
| 22 |
+
3. Redistribution
|
| 23 |
+
Subject to Section 2 (Grant of Rights), you may distribute copies or make the Materials, or derivative works thereof, available as part of a product or service that contains any of them, with or without modifications, and in Source or Object form, provided that you meet the following conditions:
|
| 24 |
+
a. You shall give any other recipients of the Materials or derivative works a copy of this Agreement;
|
| 25 |
+
b. You shall cause any modified files to carry prominent notices stating that you changed the files;
|
| 26 |
+
c. You shall retain in all copies of the Materials that you distribute the following attribution notices within a "Notice" text file distributed as a part of such copies: "Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved."; and
|
| 27 |
+
d. You may add your own copyright statement to your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of your modifications, or for any such derivative works as a whole, provided your use, reproduction, and distribution of the work otherwise complies with the terms and conditions of this Agreement.
|
| 28 |
+
|
| 29 |
+
4. Rules of use
|
| 30 |
+
a. The Materials may be subject to export controls or restrictions in China, the United States or other countries or regions. You shall comply with applicable laws and regulations in your use of the Materials.
|
| 31 |
+
b. If you use the Materials or any outputs or results therefrom to create, train, fine-tune, or improve an AI model that is distributed or made available, you shall prominently display “Built with Qwen” or “Improved using Qwen” in the related product documentation.
|
| 32 |
+
c. You shall not use "Qwen" as the primary name or identifier of any derivative works or products; reasonable descriptive use (e.g., "fine-tuned from Qwen Image") is permitted.
|
| 33 |
+
|
| 34 |
+
5. Intellectual Property
|
| 35 |
+
a. We retain ownership of all intellectual property rights in and to the Materials and derivatives made by or for us. Conditioned upon compliance with the terms and conditions of this Agreement, with respect to any derivative works and modifications of the Materials that are made by you, you are and will be the owner of such derivative works and modifications.
|
| 36 |
+
b. No trademark license is granted to use the trade names, trademarks, service marks, or product names of us, except as required to fulfill notice requirements under this Agreement or as required for reasonable and customary use in describing and redistributing the Materials.
|
| 37 |
+
c. If you commence a lawsuit or other proceedings (including a cross-claim or counterclaim in a lawsuit) against us or any entity alleging that the Materials or any output therefrom, or any part of the foregoing, infringe any intellectual property or other right owned or licensable by you, then all licenses granted to you under this Agreement shall terminate as of the date such lawsuit or other proceeding is commenced or brought.
|
| 38 |
+
|
| 39 |
+
6. Disclaimer of Warranty and Limitation of Liability
|
| 40 |
+
a. We are not obligated to support, update, provide training for, or develop any further version of the Qwen Materials or to grant any license thereto.
|
| 41 |
+
b. THE MATERIALS ARE PROVIDED "AS IS" WITHOUT ANY EXPRESS OR IMPLIED WARRANTY OF ANY KIND INCLUDING WARRANTIES OF MERCHANTABILITY, NONINFRINGEMENT, OR FITNESS FOR A PARTICULAR PURPOSE. WE MAKE NO WARRANTY AND ASSUME NO RESPONSIBILITY FOR THE SAFETY OR STABILITY OF THE MATERIALS AND ANY OUTPUT THEREFROM.
|
| 42 |
+
c. IN NO EVENT SHALL WE BE LIABLE TO YOU FOR ANY DAMAGES, INCLUDING, BUT NOT LIMITED TO ANY DIRECT, OR INDIRECT, SPECIAL OR CONSEQUENTIAL DAMAGES ARISING FROM YOUR USE OR INABILITY TO USE THE MATERIALS OR ANY OUTPUT OF IT, NO MATTER HOW IT’S CAUSED.
|
| 43 |
+
d. You will defend, indemnify and hold harmless us from and against any claim by any third party arising out of or related to your use or distribution of the Materials.
|
| 44 |
+
|
| 45 |
+
7. Survival and Termination.
|
| 46 |
+
a. The term of this Agreement shall commence upon your acceptance of this Agreement or access to the Materials and will continue in full force and effect until terminated in accordance with the terms and conditions herein.
|
| 47 |
+
b. We may terminate this Agreement if you breach any of the terms or conditions of this Agreement. Upon termination of this Agreement, you must delete and cease use of the Materials. Sections 6 and 8 shall survive the termination of this Agreement.
|
| 48 |
+
|
| 49 |
+
8. Governing Law and Jurisdiction.
|
| 50 |
+
a. This Agreement and any dispute arising out of or relating to it will be governed by the laws of China, without regard to conflict of law principles, and the UN Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
|
| 51 |
+
b. The People's Courts in Hangzhou City shall have exclusive jurisdiction over any dispute arising out of this Agreement.
|
| 52 |
+
|
| 53 |
+
9. Other Terms and Conditions.
|
| 54 |
+
a. Any arrangements, understandings, or agreements regarding the Material not stated herein are separate from and independent of the terms and conditions of this Agreement. You shall request a separate license from us, if you use the Materials in ways not expressly agreed to in this Agreement.
|
| 55 |
+
b. We shall not be bound by any additional or different terms or conditions communicated by you unless expressly agreed.
|
NOTICE
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved.
|
| 2 |
+
|
| 3 |
+
The model files in this repository were converted and re-authored from the Qwen-Image-2.1 weights
|
| 4 |
+
(Qwen/Qwen-Image-2.1, revision 790c92633540aa0cb11d9abf19eb46d861714758) into Apple Core AI
|
| 5 |
+
.aimodel graphs. They were changed as follows.
|
| 6 |
+
|
| 7 |
+
qi21_dit_full_bf16_dyn_iofp32.aimodel — the diffusion transformer (from transformer/)
|
| 8 |
+
- Re-authored in plain PyTorch with the checkpoint's parameter names, then exported as one graph.
|
| 9 |
+
- Block-causal attention is computed as two scaled-dot-product-attention calls per block
|
| 10 |
+
(text tokens attend causally to text; image tokens attend to every token).
|
| 11 |
+
- Rotary position embeddings are applied as real-valued cos/sin pairs that the host supplies.
|
| 12 |
+
- The prefix key/value cache is removed: the text prefix is recomputed at every step.
|
| 13 |
+
- Weights and computation are bfloat16; every graph input and output is float32.
|
| 14 |
+
- Both sequence axes are dynamic (text 8-512 tokens, image 64-4096 tokens). Text-to-image only:
|
| 15 |
+
the graph has no condition-image input.
|
| 16 |
+
|
| 17 |
+
qi21_encoder_dynL_w16a32_ids_iofp32.aimodel — the text encoder (from text_encoder/)
|
| 18 |
+
- Only the text path of the Qwen3-VL-8B encoder is kept. The vision tower, the language-model head
|
| 19 |
+
and the final RMSNorm are removed; the output is the residual stream after the last layer.
|
| 20 |
+
- Re-authored in plain PyTorch; the token embedding is inside the graph (int32 token ids in).
|
| 21 |
+
- Weights are stored in bfloat16 and all computation is float32; the output is float32.
|
| 22 |
+
- The sequence axis is dynamic (16-512 tokens).
|
| 23 |
+
|
| 24 |
+
qi21_vae_256_fp32.aimodel, qi21_vae_512_fp32.aimodel, qi21_vae_1024_fp32.aimodel — the VAE (from vae/)
|
| 25 |
+
- Decoder only, one frame; the VAE encoder is not included.
|
| 26 |
+
- The latent un-normalisation (latents * std + mean) and the token-to-grid unpacking are inside
|
| 27 |
+
the graph.
|
| 28 |
+
- One graph per output size (256x256, 512x512, 1024x1024), float32.
|
| 29 |
+
|
| 30 |
+
Copied without changes: LICENSE; config.json (= transformer/config.json); tokenizer/ (= processor/
|
| 31 |
+
tokenizer.json, tokenizer_config.json, vocab.json, merges.txt, special_tokens_map.json,
|
| 32 |
+
added_tokens.json, chat_template.jinja).
|
| 33 |
+
|
| 34 |
+
Added by the conversion: host/ (rotary tables and sampler constants computed from the model
|
| 35 |
+
definition, and host_contract.md) and README.md.
|
README.md
ADDED
|
@@ -0,0 +1,248 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: other
|
| 3 |
+
license_name: qwen-research
|
| 4 |
+
license_link: LICENSE
|
| 5 |
+
base_model: Qwen/Qwen-Image-2.1
|
| 6 |
+
pipeline_tag: text-to-image
|
| 7 |
+
library_name: coreai
|
| 8 |
+
tags:
|
| 9 |
+
- coreai
|
| 10 |
+
- rgba
|
| 11 |
+
- text-to-image
|
| 12 |
+
- apple-silicon
|
| 13 |
+
---
|
| 14 |
+
|
| 15 |
+
# RGBA-Image-2.1 — Core AI port of Qwen-Image-2.1
|
| 16 |
+
|
| 17 |
+
**Built with Qwen.**
|
| 18 |
+
|
| 19 |
+
**Non-commercial use only** — Qwen Research License (research or evaluation purposes).
|
| 20 |
+
|
| 21 |
+
**macOS 27 on Apple silicon only.** Five Core AI bundles, 32.41 GB (30.18 GiB) in total.
|
| 22 |
+
|
| 23 |
+
This is [`Qwen/Qwen-Image-2.1`](https://huggingface.co/Qwen/Qwen-Image-2.1) (revision `790c926`)
|
| 24 |
+
converted to Core AI `.aimodel` graphs that run on the Mac GPU. A Qwen3-VL-8B text encoder (text
|
| 25 |
+
path only) conditions a 7B single-stream DiT (32 blocks, block-causal attention). A 64-channel VAE
|
| 26 |
+
decodes to four channels, RGBA. The sampler is the pipeline default: 40 FlowMatch Euler steps, no
|
| 27 |
+
CFG. Ask for a transparent background in the prompt and the fourth channel is a real alpha.
|
| 28 |
+
|
| 29 |
+
Sizes: 256², 512² and 1024², square. The DiT and the text encoder have dynamic axes (see Graph
|
| 30 |
+
contracts); the VAE has one graph per size. The upstream default of 2048² needs 16,384 image
|
| 31 |
+
tokens, outside this DiT's 64–4096-token axis.
|
| 32 |
+
|
| 33 |
+
## Bundle
|
| 34 |
+
|
| 35 |
+
[🤗 mlboydaisuke/RGBA-Image-2.1-CoreAI](https://huggingface.co/mlboydaisuke/RGBA-Image-2.1-CoreAI)
|
| 36 |
+
|
| 37 |
+
| file | what it is | size |
|
| 38 |
+
| --- | --- | --- |
|
| 39 |
+
| `qi21_dit_full_bf16_dyn_iofp32.aimodel` | the DiT: bf16 weights and compute, fp32 inputs and outputs | 14.23 GB (13.25 GiB) |
|
| 40 |
+
| `qi21_encoder_dynL_w16a32_ids_iofp32.aimodel` | the text encoder: bf16 weights, fp32 compute; token ids in, `embed_tokens` inside | 15.14 GB (14.10 GiB) |
|
| 41 |
+
| `qi21_vae_{256,512,1024}_fp32.aimodel` | the VAE decoder, fp32, one per size | 1.01 GB (0.94 GiB) each |
|
| 42 |
+
| `host/` | per-axis RoPE tables, `scheduler.json`, [`host_contract.md`](https://huggingface.co/mlboydaisuke/RGBA-Image-2.1-CoreAI/blob/main/host/host_contract.md) | |
|
| 43 |
+
| `tokenizer/` | the source's `processor/` tokenizer files, unchanged | |
|
| 44 |
+
| `config.json` | the source's `transformer/config.json`, unchanged | |
|
| 45 |
+
| `LICENSE`, `NOTICE` | the Qwen Research License and the change notice | |
|
| 46 |
+
|
| 47 |
+
No quantized variant ships. The int8 DiT does not compile on the one GPU path that works (next
|
| 48 |
+
paragraph). int8 weights break the text encoder's `<|im_start|>` tokens (Lessons, 2). Compiled, the int8
|
| 49 |
+
encoder is also 34.32 GiB; this one is 14.10 GiB.
|
| 50 |
+
|
| 51 |
+
**Compile before you run.** On macOS 27.0 (26A428) the DiT crashes in the Python runtime when its
|
| 52 |
+
`.aimodel` is compiled just-in-time. What runs is an ahead-of-time compile with
|
| 53 |
+
`--expect-frequent-reshapes` (step 2 below). The compiled copies need their own disk: 27 GB for
|
| 54 |
+
the DiT and 14.10 GiB for the encoder.
|
| 55 |
+
|
| 56 |
+
## Use it
|
| 57 |
+
|
| 58 |
+
There is no Swift app for this model yet. The Python engine
|
| 59 |
+
[`conversion/qwenimage21/pipeline_engine.py`](https://github.com/john-rocky/coreai-model-zoo/blob/main/conversion/qwenimage21/pipeline_engine.py)
|
| 60 |
+
runs the whole loop on the three bundles: tokenize, encode, 40 DiT steps, decode, write a PNG.
|
| 61 |
+
|
| 62 |
+
```
|
| 63 |
+
# 1. the bundles, and the tokenizer the engine reads from the source repo's processor/
|
| 64 |
+
hf download mlboydaisuke/RGBA-Image-2.1-CoreAI --local-dir RGBA-Image-2.1-CoreAI
|
| 65 |
+
hf download Qwen/Qwen-Image-2.1 --revision 790c92633540aa0cb11d9abf19eb46d861714758 --include "processor/*"
|
| 66 |
+
|
| 67 |
+
# 2. compile for the Mac GPU (once per bundle; the gates used --architecture h16c on an M4 Max)
|
| 68 |
+
cd RGBA-Image-2.1-CoreAI
|
| 69 |
+
for m in qi21_encoder_dynL_w16a32_ids_iofp32 qi21_dit_full_bf16_dyn_iofp32 qi21_vae_512_fp32; do
|
| 70 |
+
xcrun coreai-build compile $m.aimodel --output aot/$m --platform macOS --architecture h16c \
|
| 71 |
+
--preferred-compute gpu --expect-frequent-reshapes
|
| 72 |
+
done
|
| 73 |
+
A=$PWD/aot
|
| 74 |
+
|
| 75 |
+
# 3. generate, from a checkout of the zoo
|
| 76 |
+
# (Python with coreai-core 1.0.0b2, torch, numpy, tokenizers, pillow)
|
| 77 |
+
cd /path/to/coreai-model-zoo/conversion/qwenimage21
|
| 78 |
+
python pipeline_engine.py \
|
| 79 |
+
--prompt "This is an RGBA image with transparency. A cute cartoon dragon sticker. The image has alpha channel and the background is transparent." \
|
| 80 |
+
--size 512 --seed 42 --tag dragon \
|
| 81 |
+
--encoder $A/qi21_encoder_dynL_w16a32_ids_iofp32/qi21_encoder_dynL_w16a32_ids_iofp32.h16c.aimodelc \
|
| 82 |
+
--dit $A/qi21_dit_full_bf16_dyn_iofp32/qi21_dit_full_bf16_dyn_iofp32.h16c.aimodelc \
|
| 83 |
+
--vae $A/qi21_vae_512_fp32/qi21_vae_512_fp32.h16c.aimodelc
|
| 84 |
+
# -> _work/samples/dragon.png (RGBA) and dragon_rgb.png (composited on white)
|
| 85 |
+
```
|
| 86 |
+
|
| 87 |
+
The prompt above is the upstream card's recommended form for transparent images: "This is an RGBA
|
| 88 |
+
image with transparency. {subject}. The image has alpha channel and the background is
|
| 89 |
+
transparent." The noise is `torch.randn` on a CPU generator seeded with `--seed`, the same draw the
|
| 90 |
+
reference pipeline makes with a CPU generator and that seed.
|
| 91 |
+
|
| 92 |
+
To check the port against the fp32 reference, record the reference once and run the engine in
|
| 93 |
+
oracle mode. `capture_oracle.py` runs the diffusers-main pipeline in fp32 on the CPU, about 2 min
|
| 94 |
+
at 256². It needs the full `Qwen/Qwen-Image-2.1` snapshot and a venv with diffusers main 4295ee3
|
| 95 |
+
and transformers 5.17.
|
| 96 |
+
|
| 97 |
+
```
|
| 98 |
+
python capture_oracle.py --size 256 --steps 40 # -> oracle/256/
|
| 99 |
+
python pipeline_engine.py --oracle oracle/256 \
|
| 100 |
+
--encoder $A/qi21_encoder_dynL_w16a32_ids_iofp32/qi21_encoder_dynL_w16a32_ids_iofp32.h16c.aimodelc \
|
| 101 |
+
--dit $A/qi21_dit_full_bf16_dyn_iofp32/qi21_dit_full_bf16_dyn_iofp32.h16c.aimodelc \
|
| 102 |
+
--vae $A/qi21_vae_256_fp32/qi21_vae_256_fp32.h16c.aimodelc
|
| 103 |
+
# prints the latent corr vs the reference after every step, then the RGBA and white-composited PSNR
|
| 104 |
+
```
|
| 105 |
+
|
| 106 |
+
## Which image model should I use?
|
| 107 |
+
|
| 108 |
+
The zoo's Mac text-to-image models are not ranked; pick by trade-off. RGBA-Image-2.1 writes RGBA
|
| 109 |
+
natively, so it fits images that need an alpha channel.
|
| 110 |
+
|
| 111 |
+
| | params | sampler | time @1024 | precision |
|
| 112 |
+
| --- | --- | --- | --- | --- |
|
| 113 |
+
| **FLUX.2 klein** | 4B | 4 steps, guidance-distilled (no CFG) | ~17 s | int4 |
|
| 114 |
+
| **Z-Image-Turbo** | 6B | 8 steps + CFG (16 forwards) | ~70 s | bf16, near-lossless |
|
| 115 |
+
| **GLM-Image** | 16B (9B AR + 7B DiT) | AR prior + 20-step DiT | ~208 s | int8 |
|
| 116 |
+
| **RGBA-Image-2.1** | 7B | 40 steps, no CFG | ~190 s | bf16 |
|
| 117 |
+
|
| 118 |
+
Times are the ones each card reports, on an M4 Max. For RGBA-Image-2.1 it is 40 DiT steps at the
|
| 119 |
+
warm median of 4.757 s per forward. It leaves out the encoder, the VAE and loading.
|
| 120 |
+
|
| 121 |
+
## Graph contracts
|
| 122 |
+
|
| 123 |
+
Every graph has one function, `main`, and fp32 inputs and outputs except `input_ids`.
|
| 124 |
+
|
| 125 |
+
| graph | inputs | output |
|
| 126 |
+
| --- | --- | --- |
|
| 127 |
+
| encoder | `input_ids [1,Lfull]` int32, `Lfull` 16..512 | `hidden [1,Lfull,4096]`: the last layer's residual stream, before the final norm |
|
| 128 |
+
| DiT | `img_tokens [1,N,64]`, `txt_feats [1,L,4096]`, `timestep [1]`, `txt_cos`/`txt_sin [1,L,64]`, `img_cos`/`img_sin [1,N,64]`; `L` 8..512, `N` 64..4096 | `vel [1,N,64]` |
|
| 129 |
+
| VAE | `latents_packed [1,N,64]`, the sampler's latent unchanged | `image [1,4,S,S]`, RGBA in [-1, 1] |
|
| 130 |
+
|
| 131 |
+
- **Prompt.** One tokenization of the text-to-image chat template around the prompt, no padding.
|
| 132 |
+
The DiT reads `hidden[:, drop_idx:]`. `drop_idx` is the token count of the template's system
|
| 133 |
+
part: 14 with this tokenizer. Compute it; do not hard-code it.
|
| 134 |
+
- **Sequence.** `[text L | image N]`, image tokens in raster order, one token per 16×16 px tile
|
| 135 |
+
(256² → 256 tokens, 512² → 1024, 1024² → 4096). There is no 2×2 latent packing.
|
| 136 |
+
- **RoPE.** Text token `i` sits at `(i, i, i)`. Image token `(y, x)` sits at `(L, y − (h − h//2),
|
| 137 |
+
x − (w − w//2))`. `host/` has the per-axis tables for positions −1024..8191.
|
| 138 |
+
- **Sampler.** Shifted sigmas: `μ` from the image-token count, exponential shift, terminal stretch
|
| 139 |
+
to 0.02. The DiT's `timestep` input is `timesteps[i] / 1000` in fp32 (σ to within 1 ulp), and each step is `x += (σᵢ₊₁ − σᵢ)·v`.
|
| 140 |
+
- **VAE.** The graph unpacks the tokens and applies `latents·std + mean` itself. Feed it the raw
|
| 141 |
+
sampler latent.
|
| 142 |
+
|
| 143 |
+
The full contract, with the formulas a Swift host needs:
|
| 144 |
+
[`host/host_contract.md`](https://huggingface.co/mlboydaisuke/RGBA-Image-2.1-CoreAI/blob/main/host/host_contract.md).
|
| 145 |
+
|
| 146 |
+
## Measured
|
| 147 |
+
|
| 148 |
+
M4 Max, macOS 27.0 (26A428). Bundles compiled ahead of time with `--expect-frequent-reshapes`,
|
| 149 |
+
run with `SpecializationOptions.default()`.
|
| 150 |
+
|
| 151 |
+
**DiT speed** (`bench_dit.py`, text L = 40, GPU lock held, load average ~10):
|
| 152 |
+
|
| 153 |
+
| size | image tokens | first call | s/forward (warm median of 5) | 40 steps |
|
| 154 |
+
| --- | --- | --- | --- | --- |
|
| 155 |
+
| 256² | 256 | 2.94 s | 0.343 s | 13.7 s |
|
| 156 |
+
| 512² | 1024 | 1.14 s | 1.104 s | 44 s |
|
| 157 |
+
| 1024² | 4096 | 4.77 s | 4.757 s | 190 s |
|
| 158 |
+
|
| 159 |
+
**Text encoder:** 0.181 s per call at 32 tokens and 0.271 s at 128 (warm median). The first call
|
| 160 |
+
takes 0.37 s; loading takes 19.6 s.
|
| 161 |
+
|
| 162 |
+
**One 256² image end to end:** encoder 1.15 s, DiT 40 steps 45 s (the first step 32 s, then
|
| 163 |
+
0.34 s per step), VAE 0.16 s.
|
| 164 |
+
|
| 165 |
+
**Memory.** A 512² run with all three compiled bundles loaded in one process (`pipeline_engine.py`,
|
| 166 |
+
free prompt): peak resident set 58.3 GB (`/usr/bin/time -l`), wall 108 s of which loading is 43 s,
|
| 167 |
+
the DiT 40 steps 63 s (first step 21 s, then 1.08 s), encoder 0.5 s, VAE 0.4 s.
|
| 168 |
+
|
| 169 |
+
**Fidelity** against the fp32 diffusers reference (prompt "a red apple on a wooden table, studio
|
| 170 |
+
lighting", seed 1234, the same noise, 40 steps):
|
| 171 |
+
|
| 172 |
+
| size | white-composited RGB PSNR | alpha max\|Δ\| | final latent corr |
|
| 173 |
+
| --- | --- | --- | --- |
|
| 174 |
+
| 256² | 46.72 dB (46.7–50.4 over 5 runs) | 1/255 | 0.999987 |
|
| 175 |
+
| 512² | 35.49 dB (23.6–43.4 over 6 runs, 4 of them ≥ 30 dB) | 2/255 | 0.999559 |
|
| 176 |
+
|
| 177 |
+
The ranges are over runs whose prompt embeddings differ at the 1e-5 level.
|
| 178 |
+
|
| 179 |
+
**The model's own bf16 band.** The official pipeline run in bf16 (diffusers main, MPS), scored
|
| 180 |
+
against the same fp32 reference:
|
| 181 |
+
|
| 182 |
+
| size | official pipeline in bf16 | this port |
|
| 183 |
+
| --- | --- | --- |
|
| 184 |
+
| 256² | 43.47 dB, final latent corr 0.999947 | 46.72 dB, 0.999987 |
|
| 185 |
+
| 512² | 33.19 dB, 0.999053 | 35.49 dB, 0.999559 |
|
| 186 |
+
|
| 187 |
+
**Gates:**
|
| 188 |
+
|
| 189 |
+
- DiT re-authored in plain PyTorch vs diffusers (fp32, 2 random layers): max|Δ| 7.2e-7. With the
|
| 190 |
+
real weights in fp32, teacher-forced on all 40 steps: corr 1.000000000.
|
| 191 |
+
- DiT bundle (bf16, GPU), teacher-forced against the fp32 reference: 40/40 steps corr ≥ 0.999854
|
| 192 |
+
at 256² (NaN 0) and ≥ 0.999844 at 512².
|
| 193 |
+
- Text encoder re-authored in fp32: bit-exact with the reference on all 32 tokens. The bundle: min
|
| 194 |
+
per-token corr 0.999999999 on the reference prompt, and ≥ 0.99999998 over 3 prompts × 7 lengths.
|
| 195 |
+
The host tokenizer gives the processor's ids on all 3 prompts, `drop_idx` 14.
|
| 196 |
+
- VAE bundles on all four channels: corr 1.0000000, max|Δ| 1.35e-5 at 256² and 1.29e-5 at 512²;
|
| 197 |
+
2.33e-5 at 1024² against torch on a synthetic latent.
|
| 198 |
+
- Sampler: sigmas, timesteps and all 40 Euler steps bit-exact at 256² and 512².
|
| 199 |
+
- Transparency: the dragon prompt above at 512², seed 42: alpha min 0, mean 126; the background
|
| 200 |
+
corners average 0.96/255.
|
| 201 |
+
|
| 202 |
+
## Lessons
|
| 203 |
+
|
| 204 |
+
1. **A 2-layer probe does not clear a 32-layer graph.** On 26A428 the Python runtime puts a Neural
|
| 205 |
+
Engine region inside the 32-block bf16 DiT, and the ANE inference fails (`Code=-19`). It fails
|
| 206 |
+
under JIT and under a plain AOT compile; the 2-layer probe never triggered it. The GPU path that
|
| 207 |
+
runs is AOT with `--expect-frequent-reshapes`: 2 min 9 s to compile, 27 GB, MPSGraph delegates
|
| 208 |
+
only. The int8 DiT does not compile with that flag (`Pass failed: MPSMemrefAllocFusion`).
|
| 209 |
+
2. **The encoder's `<|im_start|>` tokens need fp32 compute, not fp32 storage.** Token 14, the `<|im_start|>`
|
| 210 |
+
that opens the user turn, is the first token the DiT reads. Its residual grows to |h| ≈ 9,100 in
|
| 211 |
+
layers 17–34, and the last two layers cancel it to ≈ 100. bf16 cannot hold that cancellation:
|
| 212 |
+
per-token corr 0.968 in torch bf16, 0.976 on the engine. An fp32 residual stream alone did not
|
| 213 |
+
hold across prompts and lengths. Full fp32 compute over bf16-stored weights did: min token corr
|
| 214 |
+
0.999999999 at the same 14.10 GiB. The DiT barely notices the bf16 error (velocity corr
|
| 215 |
+
≥ 0.99998); only a per-token gate catches it.
|
| 216 |
+
3. **Judge a bf16 port by the model's own bf16 band, and look at the image when PSNR drops.** At
|
| 217 |
+
512², six runs whose prompt embeddings differ at the 1e-5 level span 23.6–43.4 dB. The two runs
|
| 218 |
+
near 24 dB show the same apple with one extra leaf on the stem: a semantic fork, not noise. The
|
| 219 |
+
fp32 reference stays at 82 dB under a 1e-5 perturbation, so the fork comes from the bf16 DiT's
|
| 220 |
+
per-step error (0.4–1.8 %). The official pipeline in bf16 scores 33.19 dB at 512²; this port's
|
| 221 |
+
35.49 dB is in that band.
|
| 222 |
+
|
| 223 |
+
Port notes, every gate and the dead ends:
|
| 224 |
+
[`knowledge/qwenimage21-port.md`](https://github.com/john-rocky/coreai-model-zoo/blob/main/knowledge/qwenimage21-port.md).
|
| 225 |
+
Scripts: [`conversion/qwenimage21/`](https://github.com/john-rocky/coreai-model-zoo/tree/main/conversion/qwenimage21).
|
| 226 |
+
|
| 227 |
+
## Licence
|
| 228 |
+
|
| 229 |
+
The weights are Qwen Materials under the **Qwen RESEARCH LICENSE AGREEMENT** (release date
|
| 230 |
+
2026-09-20), included unchanged as
|
| 231 |
+
[`LICENSE`](https://huggingface.co/mlboydaisuke/RGBA-Image-2.1-CoreAI/blob/main/LICENSE). A summary
|
| 232 |
+
follows; the Agreement is what binds.
|
| 233 |
+
|
| 234 |
+
- **§2, non-commercial only.** You may use, copy, modify and redistribute the Materials for
|
| 235 |
+
research or evaluation purposes only. Commercial use needs a separate licence from Hangzhou
|
| 236 |
+
Tongyi Laboratory Technology Co., Ltd.; §2(b) gives the contact.
|
| 237 |
+
- **§3, redistribution,** under three conditions: every recipient gets a copy of the Agreement
|
| 238 |
+
(`LICENSE`); modified files carry a notice that they were changed (`NOTICE` lists every change);
|
| 239 |
+
and copies keep the attribution text below in a "Notice" file (`NOTICE`, first line).
|
| 240 |
+
- **§4(b).** An AI model created from the Materials and made available displays "Built with Qwen"
|
| 241 |
+
prominently in its documentation. This card does, at the top.
|
| 242 |
+
- **§4(c).** "Qwen" is not the primary name of a derivative. This port is named RGBA-Image-2.1;
|
| 243 |
+
"Core AI port of Qwen-Image-2.1" is the descriptive use the Agreement permits.
|
| 244 |
+
|
| 245 |
+
[`NOTICE`](https://huggingface.co/mlboydaisuke/RGBA-Image-2.1-CoreAI/blob/main/NOTICE) begins with
|
| 246 |
+
the attribution text §3(c) requires:
|
| 247 |
+
|
| 248 |
+
> Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved.
|
config.json
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_class_name": "QwenImage21Transformer2DModel",
|
| 3 |
+
"_diffusers_version": "0.37.0.dev0",
|
| 4 |
+
"attention_head_dim": 128,
|
| 5 |
+
"axes_dims_rope": [
|
| 6 |
+
16,
|
| 7 |
+
56,
|
| 8 |
+
56
|
| 9 |
+
],
|
| 10 |
+
"context_in_dim": 4096,
|
| 11 |
+
"in_channels": 64,
|
| 12 |
+
"num_attention_heads": 32,
|
| 13 |
+
"num_layers": 32,
|
| 14 |
+
"out_channels": 64,
|
| 15 |
+
"patch_size": 1,
|
| 16 |
+
"mlp_ratio": 3,
|
| 17 |
+
"eps": 1e-06,
|
| 18 |
+
"causal_condition": true
|
| 19 |
+
}
|
host/host_contract.md
ADDED
|
@@ -0,0 +1,153 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Host contract — RGBA-Image-2.1 (Core AI port of Qwen-Image-2.1)
|
| 2 |
+
|
| 3 |
+
Everything a host computes around the three graphs, for text-to-image, batch 1, no CFG.
|
| 4 |
+
The Python reference of every step below is in the zoo's
|
| 5 |
+
[`conversion/qwenimage21/`](https://github.com/john-rocky/coreai-model-zoo/tree/main/conversion/qwenimage21):
|
| 6 |
+
`qi21_tokenize.py` (tokenizer), `qi21_host.py` (RoPE, pack/unpack), `qi21_sched.py` (sampler),
|
| 7 |
+
`pipeline_engine.py` (the whole loop on the three bundles). The first three match the diffusers-main
|
| 8 |
+
pipeline (`Qwen/Qwen-Image-2.1` @ `790c926`) exactly: the same token ids, bit-identical RoPE tables,
|
| 9 |
+
bit-identical sigmas and Euler steps. `pipeline_engine.py` is scored against the pipeline's fp32 images.
|
| 10 |
+
|
| 11 |
+
## 1. The graphs
|
| 12 |
+
|
| 13 |
+
Every graph has one function, `main`. All tensors crossing a graph boundary are fp32, except
|
| 14 |
+
`input_ids` (int32).
|
| 15 |
+
|
| 16 |
+
| bundle | inputs | output | axes |
|
| 17 |
+
| --- | --- | --- | --- |
|
| 18 |
+
| `qi21_encoder_dynL_w16a32_ids_iofp32.aimodel` | `input_ids [1,Lfull]` int32 | `hidden [1,Lfull,4096]` | `Lfull` 16..512 |
|
| 19 |
+
| `qi21_dit_full_bf16_dyn_iofp32.aimodel` | `img_tokens [1,N,64]`, `txt_feats [1,L,4096]`, `timestep [1]`, `txt_cos [1,L,64]`, `txt_sin [1,L,64]`, `img_cos [1,N,64]`, `img_sin [1,N,64]` | `vel [1,N,64]` | `L` 8..512, `N` 64..4096 |
|
| 20 |
+
| `qi21_vae_{256,512,1024}_fp32.aimodel` | `latents_packed [1,N,64]` | `image [1,4,S,S]` | fixed: `N = (S/16)²` |
|
| 21 |
+
|
| 22 |
+
- Encoder: the Qwen3-VL-8B text stack (36 layers). bf16 weights, fp32 compute. `embed_tokens` is
|
| 23 |
+
inside the graph. The output is the residual stream after the last layer, **before** the final
|
| 24 |
+
RMSNorm. Do not apply a norm on the host.
|
| 25 |
+
- DiT: 32 blocks, bf16 weights and compute, fp32 boundary.
|
| 26 |
+
- VAE: decoder only, fp32. The graph unpacks the tokens and applies `latents * std + mean` itself.
|
| 27 |
+
|
| 28 |
+
## 2. Tokenize
|
| 29 |
+
|
| 30 |
+
Tokenizer: `tokenizer/tokenizer.json` (Qwen2 BPE, byte-level). No BOS token. An empty prompt is
|
| 31 |
+
replaced by a single space `" "`.
|
| 32 |
+
|
| 33 |
+
Template (text-to-image):
|
| 34 |
+
|
| 35 |
+
```
|
| 36 |
+
<|im_start|>system\nComprehend and analyze the provided prompt.<|im_end|>\n<|im_start|>user\n{prompt}<|im_end|>\n<|im_start|>assistant\n
|
| 37 |
+
```
|
| 38 |
+
|
| 39 |
+
`\n` is a newline character. Encode the whole string once, with no padding and no truncation, to get
|
| 40 |
+
`ids` (length `Lfull`).
|
| 41 |
+
|
| 42 |
+
**`drop_idx`** is the number of tokens of the system part alone,
|
| 43 |
+
`<|im_start|>system\nComprehend and analyze the provided prompt.<|im_end|>\n`. Compute it by encoding
|
| 44 |
+
that string. With this tokenizer it is 14:
|
| 45 |
+
`[151644, 8948, 198, 1092, 30782, 408, 323, 23643, 279, 3897, 9934, 13, 151645, 198]`.
|
| 46 |
+
Check that `ids[:drop_idx]` equals those tokens.
|
| 47 |
+
|
| 48 |
+
`Lfull` must be 16..512 (the encoder's axis), and the DiT needs `L = Lfull − drop_idx` in 8..512.
|
| 49 |
+
The template around the empty-prompt substitute `" "` is 23 tokens (`L` = 9).
|
| 50 |
+
|
| 51 |
+
## 3. Encode
|
| 52 |
+
|
| 53 |
+
```
|
| 54 |
+
hidden = encoder(input_ids = ids as int32 [1, Lfull]) # [1, Lfull, 4096]
|
| 55 |
+
prompt_embeds = hidden[:, drop_idx : Lfull] # [1, L, 4096], L = Lfull - drop_idx
|
| 56 |
+
```
|
| 57 |
+
|
| 58 |
+
Pass exactly `Lfull` tokens; the axis is dynamic, so no padding is needed. (Attention is causal, so
|
| 59 |
+
pad tokens never reach the first `Lfull` outputs, but a longer input moves the GPU result by about
|
| 60 |
+
4e-6 relative.) Run the encoder once per image.
|
| 61 |
+
|
| 62 |
+
## 4. Sequence layout and RoPE
|
| 63 |
+
|
| 64 |
+
The DiT's sequence is `[text L | image N]`: the `L` prompt tokens, then the image tokens in raster
|
| 65 |
+
order. One image token covers one 16×16 px tile of the output: for an `S×S` image,
|
| 66 |
+
`h = w = S/16` and `N = h·w` (256² → 256 tokens, 512² → 1024, 1024² → 4096). There is no 2×2 packing.
|
| 67 |
+
|
| 68 |
+
Each token has a 3-axis position `(frame, height, width)`:
|
| 69 |
+
|
| 70 |
+
- text token `i` (0-based) → `(i, i, i)`;
|
| 71 |
+
- image token at row `y`, column `x` (0-based, token index `y·w + x`) →
|
| 72 |
+
`(L, y − (h − h//2), x − (w − w//2))`. The grid is centred on zero. For `h = 16`, rows go −8..7.
|
| 73 |
+
|
| 74 |
+
`host/rope_axis{a}_{cos,sin}.f32` hold one row per position −1024..8191: fp32, little-endian,
|
| 75 |
+
row-major `[9216, P_a]` with `P = (8, 28, 28)`. The row for position `p` is `p + 1024`. A token's
|
| 76 |
+
64 values are the three rows concatenated in axis order:
|
| 77 |
+
|
| 78 |
+
```
|
| 79 |
+
cos[token] = axis0_cos[f + 1024] ++ axis1_cos[hpos + 1024] ++ axis2_cos[wpos + 1024] # 8 + 28 + 28
|
| 80 |
+
sin[token] = the same with the _sin tables
|
| 81 |
+
txt_cos, txt_sin = rows of the L text tokens # [1, L, 64]
|
| 82 |
+
img_cos, img_sin = rows of the N image tokens # [1, N, 64]
|
| 83 |
+
```
|
| 84 |
+
|
| 85 |
+
The tables are `torch.polar(1, outer(pos, 1 / 10000^(arange(0, d, 2)/d)))` for axis widths
|
| 86 |
+
`d = 16, 56, 56`, the same numbers as `QwenImage21Rope.freqs` of diffusers main (bit-exact). Axes 1
|
| 87 |
+
and 2 have the same width, so their files are byte-identical. The four tensors depend only on `L`,
|
| 88 |
+
`h` and `w`: build them once per image, not per step.
|
| 89 |
+
|
| 90 |
+
## 5. Noise and pack
|
| 91 |
+
|
| 92 |
+
The initial latent is standard normal noise: 64 channels × `h` rows × `w` columns, channel-major.
|
| 93 |
+
The reference pipeline draws it as `randn((1, 1, 64, h, w))` and packs it into DiT tokens:
|
| 94 |
+
|
| 95 |
+
```
|
| 96 |
+
x[0, y·w + x_, c] = z[c, y, x_] # = z.view(1, 64, h·w).transpose(1, 2)
|
| 97 |
+
```
|
| 98 |
+
|
| 99 |
+
Any N(0, 1) noise generates an image. Matching the Python engine for a given seed needs its exact
|
| 100 |
+
draw: `torch.randn((1, 1, 64, h, w), generator=torch.Generator("cpu").manual_seed(seed))`.
|
| 101 |
+
|
| 102 |
+
## 6. Sampler (FlowMatch Euler, 40 steps, no CFG)
|
| 103 |
+
|
| 104 |
+
Constants in `host/scheduler.json` (from the checkpoint's `scheduler_config.json`). All arrays fp32.
|
| 105 |
+
|
| 106 |
+
```
|
| 107 |
+
sigmas = linspace(1, 1/steps, steps) # steps = 40
|
| 108 |
+
mu = N · m + b, m = (max_shift − base_shift) / (max_image_seq_len − base_image_seq_len),
|
| 109 |
+
b = base_shift − m · base_image_seq_len # N = image tokens
|
| 110 |
+
sigmas = e^mu / (e^mu + (1/sigmas − 1)) # time_shift_type "exponential"
|
| 111 |
+
sigmas = 1 − (1 − sigmas) / ((1 − sigmas[-1]) / (1 − shift_terminal))
|
| 112 |
+
timesteps = sigmas · num_train_timesteps # fp32
|
| 113 |
+
sigmas = sigmas ++ [0] # 41 values
|
| 114 |
+
for i in 0 ..< steps:
|
| 115 |
+
t = timesteps[i] / 1000 # fp32; this is the DiT `timestep` input
|
| 116 |
+
vel = dit(img_tokens = x, txt_feats = prompt_embeds, timestep = [t], txt_cos, txt_sin, img_cos, img_sin)
|
| 117 |
+
x = x + (sigmas[i+1] − sigmas[i]) · vel # fp32
|
| 118 |
+
```
|
| 119 |
+
|
| 120 |
+
`mu` is 0.5 at 256² (N = 256), 0.5387… at 512², 0.6935… at 1024². Keep `t = timesteps[i] / 1000` as
|
| 121 |
+
written (multiply by 1000, then divide) to match the reference bit for bit. `qi21_sched.py`
|
| 122 |
+
reproduces the reference sigmas, timesteps and every Euler step bit-exactly at 256² and 512².
|
| 123 |
+
|
| 124 |
+
## 7. Decode
|
| 125 |
+
|
| 126 |
+
```
|
| 127 |
+
image = vae_S(latents_packed = x) # x after the last step, [1, N, 64], exactly as the sampler holds it
|
| 128 |
+
rgba8 = round(clip(image · 0.5 + 0.5, 0, 1) · 255) # [1, 4, S, S] -> channels R, G, B, A
|
| 129 |
+
```
|
| 130 |
+
|
| 131 |
+
Feed the sampler's latent unchanged: no unpacking and no `* std + mean` on the host (the graph does
|
| 132 |
+
both). The four channels are what the reference pipeline saves as an RGBA PNG.
|
| 133 |
+
|
| 134 |
+
## 8. Compile before running on the GPU
|
| 135 |
+
|
| 136 |
+
On macOS 27.0 (26A428) the Python runtime crashes on the 32-block DiT when the `.aimodel` is
|
| 137 |
+
compiled just-in-time: the MPSGraph delegate forms a Neural Engine region inside the graph and the
|
| 138 |
+
ANE inference fails (`ANERegion.mm:414 failed assertion … Code=-19`). A plain ahead-of-time compile
|
| 139 |
+
fails the same way. What runs is an ahead-of-time compile with `--expect-frequent-reshapes`, loaded
|
| 140 |
+
with `SpecializationOptions.default()`:
|
| 141 |
+
|
| 142 |
+
```
|
| 143 |
+
xcrun coreai-build compile qi21_dit_full_bf16_dyn_iofp32.aimodel \
|
| 144 |
+
--output aot/qi21_dit_full_bf16_dyn_iofp32 \
|
| 145 |
+
--platform macOS --architecture h16c --preferred-compute gpu --expect-frequent-reshapes
|
| 146 |
+
# -> aot/qi21_dit_full_bf16_dyn_iofp32/qi21_dit_full_bf16_dyn_iofp32.h16c.aimodelc
|
| 147 |
+
```
|
| 148 |
+
|
| 149 |
+
Do the same for the encoder and the VAE bundle you use; every gate of this port ran all three this
|
| 150 |
+
way, on an M4 Max with `--architecture h16c` (without `--architecture`, `coreai-build` compiles for
|
| 151 |
+
every supported architecture). The compiled DiT is about 1.9× its `.aimodel` (27 GB); the encoder's
|
| 152 |
+
stays at 14.10 GiB. Whether a Swift host using `GraphModel(computeUnits: .gpu)` hits the same Neural
|
| 153 |
+
Engine region has not been tested.
|
host/rope_axis0_cos.f32
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2a67b7ed9fea4c3f775dea442cf921f6cfa25be5d7da2671a87a01b154af3b5b
|
| 3 |
+
size 294912
|
host/rope_axis0_sin.f32
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a6da418893c5551a329985da7efbeb20858736f3fb21077239405f9ffb5f578f
|
| 3 |
+
size 294912
|
host/rope_axis1_cos.f32
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8152a557c6febd21a2baeded29590c6b5bf18c439da5ddbe110c1273379c6ee3
|
| 3 |
+
size 1032192
|
host/rope_axis1_sin.f32
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a9e468de85274824afd0a30ecf8ca8bb53005ed0d606b656a65703e51c1b21cc
|
| 3 |
+
size 1032192
|
host/rope_axis2_cos.f32
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8152a557c6febd21a2baeded29590c6b5bf18c439da5ddbe110c1273379c6ee3
|
| 3 |
+
size 1032192
|
host/rope_axis2_sin.f32
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a9e468de85274824afd0a30ecf8ca8bb53005ed0d606b656a65703e51c1b21cc
|
| 3 |
+
size 1032192
|
host/scheduler.json
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"base_image_seq_len": 256,
|
| 3 |
+
"max_image_seq_len": 8192,
|
| 4 |
+
"base_shift": 0.5,
|
| 5 |
+
"max_shift": 0.9,
|
| 6 |
+
"shift_terminal": 0.02,
|
| 7 |
+
"time_shift_type": "exponential",
|
| 8 |
+
"num_train_timesteps": 1000,
|
| 9 |
+
"steps": 40
|
| 10 |
+
}
|
qi21_dit_full_bf16_dyn_iofp32.aimodel/main.hash
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
���6l�A��T����?*b���coDw��bN
|
qi21_dit_full_bf16_dyn_iofp32.aimodel/main.mlirb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9c87e0b2366cc61e4185b154dfece8d23f2a628483bc636f4477fcce624e1310
|
| 3 |
+
size 14231004496
|
qi21_dit_full_bf16_dyn_iofp32.aimodel/metadata.json
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"producer" : "coreai-core 1.0.0b2",
|
| 3 |
+
"assetVersion" : "2.0",
|
| 4 |
+
"license" : "qwen-research",
|
| 5 |
+
"description" : "Qwen-Image-2.1 DiT (full), bf16 weights+compute, fp32 I\/O, dynamic text\/image axes. Source: Qwen\/Qwen-Image-2.1@790c926.",
|
| 6 |
+
"creationDate" : "20260925T055156Z"
|
| 7 |
+
}
|
qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/main.hash
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
�{�dv6H�j2�:!��-�?�,8��]�d�4V2C
|
qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/main.mlirb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d37b8e64763648cb6a32b23a2199b42d0ee93fac2c3899835d8e64d534563243
|
| 3 |
+
size 15138226830
|
qi21_encoder_dynL_w16a32_ids_iofp32.aimodel/metadata.json
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"creationDate" : "20260925T065421Z",
|
| 3 |
+
"assetVersion" : "2.0",
|
| 4 |
+
"description" : "Qwen-Image-2.1 text encoder (Qwen3-VL-8B text stack, all 36 layers), bf16 weights, fp32 compute, int32 input_ids -> fp32 hidden (last layer, before the final norm), dynamic L 16..512. Source: Qwen\/Qwen-Image-2.1@790c926.",
|
| 5 |
+
"license" : "qwen-research",
|
| 6 |
+
"producer" : "coreai-core 1.0.0b2"
|
| 7 |
+
}
|
qi21_vae_1024_fp32.aimodel/main.hash
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
�/�6�6ޡrr�N��g�}�H[�%1��9D
|
qi21_vae_1024_fp32.aimodel/main.mlirb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d82fc436a51e36dea17272c64e8dd667df7dcd0c1d485bf225319303ed823944
|
| 3 |
+
size 1012440351
|
qi21_vae_1024_fp32.aimodel/metadata.json
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"description" : "Qwen-Image-2.1 VAE decoder, 1024x1024, fp32: latents_packed [1,4096,64] (sampler output, normalised) -> image [1,4,1024,1024] RGBA in [-1,1]; unpack and latents*std+mean inside the graph. Source: Qwen\/Qwen-Image-2.1@790c926.",
|
| 3 |
+
"assetVersion" : "2.0",
|
| 4 |
+
"license" : "qwen-research",
|
| 5 |
+
"creationDate" : "20260925T071221Z",
|
| 6 |
+
"producer" : "coreai-core 1.0.0b2"
|
| 7 |
+
}
|
qi21_vae_256_fp32.aimodel/main.hash
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
�9bf�DN�k�G�ø�&+U"�w�?Rwb
|
qi21_vae_256_fp32.aimodel/main.mlirb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:82396266f39abd444e11fd6b17b20347eeb7c3b8d5262b5522f877b13f527762
|
| 3 |
+
size 1012434454
|
qi21_vae_256_fp32.aimodel/metadata.json
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"creationDate" : "20260925T071101Z",
|
| 3 |
+
"producer" : "coreai-core 1.0.0b2",
|
| 4 |
+
"assetVersion" : "2.0",
|
| 5 |
+
"license" : "qwen-research",
|
| 6 |
+
"description" : "Qwen-Image-2.1 VAE decoder, 256x256, fp32: latents_packed [1,256,64] (sampler output, normalised) -> image [1,4,256,256] RGBA in [-1,1]; unpack and latents*std+mean inside the graph. Source: Qwen\/Qwen-Image-2.1@790c926."
|
| 7 |
+
}
|
qi21_vae_512_fp32.aimodel/main.hash
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
ϋpIc��(f����$�6w�����~��(.V
|
qi21_vae_512_fp32.aimodel/main.mlirb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cf8b1b14704963bc912866b81482a2ac24b1367796c6f60eeed17e95e9282e56
|
| 3 |
+
size 1012436431
|
qi21_vae_512_fp32.aimodel/metadata.json
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"description" : "Qwen-Image-2.1 VAE decoder, 512x512, fp32: latents_packed [1,1024,64] (sampler output, normalised) -> image [1,4,512,512] RGBA in [-1,1]; unpack and latents*std+mean inside the graph. Source: Qwen\/Qwen-Image-2.1@790c926.",
|
| 3 |
+
"assetVersion" : "2.0",
|
| 4 |
+
"license" : "qwen-research",
|
| 5 |
+
"creationDate" : "20260925T071145Z",
|
| 6 |
+
"producer" : "coreai-core 1.0.0b2"
|
| 7 |
+
}
|
tokenizer/added_tokens.json
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"</think>": 151668,
|
| 3 |
+
"</tool_call>": 151658,
|
| 4 |
+
"</tool_response>": 151666,
|
| 5 |
+
"<think>": 151667,
|
| 6 |
+
"<tool_call>": 151657,
|
| 7 |
+
"<tool_response>": 151665,
|
| 8 |
+
"<|box_end|>": 151649,
|
| 9 |
+
"<|box_start|>": 151648,
|
| 10 |
+
"<|endoftext|>": 151643,
|
| 11 |
+
"<|file_sep|>": 151664,
|
| 12 |
+
"<|fim_middle|>": 151660,
|
| 13 |
+
"<|fim_pad|>": 151662,
|
| 14 |
+
"<|fim_prefix|>": 151659,
|
| 15 |
+
"<|fim_suffix|>": 151661,
|
| 16 |
+
"<|im_end|>": 151645,
|
| 17 |
+
"<|im_start|>": 151644,
|
| 18 |
+
"<|image_pad|>": 151655,
|
| 19 |
+
"<|object_ref_end|>": 151647,
|
| 20 |
+
"<|object_ref_start|>": 151646,
|
| 21 |
+
"<|quad_end|>": 151651,
|
| 22 |
+
"<|quad_start|>": 151650,
|
| 23 |
+
"<|repo_name|>": 151663,
|
| 24 |
+
"<|video_pad|>": 151656,
|
| 25 |
+
"<|vision_end|>": 151653,
|
| 26 |
+
"<|vision_pad|>": 151654,
|
| 27 |
+
"<|vision_start|>": 151652
|
| 28 |
+
}
|
tokenizer/chat_template.jinja
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{%- if tools %}
|
| 2 |
+
{{- '<|im_start|>system\n' }}
|
| 3 |
+
{%- if messages[0].role == 'system' %}
|
| 4 |
+
{%- if messages[0].content is string %}
|
| 5 |
+
{{- messages[0].content }}
|
| 6 |
+
{%- else %}
|
| 7 |
+
{%- for content in messages[0].content %}
|
| 8 |
+
{%- if 'text' in content %}
|
| 9 |
+
{{- content.text }}
|
| 10 |
+
{%- endif %}
|
| 11 |
+
{%- endfor %}
|
| 12 |
+
{%- endif %}
|
| 13 |
+
{{- '\n\n' }}
|
| 14 |
+
{%- endif %}
|
| 15 |
+
{{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
| 16 |
+
{%- for tool in tools %}
|
| 17 |
+
{{- "\n" }}
|
| 18 |
+
{{- tool | tojson }}
|
| 19 |
+
{%- endfor %}
|
| 20 |
+
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
| 21 |
+
{%- else %}
|
| 22 |
+
{%- if messages[0].role == 'system' %}
|
| 23 |
+
{{- '<|im_start|>system\n' }}
|
| 24 |
+
{%- if messages[0].content is string %}
|
| 25 |
+
{{- messages[0].content }}
|
| 26 |
+
{%- else %}
|
| 27 |
+
{%- for content in messages[0].content %}
|
| 28 |
+
{%- if 'text' in content %}
|
| 29 |
+
{{- content.text }}
|
| 30 |
+
{%- endif %}
|
| 31 |
+
{%- endfor %}
|
| 32 |
+
{%- endif %}
|
| 33 |
+
{{- '<|im_end|>\n' }}
|
| 34 |
+
{%- endif %}
|
| 35 |
+
{%- endif %}
|
| 36 |
+
{%- set image_count = namespace(value=0) %}
|
| 37 |
+
{%- set video_count = namespace(value=0) %}
|
| 38 |
+
{%- for message in messages %}
|
| 39 |
+
{%- if message.role == "user" %}
|
| 40 |
+
{{- '<|im_start|>' + message.role + '\n' }}
|
| 41 |
+
{%- if message.content is string %}
|
| 42 |
+
{{- message.content }}
|
| 43 |
+
{%- else %}
|
| 44 |
+
{%- for content in message.content %}
|
| 45 |
+
{%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
|
| 46 |
+
{%- set image_count.value = image_count.value + 1 %}
|
| 47 |
+
{%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
|
| 48 |
+
<|vision_start|><|image_pad|><|vision_end|>
|
| 49 |
+
{%- elif content.type == 'video' or 'video' in content %}
|
| 50 |
+
{%- set video_count.value = video_count.value + 1 %}
|
| 51 |
+
{%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
|
| 52 |
+
<|vision_start|><|video_pad|><|vision_end|>
|
| 53 |
+
{%- elif 'text' in content %}
|
| 54 |
+
{{- content.text }}
|
| 55 |
+
{%- endif %}
|
| 56 |
+
{%- endfor %}
|
| 57 |
+
{%- endif %}
|
| 58 |
+
{{- '<|im_end|>\n' }}
|
| 59 |
+
{%- elif message.role == "assistant" %}
|
| 60 |
+
{{- '<|im_start|>' + message.role + '\n' }}
|
| 61 |
+
{%- if message.content is string %}
|
| 62 |
+
{{- message.content }}
|
| 63 |
+
{%- else %}
|
| 64 |
+
{%- for content_item in message.content %}
|
| 65 |
+
{%- if 'text' in content_item %}
|
| 66 |
+
{{- content_item.text }}
|
| 67 |
+
{%- endif %}
|
| 68 |
+
{%- endfor %}
|
| 69 |
+
{%- endif %}
|
| 70 |
+
{%- if message.tool_calls %}
|
| 71 |
+
{%- for tool_call in message.tool_calls %}
|
| 72 |
+
{%- if (loop.first and message.content) or (not loop.first) %}
|
| 73 |
+
{{- '\n' }}
|
| 74 |
+
{%- endif %}
|
| 75 |
+
{%- if tool_call.function %}
|
| 76 |
+
{%- set tool_call = tool_call.function %}
|
| 77 |
+
{%- endif %}
|
| 78 |
+
{{- '<tool_call>\n{"name": "' }}
|
| 79 |
+
{{- tool_call.name }}
|
| 80 |
+
{{- '", "arguments": ' }}
|
| 81 |
+
{%- if tool_call.arguments is string %}
|
| 82 |
+
{{- tool_call.arguments }}
|
| 83 |
+
{%- else %}
|
| 84 |
+
{{- tool_call.arguments | tojson }}
|
| 85 |
+
{%- endif %}
|
| 86 |
+
{{- '}\n</tool_call>' }}
|
| 87 |
+
{%- endfor %}
|
| 88 |
+
{%- endif %}
|
| 89 |
+
{{- '<|im_end|>\n' }}
|
| 90 |
+
{%- elif message.role == "tool" %}
|
| 91 |
+
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
|
| 92 |
+
{{- '<|im_start|>user' }}
|
| 93 |
+
{%- endif %}
|
| 94 |
+
{{- '\n<tool_response>\n' }}
|
| 95 |
+
{%- if message.content is string %}
|
| 96 |
+
{{- message.content }}
|
| 97 |
+
{%- else %}
|
| 98 |
+
{%- for content in message.content %}
|
| 99 |
+
{%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
|
| 100 |
+
{%- set image_count.value = image_count.value + 1 %}
|
| 101 |
+
{%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
|
| 102 |
+
<|vision_start|><|image_pad|><|vision_end|>
|
| 103 |
+
{%- elif content.type == 'video' or 'video' in content %}
|
| 104 |
+
{%- set video_count.value = video_count.value + 1 %}
|
| 105 |
+
{%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
|
| 106 |
+
<|vision_start|><|video_pad|><|vision_end|>
|
| 107 |
+
{%- elif 'text' in content %}
|
| 108 |
+
{{- content.text }}
|
| 109 |
+
{%- endif %}
|
| 110 |
+
{%- endfor %}
|
| 111 |
+
{%- endif %}
|
| 112 |
+
{{- '\n</tool_response>' }}
|
| 113 |
+
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
| 114 |
+
{{- '<|im_end|>\n' }}
|
| 115 |
+
{%- endif %}
|
| 116 |
+
{%- endif %}
|
| 117 |
+
{%- endfor %}
|
| 118 |
+
{%- if add_generation_prompt %}
|
| 119 |
+
{{- '<|im_start|>assistant\n' }}
|
| 120 |
+
{%- endif %}
|
tokenizer/merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
tokenizer/special_tokens_map.json
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"additional_special_tokens": [
|
| 3 |
+
"<|im_start|>",
|
| 4 |
+
"<|im_end|>",
|
| 5 |
+
"<|object_ref_start|>",
|
| 6 |
+
"<|object_ref_end|>",
|
| 7 |
+
"<|box_start|>",
|
| 8 |
+
"<|box_end|>",
|
| 9 |
+
"<|quad_start|>",
|
| 10 |
+
"<|quad_end|>",
|
| 11 |
+
"<|vision_start|>",
|
| 12 |
+
"<|vision_end|>",
|
| 13 |
+
"<|vision_pad|>",
|
| 14 |
+
"<|image_pad|>",
|
| 15 |
+
"<|video_pad|>"
|
| 16 |
+
],
|
| 17 |
+
"eos_token": {
|
| 18 |
+
"content": "<|im_end|>",
|
| 19 |
+
"lstrip": false,
|
| 20 |
+
"normalized": false,
|
| 21 |
+
"rstrip": false,
|
| 22 |
+
"single_word": false
|
| 23 |
+
},
|
| 24 |
+
"pad_token": {
|
| 25 |
+
"content": "<|endoftext|>",
|
| 26 |
+
"lstrip": false,
|
| 27 |
+
"normalized": false,
|
| 28 |
+
"rstrip": false,
|
| 29 |
+
"single_word": false
|
| 30 |
+
}
|
| 31 |
+
}
|
tokenizer/tokenizer.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:aeb13307a71acd8fe81861d94ad54ab689df773318809eed3cbe794b4492dae4
|
| 3 |
+
size 11422654
|
tokenizer/tokenizer_config.json
ADDED
|
@@ -0,0 +1,240 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_bos_token": false,
|
| 3 |
+
"add_prefix_space": false,
|
| 4 |
+
"added_tokens_decoder": {
|
| 5 |
+
"151643": {
|
| 6 |
+
"content": "<|endoftext|>",
|
| 7 |
+
"lstrip": false,
|
| 8 |
+
"normalized": false,
|
| 9 |
+
"rstrip": false,
|
| 10 |
+
"single_word": false,
|
| 11 |
+
"special": true
|
| 12 |
+
},
|
| 13 |
+
"151644": {
|
| 14 |
+
"content": "<|im_start|>",
|
| 15 |
+
"lstrip": false,
|
| 16 |
+
"normalized": false,
|
| 17 |
+
"rstrip": false,
|
| 18 |
+
"single_word": false,
|
| 19 |
+
"special": true
|
| 20 |
+
},
|
| 21 |
+
"151645": {
|
| 22 |
+
"content": "<|im_end|>",
|
| 23 |
+
"lstrip": false,
|
| 24 |
+
"normalized": false,
|
| 25 |
+
"rstrip": false,
|
| 26 |
+
"single_word": false,
|
| 27 |
+
"special": true
|
| 28 |
+
},
|
| 29 |
+
"151646": {
|
| 30 |
+
"content": "<|object_ref_start|>",
|
| 31 |
+
"lstrip": false,
|
| 32 |
+
"normalized": false,
|
| 33 |
+
"rstrip": false,
|
| 34 |
+
"single_word": false,
|
| 35 |
+
"special": true
|
| 36 |
+
},
|
| 37 |
+
"151647": {
|
| 38 |
+
"content": "<|object_ref_end|>",
|
| 39 |
+
"lstrip": false,
|
| 40 |
+
"normalized": false,
|
| 41 |
+
"rstrip": false,
|
| 42 |
+
"single_word": false,
|
| 43 |
+
"special": true
|
| 44 |
+
},
|
| 45 |
+
"151648": {
|
| 46 |
+
"content": "<|box_start|>",
|
| 47 |
+
"lstrip": false,
|
| 48 |
+
"normalized": false,
|
| 49 |
+
"rstrip": false,
|
| 50 |
+
"single_word": false,
|
| 51 |
+
"special": true
|
| 52 |
+
},
|
| 53 |
+
"151649": {
|
| 54 |
+
"content": "<|box_end|>",
|
| 55 |
+
"lstrip": false,
|
| 56 |
+
"normalized": false,
|
| 57 |
+
"rstrip": false,
|
| 58 |
+
"single_word": false,
|
| 59 |
+
"special": true
|
| 60 |
+
},
|
| 61 |
+
"151650": {
|
| 62 |
+
"content": "<|quad_start|>",
|
| 63 |
+
"lstrip": false,
|
| 64 |
+
"normalized": false,
|
| 65 |
+
"rstrip": false,
|
| 66 |
+
"single_word": false,
|
| 67 |
+
"special": true
|
| 68 |
+
},
|
| 69 |
+
"151651": {
|
| 70 |
+
"content": "<|quad_end|>",
|
| 71 |
+
"lstrip": false,
|
| 72 |
+
"normalized": false,
|
| 73 |
+
"rstrip": false,
|
| 74 |
+
"single_word": false,
|
| 75 |
+
"special": true
|
| 76 |
+
},
|
| 77 |
+
"151652": {
|
| 78 |
+
"content": "<|vision_start|>",
|
| 79 |
+
"lstrip": false,
|
| 80 |
+
"normalized": false,
|
| 81 |
+
"rstrip": false,
|
| 82 |
+
"single_word": false,
|
| 83 |
+
"special": true
|
| 84 |
+
},
|
| 85 |
+
"151653": {
|
| 86 |
+
"content": "<|vision_end|>",
|
| 87 |
+
"lstrip": false,
|
| 88 |
+
"normalized": false,
|
| 89 |
+
"rstrip": false,
|
| 90 |
+
"single_word": false,
|
| 91 |
+
"special": true
|
| 92 |
+
},
|
| 93 |
+
"151654": {
|
| 94 |
+
"content": "<|vision_pad|>",
|
| 95 |
+
"lstrip": false,
|
| 96 |
+
"normalized": false,
|
| 97 |
+
"rstrip": false,
|
| 98 |
+
"single_word": false,
|
| 99 |
+
"special": true
|
| 100 |
+
},
|
| 101 |
+
"151655": {
|
| 102 |
+
"content": "<|image_pad|>",
|
| 103 |
+
"lstrip": false,
|
| 104 |
+
"normalized": false,
|
| 105 |
+
"rstrip": false,
|
| 106 |
+
"single_word": false,
|
| 107 |
+
"special": true
|
| 108 |
+
},
|
| 109 |
+
"151656": {
|
| 110 |
+
"content": "<|video_pad|>",
|
| 111 |
+
"lstrip": false,
|
| 112 |
+
"normalized": false,
|
| 113 |
+
"rstrip": false,
|
| 114 |
+
"single_word": false,
|
| 115 |
+
"special": true
|
| 116 |
+
},
|
| 117 |
+
"151657": {
|
| 118 |
+
"content": "<tool_call>",
|
| 119 |
+
"lstrip": false,
|
| 120 |
+
"normalized": false,
|
| 121 |
+
"rstrip": false,
|
| 122 |
+
"single_word": false,
|
| 123 |
+
"special": false
|
| 124 |
+
},
|
| 125 |
+
"151658": {
|
| 126 |
+
"content": "</tool_call>",
|
| 127 |
+
"lstrip": false,
|
| 128 |
+
"normalized": false,
|
| 129 |
+
"rstrip": false,
|
| 130 |
+
"single_word": false,
|
| 131 |
+
"special": false
|
| 132 |
+
},
|
| 133 |
+
"151659": {
|
| 134 |
+
"content": "<|fim_prefix|>",
|
| 135 |
+
"lstrip": false,
|
| 136 |
+
"normalized": false,
|
| 137 |
+
"rstrip": false,
|
| 138 |
+
"single_word": false,
|
| 139 |
+
"special": false
|
| 140 |
+
},
|
| 141 |
+
"151660": {
|
| 142 |
+
"content": "<|fim_middle|>",
|
| 143 |
+
"lstrip": false,
|
| 144 |
+
"normalized": false,
|
| 145 |
+
"rstrip": false,
|
| 146 |
+
"single_word": false,
|
| 147 |
+
"special": false
|
| 148 |
+
},
|
| 149 |
+
"151661": {
|
| 150 |
+
"content": "<|fim_suffix|>",
|
| 151 |
+
"lstrip": false,
|
| 152 |
+
"normalized": false,
|
| 153 |
+
"rstrip": false,
|
| 154 |
+
"single_word": false,
|
| 155 |
+
"special": false
|
| 156 |
+
},
|
| 157 |
+
"151662": {
|
| 158 |
+
"content": "<|fim_pad|>",
|
| 159 |
+
"lstrip": false,
|
| 160 |
+
"normalized": false,
|
| 161 |
+
"rstrip": false,
|
| 162 |
+
"single_word": false,
|
| 163 |
+
"special": false
|
| 164 |
+
},
|
| 165 |
+
"151663": {
|
| 166 |
+
"content": "<|repo_name|>",
|
| 167 |
+
"lstrip": false,
|
| 168 |
+
"normalized": false,
|
| 169 |
+
"rstrip": false,
|
| 170 |
+
"single_word": false,
|
| 171 |
+
"special": false
|
| 172 |
+
},
|
| 173 |
+
"151664": {
|
| 174 |
+
"content": "<|file_sep|>",
|
| 175 |
+
"lstrip": false,
|
| 176 |
+
"normalized": false,
|
| 177 |
+
"rstrip": false,
|
| 178 |
+
"single_word": false,
|
| 179 |
+
"special": false
|
| 180 |
+
},
|
| 181 |
+
"151665": {
|
| 182 |
+
"content": "<tool_response>",
|
| 183 |
+
"lstrip": false,
|
| 184 |
+
"normalized": false,
|
| 185 |
+
"rstrip": false,
|
| 186 |
+
"single_word": false,
|
| 187 |
+
"special": false
|
| 188 |
+
},
|
| 189 |
+
"151666": {
|
| 190 |
+
"content": "</tool_response>",
|
| 191 |
+
"lstrip": false,
|
| 192 |
+
"normalized": false,
|
| 193 |
+
"rstrip": false,
|
| 194 |
+
"single_word": false,
|
| 195 |
+
"special": false
|
| 196 |
+
},
|
| 197 |
+
"151667": {
|
| 198 |
+
"content": "<think>",
|
| 199 |
+
"lstrip": false,
|
| 200 |
+
"normalized": false,
|
| 201 |
+
"rstrip": false,
|
| 202 |
+
"single_word": false,
|
| 203 |
+
"special": false
|
| 204 |
+
},
|
| 205 |
+
"151668": {
|
| 206 |
+
"content": "</think>",
|
| 207 |
+
"lstrip": false,
|
| 208 |
+
"normalized": false,
|
| 209 |
+
"rstrip": false,
|
| 210 |
+
"single_word": false,
|
| 211 |
+
"special": false
|
| 212 |
+
}
|
| 213 |
+
},
|
| 214 |
+
"additional_special_tokens": [
|
| 215 |
+
"<|im_start|>",
|
| 216 |
+
"<|im_end|>",
|
| 217 |
+
"<|object_ref_start|>",
|
| 218 |
+
"<|object_ref_end|>",
|
| 219 |
+
"<|box_start|>",
|
| 220 |
+
"<|box_end|>",
|
| 221 |
+
"<|quad_start|>",
|
| 222 |
+
"<|quad_end|>",
|
| 223 |
+
"<|vision_start|>",
|
| 224 |
+
"<|vision_end|>",
|
| 225 |
+
"<|vision_pad|>",
|
| 226 |
+
"<|image_pad|>",
|
| 227 |
+
"<|video_pad|>"
|
| 228 |
+
],
|
| 229 |
+
"bos_token": null,
|
| 230 |
+
"clean_up_tokenization_spaces": false,
|
| 231 |
+
"eos_token": "<|im_end|>",
|
| 232 |
+
"errors": "replace",
|
| 233 |
+
"extra_special_tokens": {},
|
| 234 |
+
"model_max_length": 262144,
|
| 235 |
+
"pad_token": "<|endoftext|>",
|
| 236 |
+
"processor_class": "Qwen3VLProcessor",
|
| 237 |
+
"split_special_tokens": false,
|
| 238 |
+
"tokenizer_class": "Qwen2Tokenizer",
|
| 239 |
+
"unk_token": null
|
| 240 |
+
}
|
tokenizer/vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|