danielhanchen commited on
Commit
555910a
·
0 Parent(s):

Initial commit

Browse files
.gitattributes ADDED
@@ -0,0 +1,63 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ assets/unsloth_baking.png filter=lfs diff=lfs merge=lfs -text
37
+ assets/unsloth_cartoon.png filter=lfs diff=lfs merge=lfs -text
38
+ assets/unsloth_field.png filter=lfs diff=lfs merge=lfs -text
39
+ assets/unsloth_headshot.png filter=lfs diff=lfs merge=lfs -text
40
+ qwen-image-2.1-UD-7.0bpw.gguf filter=lfs diff=lfs merge=lfs -text
41
+ qwen-image-2.1-UD-7.5bpw.gguf filter=lfs diff=lfs merge=lfs -text
42
+ qwen-image-2.1-UD-2.63bpw.gguf filter=lfs diff=lfs merge=lfs -text
43
+ qwen-image-2.1-UD-3.0bpw.gguf filter=lfs diff=lfs merge=lfs -text
44
+ qwen-image-2.1-UD-3.5bpw.gguf filter=lfs diff=lfs merge=lfs -text
45
+ qwen-image-2.1-UD-4.0bpw.gguf filter=lfs diff=lfs merge=lfs -text
46
+ qwen-image-2.1-UD-4.33bpw.gguf filter=lfs diff=lfs merge=lfs -text
47
+ qwen-image-2.1-UD-4.66bpw.gguf filter=lfs diff=lfs merge=lfs -text
48
+ qwen-image-2.1-UD-5.0bpw.gguf filter=lfs diff=lfs merge=lfs -text
49
+ qwen-image-2.1-UD-6.0bpw.gguf filter=lfs diff=lfs merge=lfs -text
50
+ qwen-image-2.1-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
51
+ qwen-image-2.1-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
52
+ qwen-image-2.1-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
53
+ qwen-image-2.1-Q3_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
54
+ qwen-image-2.1-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
55
+ qwen-image-2.1-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
56
+ qwen-image-2.1-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
57
+ qwen-image-2.1-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
58
+ qwen-image-2.1-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
59
+ qwen-image-2.1-Q6_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
60
+ qwen-image-2.1-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
61
+ qwen-image-2.1-F16.gguf filter=lfs diff=lfs merge=lfs -text
62
+ assets/qwen21_pareto.png filter=lfs diff=lfs merge=lfs -text
63
+ assets/qwen21_pareto_v2.png filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,216 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen-Image-2.1
3
+ base_model_relation: quantized
4
+ license: other
5
+ license_name: qwen-research
6
+ license_link: https://huggingface.co/Qwen/Qwen-Image-2.1/blob/main/LICENSE
7
+ language:
8
+ - en
9
+ - zh
10
+ pipeline_tag: text-to-image
11
+ tags:
12
+ - gguf
13
+ - quantized
14
+ - unsloth
15
+ - qwen
16
+ - image-generation
17
+ widget:
18
+ - text: a cartoon sloth mascot waving, flat vector illustration, bright colours
19
+ output:
20
+ url: assets/unsloth_cartoon.png
21
+ - text: a sloth baking bread in a rustic kitchen, warm morning light, photorealistic
22
+ output:
23
+ url: assets/unsloth_baking.png
24
+ ---
25
+ This is a GGUF quantized version of [Qwen-Image-2.1](https://huggingface.co/Qwen/Qwen-Image-2.1). <br>
26
+ unsloth/Qwen-Image-2.1-GGUF uses [Unsloth Dynamic 2.0](https://docs.unsloth.ai/basics/unsloth-dynamic-2.0-ggufs) methodology for SOTA performance.
27
+
28
+ - Important layers are upcasted to higher precision, per tensor, from a measured sensitivity scan.
29
+ - Run these with [stable-diffusion.cpp](https://github.com/unslothai/stable-diffusion.cpp). A GGUF is the denoiser only, so it needs the VAE and the Qwen3-VL text encoder alongside it.
30
+ - VAE: [unsloth/Qwen-Image-2.1-FP8](https://huggingface.co/unsloth/Qwen-Image-2.1-FP8) `vae/qwen_image_2.1_vae_bf16.safetensors`. Text encoder: [unsloth/Qwen3-VL-8B-Instruct-GGUF](https://huggingface.co/unsloth/Qwen3-VL-8B-Instruct-GGUF) `Qwen3-VL-8B-Instruct-Q4_K_M.gguf`.
31
+
32
+ ```bash
33
+ sd-cli --diffusion-model qwen-image-2.1-Q4_K_M.gguf \
34
+ --vae qwen_image_2.1_vae_bf16.safetensors \
35
+ --llm Qwen3-VL-8B-Instruct-Q4_K_M.gguf \
36
+ -p "a cartoon sloth mascot waving, flat vector illustration, bright colours" \
37
+ --steps 20 --cfg-scale 6.0 --sampling-method euler -W 1024 -H 1024 --diffusion-fa \
38
+ -o out.png
39
+ ```
40
+
41
+
42
+ <div>
43
+ <div style="display: flex; gap: 5px; align-items: center; ">
44
+ <a href="https://github.com/unslothai/unsloth/">
45
+ <img src="https://github.com/unslothai/unsloth/raw/main/images/unsloth%20new%20logo.png" width="133">
46
+ </a>
47
+ <a href="https://discord.gg/unsloth">
48
+ <img src="https://github.com/unslothai/unsloth/raw/main/images/Discord%20button.png" width="173">
49
+ </a>
50
+ <a href="https://docs.unsloth.ai/">
51
+ <img src="https://raw.githubusercontent.com/unslothai/unsloth/refs/heads/main/images/documentation%20green%20button.png" width="143">
52
+ </a>
53
+ </div>
54
+ </div>
55
+
56
+ ### Samples
57
+
58
+ Rendered with the Q4_K_M file, 1024x1024, 20 steps, cfg 6.0, euler.
59
+
60
+ <table>
61
+ <tr>
62
+ <td><img src="assets/unsloth_cartoon.png" width="200"></td>
63
+ <td><img src="assets/unsloth_headshot.png" width="200"></td>
64
+ </tr>
65
+ <tr>
66
+ <td><img src="assets/unsloth_baking.png" width="200"></td>
67
+ <td><img src="assets/unsloth_field.png" width="200"></td>
68
+ </tr>
69
+ </table>
70
+
71
+ ---
72
+
73
+ <p align="center">
74
+ <img src="https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen-Image/image2.1/logo.png" width="400"/>
75
+ </p>
76
+ <p align="center">
77
+ 🤖 <a href="https://modelscope.cn/models/Qwen/Qwen-Image-2.1">ModelScope</a>&nbsp;&nbsp;|
78
+ &nbsp;&nbsp;🤗 <a href="https://huggingface.co/Qwen/Qwen-Image-2.1">HuggingFace</a>&nbsp;&nbsp;|
79
+ &nbsp;&nbsp;📑 <a href="https://qwen.ai/blog?id=qwen-image-2.1">Blog</a>&nbsp;&nbsp;|
80
+ &nbsp;&nbsp;🖥️ <a href="https://huggingface.co/spaces/Qwen/Qwen-Image-2.1">Demo</a>&nbsp;&nbsp;|
81
+ &nbsp;&nbsp;🫨 <a href="https://discord.gg/BEYSk3pkSu">Discord</a>&nbsp;&nbsp;|
82
+ &nbsp;&nbsp;💬 <a href="https://huggingface.co/Qwen/Qwen-Image-2.1/blob/main/assets/qr.png">WeChat</a>
83
+ </p>
84
+
85
+ ## Introduction
86
+
87
+ We are excited to open-source **Qwen-Image-2.1**, a unified text-to-image generation and image editing model in the Qwen family. With just **7B parameters in its visual generation component** (32 Single-Stream DiT layers), Qwen-Image-2.1 balances generation quality, inference efficiency, and versatility.
88
+
89
+ Four key improvements define this release:
90
+
91
+ - **Compact and Efficient**: a lightweight architecture with mixed-granularity attention and prefix KV cache reuse delivers strong image quality at low computational cost.
92
+ - **Native Transparency, Unified Creation and Editing**: generate regular or transparent (RGBA) images from text, edit transparent layers, and extract subjects from photographs, all in one model.
93
+ - **Versatile Editing**: support up to **10 reference images**, specify local edits via circles, painted annotations, or separate masks, and preserve identity for people and products.
94
+ - **Realistic Textures and Refined Aesthetics**: improved typography, portrait lighting, and fine details for more visually compelling results.
95
+
96
+ <p align="center">
97
+ <img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen-Image/image2.1/images/example-01.png" width="100%"/>
98
+ </p>
99
+
100
+ For more details, see the [GitHub repo](https://github.com/QwenLM/Qwen-Image-2.1) and [Blog](https://qwen.ai/blog?id=qwen-image-2.1).
101
+
102
+ ## Quick Start
103
+
104
+ ### Installation
105
+
106
+ ```bash
107
+ pip install torch>=2.4.0
108
+ pip install transformers>=5.17
109
+ pip install git+https://github.com/huggingface/diffusers
110
+ pip install accelerate pillow
111
+ ```
112
+
113
+ ### Text-to-Image
114
+
115
+ ```python
116
+ import torch
117
+ from diffusers import QwenImage21Pipeline
118
+
119
+ pipe = QwenImage21Pipeline.from_pretrained(
120
+ "Qwen/Qwen-Image-2.1", torch_dtype=torch.bfloat16
121
+ ).to("cuda")
122
+
123
+ image = pipe(
124
+ prompt="A neon shop sign that reads \"QWEN IMAGE 2.1\", rainy night, reflections on wet pavement",
125
+ width=2048, height=2048,
126
+ num_inference_steps=40,
127
+ generator=torch.Generator("cuda").manual_seed(42),
128
+ ).images[0]
129
+
130
+ image.save("t2i_example.png")
131
+ ```
132
+
133
+ ### Image Editing
134
+
135
+ ```python
136
+ import torch
137
+ from PIL import Image
138
+ from diffusers import QwenImage21Pipeline
139
+
140
+ pipe = QwenImage21Pipeline.from_pretrained(
141
+ "Qwen/Qwen-Image-2.1", torch_dtype=torch.bfloat16
142
+ ).to("cuda")
143
+
144
+ input_image = Image.open("input.png")
145
+
146
+ image = pipe(
147
+ prompt="Change the background to a sunset beach",
148
+ image=input_image,
149
+ num_inference_steps=40,
150
+ generator=torch.Generator("cuda").manual_seed(42),
151
+ ).images[0]
152
+
153
+ image.save("edit_example.png")
154
+ ```
155
+
156
+ ### Transparent Image Generation (RGBA)
157
+
158
+ Use the recommended prompt format for transparent images:
159
+
160
+ ```python
161
+ image = pipe(
162
+ prompt="This is an RGBA image with transparency. A cute cartoon dragon sticker. The image has alpha channel and the background is transparent.",
163
+ width=2048, height=2048,
164
+ num_inference_steps=40,
165
+ generator=torch.Generator("cuda").manual_seed(42),
166
+ ).images[0]
167
+
168
+ image.save("transparent_example.png")
169
+ ```
170
+
171
+ ### Supported Aspect Ratios
172
+
173
+ ```python
174
+ aspect_ratios = {
175
+ "1:1": (2048, 2048),
176
+ "4:3": (2400, 1792),
177
+ "3:4": (1792, 2400),
178
+ "3:2": (2528, 1696),
179
+ "2:3": (1696, 2528),
180
+ "16:9": (2752, 1536),
181
+ "9:16": (1536, 2752),
182
+ }
183
+ ```
184
+
185
+ ### Memory Optimization
186
+
187
+ ```python
188
+ pipe = QwenImage21Pipeline.from_pretrained(
189
+ "Qwen/Qwen-Image-2.1", torch_dtype=torch.bfloat16
190
+ )
191
+ pipe.enable_model_cpu_offload()
192
+ ```
193
+
194
+ ## Showcase
195
+
196
+ <p align="center">
197
+ <img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen-Image/image2.1/images/example-04.png" width="30%"/>
198
+ <img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen-Image/image2.1/images/example-05.png" width="30%"/>
199
+ <img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen-Image/image2.1/images/example-06.png" width="30%"/>
200
+ </p>
201
+ <p align="center"><em>Native transparent image generation</em></p>
202
+
203
+ <p align="center">
204
+ <img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen-Image/image2.1/images/example-15.png" width="100%"/>
205
+ </p>
206
+ <p align="center"><em>Group photograph generated from six portrait references</em></p>
207
+
208
+ <p align="center">
209
+ <img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen-Image/image2.1/images/example-43.png" width="48%"/>
210
+ <img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen-Image/image2.1/images/example-44.png" width="48%"/>
211
+ </p>
212
+ <p align="center"><em>Text rendering</em></p>
213
+
214
+ ## License
215
+
216
+ This model is licensed under the [Qwen Research License Agreement](https://huggingface.co/Qwen/Qwen-Image-2.1/blob/main/LICENSE).
assets/unsloth_baking.png ADDED

Git LFS Details

  • SHA256: 4500fb8fafad61555471aa1e91e9e1bd9028aeac17ce66f43a0081783dc6d50a
  • Pointer size: 132 Bytes
  • Size of remote file: 2.36 MB
assets/unsloth_cartoon.png ADDED

Git LFS Details

  • SHA256: 1f1f01e7d96ba79b0fc3daf02fb70ac736da97d47f43faf7034c91f845aba3f8
  • Pointer size: 132 Bytes
  • Size of remote file: 1.39 MB
assets/unsloth_field.png ADDED

Git LFS Details

  • SHA256: 9c9c1b0aa4f51a9adff90fe3f82e274d54961d0a50e4c6fb69203edc0354c262
  • Pointer size: 132 Bytes
  • Size of remote file: 2.21 MB
assets/unsloth_headshot.png ADDED

Git LFS Details

  • SHA256: 7dfb9dd5ec8cc69b26aa8f8aedf2a4b1dba50b971d9049fa0d1af1dba78f0840
  • Pointer size: 132 Bytes
  • Size of remote file: 1.67 MB
qwen-image-2.1-F16.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8aa239e5ddb28d4641942231396d7171da41ad305ea401dd92a7d6410bde798d
3
+ size 14230275808
qwen-image-2.1-Q2_K.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:053e2cf11059e9f5091c03853025b18bc64d18fff4426b1a17da09ed29cc09c9
3
+ size 2466137824
qwen-image-2.1-Q3_K_M.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f688e47c32c033aa8d059c6b7e27d668b2cef0f1dd3d667f97f544954d541238
3
+ size 3168290528
qwen-image-2.1-Q3_K_S.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e71af4723442d004b469425ddbdbbf5e4f5f6068c2544cd82d9bb49791b54be3
3
+ size 2724742880
qwen-image-2.1-Q3_K_XL.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ef42f3a1b0ccf30be0a068341d383beeea86173e6aa1aa1a33749b337e55e0a9
3
+ size 3612493536
qwen-image-2.1-Q4_K_M.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:631d532e7ca71e8d90a87c71d3699761a812039d22e3370e87498d87754660fe
3
+ size 4199565024
qwen-image-2.1-Q4_K_S.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:328a2b2203ceb45a7f14515fb902c5ec28e9df901d43f49020b4fd251234587f
3
+ size 3906356960
qwen-image-2.1-Q5_K_M.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4b53321654dd3bf0aa8dd6bb821fdcb1c9053ca735070e359a23cbb9f97268ce
3
+ size 5390223072
qwen-image-2.1-Q5_K_S.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ba9399e07005c929d2cba313d73f0658e7893219825f29dc609130b75803ea75
3
+ size 4501948128
qwen-image-2.1-Q6_K.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2ead1ee665e8ebc5fec517d287a167f5c6a7e985acbf3b1b93143437a04616ac
3
+ size 6271551200
qwen-image-2.1-Q6_K_XL.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9410ad28f20018e7f2f300a06eed7e48db144286d02f0a6d4208d4dbd266691b
3
+ size 6718506720
qwen-image-2.1-Q8_0.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c0ed4b2ffd56cbe9c3df1e4a4098045256484ebe94ba5a7e4338d35ec046baa5
3
+ size 7640860384