baka999 commited on
Commit
398e16e
·
verified ·
1 Parent(s): ae876a1

Upload 7 files

Browse files
Files changed (7) hide show
  1. .gitattributes +35 -35
  2. .gitignore +5 -5
  3. README.md +49 -38
  4. app.py +1224 -676
  5. pe_i2i_system_prompt.txt +14 -0
  6. pe_t2i_system_prompt.txt +25 -0
  7. requirements.txt +11 -10
.gitattributes CHANGED
@@ -1,35 +1,35 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
- *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
- *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
- *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
- *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
- *.xz filter=lfs diff=lfs merge=lfs -text
33
- *.zip filter=lfs diff=lfs merge=lfs -text
34
- *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
.gitignore CHANGED
@@ -1,5 +1,5 @@
1
- __pycache__/
2
- *.pyc
3
- .gradio/
4
- *.safetensors
5
- *.gguf
 
1
+ __pycache__/
2
+ *.pyc
3
+ .gradio/
4
+ *.safetensors
5
+ *.gguf
README.md CHANGED
@@ -1,38 +1,49 @@
1
- ---
2
- title: Qwen Image 2.1 Uncensored All-In-One LoRA Studio
3
- emoji: 🚀
4
- colorFrom: purple
5
- colorTo: pink
6
- sdk: gradio
7
- sdk_version: 6.28.0
8
- python_version: '3.12'
9
- app_file: app.py
10
- short_description: Uncensored Qwen 2.1 with All-In-One LoRAs
11
- startup_duration_timeout: 1h
12
- models:
13
- - KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF
14
- - Qwen/Qwen-Image-2.1
15
- - WarmBloodAban/Qwen-Image-2.1-LoRAs
16
- - prithivMLmods/Qwen-Image-2.1-Natural-Exposure-LoRA
17
- - alibaba-pai/Qwen-Image-2.1-Fun-Acc-LoRAs
18
- - Viggle/Qwen-Image-2.1-viggle-turbo
19
- - Alissonerdx/BFS-Best-Face-Swap
20
- - lilylilith/AnyPose
21
- pinned: true
22
- ---
23
-
24
- # 🚀 Qwen-Image-2.1 Uncensored All-In-One LoRA Studio
25
-
26
- An all-in-one generative AI suite running [KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF](https://huggingface.co/KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF) on Hugging Face ZeroGPU (`zero-a10g`), equipped with on-demand **All-In-One LoRA Adapters**.
27
-
28
- ## ✨ Features
29
- - **Uncensored GGUF Base**: Native fast BF16-packed inference powered by `qwen-image-2.1-UC-Q4_K_M.gguf`.
30
- - **All-In-One LoRA Suite**:
31
- - ⚡ **Turbo Acceleration**: 4-step / 5-step fast inference with Viggle Turbo & Pai Fun-Acc.
32
- - 🎨 **Aesthetic & Style LoRAs**: Anime Consistency, Natural Exposure Photorealism, Hyperrealistic & Ultrarealistic Portraits, Flat-Log Film Grade.
33
- - 🎭 **Face Swap & Pose Transfer**: BFS Best Face Swap and AnyPose 2-image guided synthesis.
34
- - 💡 **Relighting & Atmosphere**: Directional studio lighting, light removal, and scene relighting.
35
- - 🔍 **Detail & Upscaling**: Semi-realistic detailer, skin retouching, and 2K resolution enhancement.
36
- - 🌐 **Custom Hugging Face LoRA**: Dynamically test ANY community LoRA simply by pasting its Hugging Face repository and filename!
37
- - **Privacy-First**: Zero server-side logging or storage. Ephemeral session generation.
38
- - **Full PNG Metadata**: Prompts, seeds, and active LoRAs are embedded into the downloaded image's chunks.
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ title: Qwen Image 2.1 Uncensored All-In-One LoRA Studio
3
+ emoji: 🚀
4
+ colorFrom: purple
5
+ colorTo: pink
6
+ sdk: gradio
7
+ sdk_version: 6.28.0
8
+ python_version: '3.12'
9
+ app_file: app.py
10
+ short_description: Uncensored Qwen 2.1 with All-In-One LoRAs
11
+ startup_duration_timeout: 1h
12
+ models:
13
+ - KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF
14
+ - Qwen/Qwen-Image-2.1
15
+ - WarmBloodAban/Qwen-Image-2.1-LoRAs
16
+ - prithivMLmods/Qwen-Image-2.1-Natural-Exposure-LoRA
17
+ - alibaba-pai/Qwen-Image-2.1-Fun-Acc-LoRAs
18
+ - Viggle/Qwen-Image-2.1-viggle-turbo
19
+ - Alissonerdx/BFS-Best-Face-Swap
20
+ - reverentelusarca/elusarcas-qwen-2.1-detail-enhancer-lora
21
+ pinned: true
22
+ ---
23
+
24
+ # 🚀 Qwen-Image-2.1 Uncensored All-In-One LoRA Studio
25
+
26
+ An all-in-one generative AI suite running [KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF](https://huggingface.co/KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF) on Hugging Face ZeroGPU (`zero-a10g`), equipped with on-demand **All-In-One LoRA Adapters**.
27
+
28
+ ## ✨ Features
29
+ - **Uncensored GGUF Base**: Native fast BF16-packed inference powered by `qwen-image-2.1-UC-Q4_K_M.gguf`.
30
+ - **All-In-One LoRA Suite**:
31
+ - ⚡ **Turbo Acceleration**: 4-step / 5-step fast inference with Viggle Turbo & Pai Fun-Acc.
32
+ - 🎨 **Aesthetic & Style LoRAs**: Anime Consistency, Natural Exposure Photorealism, Hyperrealistic & Ultrarealistic Portraits, Flat-Log Film Grade.
33
+ - 🎭 **Face Swap & Pose Transfer**: BFS Best Face Swap and AnyPose 2-image guided synthesis.
34
+ - 💡 **Relighting & Atmosphere**: Directional studio lighting, light removal, and scene relighting.
35
+ - 🔍 **Detail & Upscaling**: Semi-realistic detailer, skin retouching, and 2K resolution enhancement.
36
+ - 🌐 **Custom Hugging Face LoRA**: Dynamically test ANY community LoRA simply by pasting its Hugging Face repository and filename!
37
+ - **Prompt Enhancement (Viggle Turbo Style)**:
38
+ - 🪄 **DeepSeek V4.1 Flash Prompt Enhancer**: Integrates Viggle's prompt rewriting system for text-to-image and editing instructions.
39
+ - ⚙️ **Auto / On / Off Control**: "Auto" enriches short prompts (< 30 tokens) while leaving long prompts intact; "On" always enriches; "Off" keeps prompt as written.
40
+ - 🚀 **ZeroGPU Optimized**: Prompt rewriting runs off-GPU before diffusion inference, preserving GPU quota.
41
+ - 🔑 **Flexible API Support**: Use `OPENROUTER_API_KEY`, `OPENAI_API_KEY`, or `DEEPSEEK_API_KEY` in environment variables or enter keys directly in the UI.
42
+ - **Multi-Image Reference Synthesis (1-6 Images)**:
43
+ - 🖼️ **Native Multi-Condition Processing**: Upload up to 6 reference images concurrently; Qwen-3-VL multimodal encoder processes `<image1>`, `<image2>`, ... seamlessly.
44
+ - 🔄 **Intelligent Prompt Integration**: Describe character, style, lighting, and composition transfers across multiple inputs in a single generation.
45
+ - **Customizable CFG & Negative Guidance**:
46
+ - 🎚️ **Adjustable CFG Scale (1.0 - 10.0)**: Supports 1.0 (native guidance-free mode, ideal for Turbo models) up to 10.0 for stronger prompt adherence.
47
+ - 🚫 **Negative Prompt Support**: Activates dual-pass unconditional guidance when CFG > 1.0 to eliminate unwanted artifacts.
48
+ - **Privacy-First**: Zero server-side logging or storage. Ephemeral session generation.
49
+ - **Full PNG Metadata**: Prompts, seeds, active LoRAs, CFG scale, and negative prompt are embedded into the downloaded image's chunks.
app.py CHANGED
@@ -1,676 +1,1224 @@
1
- import os
2
- import hashlib
3
- import json
4
- import random
5
- import tempfile
6
- import time
7
- from pathlib import Path
8
-
9
- # Configure memory allocator before heavy operations
10
- os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True")
11
-
12
- # Rule 1: import spaces FIRST before importing torch or any CUDA-touching library
13
- import spaces
14
- import gradio as gr
15
- import torch
16
- from accelerate import init_empty_weights
17
- from diffusers import QwenImage21Pipeline, QwenImage21Transformer2DModel
18
- from diffusers.models.model_loading_utils import load_gguf_checkpoint
19
- from diffusers.quantizers.gguf.utils import dequantize_gguf_tensor
20
- from huggingface_hub import hf_hub_download
21
- from PIL import Image, ImageOps, PngImagePlugin
22
- from safetensors.torch import load_file as safetensors_load_file
23
-
24
- # Model Configuration
25
- MODEL_ID = "KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF"
26
- CHECKPOINT = os.environ.get("QWEN_GGUF_CHECKPOINT", "qwen-image-2.1-UC-Q4_K_M.gguf")
27
- COMPANION_ID = "Qwen/Qwen-Image-2.1"
28
- COMPANION_REVISION = "790c92633540aa0cb11d9abf19eb46d861714758"
29
-
30
- SHA256_CHECKSUMS = {
31
- # Uncensored (UC) GGUFs
32
- "qwen-image-2.1-UC-Q4_K_M.gguf": "e79c8a009f2ecbdb6c70fd663d9aea9ee304a0d91f347e4169a756b8ad141b41",
33
- "qwen-image-2.1-UC-Q4_0.gguf": "13f59f20656efc0aa385d03c1fcac1a9dc2ad6e5ccc0ea9bfe1d6ac636f2c5b9",
34
- "qwen-image-2.1-UC-Q5_K_M.gguf": "af0bf278cf16d204fb31c384dc82fd41dca82d976b15fe9305a60c726fd6f821",
35
- "qwen-image-2.1-UC-Q6_K.gguf": "e14bb312109333b3d73b92ad9b9ac8b29b51f2edca1e7b86d1af1981bf1c4ee3",
36
- "qwen-image-2.1-UC-Q8_0.gguf": "cde456c72ea3ecebfc1be783300e972711d875e0c5f1bed33d42b66b156affa8",
37
- "qwen-image-2.1-UC-BF16.gguf": "f151c683a8aed4b310777017ebbbe3f2180f1180f7867115171adb7d50b0762a",
38
- }
39
-
40
- MODES = ["Text to Image", "Edit Image (1 Ref)", "Transform & Swap (2 Refs)", "Transparent PNG"]
41
- SIZES = {
42
- "Square · 1:1 (1024x1024)": (1024, 1024),
43
- "Landscape · 16:9 (1344x768)": (1344, 768),
44
- "Portrait · 9:16 (768x1344)": (768, 1344),
45
- "Landscape · 4:3 (1152x864)": (1152, 864),
46
- "Portrait · 3:4 (864x1152)": (864, 1152),
47
- }
48
- MAX_SEED = 2**31 - 1
49
-
50
- # ============================================================
51
- # Curated All-In-One LoRA Specifications
52
- # ============================================================
53
- NONE_LORA = "None (Base Uncensored)"
54
-
55
- ADAPTER_SPECS = {
56
- "Anime Consistency": {
57
- "repo": "WarmBloodAban/Qwen-Image-2.1-LoRAs",
58
- "weights": "Qwen2.1_Anime_consistency.safetensors",
59
- "adapter_name": "anime_consistency",
60
- "default_strength": 0.85,
61
- "default_steps": 28,
62
- "preset_prompt": "masterpiece, highly detailed anime illustration, vibrant colors, expressive eyes",
63
- "category": "Style",
64
- },
65
- "Natural Exposure (Photorealism)": {
66
- "repo": "prithivMLmods/Qwen-Image-2.1-Natural-Exposure-LoRA",
67
- "weights": "Qwen-Image-2.1-Natural-Exposure-LoRA-4000.safetensors",
68
- "adapter_name": "natural_exposure",
69
- "default_strength": 0.8,
70
- "default_steps": 30,
71
- "preset_prompt": "natural daylight exposure, authentic colors, unedited 35mm photograph, soft organic textures",
72
- "category": "Style",
73
- },
74
- "Viggle Turbo (4-Step Acceleration)": {
75
- "repo": "Viggle/Qwen-Image-2.1-viggle-turbo",
76
- "weights": "Qwen-Image-2.1-viggle-turbo-4step-lora-r64.safetensors",
77
- "adapter_name": "viggle_turbo",
78
- "default_strength": 1.0,
79
- "default_steps": 4,
80
- "preset_prompt": "",
81
- "category": "Turbo Speed",
82
- },
83
- "Fun-Acc (4-Step Turbo)": {
84
- "repo": "alibaba-pai/Qwen-Image-2.1-Fun-Acc-LoRAs",
85
- "weights": "models/Qwen-Image-2.1-Fun-Acc-4Step.safetensors",
86
- "adapter_name": "fun_acc_turbo",
87
- "default_strength": 1.0,
88
- "default_steps": 4,
89
- "preset_prompt": "",
90
- "category": "Turbo Speed",
91
- },
92
- "Hyperrealistic Portrait": {
93
- "repo": "prithivMLmods/Qwen-Image-Edit-2511-Hyper-Realistic-Portrait",
94
- "weights": "HRP_20.safetensors",
95
- "adapter_name": "hyper_portrait",
96
- "default_strength": 0.9,
97
- "default_steps": 30,
98
- "preset_prompt": "ultra-realistic photorealistic portrait, strict identity preservation, facing camera, pore-level skin texture, soft-box studio lighting, 85mm portrait lens",
99
- "category": "Style",
100
- },
101
- "Ultrarealistic Glamour Portrait": {
102
- "repo": "prithivMLmods/Qwen-Image-Edit-2511-Ultra-Realistic-Portrait",
103
- "weights": "URP_20.safetensors",
104
- "adapter_name": "ultra_glamour",
105
- "default_strength": 0.9,
106
- "default_steps": 30,
107
- "preset_prompt": "luxury fashion magazine glamour portrait, luminous skin highlighter, dramatic studio lighting, glossy lips, natural epidermal textures",
108
- "category": "Style",
109
- },
110
- "Anything to Real Photo": {
111
- "repo": "lrzjason/Anything2Real_2601",
112
- "weights": "anything2real_2601_A_final_patched.safetensors",
113
- "adapter_name": "any2real",
114
- "default_strength": 1.0,
115
- "default_steps": 30,
116
- "preset_prompt": "change the picture to a realistic high-definition photograph, authentic skin and materials",
117
- "category": "Transform",
118
- },
119
- "Semi-Realistic Photo Detailer": {
120
- "repo": "rzgar/Qwen-Image-Edit-semi-realistic-detailer",
121
- "weights": "Qwen-Image-Edit-Anime-Semi-Realistic-Detailer-v1.safetensors",
122
- "adapter_name": "semireal_detailer",
123
- "default_strength": 0.9,
124
- "default_steps": 30,
125
- "preset_prompt": "transform the image into a detailed semi-realistic rendering, refined lighting and depth",
126
- "category": "Transform",
127
- },
128
- "Relight & Atmosphere": {
129
- "repo": "dx8152/Qwen-Image-Edit-2509-Relight",
130
- "weights": "Qwen-Edit-Relight.safetensors",
131
- "adapter_name": "relight",
132
- "default_strength": 0.85,
133
- "default_steps": 30,
134
- "preset_prompt": "cinematic dramatic lighting, warm amber key light, subtle cyan rim light, soft volumetric glow",
135
- "category": "Lighting",
136
- },
137
- "Multi-Angle Lighting": {
138
- "repo": "dx8152/Qwen-Edit-2509-Multi-Angle-Lighting",
139
- "weights": "多角度灯光-251116.safetensors",
140
- "adapter_name": "multi_angle_lighting",
141
- "default_strength": 0.85,
142
- "default_steps": 30,
143
- "preset_prompt": "studio portrait lighting from side angle, sharp highlights and balanced shadow contours",
144
- "category": "Lighting",
145
- },
146
- "Light Restoration": {
147
- "repo": "dx8152/Qwen-Image-Edit-2509-Light_restoration",
148
- "weights": "移除光影.safetensors",
149
- "adapter_name": "light_restore",
150
- "default_strength": 0.8,
151
- "default_steps": 28,
152
- "preset_prompt": "remove harsh shadows and uneven lighting, restore clean even illumination across the subject",
153
- "category": "Lighting",
154
- },
155
- "Flat Log Filmic Grade": {
156
- "repo": "tlennon-ie/QwenEdit2509-FlatLogColor",
157
- "weights": "QwenEdit2509-FlatLogColor.safetensors",
158
- "adapter_name": "flat_log",
159
- "default_strength": 0.8,
160
- "default_steps": 28,
161
- "preset_prompt": "cinematic flat log color profile, wide dynamic range, muted contrast, cinema grade palette",
162
- "category": "Color",
163
- },
164
- "Skin Retouch & Texture": {
165
- "repo": "tlennon-ie/qwen-edit-skin",
166
- "weights": "qwen-edit-skin_1.1_000002750.safetensors",
167
- "adapter_name": "edit_skin",
168
- "default_strength": 0.85,
169
- "default_steps": 28,
170
- "preset_prompt": "clean natural skin complexion, pore clarity, smooth texture without synthetic plastic appearance",
171
- "category": "Transform",
172
- },
173
- "Upscale 2K / Enhance": {
174
- "repo": "valiantcat/Qwen-Image-Edit-2509-Upscale2K",
175
- "weights": "qwen_image_edit_2509_upscale.safetensors",
176
- "adapter_name": "upscale_2k",
177
- "default_strength": 0.8,
178
- "default_steps": 28,
179
- "preset_prompt": "upscale this image to sharp high definition 4K resolution, enhanced edges and textures",
180
- "category": "Utility",
181
- },
182
- "BFS Best Face Swap (2 Images)": {
183
- "repo": "Alissonerdx/BFS-Best-Face-Swap",
184
- "weights": "bfs_head_v5_2511_original.safetensors",
185
- "adapter_name": "bfs_faceswap",
186
- "default_strength": 1.0,
187
- "default_steps": 32,
188
- "requires_two_images": True,
189
- "image2_label": "Upload Head/Face Donor (Image 2)",
190
- "needs_alpha_fix": True,
191
- "preset_prompt": "head_swap: start with Picture 1 as the base image, keeping its lighting and environment. Replace the head with the head from Picture 2, strictly preserving identity, eye color, and nose structure. Sharp details, 4k",
192
- "category": "Two Images",
193
- },
194
- "AnyPose Pose Transfer (2 Images)": {
195
- "repo": "lilylilith/AnyPose",
196
- "weights": "2511-AnyPose-base-000006250.safetensors",
197
- "adapter_name": "anypose",
198
- "default_strength": 0.85,
199
- "default_steps": 32,
200
- "requires_two_images": True,
201
- "image2_label": "Upload Target Pose Reference (Image 2)",
202
- "preset_prompt": "Make the person in image 1 match the exact pose of the person in image 2. The arms, head, and legs should match image 2 while keeping the character identity and clothing from image 1.",
203
- "category": "Two Images",
204
- },
205
- }
206
-
207
- LORA_CHOICES = [NONE_LORA] + list(ADAPTER_SPECS.keys()) + ["Custom HuggingFace LoRA..."]
208
-
209
- # Track dynamically loaded adapters
210
- LOADED_ADAPTERS = set()
211
-
212
- # ============================================================
213
- # GGUF Checkpoint Initialization on CUDA
214
- # ============================================================
215
- print(f"Loading checkpoint {CHECKPOINT} from {MODEL_ID}...", flush=True)
216
- checkpoint_path = hf_hub_download(MODEL_ID, CHECKPOINT)
217
-
218
- if CHECKPOINT in SHA256_CHECKSUMS:
219
- with open(checkpoint_path, "rb") as f:
220
- checksum = hashlib.file_digest(f, "sha256").hexdigest()
221
- if checksum != SHA256_CHECKSUMS[CHECKPOINT]:
222
- raise RuntimeError(f"Checksum verification failed for {CHECKPOINT}! Got {checksum}")
223
- print(f"Checksum verified: {checksum}", flush=True)
224
-
225
- # Expand the GGUF weights to bfloat16 once at startup for ZeroGPU eager packing
226
- weights = load_gguf_checkpoint(checkpoint_path)
227
- for name in list(weights.keys()):
228
- weights[name] = dequantize_gguf_tensor(weights[name]).to(torch.bfloat16)
229
-
230
- config = QwenImage21Transformer2DModel.load_config(
231
- COMPANION_ID, subfolder="transformer", revision=COMPANION_REVISION
232
- )
233
- with init_empty_weights():
234
- transformer = QwenImage21Transformer2DModel.from_config(config)
235
-
236
- transformer.load_state_dict(weights, strict=True, assign=True)
237
- transformer.eval().requires_grad_(False)
238
- print(f"Loaded {len(weights)} tensors into QwenImage21Transformer2DModel.", flush=True)
239
- del weights
240
-
241
- # Load full pipeline with companion VAE and text encoder, eagerly placed on 'cuda'
242
- pipe = QwenImage21Pipeline.from_pretrained(
243
- COMPANION_ID,
244
- revision=COMPANION_REVISION,
245
- transformer=transformer,
246
- torch_dtype=torch.bfloat16,
247
- ).to("cuda")
248
- print(f"{MODEL_ID} successfully initialized on CUDA for ZeroGPU.", flush=True)
249
-
250
-
251
- # ============================================================
252
- # LoRA Adapter Loading Helpers
253
- # ============================================================
254
- def _inject_missing_alpha_keys(state_dict: dict) -> dict:
255
- bases = {}
256
- for k, v in state_dict.items():
257
- if not isinstance(v, torch.Tensor):
258
- continue
259
- if k.endswith(".lora_down.weight") and v.ndim >= 1:
260
- base = k[:-len(".lora_down.weight")]
261
- rank = int(v.shape[0])
262
- bases[base] = rank
263
-
264
- for base, rank in bases.items():
265
- alpha_tensor = torch.tensor(float(rank), dtype=torch.float32)
266
- full_alpha = f"{base}.alpha"
267
- if full_alpha not in state_dict:
268
- state_dict[full_alpha] = alpha_tensor
269
- if base.startswith("diffusion_model."):
270
- stripped_base = base[len("diffusion_model."):]
271
- stripped_alpha = f"{stripped_base}.alpha"
272
- if stripped_alpha not in state_dict:
273
- state_dict[stripped_alpha] = alpha_tensor
274
- return state_dict
275
-
276
-
277
- def _filter_to_diffusers_lora_keys(state_dict: dict) -> dict:
278
- keep_suffixes = (
279
- ".lora_up.weight",
280
- ".lora_down.weight",
281
- ".lora_mid.weight",
282
- ".alpha",
283
- ".lora_alpha",
284
- )
285
- out: dict[str, torch.Tensor] = {}
286
- for k, v in state_dict.items():
287
- if not isinstance(v, torch.Tensor):
288
- continue
289
- if k.endswith(".diff") or k.endswith(".diff_b"):
290
- continue
291
- if not k.endswith(keep_suffixes):
292
- continue
293
- if k.endswith(".lora_alpha"):
294
- base = k[:-len(".lora_alpha")]
295
- k2 = f"{base}.alpha"
296
- out[k2] = v.float() if v.dtype != torch.float32 else v
297
- continue
298
- out[k] = v
299
- return out
300
-
301
-
302
- def _duplicate_stripped_prefix_keys(state_dict: dict, prefix: str = "diffusion_model.") -> dict:
303
- out = dict(state_dict)
304
- for k, v in list(state_dict.items()):
305
- if not k.startswith(prefix):
306
- continue
307
- stripped = k[len(prefix):]
308
- if stripped not in out:
309
- out[stripped] = v
310
- return out
311
-
312
-
313
- def _load_lora_with_fallback(repo: str, weight_name: str, adapter_name: str, needs_alpha_fix: bool = False):
314
- try:
315
- pipe.load_lora_weights(repo, weight_name=weight_name, adapter_name=adapter_name)
316
- return
317
- except Exception as e:
318
- print(f"Direct LoRA load failed ({e}), attempting safetensors fallback...", flush=True)
319
- local_path = hf_hub_download(repo_id=repo, filename=weight_name)
320
- sd = safetensors_load_file(local_path)
321
- if needs_alpha_fix:
322
- sd = _inject_missing_alpha_keys(sd)
323
- sd = _filter_to_diffusers_lora_keys(sd)
324
- sd = _duplicate_stripped_prefix_keys(sd)
325
- pipe.load_lora_weights(sd, adapter_name=adapter_name)
326
-
327
-
328
- def ensure_adapter_ready(selected_lora: str, custom_repo: str = "", custom_file: str = "") -> tuple[str, float]:
329
- if selected_lora == NONE_LORA:
330
- return "", 1.0
331
-
332
- if selected_lora == "Custom HuggingFace LoRA...":
333
- custom_repo = (custom_repo or "").strip()
334
- custom_file = (custom_file or "").strip()
335
- if not custom_repo or not custom_file:
336
- raise gr.Error("Please enter both a Hugging Face Repo ID and LoRA filename for Custom LoRA.")
337
- adapter_name = f"custom_{hashlib.md5((custom_repo + custom_file).encode()).hexdigest()[:8]}"
338
- if adapter_name not in LOADED_ADAPTERS:
339
- print(f"Loading custom LoRA from {custom_repo} / {custom_file}...", flush=True)
340
- _load_lora_with_fallback(custom_repo, custom_file, adapter_name, needs_alpha_fix=True)
341
- LOADED_ADAPTERS.add(adapter_name)
342
- return adapter_name, 1.0
343
-
344
- spec = ADAPTER_SPECS.get(selected_lora)
345
- if not spec:
346
- return "", 1.0
347
-
348
- adapter_name = spec["adapter_name"]
349
- if adapter_name not in LOADED_ADAPTERS:
350
- print(f"Loading LoRA {selected_lora} ({spec['repo']} / {spec['weights']})...", flush=True)
351
- _load_lora_with_fallback(
352
- spec["repo"],
353
- spec["weights"],
354
- adapter_name,
355
- needs_alpha_fix=spec.get("needs_alpha_fix", False),
356
- )
357
- LOADED_ADAPTERS.add(adapter_name)
358
-
359
- return adapter_name, float(spec.get("default_strength", 1.0))
360
-
361
-
362
- # ============================================================
363
- # Inference Function
364
- # ============================================================
365
- @spaces.GPU(duration=60)
366
- def generate(
367
- prompt: str,
368
- mode: str = "Text to Image",
369
- ref_image_1: Image.Image | None = None,
370
- ref_image_2: Image.Image | None = None,
371
- lora_adapter: str = NONE_LORA,
372
- lora_strength: float = 1.0,
373
- custom_repo: str = "",
374
- custom_file: str = "",
375
- aspect_ratio: str = "Square · 1:1 (1024x1024)",
376
- steps: int = 30,
377
- seed: int = 42,
378
- randomize_seed: bool = True,
379
- progress: gr.Progress = gr.Progress(track_tqdm=True),
380
- ) -> tuple[str, str, int, str]:
381
- prompt = (prompt or "").strip()
382
- if not prompt:
383
- raise gr.Error("Please enter a prompt describing your image.")
384
- if len(prompt) > 4000:
385
- raise gr.Error("Prompt is too long. Please keep under 4000 characters.")
386
- if mode not in MODES:
387
- raise gr.Error(f"Invalid mode: {mode}")
388
- if aspect_ratio not in SIZES:
389
- raise gr.Error(f"Invalid aspect ratio: {aspect_ratio}")
390
- if steps is None or not (4 <= int(steps) <= 40):
391
- raise gr.Error("Inference steps must be between 4 and 40.")
392
-
393
- if mode == "Edit Image (1 Ref)" and ref_image_1 is None:
394
- raise gr.Error("Please upload a reference image for single-image edit mode.")
395
- if mode == "Transform & Swap (2 Refs)":
396
- if ref_image_1 is None or ref_image_2 is None:
397
- raise gr.Error("Please upload both Image 1 (Base) and Image 2 (Donor/Pose) for this mode.")
398
-
399
- actual_seed = random.randint(0, MAX_SEED) if randomize_seed else int(seed)
400
- width, height = SIZES[aspect_ratio]
401
-
402
- # Prepare LoRA
403
- active_adapter, base_strength = ensure_adapter_ready(lora_adapter, custom_repo, custom_file)
404
- if active_adapter:
405
- effective_strength = float(lora_strength) * base_strength
406
- pipe.set_adapters([active_adapter], adapter_weights=[effective_strength])
407
- active_lora_desc = f"{lora_adapter} (scale={round(effective_strength, 2)})"
408
- else:
409
- pipe.disable_lora()
410
- active_lora_desc = "None"
411
-
412
- effective_prompt = prompt
413
- if mode == "Transparent PNG":
414
- effective_prompt = (
415
- "This is an RGBA image with transparency. " + prompt
416
- + " The image has alpha channel and the background is transparent."
417
- )
418
-
419
- call_kwargs = {}
420
- if mode == "Edit Image (1 Ref)":
421
- img = ImageOps.exif_transpose(ref_image_1).convert("RGBA")
422
- img.thumbnail((2048, 2048))
423
- call_kwargs["image"] = img
424
- elif mode == "Transform & Swap (2 Refs)":
425
- # Multi-image edit pipeline passes images list
426
- img1 = ImageOps.exif_transpose(ref_image_1).convert("RGBA")
427
- img1.thumbnail((2048, 2048))
428
- img2 = ImageOps.exif_transpose(ref_image_2).convert("RGBA")
429
- img2.thumbnail((2048, 2048))
430
- call_kwargs["image"] = [img1, img2]
431
-
432
- start_time = time.perf_counter()
433
- with torch.inference_mode():
434
- result = pipe(
435
- prompt=effective_prompt,
436
- width=width,
437
- height=height,
438
- num_inference_steps=int(steps),
439
- generator=torch.Generator("cuda").manual_seed(actual_seed),
440
- **call_kwargs,
441
- ).images[0]
442
- elapsed = time.perf_counter() - start_time
443
-
444
- # Embed metadata into PNG chunks
445
- metadata = {
446
- "model": MODEL_ID,
447
- "checkpoint": CHECKPOINT,
448
- "lora_adapter": active_lora_desc,
449
- "prompt": prompt,
450
- "mode": mode,
451
- "seed": actual_seed,
452
- "steps": int(steps),
453
- "dimensions": f"{result.width}x{result.height}",
454
- "elapsed_seconds": round(elapsed, 2),
455
- }
456
- png_info = PngImagePlugin.PngInfo()
457
- png_info.add_text("parameters", json.dumps(metadata, ensure_ascii=False))
458
-
459
- temp_dir = tempfile.mkdtemp(prefix="qwen_aio_lora_")
460
- out_png_path = os.path.join(temp_dir, f"qwen_aio_{actual_seed}.png")
461
- result.save(out_png_path, "PNG", pnginfo=png_info, optimize=True)
462
-
463
- details = (
464
- f"⚡ Time: {elapsed:.2f}s | Seed: {actual_seed} | Steps: {steps}\n"
465
- f"🎛️ LoRA: {active_lora_desc} | Size: {result.width}x{result.height}\n"
466
- f"🔒 Privacy: Zero data retention (session ephemeral)"
467
- )
468
-
469
- return out_png_path, out_png_path, actual_seed, details
470
-
471
-
472
- # ============================================================
473
- # UI Helpers
474
- # ============================================================
475
- def update_lora_selection(selected_lora: str, current_prompt: str, current_steps: int):
476
- spec = ADAPTER_SPECS.get(selected_lora)
477
- show_custom = gr.update(visible=(selected_lora == "Custom HuggingFace LoRA..."))
478
- if not spec:
479
- return current_prompt, current_steps, 1.0, show_custom
480
-
481
- preset = spec.get("preset_prompt", "")
482
- new_prompt = current_prompt
483
- if preset and not current_prompt.strip():
484
- new_prompt = preset
485
- elif preset and preset not in current_prompt:
486
- new_prompt = f"{current_prompt}, {preset}" if current_prompt.strip() else preset
487
-
488
- recommended_steps = spec.get("default_steps", current_steps)
489
- recommended_strength = spec.get("default_strength", 1.0)
490
-
491
- return new_prompt, recommended_steps, recommended_strength, show_custom
492
-
493
-
494
- def update_mode_ui(mode: str):
495
- is_edit_1 = mode == "Edit Image (1 Ref)"
496
- is_swap_2 = mode == "Transform & Swap (2 Refs)"
497
- return (
498
- gr.update(visible=(is_edit_1 or is_swap_2)),
499
- gr.update(visible=is_swap_2),
500
- )
501
-
502
-
503
- # ============================================================
504
- # Gradio Interface
505
- # ============================================================
506
- CUSTOM_CSS = """
507
- .gradio-container { max-width: 1280px !important; margin: 0 auto !important; }
508
- .badge { display: inline-block; padding: 2px 8px; border-radius: 6px; font-size: 0.8rem; font-weight: 600; margin-right: 6px; }
509
- .badge-turbo { background: #fee2e2; color: #991b1b; }
510
- .badge-style { background: #ede9fe; color: #5b21b6; }
511
- .badge-tool { background: #e0f2fe; color: #075985; }
512
- #generate-btn { font-weight: 700; font-size: 1.1rem; }
513
- """
514
-
515
- with gr.Blocks(title="Qwen Image 2.1 Uncensored All-In-One LoRA Studio", delete_cache=(3600, 86400)) as demo:
516
- gr.Markdown(
517
- "# 🚀 Qwen Image 2.1 Uncensored All-In-One LoRA Studio\n"
518
- "### Supercharged Text-to-Image, Image Editing & Guided Synthesis powered by ZeroGPU\n"
519
- "Uncensored Base (`KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF`) + 15+ On-Demand Style, Speed, and Face/Pose LoRAs"
520
- )
521
-
522
- with gr.Row():
523
- with gr.Column(scale=6):
524
- mode_selector = gr.Radio(
525
- choices=MODES,
526
- value=MODES[0],
527
- label="Generation Mode",
528
- )
529
-
530
- prompt_input = gr.Textbox(
531
- label="Prompt",
532
- placeholder="Describe your vision, character, scene, or requested edit...",
533
- lines=3,
534
- max_lines=6,
535
- )
536
-
537
- with gr.Row():
538
- ref_img_1 = gr.Image(
539
- label="Image 1 (Base / Target)",
540
- type="pil",
541
- visible=False,
542
- )
543
- ref_img_2 = gr.Image(
544
- label="Image 2 (Face Donor / Pose Reference)",
545
- type="pil",
546
- visible=False,
547
- )
548
-
549
- with gr.Accordion("🎨 All-In-One LoRA Adapters", open=True):
550
- lora_dropdown = gr.Dropdown(
551
- choices=LORA_CHOICES,
552
- value=NONE_LORA,
553
- label="Select LoRA Adapter",
554
- info="Pick an on-demand style, 4-step turbo speed booster, or face/pose transfer model.",
555
- )
556
- lora_strength_slider = gr.Slider(
557
- minimum=0.0,
558
- maximum=1.5,
559
- value=1.0,
560
- step=0.05,
561
- label="LoRA Strength / Weight",
562
- )
563
-
564
- with gr.Row(visible=False) as custom_lora_box:
565
- custom_repo_input = gr.Textbox(
566
- label="Hugging Face LoRA Repo",
567
- placeholder="e.g. prithivMLmods/Qwen-Image-2.1-Natural-Exposure-LoRA",
568
- )
569
- custom_file_input = gr.Textbox(
570
- label="LoRA Weights Filename",
571
- placeholder="e.g. Qwen-Image-2.1-Natural-Exposure-LoRA-4000.safetensors",
572
- )
573
-
574
- with gr.Accordion("⚙️ Advanced Generation Settings", open=False):
575
- with gr.Row():
576
- aspect_ratio_dropdown = gr.Dropdown(
577
- choices=list(SIZES.keys()),
578
- value=list(SIZES.keys())[0],
579
- label="Aspect Ratio",
580
- )
581
- steps_slider = gr.Slider(
582
- minimum=4,
583
- maximum=40,
584
- value=30,
585
- step=1,
586
- label="Inference Steps",
587
- info="4 steps for Turbo LoRAs, 25-35 for regular generation",
588
- )
589
- with gr.Row():
590
- seed_number = gr.Number(value=42, label="Seed", precision=0)
591
- randomize_seed_cb = gr.Checkbox(value=True, label="Randomize Seed")
592
-
593
- generate_btn = gr.Button(
594
- "✨ Generate Image",
595
- variant="primary",
596
- elem_id="generate-btn",
597
- )
598
-
599
- with gr.Column(scale=6):
600
- output_image = gr.Image(
601
- label="Generated Output",
602
- type="filepath",
603
- interactive=False,
604
- )
605
- with gr.Row():
606
- download_file = gr.File(
607
- label="Download High-Res PNG (Includes Parameters)",
608
- interactive=False,
609
- )
610
- details_box = gr.Textbox(
611
- label="Execution Details & Privacy",
612
- interactive=False,
613
- )
614
-
615
- # Event Bindings
616
- mode_selector.change(
617
- fn=update_mode_ui,
618
- inputs=[mode_selector],
619
- outputs=[ref_img_1, ref_img_2],
620
- )
621
-
622
- lora_dropdown.change(
623
- fn=update_lora_selection,
624
- inputs=[lora_dropdown, prompt_input, steps_slider],
625
- outputs=[prompt_input, steps_slider, lora_strength_slider, custom_lora_box],
626
- )
627
-
628
- generate_btn.click(
629
- fn=generate,
630
- inputs=[
631
- prompt_input,
632
- mode_selector,
633
- ref_img_1,
634
- ref_img_2,
635
- lora_dropdown,
636
- lora_strength_slider,
637
- custom_repo_input,
638
- custom_file_input,
639
- aspect_ratio_dropdown,
640
- steps_slider,
641
- seed_number,
642
- randomize_seed_cb,
643
- ],
644
- outputs=[output_image, download_file, seed_number, details_box],
645
- api_name="generate",
646
- concurrency_limit=1,
647
- concurrency_id="qwen-aio-pipeline",
648
- )
649
-
650
- gr.Examples(
651
- examples=[
652
- [
653
- "A futuristic neon cyberpunk samurai in a rain-soaked Tokyo alleyway, glowing katana, volumetric mist",
654
- "Text to Image",
655
- "Anime Consistency",
656
- ],
657
- [
658
- "A tranquil highland mountain lake surrounded by autumn pines at golden hour, reflections in water",
659
- "Text to Image",
660
- "Natural Exposure (Photorealism)",
661
- ],
662
- [
663
- "A charming little steampunk robot holding a delicate glass flower, highly detailed gears and brass clockwork",
664
- "Text to Image",
665
- "Viggle Turbo (4-Step Acceleration)",
666
- ],
667
- ],
668
- inputs=[prompt_input, mode_selector, lora_dropdown],
669
- label="Try an Idea",
670
- )
671
-
672
- if __name__ == "__main__":
673
- demo.queue(max_size=16, default_concurrency_limit=1).launch(
674
- css=CUSTOM_CSS,
675
- mcp_server=True,
676
- )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import base64
2
+ import hashlib
3
+ import importlib.util
4
+ import io
5
+ import json
6
+ import os
7
+ import random
8
+ import re
9
+ import tempfile
10
+ import time
11
+ from pathlib import Path
12
+
13
+ # Configure memory allocator before heavy operations
14
+ os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True")
15
+
16
+ # Rule 1: import spaces FIRST before importing torch or any CUDA-touching library
17
+ if importlib.util.find_spec("spaces"):
18
+ import spaces
19
+ gpu = spaces.GPU(duration=90)
20
+ else:
21
+ gpu = lambda fn: fn
22
+
23
+ import httpx
24
+ import gradio as gr
25
+ import torch
26
+ from accelerate import init_empty_weights
27
+ from diffusers import QwenImage21Pipeline, QwenImage21Transformer2DModel
28
+ from diffusers.models.model_loading_utils import load_gguf_checkpoint
29
+ from diffusers.quantizers.gguf.utils import dequantize_gguf_tensor
30
+ from huggingface_hub import hf_hub_download
31
+ from PIL import Image, ImageOps, PngImagePlugin
32
+ from safetensors.torch import load_file as safetensors_load_file
33
+
34
+ # Model Configuration
35
+ MODEL_ID = "KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF"
36
+ CHECKPOINT = os.environ.get("QWEN_GGUF_CHECKPOINT", "qwen-image-2.1-UC-Q4_K_M.gguf")
37
+ COMPANION_ID = "Qwen/Qwen-Image-2.1"
38
+ COMPANION_REVISION = "790c92633540aa0cb11d9abf19eb46d861714758"
39
+
40
+ SHA256_CHECKSUMS = {
41
+ # Uncensored (UC) GGUFs
42
+ "qwen-image-2.1-UC-Q4_K_M.gguf": "e79c8a009f2ecbdb6c70fd663d9aea9ee304a0d91f347e4169a756b8ad141b41",
43
+ "qwen-image-2.1-UC-Q4_0.gguf": "13f59f20656efc0aa385d03c1fcac1a9dc2ad6e5ccc0ea9bfe1d6ac636f2c5b9",
44
+ "qwen-image-2.1-UC-Q5_K_M.gguf": "af0bf278cf16d204fb31c384dc82fd41dca82d976b15fe9305a60c726fd6f821",
45
+ "qwen-image-2.1-UC-Q6_K.gguf": "e14bb312109333b3d73b92ad9b9ac8b29b51f2edca1e7b86d1af1981bf1c4ee3",
46
+ "qwen-image-2.1-UC-Q8_0.gguf": "cde456c72ea3ecebfc1be783300e972711d875e0c5f1bed33d42b66b156affa8",
47
+ "qwen-image-2.1-UC-BF16.gguf": "f151c683a8aed4b310777017ebbbe3f2180f1180f7867115171adb7d50b0762a",
48
+ }
49
+
50
+ MODES = [
51
+ "Text to Image",
52
+ "Multi-Image Reference (1-6 Refs)",
53
+ "Edit Image (1 Ref)",
54
+ "Transform & Swap (2 Refs)",
55
+ "Transparent PNG",
56
+ ]
57
+ SIZES = {
58
+ "Square · 1:1 (1024x1024)": (1024, 1024),
59
+ "Landscape · 16:9 (1344x768)": (1344, 768),
60
+ "Portrait · 9:16 (768x1344)": (768, 1344),
61
+ "Landscape · 4:3 (1152x864)": (1152, 864),
62
+ "Portrait · 3:4 (864x1152)": (864, 1152),
63
+ }
64
+ MAX_SEED = 2**31 - 1
65
+
66
+ # ============================================================
67
+ # Curated All-In-One LoRA Specifications
68
+ # ============================================================
69
+ NONE_LORA = "None (Base Uncensored)"
70
+
71
+ ADAPTER_SPECS = {
72
+ "Anime Consistency": {
73
+ "repo": "WarmBloodAban/Qwen-Image-2.1-LoRAs",
74
+ "weights": "Qwen2.1_Anime_consistency.safetensors",
75
+ "adapter_name": "anime_consistency",
76
+ "default_strength": 0.85,
77
+ "default_steps": 28,
78
+ "preset_prompt": "masterpiece, highly detailed anime illustration, vibrant colors, expressive eyes",
79
+ "category": "Style",
80
+ },
81
+ "Natural Exposure (Photorealism)": {
82
+ "repo": "prithivMLmods/Qwen-Image-2.1-Natural-Exposure-LoRA",
83
+ "weights": "Qwen-Image-2.1-Natural-Exposure-LoRA-4000.safetensors",
84
+ "adapter_name": "natural_exposure",
85
+ "default_strength": 0.8,
86
+ "default_steps": 30,
87
+ "preset_prompt": "natural daylight exposure, authentic colors, unedited 35mm photograph, soft organic textures",
88
+ "category": "Style",
89
+ },
90
+ "Viggle Turbo (4-Step Acceleration)": {
91
+ "repo": "Viggle/Qwen-Image-2.1-viggle-turbo",
92
+ "weights": "Qwen-Image-2.1-viggle-turbo-4step-lora-r64.safetensors",
93
+ "adapter_name": "viggle_turbo",
94
+ "default_strength": 1.0,
95
+ "default_steps": 4,
96
+ "preset_prompt": "",
97
+ "category": "Turbo Speed",
98
+ },
99
+ "Viggle Turbo v0.2.1 (6-Step Acceleration)": {
100
+ "repo": "Viggle/Qwen-Image-2.1-viggle-turbo",
101
+ "weights": "Qwen-Image-2.1-viggle-turbo-v0.2.1-6step-lora-r256.safetensors",
102
+ "adapter_name": "viggle_turbo_v021",
103
+ "default_strength": 1.0,
104
+ "default_steps": 6,
105
+ "preset_prompt": "",
106
+ "category": "Turbo Speed",
107
+ },
108
+ "Fun-Acc (4-Step Turbo)": {
109
+ "repo": "alibaba-pai/Qwen-Image-2.1-Fun-Acc-LoRAs",
110
+ "weights": "models/Qwen-Image-2.1-Fun-Acc-4Step.safetensors",
111
+ "adapter_name": "fun_acc_turbo",
112
+ "default_strength": 1.0,
113
+ "default_steps": 4,
114
+ "preset_prompt": "",
115
+ "category": "Turbo Speed",
116
+ },
117
+ "BFS Best Face Swap (2 Images)": {
118
+ "repo": "Alissonerdx/BFS-Best-Face-Swap",
119
+ "weights": "bfs_head_v1.1_qwen_2.1.safetensors",
120
+ "adapter_name": "bfs_faceswap",
121
+ "default_strength": 1.0,
122
+ "default_steps": 32,
123
+ "requires_two_images": True,
124
+ "image2_label": "Upload Head/Face Donor (Image 2)",
125
+ "needs_alpha_fix": True,
126
+ "preset_prompt": "head_swap: start with <image1> as the base image, keeping its lighting, environment, and background. remove the head from <image1> completely and replace it with the head from <image2>, strictly preserving the hair, eye color, nose structure from <image2>. copy the direction of the eye, head rotation, micro expressions from <image1>, high quality, sharp details, 4k",
127
+ "category": "Two Images",
128
+ },
129
+ "Anything to Real Photo": {
130
+ "repo": "WarmBloodAban/Qwen-Image-2.1-LoRAs",
131
+ "weights": "Qwen2.1_Anything2RealCharacters.safetensors",
132
+ "adapter_name": "any2real",
133
+ "default_strength": 1.0,
134
+ "default_steps": 30,
135
+ "preset_prompt": "change the picture to a realistic high-definition photograph, authentic skin and materials",
136
+ "category": "Transform",
137
+ },
138
+ "Detail Enhancer": {
139
+ "repo": "reverentelusarca/elusarcas-qwen-2.1-detail-enhancer-lora",
140
+ "weights": "elusarcas-qwen2-1-detailer-v1.safetensors",
141
+ "adapter_name": "detail_enhancer",
142
+ "default_strength": 0.85,
143
+ "default_steps": 30,
144
+ "preset_prompt": "ultra high detail, intricate textures, sharp focus, refined clarity",
145
+ "category": "Enhance",
146
+ },
147
+ }
148
+
149
+ LORA_CHOICES = [NONE_LORA] + list(ADAPTER_SPECS.keys()) + ["Custom HuggingFace LoRA..."]
150
+
151
+ # Track dynamically loaded adapters
152
+ LOADED_ADAPTERS = set()
153
+
154
+ # ============================================================
155
+ # GGUF Checkpoint Initialization on CUDA
156
+ # ============================================================
157
+ print(f"Loading checkpoint {CHECKPOINT} from {MODEL_ID}...", flush=True)
158
+ checkpoint_path = hf_hub_download(MODEL_ID, CHECKPOINT)
159
+
160
+ if CHECKPOINT in SHA256_CHECKSUMS:
161
+ with open(checkpoint_path, "rb") as f:
162
+ checksum = hashlib.file_digest(f, "sha256").hexdigest()
163
+ if checksum != SHA256_CHECKSUMS[CHECKPOINT]:
164
+ raise RuntimeError(f"Checksum verification failed for {CHECKPOINT}! Got {checksum}")
165
+ print(f"Checksum verified: {checksum}", flush=True)
166
+
167
+ # Expand the GGUF weights to bfloat16 once at startup for ZeroGPU eager packing
168
+ weights = load_gguf_checkpoint(checkpoint_path)
169
+ for name in list(weights.keys()):
170
+ weights[name] = dequantize_gguf_tensor(weights[name]).to(torch.bfloat16)
171
+
172
+ config = QwenImage21Transformer2DModel.load_config(
173
+ COMPANION_ID, subfolder="transformer", revision=COMPANION_REVISION
174
+ )
175
+ with init_empty_weights():
176
+ transformer = QwenImage21Transformer2DModel.from_config(config)
177
+
178
+ transformer.load_state_dict(weights, strict=True, assign=True)
179
+ transformer.eval().requires_grad_(False)
180
+ print(f"Loaded {len(weights)} tensors into QwenImage21Transformer2DModel.", flush=True)
181
+ del weights
182
+
183
+ # Load full pipeline with companion VAE and text encoder, eagerly placed on 'cuda'
184
+ pipe = QwenImage21Pipeline.from_pretrained(
185
+ COMPANION_ID,
186
+ revision=COMPANION_REVISION,
187
+ transformer=transformer,
188
+ torch_dtype=torch.bfloat16,
189
+ ).to("cuda")
190
+ print(f"{MODEL_ID} successfully initialized on CUDA for ZeroGPU.", flush=True)
191
+
192
+
193
+ # ============================================================
194
+ # LoRA Adapter Loading Helpers
195
+ # ============================================================
196
+ def _inject_missing_alpha_keys(state_dict: dict) -> dict:
197
+ bases = {}
198
+ for k, v in state_dict.items():
199
+ if not isinstance(v, torch.Tensor):
200
+ continue
201
+ if k.endswith(".lora_down.weight") and v.ndim >= 1:
202
+ base = k[:-len(".lora_down.weight")]
203
+ rank = int(v.shape[0])
204
+ bases[base] = rank
205
+ elif k.endswith(".lora_A.weight") and v.ndim >= 1:
206
+ base = k[:-len(".lora_A.weight")]
207
+ rank = int(v.shape[0])
208
+ bases[base] = rank
209
+
210
+ for base, rank in bases.items():
211
+ alpha_tensor = torch.tensor(float(rank), dtype=torch.float32)
212
+ full_alpha = f"{base}.alpha"
213
+ if full_alpha not in state_dict:
214
+ state_dict[full_alpha] = alpha_tensor
215
+ if base.startswith("diffusion_model."):
216
+ stripped_base = base[len("diffusion_model."):]
217
+ stripped_alpha = f"{stripped_base}.alpha"
218
+ if stripped_alpha not in state_dict:
219
+ state_dict[stripped_alpha] = alpha_tensor
220
+ return state_dict
221
+
222
+
223
+ def _filter_to_diffusers_lora_keys(state_dict: dict) -> dict:
224
+ keep_suffixes = (
225
+ ".lora_up.weight",
226
+ ".lora_down.weight",
227
+ ".lora_mid.weight",
228
+ ".lora_A.weight",
229
+ ".lora_B.weight",
230
+ ".alpha",
231
+ ".lora_alpha",
232
+ )
233
+ out: dict[str, torch.Tensor] = {}
234
+ for k, v in state_dict.items():
235
+ if not isinstance(v, torch.Tensor):
236
+ continue
237
+ if k.endswith(".diff") or k.endswith(".diff_b"):
238
+ continue
239
+ if not k.endswith(keep_suffixes):
240
+ continue
241
+ if k.endswith(".lora_alpha"):
242
+ base = k[:-len(".lora_alpha")]
243
+ k2 = f"{base}.alpha"
244
+ out[k2] = v.float() if v.dtype != torch.float32 else v
245
+ continue
246
+ out[k] = v
247
+ return out
248
+
249
+
250
+ def _duplicate_stripped_prefix_keys(state_dict: dict, prefix: str = "diffusion_model.") -> dict:
251
+ out = dict(state_dict)
252
+ for k, v in list(state_dict.items()):
253
+ if not k.startswith(prefix):
254
+ continue
255
+ stripped = k[len(prefix):]
256
+ if stripped not in out:
257
+ out[stripped] = v
258
+ return out
259
+
260
+
261
+ def _split_gate_up_keys(state_dict: dict) -> dict:
262
+ """
263
+ ai-toolkit / ostris trains fused 'gate_up' linear layers of shape [24576, rank].
264
+ Diffusers QwenImage21Transformer2DModel splits this into two separate layers:
265
+ - 'gate_layer' (first 12288 rows of B)
266
+ - 'proj' (last 12288 rows of B)
267
+ Both gate_layer and proj share the same input projection A of shape [rank, 4096].
268
+ """
269
+ out = {}
270
+ for k, v in state_dict.items():
271
+ if "img_mlp.gate_up" in k:
272
+ if k.endswith(".lora_B.weight") or k.endswith(".lora_up.weight"):
273
+ if isinstance(v, torch.Tensor) and v.ndim == 2 and v.shape[0] == 24576:
274
+ half = 12288
275
+ k_gate = k.replace("img_mlp.gate_up", "img_mlp.gate_layer")
276
+ k_proj = k.replace("img_mlp.gate_up", "img_mlp.proj")
277
+ out[k_gate] = v[:half, :].clone()
278
+ out[k_proj] = v[half:, :].clone()
279
+ continue
280
+ elif k.endswith(".lora_A.weight") or k.endswith(".lora_down.weight"):
281
+ k_gate = k.replace("img_mlp.gate_up", "img_mlp.gate_layer")
282
+ k_proj = k.replace("img_mlp.gate_up", "img_mlp.proj")
283
+ out[k_gate] = v.clone() if isinstance(v, torch.Tensor) else v
284
+ out[k_proj] = v.clone() if isinstance(v, torch.Tensor) else v
285
+ continue
286
+ elif k.endswith(".alpha") or k.endswith(".lora_alpha"):
287
+ k_gate = k.replace("img_mlp.gate_up", "img_mlp.gate_layer")
288
+ k_proj = k.replace("img_mlp.gate_up", "img_mlp.proj")
289
+ out[k_gate] = v.clone() if isinstance(v, torch.Tensor) else v
290
+ out[k_proj] = v.clone() if isinstance(v, torch.Tensor) else v
291
+ continue
292
+ out[k] = v
293
+ return out
294
+
295
+
296
+ def _check_lora_compatibility(state_dict: dict, adapter_name: str, weight_name: str):
297
+ """Detect if a LoRA was trained for the legacy 3072-dim Qwen-Image-Edit rather than 4096-dim Qwen-Image-2.1."""
298
+ for k, v in state_dict.items():
299
+ if isinstance(v, torch.Tensor) and v.ndim == 2:
300
+ if 3072 in v.shape and 4096 not in v.shape:
301
+ raise gr.Error(
302
+ f"LoRA '{adapter_name}' ({weight_name}) is incompatible with Qwen-Image-2.1: "
303
+ f"parameter shape {list(v.shape)} indicates it was trained for legacy 3072-dim Qwen-Image-Edit, but Qwen-Image-2.1 requires 4096-dim weights."
304
+ )
305
+
306
+
307
+ def _load_lora_with_fallback(repo: str, weight_name: str, adapter_name: str, needs_alpha_fix: bool = False):
308
+ try:
309
+ local_path = hf_hub_download(repo_id=repo, filename=weight_name)
310
+ sd = safetensors_load_file(local_path)
311
+ _check_lora_compatibility(sd, adapter_name, weight_name)
312
+
313
+ if any("img_mlp.gate_up" in k for k in sd.keys()):
314
+ sd = _split_gate_up_keys(sd)
315
+
316
+ if needs_alpha_fix:
317
+ sd = _inject_missing_alpha_keys(sd)
318
+ sd = _filter_to_diffusers_lora_keys(sd)
319
+ sd = _duplicate_stripped_prefix_keys(sd)
320
+ if not sd:
321
+ raise gr.Error(f"No valid LoRA weights found in '{weight_name}'.")
322
+
323
+ pipe.load_lora_weights(sd, adapter_name=adapter_name)
324
+ except Exception as e:
325
+ err_str = str(e)
326
+ if "size mismatch" in err_str or "3072" in err_str:
327
+ raise gr.Error(
328
+ f"LoRA '{adapter_name}' ({weight_name}) cannot be loaded due to dimension mismatch: "
329
+ "it was trained for legacy 3072-dim Qwen-Image-Edit, whereas Qwen-Image-2.1 has 4096 hidden dimensions."
330
+ )
331
+ print(f"Pre-processed LoRA load failed ({e}), attempting direct repo load...", flush=True)
332
+ pipe.load_lora_weights(repo, weight_name=weight_name, adapter_name=adapter_name)
333
+
334
+
335
+ def ensure_adapter_ready(selected_lora: str, custom_repo: str = "", custom_file: str = "") -> tuple[str, float]:
336
+ if selected_lora == NONE_LORA:
337
+ return "", 1.0
338
+
339
+ if selected_lora == "Custom HuggingFace LoRA...":
340
+ custom_repo = (custom_repo or "").strip()
341
+ custom_file = (custom_file or "").strip()
342
+ if not custom_repo or not custom_file:
343
+ raise gr.Error("Please enter both a Hugging Face Repo ID and LoRA filename for Custom LoRA.")
344
+ adapter_name = f"custom_{hashlib.md5((custom_repo + custom_file).encode()).hexdigest()[:8]}"
345
+ if adapter_name not in LOADED_ADAPTERS:
346
+ print(f"Loading custom LoRA from {custom_repo} / {custom_file}...", flush=True)
347
+ _load_lora_with_fallback(custom_repo, custom_file, adapter_name, needs_alpha_fix=True)
348
+ LOADED_ADAPTERS.add(adapter_name)
349
+ return adapter_name, 1.0
350
+
351
+ spec = ADAPTER_SPECS.get(selected_lora)
352
+ if not spec:
353
+ return "", 1.0
354
+
355
+ adapter_name = spec["adapter_name"]
356
+ if adapter_name not in LOADED_ADAPTERS:
357
+ print(f"Loading LoRA {selected_lora} ({spec['repo']} / {spec['weights']})...", flush=True)
358
+ _load_lora_with_fallback(
359
+ spec["repo"],
360
+ spec["weights"],
361
+ adapter_name,
362
+ needs_alpha_fix=spec.get("needs_alpha_fix", False),
363
+ )
364
+ LOADED_ADAPTERS.add(adapter_name)
365
+
366
+ return adapter_name, float(spec.get("default_strength", 1.0))
367
+
368
+
369
+ # ============================================================
370
+ # Prompt Enhancement (PE) Setup (Viggle Turbo Style)
371
+ # ============================================================
372
+ PE_MODEL_DEFAULT = os.environ.get("PE_MODEL", "deepseek/deepseek-v4.1-flash")
373
+ PE_T2I_FILE = Path(__file__).parent / "pe_t2i_system_prompt.txt"
374
+ PE_I2I_FILE = Path(__file__).parent / "pe_i2i_system_prompt.txt"
375
+
376
+ PE_T2I = PE_T2I_FILE.read_text(encoding="utf-8").strip() if PE_T2I_FILE.exists() else (
377
+ "You rewrite a user's image request into a prompt for a text-to-image model. "
378
+ "Reply with the prompt only: one paragraph of plain English, with nothing before or after it: "
379
+ "no JSON, no label, no markdown, no quotes around it."
380
+ )
381
+ PE_I2I = PE_I2I_FILE.read_text(encoding="utf-8").strip() if PE_I2I_FILE.exists() else (
382
+ "You rewrite a user's image-editing instruction into a clear instruction for an image-editing model. "
383
+ "The input image(s) come with the request. Reply with the instruction only: one paragraph of plain text."
384
+ )
385
+
386
+ PE_MAX_PIXELS = 500_000
387
+ PE_AUTO_TOKENS = 30
388
+ CJK = re.compile(r"[\u4e00-\u9fff]")
389
+ KANA_HANGUL = re.compile(r"[\u3040-\u30ff\uac00-\ud7af]")
390
+
391
+
392
+ def n_tokens(prompt: str) -> int:
393
+ """Estimate token count in prompt using Qwen tokenizer or regex fallback."""
394
+ try:
395
+ if hasattr(pipe, "processor") and hasattr(pipe.processor, "tokenizer"):
396
+ return len(pipe.processor.tokenizer(prompt)["input_ids"])
397
+ if hasattr(pipe, "tokenizer") and pipe.tokenizer:
398
+ return len(pipe.tokenizer(prompt)["input_ids"])
399
+ except Exception:
400
+ pass
401
+ return len(prompt.split()) + len(CJK.findall(prompt))
402
+
403
+
404
+ def enhance_prompt(
405
+ prompt: str,
406
+ images: list[Image.Image] | None = None,
407
+ ratio_name: str = "1:1",
408
+ api_key: str | None = None,
409
+ model_name: str | None = None,
410
+ base_url: str | None = None,
411
+ ) -> tuple[str | None, str | None]:
412
+ """Rewrite prompt using DeepSeek V4.1 Flash via OpenRouter/DeepSeek/OpenAI-compatible API."""
413
+ images = [img for img in (images or []) if img is not None]
414
+ if images:
415
+ chinese = CJK.search(prompt) and not KANA_HANGUL.search(prompt)
416
+ user_text = f"{prompt}\n\nDescription language: {'Chinese' if chinese else 'English'}"
417
+ else:
418
+ user_text = f"{prompt}\nAspect ratio: {ratio_name}"
419
+
420
+ resolved_key = (
421
+ (api_key or "").strip()
422
+ or os.environ.get("OPENROUTER_API_KEY", "").strip()
423
+ or os.environ.get("DEEPSEEK_API_KEY", "").strip()
424
+ or os.environ.get("OPENAI_API_KEY", "").strip()
425
+ or os.environ.get("HF_TOKEN", "").strip()
426
+ )
427
+ if not resolved_key:
428
+ msg = "No API key provided. Please fill in the API Key in Prompt Enhancer settings, or set OPENROUTER_API_KEY / DEEPSEEK_API_KEY."
429
+ print(f"[pe] skipped: {msg}", flush=True)
430
+ return None, msg
431
+
432
+ raw_url = (base_url or "").strip() or os.environ.get("PE_BASE_URL", "").strip()
433
+ raw_model = (model_name or "").strip() or os.environ.get("PE_MODEL", "").strip()
434
+
435
+ # Determine default endpoint if URL is not provided
436
+ if not raw_url:
437
+ if resolved_key.startswith("sk-or-"):
438
+ resolved_url = "https://openrouter.ai/api/v1/chat/completions"
439
+ elif (os.environ.get("DEEPSEEK_API_KEY") and resolved_key == os.environ.get("DEEPSEEK_API_KEY")) or (raw_model and "deepseek-chat" in raw_model):
440
+ resolved_url = "https://api.deepseek.com/chat/completions"
441
+ else:
442
+ resolved_url = "https://openrouter.ai/api/v1/chat/completions"
443
+ else:
444
+ resolved_url = raw_url
445
+
446
+ # Normalize URL: support base URLs (e.g. https://api.deepseek.com or https://api.deepseek.com/v1) and full endpoint
447
+ resolved_url = resolved_url.rstrip("/")
448
+ if not resolved_url.endswith("/chat/completions"):
449
+ resolved_url = f"{resolved_url}/chat/completions"
450
+
451
+ is_openrouter = "openrouter" in resolved_url.lower() or resolved_key.startswith("sk-or-")
452
+ is_deepseek_official = "api.deepseek.com" in resolved_url.lower()
453
+
454
+ # Model resolution
455
+ resolved_model = raw_model
456
+ if not resolved_model:
457
+ resolved_model = "deepseek-chat" if is_deepseek_official else PE_MODEL_DEFAULT
458
+
459
+ # If user selected DeepSeek official API, but left default OpenRouter model name (deepseek/deepseek-v4.1-flash), auto-switch to deepseek-chat
460
+ if is_deepseek_official and ("deepseek/" in resolved_model.lower() or "flash" in resolved_model.lower()):
461
+ resolved_model = "deepseek-chat"
462
+
463
+ # Multimodal image packaging: DeepSeek official API only supports text; OpenRouter/OpenAI support vision
464
+ content = []
465
+ if images and not is_deepseek_official:
466
+ for image in images:
467
+ img_rgb = ImageOps.exif_transpose(image).convert("RGB")
468
+ scale = min(1.0, (PE_MAX_PIXELS / (img_rgb.width * img_rgb.height)) ** 0.5)
469
+ buf = io.BytesIO()
470
+ img_rgb.resize(
471
+ (max(32, int(img_rgb.width * scale)), max(32, int(img_rgb.height * scale))),
472
+ Image.LANCZOS,
473
+ ).save(buf, "JPEG", quality=90)
474
+ content.append({
475
+ "type": "image_url",
476
+ "image_url": {"url": "data:image/jpeg;base64," + base64.b64encode(buf.getvalue()).decode()},
477
+ })
478
+ content.append({"type": "text", "text": user_text})
479
+ user_message_content = content
480
+ elif images and is_deepseek_official:
481
+ user_message_content = f"{user_text}\n(Note: User provided {len(images)} reference image(s) for visual edit context)."
482
+ else:
483
+ user_message_content = user_text
484
+
485
+ body = {
486
+ "model": resolved_model,
487
+ "messages": [
488
+ {"role": "system", "content": PE_I2I if images else PE_T2I},
489
+ {"role": "user", "content": user_message_content},
490
+ ],
491
+ }
492
+ if is_openrouter:
493
+ body["provider"] = {"sort": "throughput", "data_collection": "deny", "zdr": True}
494
+ if not images:
495
+ body["reasoning"] = {"effort": "low"}
496
+ else:
497
+ body["reasoning"] = {"enabled": False}
498
+
499
+ headers = {
500
+ "Authorization": f"Bearer {resolved_key}",
501
+ "Content-Type": "application/json",
502
+ }
503
+ if is_openrouter:
504
+ headers["HTTP-Referer"] = "https://huggingface.co/spaces"
505
+ headers["X-Title"] = "Qwen Image 2.1 Uncensored Studio"
506
+
507
+ print(f"[pe] Requesting enhancement via {resolved_url} (model: {resolved_model})...", flush=True)
508
+ try:
509
+ with httpx.Client(timeout=30) as client:
510
+ r = client.post(resolved_url, json=body, headers=headers)
511
+ r.raise_for_status()
512
+ data = r.json()
513
+ if "choices" in data and len(data["choices"]) > 0:
514
+ ans = data["choices"][0]["message"]["content"].strip()
515
+ return ans or None, None
516
+ return None, f"Empty choices response from {resolved_model}"
517
+ except httpx.HTTPStatusError as e:
518
+ err_body = ""
519
+ try:
520
+ err_body = f" - {e.response.text[:200]}"
521
+ except Exception:
522
+ pass
523
+ err_msg = f"HTTP {e.response.status_code}{err_body}"
524
+ print(f"[pe] failed: {err_msg}", flush=True)
525
+ return None, err_msg
526
+ except Exception as e:
527
+ err_msg = f"{type(e).__name__}: {e}"
528
+ print(f"[pe] failed: {err_msg}", flush=True)
529
+ return None, err_msg
530
+
531
+
532
+ def extract_pil_images(gallery_or_files):
533
+ """Normalize Gradio gallery/file inputs into a clean list of RGBA PIL Images."""
534
+ if not gallery_or_files:
535
+ return []
536
+ imgs = []
537
+ for item in gallery_or_files:
538
+ img_obj = None
539
+ if isinstance(item, (tuple, list)) and len(item) > 0:
540
+ img_obj = item[0]
541
+ elif isinstance(item, dict) and "image" in item:
542
+ img_obj = item["image"]
543
+ elif isinstance(item, (Image.Image, str)):
544
+ img_obj = item
545
+
546
+ if isinstance(img_obj, Image.Image):
547
+ imgs.append(img_obj)
548
+ elif isinstance(img_obj, str) and os.path.exists(img_obj):
549
+ try:
550
+ imgs.append(ImageOps.exif_transpose(Image.open(img_obj)).convert("RGBA"))
551
+ except Exception as e:
552
+ print(f"[multi_ref] Failed to read image {img_obj}: {e}", flush=True)
553
+ return imgs
554
+
555
+
556
+ def manual_enhance_prompt(
557
+ prompt: str,
558
+ mode: str,
559
+ ref_1: Image.Image | None,
560
+ ref_2: Image.Image | None,
561
+ multi_refs: list | None,
562
+ aspect_ratio: str,
563
+ api_key: str,
564
+ model_name: str,
565
+ base_url: str,
566
+ ) -> str:
567
+ """Manually enhance prompt and return the enriched string into UI."""
568
+ prompt = (prompt or "").strip()
569
+ if not prompt:
570
+ gr.Warning("Please enter a prompt first.")
571
+ return prompt
572
+
573
+ refs = []
574
+ if mode == "Edit Image (1 Ref)" and ref_1 is not None:
575
+ refs = [ref_1]
576
+ elif mode == "Transform & Swap (2 Refs)":
577
+ refs = [img for img in [ref_1, ref_2] if img is not None]
578
+ elif mode == "Multi-Image Reference (1-6 Refs)":
579
+ refs = extract_pil_images(multi_refs)
580
+
581
+ ratio_name = aspect_ratio.split(" · ")[1].split(" ")[0] if " · " in aspect_ratio else "1:1"
582
+ gr.Info("Enhancing prompt...")
583
+ enhanced, pe_err = enhance_prompt(
584
+ prompt=prompt,
585
+ images=refs,
586
+ ratio_name=ratio_name,
587
+ api_key=api_key,
588
+ model_name=model_name,
589
+ base_url=base_url,
590
+ )
591
+ if enhanced:
592
+ gr.Info("✨ Prompt enhanced successfully!")
593
+ return enhanced
594
+ else:
595
+ if pe_err:
596
+ gr.Warning(f"Prompt enhancement failed: {pe_err}")
597
+ else:
598
+ gr.Warning("⚠️ Prompt enhancement returned empty result.")
599
+ return prompt
600
+
601
+
602
+ # ============================================================
603
+ # Inference Function
604
+ # ============================================================
605
+ @gpu
606
+ def _run_diffusion_pipeline(
607
+ effective_prompt: str,
608
+ width: int,
609
+ height: int,
610
+ steps: int,
611
+ actual_seed: int,
612
+ call_kwargs: dict,
613
+ lora_slots: list[tuple[str, float, str, str]],
614
+ cfg_scale: float = 1.0,
615
+ negative_prompt: str = "",
616
+ ):
617
+ merged_adapters = {} # adapter_name -> (eff_weight, display_name)
618
+
619
+ for lora_name, lora_weight, c_repo, c_file in lora_slots:
620
+ if not lora_name or lora_name == NONE_LORA:
621
+ continue
622
+ adapter_name, base_strength = ensure_adapter_ready(lora_name, c_repo, c_file)
623
+ if adapter_name:
624
+ eff_weight = float(lora_weight) * base_strength
625
+ disp_name = lora_name if lora_name != "Custom HuggingFace LoRA..." else f"Custom({c_file or c_repo})"
626
+ if adapter_name in merged_adapters:
627
+ cur_w, cur_name = merged_adapters[adapter_name]
628
+ merged_adapters[adapter_name] = (cur_w + eff_weight, cur_name)
629
+ else:
630
+ merged_adapters[adapter_name] = (eff_weight, disp_name)
631
+
632
+ if merged_adapters:
633
+ active_adapter_names = list(merged_adapters.keys())
634
+ active_adapter_weights = [v[0] for v in merged_adapters.values()]
635
+ pipe.set_adapters(active_adapter_names, adapter_weights=active_adapter_weights)
636
+ active_lora_desc = " + ".join([f"{v[1]} (scale={round(v[0], 2)})" for v in merged_adapters.values()])
637
+ else:
638
+ pipe.disable_lora()
639
+ active_lora_desc = "None"
640
+
641
+ pipe_kwargs = dict(call_kwargs)
642
+ cfg_val = float(cfg_scale) if cfg_scale is not None else 1.0
643
+ if cfg_val > 1.0:
644
+ pipe_kwargs["true_cfg_scale"] = cfg_val
645
+ pipe_kwargs["negative_prompt"] = negative_prompt.strip() if (negative_prompt and negative_prompt.strip()) else ""
646
+ else:
647
+ pipe_kwargs["true_cfg_scale"] = 1.0
648
+
649
+ with torch.inference_mode():
650
+ result = pipe(
651
+ prompt=effective_prompt,
652
+ width=width,
653
+ height=height,
654
+ num_inference_steps=int(steps),
655
+ generator=torch.Generator("cuda").manual_seed(actual_seed),
656
+ **pipe_kwargs,
657
+ ).images[0]
658
+ return result, active_lora_desc
659
+
660
+
661
+ def generate(
662
+ prompt: str,
663
+ negative_prompt: str = "",
664
+ mode: str = "Text to Image",
665
+ ref_image_1: Image.Image | None = None,
666
+ ref_image_2: Image.Image | None = None,
667
+ multi_refs: list | None = None,
668
+ # LoRA Slot 1
669
+ lora_1: str = NONE_LORA,
670
+ lora_1_strength: float = 1.0,
671
+ custom_repo_1: str = "",
672
+ custom_file_1: str = "",
673
+ # LoRA Slot 2
674
+ lora_2: str = NONE_LORA,
675
+ lora_2_strength: float = 1.0,
676
+ custom_repo_2: str = "",
677
+ custom_file_2: str = "",
678
+ # LoRA Slot 3
679
+ lora_3: str = NONE_LORA,
680
+ lora_3_strength: float = 1.0,
681
+ custom_repo_3: str = "",
682
+ custom_file_3: str = "",
683
+ # LoRA Slot 4
684
+ lora_4: str = NONE_LORA,
685
+ lora_4_strength: float = 1.0,
686
+ custom_repo_4: str = "",
687
+ custom_file_4: str = "",
688
+ aspect_ratio: str = "Square · 1:1 (1024x1024)",
689
+ enhance_mode: str = "Auto",
690
+ cfg_scale: float = 1.0,
691
+ steps: int = 30,
692
+ seed: int = 42,
693
+ randomize_seed: bool = True,
694
+ pe_api_key: str = "",
695
+ pe_model: str = PE_MODEL_DEFAULT,
696
+ pe_base_url: str = "",
697
+ progress: gr.Progress = gr.Progress(track_tqdm=True),
698
+ ) -> tuple[str, str, int, str, str]:
699
+ prompt = (prompt or "").strip()
700
+ if not prompt:
701
+ raise gr.Error("Please enter a prompt describing your image.")
702
+ if len(prompt) > 4000:
703
+ raise gr.Error("Prompt is too long. Please keep under 4000 characters.")
704
+ if mode not in MODES:
705
+ raise gr.Error(f"Invalid mode: {mode}")
706
+ if aspect_ratio not in SIZES:
707
+ raise gr.Error(f"Invalid aspect ratio: {aspect_ratio}")
708
+ if steps is None or not (4 <= int(steps) <= 40):
709
+ raise gr.Error("Inference steps must be between 4 and 40.")
710
+
711
+ actual_seed = random.randint(0, MAX_SEED) if randomize_seed else int(seed)
712
+ width, height = SIZES[aspect_ratio]
713
+ ratio_name = aspect_ratio.split(" · ")[1].split(" ")[0] if " · " in aspect_ratio else "1:1"
714
+ cfg_val = float(cfg_scale) if cfg_scale is not None else 1.0
715
+
716
+ call_kwargs = {}
717
+ refs_for_pe = []
718
+
719
+ if mode == "Edit Image (1 Ref)":
720
+ if ref_image_1 is None:
721
+ raise gr.Error("Please upload a reference image for single-image edit mode.")
722
+ img = ImageOps.exif_transpose(ref_image_1).convert("RGBA")
723
+ img.thumbnail((2048, 2048))
724
+ call_kwargs["image"] = img
725
+ refs_for_pe = [ref_image_1]
726
+
727
+ elif mode == "Transform & Swap (2 Refs)":
728
+ if ref_image_1 is None or ref_image_2 is None:
729
+ raise gr.Error("Please upload both Image 1 (Base) and Image 2 (Donor/Pose) for this mode.")
730
+ img1 = ImageOps.exif_transpose(ref_image_1).convert("RGBA")
731
+ img1.thumbnail((2048, 2048))
732
+ img2 = ImageOps.exif_transpose(ref_image_2).convert("RGBA")
733
+ img2.thumbnail((2048, 2048))
734
+ call_kwargs["image"] = [img1, img2]
735
+ refs_for_pe = [ref_image_1, ref_image_2]
736
+
737
+ elif mode == "Multi-Image Reference (1-6 Refs)":
738
+ multi_imgs = extract_pil_images(multi_refs)
739
+ if not multi_imgs:
740
+ raise gr.Error("Please upload at least one reference image for Multi-Image Reference mode.")
741
+ if len(multi_imgs) > 6:
742
+ gr.Warning(f"Only the first 6 reference images will be used (received {len(multi_imgs)}).")
743
+ multi_imgs = multi_imgs[:6]
744
+
745
+ processed_imgs = []
746
+ for raw_img in multi_imgs:
747
+ m_img = ImageOps.exif_transpose(raw_img).convert("RGBA")
748
+ m_img.thumbnail((2048, 2048))
749
+ processed_imgs.append(m_img)
750
+
751
+ call_kwargs["image"] = processed_imgs if len(processed_imgs) > 1 else processed_imgs[0]
752
+ refs_for_pe = multi_imgs
753
+
754
+ # Step 1: Prompt Enhancement (runs before GPU acquisition to conserve ZeroGPU quota)
755
+ used_prompt = prompt
756
+ enhance_note = ""
757
+ pe_time = 0.0
758
+
759
+ should_enhance = False
760
+ if prompt.strip():
761
+ if enhance_mode == "On":
762
+ should_enhance = True
763
+ elif enhance_mode == "Auto" and n_tokens(prompt) < PE_AUTO_TOKENS:
764
+ should_enhance = True
765
+
766
+ if should_enhance:
767
+ pe_start = time.perf_counter()
768
+ enhanced, pe_err = enhance_prompt(
769
+ prompt=prompt,
770
+ images=refs_for_pe,
771
+ ratio_name=ratio_name,
772
+ api_key=pe_api_key,
773
+ model_name=pe_model,
774
+ base_url=pe_base_url,
775
+ )
776
+ pe_time = time.perf_counter() - pe_start
777
+ if enhanced and enhanced != prompt:
778
+ used_prompt = enhanced
779
+ enhance_note = f" | 🪄 Enhanced ({pe_time:.1f}s)"
780
+ elif pe_err:
781
+ enhance_note = f" | ⚠️ PE failed: {pe_err[:40]}"
782
+ else:
783
+ enhance_note = " | 🪄 Enhancement skipped/as written"
784
+
785
+ effective_prompt = used_prompt
786
+ if mode == "Transparent PNG":
787
+ effective_prompt = (
788
+ "This is an RGBA image with transparency. " + used_prompt
789
+ + " The image has alpha channel and the background is transparent."
790
+ )
791
+
792
+ lora_slots = [
793
+ (lora_1, lora_1_strength, custom_repo_1, custom_file_1),
794
+ (lora_2, lora_2_strength, custom_repo_2, custom_file_2),
795
+ (lora_3, lora_3_strength, custom_repo_3, custom_file_3),
796
+ (lora_4, lora_4_strength, custom_repo_4, custom_file_4),
797
+ ]
798
+
799
+ # Step 2: Denoise on GPU
800
+ start_time = time.perf_counter()
801
+ result, active_lora_desc = _run_diffusion_pipeline(
802
+ effective_prompt=effective_prompt,
803
+ width=width,
804
+ height=height,
805
+ steps=int(steps),
806
+ actual_seed=actual_seed,
807
+ call_kwargs=call_kwargs,
808
+ lora_slots=lora_slots,
809
+ cfg_scale=cfg_val,
810
+ negative_prompt=negative_prompt,
811
+ )
812
+ elapsed = time.perf_counter() - start_time
813
+
814
+ # Embed metadata into PNG chunks
815
+ metadata = {
816
+ "model": MODEL_ID,
817
+ "checkpoint": CHECKPOINT,
818
+ "lora_adapter": active_lora_desc,
819
+ "original_prompt": prompt,
820
+ "prompt": used_prompt,
821
+ "negative_prompt": negative_prompt if cfg_val > 1.0 else "",
822
+ "mode": mode,
823
+ "seed": actual_seed,
824
+ "steps": int(steps),
825
+ "cfg_scale": cfg_val,
826
+ "dimensions": f"{result.width}x{result.height}",
827
+ "elapsed_seconds": round(elapsed, 2),
828
+ "pe_seconds": round(pe_time, 2) if pe_time > 0 else 0,
829
+ }
830
+ png_info = PngImagePlugin.PngInfo()
831
+ png_info.add_text("parameters", json.dumps(metadata, ensure_ascii=False))
832
+
833
+ temp_dir = tempfile.mkdtemp(prefix="qwen_aio_lora_")
834
+ out_png_path = os.path.join(temp_dir, f"qwen_aio_{actual_seed}.png")
835
+ result.save(out_png_path, "PNG", pnginfo=png_info, optimize=True)
836
+
837
+ cfg_info = f" | CFG: {cfg_val}" if cfg_val > 1.0 else " | CFG: 1.0 (Native)"
838
+ details = (
839
+ f"⚡ Time: {elapsed:.2f}s{enhance_note} | Seed: {actual_seed} | Steps: {steps}{cfg_info}\n"
840
+ f"🎛️ LoRA: {active_lora_desc} | Size: {result.width}x{result.height}\n"
841
+ f"🔒 Privacy: Zero data retention (session ephemeral)"
842
+ )
843
+
844
+ return out_png_path, out_png_path, actual_seed, details, used_prompt
845
+
846
+
847
+
848
+ # ============================================================
849
+ # UI Helpers
850
+ # ============================================================
851
+ def update_lora_selection(selected_lora: str, current_prompt: str, current_steps: int):
852
+ spec = ADAPTER_SPECS.get(selected_lora)
853
+ show_custom = gr.update(visible=(selected_lora == "Custom HuggingFace LoRA..."))
854
+ if not spec:
855
+ return current_prompt, current_steps, 1.0, show_custom
856
+
857
+ preset = spec.get("preset_prompt", "")
858
+ new_prompt = current_prompt
859
+ if preset and not current_prompt.strip():
860
+ new_prompt = preset
861
+ elif preset and preset not in current_prompt:
862
+ new_prompt = f"{current_prompt}, {preset}" if current_prompt.strip() else preset
863
+
864
+ recommended_steps = spec.get("default_steps", current_steps)
865
+ recommended_strength = spec.get("default_strength", 1.0)
866
+
867
+ return new_prompt, recommended_steps, recommended_strength, show_custom
868
+
869
+
870
+ def update_mode_ui(mode: str):
871
+ is_multi = mode == "Multi-Image Reference (1-6 Refs)"
872
+ is_edit_1 = mode == "Edit Image (1 Ref)"
873
+ is_swap_2 = mode == "Transform & Swap (2 Refs)"
874
+ return (
875
+ gr.update(visible=is_multi),
876
+ gr.update(visible=(is_edit_1 or is_swap_2)),
877
+ gr.update(visible=is_swap_2),
878
+ )
879
+
880
+
881
+
882
+ # ============================================================
883
+ # Gradio Interface
884
+ # ============================================================
885
+ CUSTOM_CSS = """
886
+ .gradio-container { max-width: 1280px !important; margin: 0 auto !important; }
887
+ .badge { display: inline-block; padding: 2px 8px; border-radius: 6px; font-size: 0.8rem; font-weight: 600; margin-right: 6px; }
888
+ .badge-turbo { background: #fee2e2; color: #991b1b; }
889
+ .badge-style { background: #ede9fe; color: #5b21b6; }
890
+ .badge-tool { background: #e0f2fe; color: #075985; }
891
+ #generate-btn { font-weight: 700; font-size: 1.1rem; }
892
+ """
893
+
894
+ with gr.Blocks(title="Qwen Image 2.1 Uncensored All-In-One LoRA Studio", css=CUSTOM_CSS, delete_cache=(3600, 86400)) as demo:
895
+ gr.Markdown(
896
+ "# 🚀 Qwen Image 2.1 Uncensored All-In-One LoRA Studio\n"
897
+ "### Supercharged Text-to-Image, Image Editing & Guided Synthesis powered by ZeroGPU\n"
898
+ "Uncensored Base (`KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF`) + 15+ On-Demand Style, Speed, and Face/Pose LoRAs"
899
+ )
900
+
901
+ with gr.Row():
902
+ with gr.Column(scale=6):
903
+ mode_selector = gr.Radio(
904
+ choices=MODES,
905
+ value=MODES[0],
906
+ label="Generation Mode",
907
+ )
908
+
909
+ prompt_input = gr.Textbox(
910
+ label="Prompt",
911
+ placeholder="Describe your vision, character, scene, or requested edit...",
912
+ lines=3,
913
+ max_lines=6,
914
+ )
915
+
916
+ with gr.Row():
917
+ enhance_radio = gr.Radio(
918
+ choices=["Auto", "On", "Off"],
919
+ value="Auto",
920
+ label="Enhance Prompt (Viggle Turbo / DeepSeek V4.1 Flash)",
921
+ info=f"Rewrites the prompt using DeepSeek V4.1 Flash (+~2s). 'Auto' rewrites prompts under {PE_AUTO_TOKENS} tokens and sends longer ones as written; 'Off' sends nothing.",
922
+ scale=4,
923
+ )
924
+ manual_enhance_btn = gr.Button("🪄 Enhance Prompt", size="sm", scale=1)
925
+
926
+ with gr.Row():
927
+ multi_refs_gallery = gr.Gallery(
928
+ label="Reference Images (Upload 1-6 images, refer to as Image 1, Image 2... in prompt)",
929
+ type="pil",
930
+ columns=3,
931
+ height=280,
932
+ interactive=True,
933
+ visible=False,
934
+ )
935
+
936
+ with gr.Row():
937
+ ref_img_1 = gr.Image(
938
+ label="Image 1 (Base / Target)",
939
+ type="pil",
940
+ visible=False,
941
+ )
942
+ ref_img_2 = gr.Image(
943
+ label="Image 2 (Face Donor / Pose Reference)",
944
+ type="pil",
945
+ visible=False,
946
+ )
947
+
948
+ with gr.Accordion("🎨 Multi-LoRA Studio (Stack up to 4 LoRAs)", open=True):
949
+ gr.Markdown("💡 **Multi-LoRA Stacking**: Combine multiple adapters simultaneously (e.g. Turbo Speed + Art Style + Face Swap + Detail Enhancer).")
950
+
951
+ # Slot 1
952
+ with gr.Group():
953
+ with gr.Row():
954
+ lora_1_dropdown = gr.Dropdown(
955
+ choices=LORA_CHOICES,
956
+ value=NONE_LORA,
957
+ label="LoRA Slot 1 (Primary)",
958
+ info="Select your primary style, speed booster, or transfer LoRA",
959
+ scale=3,
960
+ )
961
+ lora_1_strength = gr.Slider(
962
+ minimum=0.0,
963
+ maximum=1.5,
964
+ value=1.0,
965
+ step=0.05,
966
+ label="Weight 1",
967
+ scale=1,
968
+ )
969
+ with gr.Row(visible=False) as custom_box_1:
970
+ custom_repo_1 = gr.Textbox(label="LoRA 1 HF Repo", placeholder="e.g. baka999/Q2.1")
971
+ custom_file_1 = gr.Textbox(label="LoRA 1 Weights File", placeholder="e.g. NSFW Qwen by TheseAlpacas V2.safetensors")
972
+
973
+ # Additional Slots 2, 3, 4
974
+ with gr.Accordion("➕ Stack Additional LoRAs (Slot 2, 3, 4)", open=False):
975
+ with gr.Group():
976
+ with gr.Row():
977
+ lora_2_dropdown = gr.Dropdown(
978
+ choices=LORA_CHOICES,
979
+ value=NONE_LORA,
980
+ label="LoRA Slot 2",
981
+ scale=3,
982
+ )
983
+ lora_2_strength = gr.Slider(
984
+ minimum=0.0,
985
+ maximum=1.5,
986
+ value=1.0,
987
+ step=0.05,
988
+ label="Weight 2",
989
+ scale=1,
990
+ )
991
+ with gr.Row(visible=False) as custom_box_2:
992
+ custom_repo_2 = gr.Textbox(label="LoRA 2 HF Repo", placeholder="e.g. reverentelusarca/elusarcas-qwen-2.1-detail-enhancer-lora")
993
+ custom_file_2 = gr.Textbox(label="LoRA 2 Weights File", placeholder="e.g. elusarcas-qwen2-1-detailer-v1.safetensors")
994
+
995
+ with gr.Group():
996
+ with gr.Row():
997
+ lora_3_dropdown = gr.Dropdown(
998
+ choices=LORA_CHOICES,
999
+ value=NONE_LORA,
1000
+ label="LoRA Slot 3",
1001
+ scale=3,
1002
+ )
1003
+ lora_3_strength = gr.Slider(
1004
+ minimum=0.0,
1005
+ maximum=1.5,
1006
+ value=1.0,
1007
+ step=0.05,
1008
+ label="Weight 3",
1009
+ scale=1,
1010
+ )
1011
+ with gr.Row(visible=False) as custom_box_3:
1012
+ custom_repo_3 = gr.Textbox(label="LoRA 3 HF Repo", placeholder="e.g. WarmBloodAban/Qwen-Image-2.1-LoRAs")
1013
+ custom_file_3 = gr.Textbox(label="LoRA 3 Weights File", placeholder="e.g. Qwen2.1_Anything2RealCharacters.safetensors")
1014
+
1015
+ with gr.Group():
1016
+ with gr.Row():
1017
+ lora_4_dropdown = gr.Dropdown(
1018
+ choices=LORA_CHOICES,
1019
+ value=NONE_LORA,
1020
+ label="LoRA Slot 4",
1021
+ scale=3,
1022
+ )
1023
+ lora_4_strength = gr.Slider(
1024
+ minimum=0.0,
1025
+ maximum=1.5,
1026
+ value=1.0,
1027
+ step=0.05,
1028
+ label="Weight 4",
1029
+ scale=1,
1030
+ )
1031
+ with gr.Row(visible=False) as custom_box_4:
1032
+ custom_repo_4 = gr.Textbox(label="LoRA 4 HF Repo", placeholder="e.g. ausboss/Qwen-Image-2.1-Outpaint-LoRA")
1033
+ custom_file_4 = gr.Textbox(label="LoRA 4 Weights File", placeholder="e.g. qwen-image-2.1-outpaint-v2.safetensors")
1034
+
1035
+ with gr.Accordion("⚙️ Advanced Generation Settings", open=False):
1036
+ with gr.Row():
1037
+ aspect_ratio_dropdown = gr.Dropdown(
1038
+ choices=list(SIZES.keys()),
1039
+ value=list(SIZES.keys())[0],
1040
+ label="Aspect Ratio",
1041
+ )
1042
+ steps_slider = gr.Slider(
1043
+ minimum=4,
1044
+ maximum=40,
1045
+ value=30,
1046
+ step=1,
1047
+ label="Inference Steps",
1048
+ info="4-6 steps for Turbo LoRAs, 25-35 for regular generation",
1049
+ )
1050
+ with gr.Row():
1051
+ cfg_slider = gr.Slider(
1052
+ minimum=1.0,
1053
+ maximum=10.0,
1054
+ value=1.0,
1055
+ step=0.1,
1056
+ label="CFG Scale (true_cfg_scale)",
1057
+ info="1.0 = native guidance-free (fastest, recommended for Turbo/Qwen 2.1). > 1.0 activates negative prompt guidance.",
1058
+ )
1059
+ seed_number = gr.Number(value=42, label="Seed", precision=0)
1060
+ randomize_seed_cb = gr.Checkbox(value=True, label="Randomize Seed")
1061
+
1062
+ negative_prompt_input = gr.Textbox(
1063
+ label="Negative Prompt (Active when CFG > 1.0)",
1064
+ placeholder="e.g. blurry, low quality, distorted, bad anatomy, deformed limbs...",
1065
+ lines=2,
1066
+ max_lines=4,
1067
+ )
1068
+
1069
+ with gr.Accordion("🤖 Prompt Enhancer API Settings (Optional)", open=False):
1070
+ pe_api_key_input = gr.Textbox(
1071
+ label="OpenRouter / OpenAI / DeepSeek API Key",
1072
+ type="password",
1073
+ placeholder="Optional if OPENROUTER_API_KEY, OPENAI_API_KEY, or DEEPSEEK_API_KEY is in env",
1074
+ )
1075
+ with gr.Row():
1076
+ pe_model_input = gr.Textbox(
1077
+ label="Enhancer Model",
1078
+ value=PE_MODEL_DEFAULT,
1079
+ placeholder="deepseek/deepseek-v4.1-flash (OpenRouter) or deepseek-chat (DeepSeek API)",
1080
+ )
1081
+ pe_base_url_input = gr.Textbox(
1082
+ label="Custom API Base URL",
1083
+ placeholder="e.g. https://api.deepseek.com/v1 (or https://openrouter.ai/api/v1)",
1084
+ )
1085
+
1086
+ generate_btn = gr.Button(
1087
+ "✨ Generate Image",
1088
+ variant="primary",
1089
+ elem_id="generate-btn",
1090
+ )
1091
+
1092
+ with gr.Column(scale=6):
1093
+ output_image = gr.Image(
1094
+ label="Generated Output",
1095
+ type="filepath",
1096
+ interactive=False,
1097
+ )
1098
+ with gr.Row():
1099
+ download_file = gr.File(
1100
+ label="Download High-Res PNG (Includes Parameters)",
1101
+ interactive=False,
1102
+ )
1103
+ used_prompt_box = gr.Textbox(
1104
+ label="Prompt Sent to the Model",
1105
+ lines=3,
1106
+ interactive=False,
1107
+ info="The final prompt sent to the image model (shows enriched prompt when enhanced).",
1108
+ )
1109
+ details_box = gr.Textbox(
1110
+ label="Execution Details & Privacy",
1111
+ interactive=False,
1112
+ )
1113
+
1114
+ # Event Bindings
1115
+ mode_selector.change(
1116
+ fn=update_mode_ui,
1117
+ inputs=[mode_selector],
1118
+ outputs=[multi_refs_gallery, ref_img_1, ref_img_2],
1119
+ )
1120
+
1121
+ lora_1_dropdown.change(
1122
+ fn=update_lora_selection,
1123
+ inputs=[lora_1_dropdown, prompt_input, steps_slider],
1124
+ outputs=[prompt_input, steps_slider, lora_1_strength, custom_box_1],
1125
+ )
1126
+ lora_2_dropdown.change(
1127
+ fn=update_lora_selection,
1128
+ inputs=[lora_2_dropdown, prompt_input, steps_slider],
1129
+ outputs=[prompt_input, steps_slider, lora_2_strength, custom_box_2],
1130
+ )
1131
+ lora_3_dropdown.change(
1132
+ fn=update_lora_selection,
1133
+ inputs=[lora_3_dropdown, prompt_input, steps_slider],
1134
+ outputs=[prompt_input, steps_slider, lora_3_strength, custom_box_3],
1135
+ )
1136
+ lora_4_dropdown.change(
1137
+ fn=update_lora_selection,
1138
+ inputs=[lora_4_dropdown, prompt_input, steps_slider],
1139
+ outputs=[prompt_input, steps_slider, lora_4_strength, custom_box_4],
1140
+ )
1141
+
1142
+ manual_enhance_btn.click(
1143
+ fn=manual_enhance_prompt,
1144
+ inputs=[
1145
+ prompt_input,
1146
+ mode_selector,
1147
+ ref_img_1,
1148
+ ref_img_2,
1149
+ multi_refs_gallery,
1150
+ aspect_ratio_dropdown,
1151
+ pe_api_key_input,
1152
+ pe_model_input,
1153
+ pe_base_url_input,
1154
+ ],
1155
+ outputs=[prompt_input],
1156
+ )
1157
+
1158
+ generate_btn.click(
1159
+ fn=generate,
1160
+ inputs=[
1161
+ prompt_input,
1162
+ negative_prompt_input,
1163
+ mode_selector,
1164
+ ref_img_1,
1165
+ ref_img_2,
1166
+ multi_refs_gallery,
1167
+ lora_1_dropdown,
1168
+ lora_1_strength,
1169
+ custom_repo_1,
1170
+ custom_file_1,
1171
+ lora_2_dropdown,
1172
+ lora_2_strength,
1173
+ custom_repo_2,
1174
+ custom_file_2,
1175
+ lora_3_dropdown,
1176
+ lora_3_strength,
1177
+ custom_repo_3,
1178
+ custom_file_3,
1179
+ lora_4_dropdown,
1180
+ lora_4_strength,
1181
+ custom_repo_4,
1182
+ custom_file_4,
1183
+ aspect_ratio_dropdown,
1184
+ enhance_radio,
1185
+ cfg_slider,
1186
+ steps_slider,
1187
+ seed_number,
1188
+ randomize_seed_cb,
1189
+ pe_api_key_input,
1190
+ pe_model_input,
1191
+ pe_base_url_input,
1192
+ ],
1193
+ outputs=[output_image, download_file, seed_number, details_box, used_prompt_box],
1194
+ api_name="generate",
1195
+ concurrency_limit=1,
1196
+ concurrency_id="qwen-aio-pipeline",
1197
+ )
1198
+
1199
+ gr.Examples(
1200
+ examples=[
1201
+ [
1202
+ "A futuristic neon cyberpunk samurai in a rain-soaked Tokyo alleyway, glowing katana, volumetric mist",
1203
+ "Text to Image",
1204
+ "Anime Consistency",
1205
+ ],
1206
+ [
1207
+ "A tranquil highland mountain lake surrounded by autumn pines at golden hour, reflections in water",
1208
+ "Text to Image",
1209
+ "Natural Exposure (Photorealism)",
1210
+ ],
1211
+ [
1212
+ "A charming little steampunk robot holding a delicate glass flower, highly detailed gears and brass clockwork",
1213
+ "Text to Image",
1214
+ "Viggle Turbo (4-Step Acceleration)",
1215
+ ],
1216
+ ],
1217
+ inputs=[prompt_input, mode_selector, lora_1_dropdown],
1218
+ label="Try an Idea",
1219
+ )
1220
+
1221
+ if __name__ == "__main__":
1222
+ demo.queue(max_size=16, default_concurrency_limit=1).launch(
1223
+ mcp_server=True,
1224
+ )
pe_i2i_system_prompt.txt ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ You rewrite a user's image-editing instruction into a clear instruction for an image-editing model. The input image(s) come with the request.
2
+
3
+ Reply with the instruction only: one paragraph of plain text, one to three sentences, at most 80 words (at most 150 characters in Chinese), with nothing before or after it: no JSON, no label, no markdown.
4
+
5
+ Write it in the language named on the last line of the request (`Description language: ...`).
6
+
7
+ - Lead with the edit, concrete and strong enough to be unmistakable: what changes, where, and how it looks afterwards.
8
+ - Then one short clause for what stays, naming kept things by type without describing them: "keep her face, pose, clothing and the background unchanged".
9
+ - Change only what was asked; do not clean up or restyle anything else.
10
+ - If the user wants a new scene built from the inputs (a group photo, a subject placed somewhere new), describe that scene briefly: setting, arrangement, lighting.
11
+ - Text in the image: copy every string the user gives exactly, in its own script, inside double quotes. New text the user asks for without giving the words follows the language of the image's existing text. Existing text the edit does not target stays unchanged; add no other text.
12
+ - With several input images, refer to them as <image1>, <image2>, ... and say what each one contributes. With one image, just say "the image".
13
+
14
+ Think briefly: identify the target and what must stay, then write.
pe_t2i_system_prompt.txt ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ You rewrite a user's image request into a prompt for a text-to-image model.
2
+
3
+ Reply with the prompt only: one paragraph of plain English, with nothing before or after it: no JSON, no label, no markdown, no quotes around it.
4
+
5
+ Length follows the content: a single subject or a simple scene gets four to six sentences; a layout with many parts (poster, infographic, app screen, chart, multi-panel comic) gets one sentence per region or panel, up to about 250 words. Do not count words.
6
+
7
+ Keep everything the user fixed: subject, names, counts, colours, positions, style. Decide what they left open with concrete, plausible choices.
8
+
9
+ Write it in this order:
10
+ 1. The medium and style (photograph, poster, flat-vector logo, app screenshot, 3D render, watercolour, ...), the subject, and the orientation.
11
+ 2. Where each main element sits in the frame, then the background.
12
+ 3. The lighting, then the palette and mood.
13
+
14
+ The request ends with a line `Aspect ratio: W:H`. Compose for it (wider is horizontal, taller is vertical, equal is square), but never write a ratio or a resolution.
15
+
16
+ Panels or shots: keep exactly the number the user gives, name the grid (for example "two rows of three"), then one sentence per panel, in order.
17
+
18
+ Text in the image:
19
+ - Copy every string the user wants shown exactly, in its own script, inside double quotes: 晨光咖啡 stays "晨光咖啡", never translated.
20
+ - If the request implies text without giving the words (labels, captions, a title, app or chart labels, timeline entries), write every readable string out in full, short and correct, in the request's language, in double quotes.
21
+ - Add no other text: no slogans, taglines, prices, dates, addresses or signs the user did not ask for. A plain scene has no text at all.
22
+
23
+ Describe what is in the frame, in the present tense. No "masterpiece", "8K" or "highly detailed". Obey instructions about the job ("no watermark", "sharp text") without repeating them.
24
+
25
+ Think briefly: settle the layout and the exact text, then write.
requirements.txt CHANGED
@@ -1,10 +1,11 @@
1
- diffusers @ git+https://github.com/huggingface/diffusers.git@9f1246971270c84dcbe71233edb7a519596a5d02
2
- transformers==5.17.0
3
- accelerate
4
- peft
5
- torchvision
6
- safetensors
7
- Pillow
8
- sentencepiece
9
- gguf==0.19.0
10
- mcp
 
 
1
+ diffusers @ git+https://github.com/huggingface/diffusers.git@9f1246971270c84dcbe71233edb7a519596a5d02
2
+ transformers==5.17.0
3
+ accelerate
4
+ peft
5
+ torchvision
6
+ safetensors
7
+ Pillow
8
+ sentencepiece
9
+ gguf==0.19.0
10
+ mcp
11
+ httpx