Spaces:
Running on Zero
Running on Zero
Add speed (1024px) / quality (2048px) toggle; decode 2K outputs with 1024px VAE tiles
Browse files
app.py
CHANGED
|
@@ -34,6 +34,7 @@ PE_SPACE_ID = os.environ.get("PE_SPACE_ID", "hugging-apps/qwen-image-2-1-prompt-
|
|
| 34 |
PE_MAX_NEW_TOKENS = {"t2i": 1536, "i2i": 2048}
|
| 35 |
GUARD_THRESHOLD = 0.5
|
| 36 |
NUM_INFERENCE_STEPS = 28
|
|
|
|
| 37 |
TRUE_CFG_SCALE = 4.0
|
| 38 |
# The prefix KV cache costs ~2 GB per 1K condition image; above this budget it is
|
| 39 |
# switched off so many-image edits still fit next to the weights.
|
|
@@ -186,6 +187,10 @@ def ratio_to_size(ratio, area=1024 * 1024, multiple=32):
|
|
| 186 |
aspect = float(match.group(1)) / float(match.group(2))
|
| 187 |
if not 1 / 4 <= aspect <= 4:
|
| 188 |
return None, None
|
|
|
|
|
|
|
|
|
|
|
|
|
| 189 |
width = round((area * aspect) ** 0.5 / multiple) * multiple
|
| 190 |
height = round((area / aspect) ** 0.5 / multiple) * multiple
|
| 191 |
return height, width
|
|
@@ -237,6 +242,13 @@ def generation_duration(prompt, image_paths, height, width, negative_prompt, see
|
|
| 237 |
def run_pipeline(prompt, image_paths, height, width, negative_prompt, seed):
|
| 238 |
images = [Image.open(p) for p in image_paths] or None
|
| 239 |
kwargs = {"use_kv_cache": kv_cache_fits(len(image_paths))}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 240 |
if negative_prompt:
|
| 241 |
kwargs.update(negative_prompt=negative_prompt, true_cfg_scale=TRUE_CFG_SCALE)
|
| 242 |
return pipe(
|
|
@@ -285,7 +297,7 @@ logger = GenerationLogger(log_dir=LOG_DIR)
|
|
| 285 |
MAX_SEED = np.iinfo(np.int32).max
|
| 286 |
|
| 287 |
# --- Stage 1: screen and rewrite the prompt (CPU, no GPU quota) ---
|
| 288 |
-
def prepare_stage(input_images, original_prompt, enable_extend, custom_size, seed, randomize_seed):
|
| 289 |
"""
|
| 290 |
校验输入、NCII 检查(仅在有输入图片时)、可选的提示词改写,全部在 GPU 之外完成。
|
| 291 |
Returns (image_paths, final_prompt, rewritten_prompt, seed, auto_height, auto_width).
|
|
@@ -310,8 +322,15 @@ def prepare_stage(input_images, original_prompt, enable_extend, custom_size, see
|
|
| 310 |
gr.Warning("提示词改写暂不可用,已使用原始提示词。 / Prompt enhancement is unavailable; using the original prompt.")
|
| 311 |
|
| 312 |
auto_height, auto_width = (None, None)
|
| 313 |
-
if not custom_size
|
| 314 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 315 |
return image_paths, rewritten_prompt or original_prompt, rewritten_prompt, seed, auto_height, auto_width
|
| 316 |
|
| 317 |
|
|
@@ -399,9 +418,9 @@ def make_placeholder_image(text, width=512, height=320, bg_color=(30, 30, 30), t
|
|
| 399 |
|
| 400 |
|
| 401 |
# --- Two-step flow: prepare (CPU) then generate (GPU) ---
|
| 402 |
-
def prepare_request(input_images, original_prompt, enable_extend, custom_size, seed, randomize_seed):
|
| 403 |
image_paths, final_prompt, rewritten_prompt, seed, auto_height, auto_width = prepare_stage(
|
| 404 |
-
input_images, original_prompt, enable_extend, custom_size, seed, randomize_seed,
|
| 405 |
)
|
| 406 |
request_state = {
|
| 407 |
"image_paths": image_paths,
|
|
@@ -576,6 +595,11 @@ with gr.Blocks(title="Qwen Image 2.1 Demo") as demo:
|
|
| 576 |
label="Enable Prompt Extend (提示词智能改写)",
|
| 577 |
value=True,
|
| 578 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 579 |
generate_button = gr.Button("Generate Image", variant="primary")
|
| 580 |
|
| 581 |
rewritten_prompt_output = localize(
|
|
@@ -720,6 +744,9 @@ with gr.Blocks(title="Qwen Image 2.1 Demo") as demo:
|
|
| 720 |
localize(result, label=("生成结果", "Result"))
|
| 721 |
localize(prompt, label=("提示词", "Prompt"), placeholder=("描述想生成或编辑的内容,可按上传顺序引用第 1–10 张图…", "Describe what to generate or edit; refer to images 1–10 in upload order…"))
|
| 722 |
localize(enable_extend, label=("智能改写提示词", "Enhance prompt"))
|
|
|
|
|
|
|
|
|
|
| 723 |
localize(generate_button, value=("生成图片", "Generate image"))
|
| 724 |
localize(advanced, label=("高级设置", "Advanced settings"))
|
| 725 |
localize(log_dir_input, label=("日志保存目录", "Log directory"), placeholder=("输入日志保存目录", "Enter a log directory"))
|
|
@@ -750,6 +777,7 @@ with gr.Blocks(title="Qwen Image 2.1 Demo") as demo:
|
|
| 750 |
prompt, # original_prompt
|
| 751 |
enable_extend, # enable_extend (是否开启提示词改写)
|
| 752 |
custom_size, # custom_size (是否自定义输出尺寸)
|
|
|
|
| 753 |
seed,
|
| 754 |
randomize_seed,
|
| 755 |
],
|
|
|
|
| 34 |
PE_MAX_NEW_TOKENS = {"t2i": 1536, "i2i": 2048}
|
| 35 |
GUARD_THRESHOLD = 0.5
|
| 36 |
NUM_INFERENCE_STEPS = 28
|
| 37 |
+
QUALITY_RESOLUTIONS = {"speed": 1024, "quality": 2048}
|
| 38 |
TRUE_CFG_SCALE = 4.0
|
| 39 |
# The prefix KV cache costs ~2 GB per 1K condition image; above this budget it is
|
| 40 |
# switched off so many-image edits still fit next to the weights.
|
|
|
|
| 187 |
aspect = float(match.group(1)) / float(match.group(2))
|
| 188 |
if not 1 / 4 <= aspect <= 4:
|
| 189 |
return None, None
|
| 190 |
+
return aspect_to_size(aspect, area, multiple)
|
| 191 |
+
|
| 192 |
+
|
| 193 |
+
def aspect_to_size(aspect, area, multiple=32):
|
| 194 |
width = round((area * aspect) ** 0.5 / multiple) * multiple
|
| 195 |
height = round((area / aspect) ** 0.5 / multiple) * multiple
|
| 196 |
return height, width
|
|
|
|
| 242 |
def run_pipeline(prompt, image_paths, height, width, negative_prompt, seed):
|
| 243 |
images = [Image.open(p) for p in image_paths] or None
|
| 244 |
kwargs = {"use_kv_cache": kv_cache_fits(len(image_paths))}
|
| 245 |
+
tile, stride = (1024, 768) if max(height or 0, width or 0) > 1536 else (1536, 1152)
|
| 246 |
+
pipe.vae.enable_tiling(
|
| 247 |
+
tile_sample_min_height=tile,
|
| 248 |
+
tile_sample_min_width=tile,
|
| 249 |
+
tile_sample_stride_height=stride,
|
| 250 |
+
tile_sample_stride_width=stride,
|
| 251 |
+
)
|
| 252 |
if negative_prompt:
|
| 253 |
kwargs.update(negative_prompt=negative_prompt, true_cfg_scale=TRUE_CFG_SCALE)
|
| 254 |
return pipe(
|
|
|
|
| 297 |
MAX_SEED = np.iinfo(np.int32).max
|
| 298 |
|
| 299 |
# --- Stage 1: screen and rewrite the prompt (CPU, no GPU quota) ---
|
| 300 |
+
def prepare_stage(input_images, original_prompt, enable_extend, custom_size, quality, seed, randomize_seed):
|
| 301 |
"""
|
| 302 |
校验输入、NCII 检查(仅在有输入图片时)、可选的提示词改写,全部在 GPU 之外完成。
|
| 303 |
Returns (image_paths, final_prompt, rewritten_prompt, seed, auto_height, auto_width).
|
|
|
|
| 322 |
gr.Warning("提示词改写暂不可用,已使用原始提示词。 / Prompt enhancement is unavailable; using the original prompt.")
|
| 323 |
|
| 324 |
auto_height, auto_width = (None, None)
|
| 325 |
+
if not custom_size:
|
| 326 |
+
area = QUALITY_RESOLUTIONS.get(quality, 1024) ** 2
|
| 327 |
+
if image_paths:
|
| 328 |
+
last_width, last_height = Image.open(image_paths[-1]).size
|
| 329 |
+
auto_height, auto_width = aspect_to_size(last_width / last_height, area)
|
| 330 |
+
else:
|
| 331 |
+
auto_height, auto_width = ratio_to_size(wh_ratio, area)
|
| 332 |
+
if auto_height is None:
|
| 333 |
+
auto_height, auto_width = aspect_to_size(1.0, area)
|
| 334 |
return image_paths, rewritten_prompt or original_prompt, rewritten_prompt, seed, auto_height, auto_width
|
| 335 |
|
| 336 |
|
|
|
|
| 418 |
|
| 419 |
|
| 420 |
# --- Two-step flow: prepare (CPU) then generate (GPU) ---
|
| 421 |
+
def prepare_request(input_images, original_prompt, enable_extend, custom_size, quality, seed, randomize_seed):
|
| 422 |
image_paths, final_prompt, rewritten_prompt, seed, auto_height, auto_width = prepare_stage(
|
| 423 |
+
input_images, original_prompt, enable_extend, custom_size, quality, seed, randomize_seed,
|
| 424 |
)
|
| 425 |
request_state = {
|
| 426 |
"image_paths": image_paths,
|
|
|
|
| 595 |
label="Enable Prompt Extend (提示词智能改写)",
|
| 596 |
value=True,
|
| 597 |
)
|
| 598 |
+
quality = gr.Radio(
|
| 599 |
+
choices=[("Speed (1024px)", "speed"), ("Quality (2048px)", "quality")],
|
| 600 |
+
value="speed",
|
| 601 |
+
show_label=False,
|
| 602 |
+
)
|
| 603 |
generate_button = gr.Button("Generate Image", variant="primary")
|
| 604 |
|
| 605 |
rewritten_prompt_output = localize(
|
|
|
|
| 744 |
localize(result, label=("生成结果", "Result"))
|
| 745 |
localize(prompt, label=("提示词", "Prompt"), placeholder=("描述想生成或编辑的内容,可按上传顺序引用第 1–10 张图…", "Describe what to generate or edit; refer to images 1–10 in upload order…"))
|
| 746 |
localize(enable_extend, label=("智能改写提示词", "Enhance prompt"))
|
| 747 |
+
localize(quality, choices=(
|
| 748 |
+
[("速度 (1024px)", "speed"), ("质量 (2048px)", "quality")],
|
| 749 |
+
[("Speed (1024px)", "speed"), ("Quality (2048px)", "quality")]))
|
| 750 |
localize(generate_button, value=("生成图片", "Generate image"))
|
| 751 |
localize(advanced, label=("高级设置", "Advanced settings"))
|
| 752 |
localize(log_dir_input, label=("日志保存目录", "Log directory"), placeholder=("输入日志保存目录", "Enter a log directory"))
|
|
|
|
| 777 |
prompt, # original_prompt
|
| 778 |
enable_extend, # enable_extend (是否开启提示词改写)
|
| 779 |
custom_size, # custom_size (是否自定义输出尺寸)
|
| 780 |
+
quality,
|
| 781 |
seed,
|
| 782 |
randomize_seed,
|
| 783 |
],
|