multimodalart HF Staff commited on
Commit
e33ac7d
·
verified ·
1 Parent(s): 91199a5

Add speed (1024px) / quality (2048px) toggle; decode 2K outputs with 1024px VAE tiles

Browse files
Files changed (1) hide show
  1. app.py +33 -5
app.py CHANGED
@@ -34,6 +34,7 @@ PE_SPACE_ID = os.environ.get("PE_SPACE_ID", "hugging-apps/qwen-image-2-1-prompt-
34
  PE_MAX_NEW_TOKENS = {"t2i": 1536, "i2i": 2048}
35
  GUARD_THRESHOLD = 0.5
36
  NUM_INFERENCE_STEPS = 28
 
37
  TRUE_CFG_SCALE = 4.0
38
  # The prefix KV cache costs ~2 GB per 1K condition image; above this budget it is
39
  # switched off so many-image edits still fit next to the weights.
@@ -186,6 +187,10 @@ def ratio_to_size(ratio, area=1024 * 1024, multiple=32):
186
  aspect = float(match.group(1)) / float(match.group(2))
187
  if not 1 / 4 <= aspect <= 4:
188
  return None, None
 
 
 
 
189
  width = round((area * aspect) ** 0.5 / multiple) * multiple
190
  height = round((area / aspect) ** 0.5 / multiple) * multiple
191
  return height, width
@@ -237,6 +242,13 @@ def generation_duration(prompt, image_paths, height, width, negative_prompt, see
237
  def run_pipeline(prompt, image_paths, height, width, negative_prompt, seed):
238
  images = [Image.open(p) for p in image_paths] or None
239
  kwargs = {"use_kv_cache": kv_cache_fits(len(image_paths))}
 
 
 
 
 
 
 
240
  if negative_prompt:
241
  kwargs.update(negative_prompt=negative_prompt, true_cfg_scale=TRUE_CFG_SCALE)
242
  return pipe(
@@ -285,7 +297,7 @@ logger = GenerationLogger(log_dir=LOG_DIR)
285
  MAX_SEED = np.iinfo(np.int32).max
286
 
287
  # --- Stage 1: screen and rewrite the prompt (CPU, no GPU quota) ---
288
- def prepare_stage(input_images, original_prompt, enable_extend, custom_size, seed, randomize_seed):
289
  """
290
  校验输入、NCII 检查(仅在有输入图片时)、可选的提示词改写,全部在 GPU 之外完成。
291
  Returns (image_paths, final_prompt, rewritten_prompt, seed, auto_height, auto_width).
@@ -310,8 +322,15 @@ def prepare_stage(input_images, original_prompt, enable_extend, custom_size, see
310
  gr.Warning("提示词改写暂不可用,已使用原始提示词。 / Prompt enhancement is unavailable; using the original prompt.")
311
 
312
  auto_height, auto_width = (None, None)
313
- if not custom_size and not image_paths:
314
- auto_height, auto_width = ratio_to_size(wh_ratio)
 
 
 
 
 
 
 
315
  return image_paths, rewritten_prompt or original_prompt, rewritten_prompt, seed, auto_height, auto_width
316
 
317
 
@@ -399,9 +418,9 @@ def make_placeholder_image(text, width=512, height=320, bg_color=(30, 30, 30), t
399
 
400
 
401
  # --- Two-step flow: prepare (CPU) then generate (GPU) ---
402
- def prepare_request(input_images, original_prompt, enable_extend, custom_size, seed, randomize_seed):
403
  image_paths, final_prompt, rewritten_prompt, seed, auto_height, auto_width = prepare_stage(
404
- input_images, original_prompt, enable_extend, custom_size, seed, randomize_seed,
405
  )
406
  request_state = {
407
  "image_paths": image_paths,
@@ -576,6 +595,11 @@ with gr.Blocks(title="Qwen Image 2.1 Demo") as demo:
576
  label="Enable Prompt Extend (提示词智能改写)",
577
  value=True,
578
  )
 
 
 
 
 
579
  generate_button = gr.Button("Generate Image", variant="primary")
580
 
581
  rewritten_prompt_output = localize(
@@ -720,6 +744,9 @@ with gr.Blocks(title="Qwen Image 2.1 Demo") as demo:
720
  localize(result, label=("生成结果", "Result"))
721
  localize(prompt, label=("提示词", "Prompt"), placeholder=("描述想生成或编辑的内容,可按上传顺序引用第 1–10 张图…", "Describe what to generate or edit; refer to images 1–10 in upload order…"))
722
  localize(enable_extend, label=("智能改写提示词", "Enhance prompt"))
 
 
 
723
  localize(generate_button, value=("生成图片", "Generate image"))
724
  localize(advanced, label=("高级设置", "Advanced settings"))
725
  localize(log_dir_input, label=("日志保存目录", "Log directory"), placeholder=("输入日志保存目录", "Enter a log directory"))
@@ -750,6 +777,7 @@ with gr.Blocks(title="Qwen Image 2.1 Demo") as demo:
750
  prompt, # original_prompt
751
  enable_extend, # enable_extend (是否开启提示词改写)
752
  custom_size, # custom_size (是否自定义输出尺寸)
 
753
  seed,
754
  randomize_seed,
755
  ],
 
34
  PE_MAX_NEW_TOKENS = {"t2i": 1536, "i2i": 2048}
35
  GUARD_THRESHOLD = 0.5
36
  NUM_INFERENCE_STEPS = 28
37
+ QUALITY_RESOLUTIONS = {"speed": 1024, "quality": 2048}
38
  TRUE_CFG_SCALE = 4.0
39
  # The prefix KV cache costs ~2 GB per 1K condition image; above this budget it is
40
  # switched off so many-image edits still fit next to the weights.
 
187
  aspect = float(match.group(1)) / float(match.group(2))
188
  if not 1 / 4 <= aspect <= 4:
189
  return None, None
190
+ return aspect_to_size(aspect, area, multiple)
191
+
192
+
193
+ def aspect_to_size(aspect, area, multiple=32):
194
  width = round((area * aspect) ** 0.5 / multiple) * multiple
195
  height = round((area / aspect) ** 0.5 / multiple) * multiple
196
  return height, width
 
242
  def run_pipeline(prompt, image_paths, height, width, negative_prompt, seed):
243
  images = [Image.open(p) for p in image_paths] or None
244
  kwargs = {"use_kv_cache": kv_cache_fits(len(image_paths))}
245
+ tile, stride = (1024, 768) if max(height or 0, width or 0) > 1536 else (1536, 1152)
246
+ pipe.vae.enable_tiling(
247
+ tile_sample_min_height=tile,
248
+ tile_sample_min_width=tile,
249
+ tile_sample_stride_height=stride,
250
+ tile_sample_stride_width=stride,
251
+ )
252
  if negative_prompt:
253
  kwargs.update(negative_prompt=negative_prompt, true_cfg_scale=TRUE_CFG_SCALE)
254
  return pipe(
 
297
  MAX_SEED = np.iinfo(np.int32).max
298
 
299
  # --- Stage 1: screen and rewrite the prompt (CPU, no GPU quota) ---
300
+ def prepare_stage(input_images, original_prompt, enable_extend, custom_size, quality, seed, randomize_seed):
301
  """
302
  校验输入、NCII 检查(仅在有输入图片时)、可选的提示词改写,全部在 GPU 之外完成。
303
  Returns (image_paths, final_prompt, rewritten_prompt, seed, auto_height, auto_width).
 
322
  gr.Warning("提示词改写暂不可用,已使用原始提示词。 / Prompt enhancement is unavailable; using the original prompt.")
323
 
324
  auto_height, auto_width = (None, None)
325
+ if not custom_size:
326
+ area = QUALITY_RESOLUTIONS.get(quality, 1024) ** 2
327
+ if image_paths:
328
+ last_width, last_height = Image.open(image_paths[-1]).size
329
+ auto_height, auto_width = aspect_to_size(last_width / last_height, area)
330
+ else:
331
+ auto_height, auto_width = ratio_to_size(wh_ratio, area)
332
+ if auto_height is None:
333
+ auto_height, auto_width = aspect_to_size(1.0, area)
334
  return image_paths, rewritten_prompt or original_prompt, rewritten_prompt, seed, auto_height, auto_width
335
 
336
 
 
418
 
419
 
420
  # --- Two-step flow: prepare (CPU) then generate (GPU) ---
421
+ def prepare_request(input_images, original_prompt, enable_extend, custom_size, quality, seed, randomize_seed):
422
  image_paths, final_prompt, rewritten_prompt, seed, auto_height, auto_width = prepare_stage(
423
+ input_images, original_prompt, enable_extend, custom_size, quality, seed, randomize_seed,
424
  )
425
  request_state = {
426
  "image_paths": image_paths,
 
595
  label="Enable Prompt Extend (提示词智能改写)",
596
  value=True,
597
  )
598
+ quality = gr.Radio(
599
+ choices=[("Speed (1024px)", "speed"), ("Quality (2048px)", "quality")],
600
+ value="speed",
601
+ show_label=False,
602
+ )
603
  generate_button = gr.Button("Generate Image", variant="primary")
604
 
605
  rewritten_prompt_output = localize(
 
744
  localize(result, label=("生成结果", "Result"))
745
  localize(prompt, label=("提示词", "Prompt"), placeholder=("描述想生成或编辑的内容,可按上传顺序引用第 1–10 张图…", "Describe what to generate or edit; refer to images 1–10 in upload order…"))
746
  localize(enable_extend, label=("智能改写提示词", "Enhance prompt"))
747
+ localize(quality, choices=(
748
+ [("速度 (1024px)", "speed"), ("质量 (2048px)", "quality")],
749
+ [("Speed (1024px)", "speed"), ("Quality (2048px)", "quality")]))
750
  localize(generate_button, value=("生成图片", "Generate image"))
751
  localize(advanced, label=("高级设置", "Advanced settings"))
752
  localize(log_dir_input, label=("日志保存目录", "Log directory"), placeholder=("输入日志保存目录", "Enter a log directory"))
 
777
  prompt, # original_prompt
778
  enable_extend, # enable_extend (是否开启提示词改写)
779
  custom_size, # custom_size (是否自定义输出尺寸)
780
+ quality,
781
  seed,
782
  randomize_seed,
783
  ],