baka999 commited on
Commit
4a0fdc9
·
verified ·
1 Parent(s): 1234d8d

Upload 4 files

Browse files
Files changed (2) hide show
  1. __pycache__/app.cpython-314.pyc +0 -0
  2. app.py +43 -23
__pycache__/app.cpython-314.pyc CHANGED
Binary files a/__pycache__/app.cpython-314.pyc and b/__pycache__/app.cpython-314.pyc differ
 
app.py CHANGED
@@ -318,7 +318,7 @@ def _parse_rewrite(gen: str, fallback: str) -> str:
318
  return lines[-1] if lines else fallback
319
 
320
 
321
- def _enhance_prompt_t2i_core(prompt: str) -> str:
322
  """Execute T2I prompt expansion using Qwen-Image-2.1-PE-T2I."""
323
  if not prompt or not prompt.strip():
324
  return prompt
@@ -337,16 +337,19 @@ def _enhance_prompt_t2i_core(prompt: str) -> str:
337
  ],
338
  tokenize=False,
339
  add_generation_prompt=True,
340
- enable_thinking=True,
341
  )
342
  inputs = tokenizer(text, return_tensors="pt").to(device)
 
 
 
343
  with torch.no_grad():
344
  out = model.generate(
345
  **inputs,
346
- max_new_tokens=4096,
347
  do_sample=True,
348
- temperature=1.0,
349
- top_p=0.95,
350
  top_k=20,
351
  )
352
  gen = tokenizer.decode(out[0, inputs["input_ids"].shape[1]:], skip_special_tokens=True)
@@ -362,7 +365,7 @@ def _enhance_prompt_t2i_core(prompt: str) -> str:
362
  return prompt.strip()
363
 
364
 
365
- def _enhance_prompt_i2i_core(image: Image.Image, instruction: str) -> str:
366
  """Execute I2I prompt expansion using Qwen-Image-2.1-PE-I2I."""
367
  if not instruction or not instruction.strip():
368
  instruction = "Describe and enhance this image."
@@ -375,8 +378,8 @@ def _enhance_prompt_i2i_core(image: Image.Image, instruction: str) -> str:
375
  model.to(device)
376
  try:
377
  pe_img = image.copy()
378
- if max(pe_img.size) > 1024:
379
- pe_img.thumbnail((1024, 1024), Image.Resampling.LANCZOS)
380
  if pe_img.mode != "RGB":
381
  pe_img = pe_img.convert("RGB")
382
 
@@ -396,15 +399,18 @@ def _enhance_prompt_i2i_core(image: Image.Image, instruction: str) -> str:
396
  tokenize=True,
397
  return_dict=True,
398
  return_tensors="pt",
399
- enable_thinking=True,
400
  ).to(device)
 
 
 
401
  with torch.no_grad():
402
  out = model.generate(
403
  **inputs,
404
- max_new_tokens=4096,
405
  do_sample=True,
406
- temperature=1.0,
407
- top_p=0.95,
408
  top_k=20,
409
  )
410
  gen = processor.tokenizer.decode(
@@ -423,7 +429,7 @@ def _enhance_prompt_i2i_core(image: Image.Image, instruction: str) -> str:
423
 
424
 
425
  @spaces.GPU(size="xlarge", duration=300)
426
- def enhance_prompt_action(prompt: str, ref_files: list) -> str:
427
  """Standalone prompt enhancement triggered by the UI button."""
428
  if not prompt or not prompt.strip():
429
  gr.Warning("Please enter a prompt to enhance.")
@@ -434,12 +440,12 @@ def enhance_prompt_action(prompt: str, ref_files: list) -> str:
434
  path = first_ref.name if hasattr(first_ref, "name") else first_ref
435
  try:
436
  img = Image.open(path)
437
- return _enhance_prompt_i2i_core(img, prompt)
438
  except Exception as e:
439
  print(f"Failed to open reference image for PE: {e}")
440
- return _enhance_prompt_t2i_core(prompt)
441
  else:
442
- return _enhance_prompt_t2i_core(prompt)
443
 
444
 
445
  # ==========================================
@@ -483,6 +489,7 @@ def generate(
483
  randomize_seed: bool,
484
  is_transparent: bool,
485
  auto_enhance: bool,
 
486
  progress=gr.Progress(track_tqdm=True),
487
  ):
488
  if not prompt or not prompt.strip():
@@ -519,11 +526,11 @@ def generate(
519
  # Optional Auto-Enhance
520
  effective_prompt = prompt.strip()
521
  if auto_enhance:
522
- print("Auto-enhancing prompt...")
523
  if input_images:
524
- effective_prompt = _enhance_prompt_i2i_core(input_images[0], effective_prompt)
525
  else:
526
- effective_prompt = _enhance_prompt_t2i_core(effective_prompt)
527
 
528
  # Process prompt for transparency
529
  final_prompt = effective_prompt
@@ -579,7 +586,8 @@ def generate(
579
  f"**Ref Images**: {len(input_images)}"
580
  )
581
  if auto_enhance:
582
- info_text += " | **Prompt Enhanced**: ✨ Yes"
 
583
 
584
  return result_image, seed, info_text, effective_prompt
585
 
@@ -647,7 +655,12 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
647
  auto_enhance = gr.Checkbox(
648
  label="Auto-Enhance on Generate",
649
  value=False,
650
- info="Automatically expands prompts using PE-T2I / PE-I2I before generation.",
 
 
 
 
 
651
  )
652
 
653
  # Reference Images Section (Up to 10)
@@ -769,7 +782,8 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
769
  ### 📌 Key Capabilities:
770
  1. **Prompt Enhancement (PE-T2I & PE-I2I)**:
771
  - Click **✨ Enhance Prompt** or enable **Auto-Enhance on Generate** to automatically expand short prompts into rich, vivid descriptions with optimal scene details, lighting, and textures.
772
- - For image editing, PE-I2I visually analyzes the reference image and instruction together.
 
773
  2. **Text-to-Image (T2I)**:
774
  - Generates ultra-high quality 2K images with realistic lighting, textures, and accurate text rendering.
775
  3. **Multi-Image Reference (Up to 10 Images)**:
@@ -796,6 +810,7 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
796
  42,
797
  False,
798
  False,
 
799
  ],
800
  [
801
  "A cute 3D cartoon baby dragon sticker, vivid colors, smooth gradient shading, trending on ArtStation.",
@@ -807,6 +822,7 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
807
  1234,
808
  True,
809
  False,
 
810
  ],
811
  [
812
  "A gourmet cheeseburger on a rustic wooden board, melting cheddar cheese, fresh lettuce, sesame bun, dramatic studio food photography, shallow depth of field.",
@@ -818,6 +834,7 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
818
  2026,
819
  False,
820
  False,
 
821
  ],
822
  [
823
  "A breathtaking majestic waterfall in a fantasy bioluminescent jungle, towering ancient glowing trees, magical mist, hyper-detailed, masterpiece.",
@@ -829,6 +846,7 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
829
  8888,
830
  False,
831
  False,
 
832
  ],
833
  ],
834
  inputs=[
@@ -841,13 +859,14 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
841
  seed,
842
  is_transparent,
843
  auto_enhance,
 
844
  ],
845
  )
846
 
847
  # Event handlers
848
  enhance_btn.click(
849
  fn=enhance_prompt_action,
850
- inputs=[prompt, ref_files],
851
  outputs=[prompt],
852
  )
853
 
@@ -866,6 +885,7 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
866
  randomize_seed,
867
  is_transparent,
868
  auto_enhance,
 
869
  ],
870
  outputs=[result_image, seed, info_text, prompt],
871
  api_name="generate",
 
318
  return lines[-1] if lines else fallback
319
 
320
 
321
+ def _enhance_prompt_t2i_core(prompt: str, enable_thinking: bool = False) -> str:
322
  """Execute T2I prompt expansion using Qwen-Image-2.1-PE-T2I."""
323
  if not prompt or not prompt.strip():
324
  return prompt
 
337
  ],
338
  tokenize=False,
339
  add_generation_prompt=True,
340
+ enable_thinking=enable_thinking,
341
  )
342
  inputs = tokenizer(text, return_tensors="pt").to(device)
343
+ max_tokens = 4096 if enable_thinking else 1024
344
+ temperature = 1.0 if enable_thinking else 0.7
345
+ top_p = 0.95 if enable_thinking else 0.9
346
  with torch.no_grad():
347
  out = model.generate(
348
  **inputs,
349
+ max_new_tokens=max_tokens,
350
  do_sample=True,
351
+ temperature=temperature,
352
+ top_p=top_p,
353
  top_k=20,
354
  )
355
  gen = tokenizer.decode(out[0, inputs["input_ids"].shape[1]:], skip_special_tokens=True)
 
365
  return prompt.strip()
366
 
367
 
368
+ def _enhance_prompt_i2i_core(image: Image.Image, instruction: str, enable_thinking: bool = False) -> str:
369
  """Execute I2I prompt expansion using Qwen-Image-2.1-PE-I2I."""
370
  if not instruction or not instruction.strip():
371
  instruction = "Describe and enhance this image."
 
378
  model.to(device)
379
  try:
380
  pe_img = image.copy()
381
+ if max(pe_img.size) > 768:
382
+ pe_img.thumbnail((768, 768), Image.Resampling.LANCZOS)
383
  if pe_img.mode != "RGB":
384
  pe_img = pe_img.convert("RGB")
385
 
 
399
  tokenize=True,
400
  return_dict=True,
401
  return_tensors="pt",
402
+ enable_thinking=enable_thinking,
403
  ).to(device)
404
+ max_tokens = 4096 if enable_thinking else 1024
405
+ temperature = 1.0 if enable_thinking else 0.7
406
+ top_p = 0.95 if enable_thinking else 0.9
407
  with torch.no_grad():
408
  out = model.generate(
409
  **inputs,
410
+ max_new_tokens=max_tokens,
411
  do_sample=True,
412
+ temperature=temperature,
413
+ top_p=top_p,
414
  top_k=20,
415
  )
416
  gen = processor.tokenizer.decode(
 
429
 
430
 
431
  @spaces.GPU(size="xlarge", duration=300)
432
+ def enhance_prompt_action(prompt: str, ref_files: list, deep_thinking: bool) -> str:
433
  """Standalone prompt enhancement triggered by the UI button."""
434
  if not prompt or not prompt.strip():
435
  gr.Warning("Please enter a prompt to enhance.")
 
440
  path = first_ref.name if hasattr(first_ref, "name") else first_ref
441
  try:
442
  img = Image.open(path)
443
+ return _enhance_prompt_i2i_core(img, prompt, enable_thinking=deep_thinking)
444
  except Exception as e:
445
  print(f"Failed to open reference image for PE: {e}")
446
+ return _enhance_prompt_t2i_core(prompt, enable_thinking=deep_thinking)
447
  else:
448
+ return _enhance_prompt_t2i_core(prompt, enable_thinking=deep_thinking)
449
 
450
 
451
  # ==========================================
 
489
  randomize_seed: bool,
490
  is_transparent: bool,
491
  auto_enhance: bool,
492
+ deep_thinking: bool,
493
  progress=gr.Progress(track_tqdm=True),
494
  ):
495
  if not prompt or not prompt.strip():
 
526
  # Optional Auto-Enhance
527
  effective_prompt = prompt.strip()
528
  if auto_enhance:
529
+ print(f"Auto-enhancing prompt (deep_thinking={deep_thinking})...")
530
  if input_images:
531
+ effective_prompt = _enhance_prompt_i2i_core(input_images[0], effective_prompt, enable_thinking=deep_thinking)
532
  else:
533
+ effective_prompt = _enhance_prompt_t2i_core(effective_prompt, enable_thinking=deep_thinking)
534
 
535
  # Process prompt for transparency
536
  final_prompt = effective_prompt
 
586
  f"**Ref Images**: {len(input_images)}"
587
  )
588
  if auto_enhance:
589
+ mode_label = "Deep Thinking" if deep_thinking else "Fast"
590
+ info_text += f" | **Prompt Enhanced**: ✨ Yes ({mode_label})"
591
 
592
  return result_image, seed, info_text, effective_prompt
593
 
 
655
  auto_enhance = gr.Checkbox(
656
  label="Auto-Enhance on Generate",
657
  value=False,
658
+ info="Automatically expands prompts before generation.",
659
+ )
660
+ deep_thinking = gr.Checkbox(
661
+ label="🧠 Deep Thinking (~60s)",
662
+ value=False,
663
+ info="Default off: Fast mode (~8s). On: Full 8-step CoT (~60s).",
664
  )
665
 
666
  # Reference Images Section (Up to 10)
 
782
  ### 📌 Key Capabilities:
783
  1. **Prompt Enhancement (PE-T2I & PE-I2I)**:
784
  - Click **✨ Enhance Prompt** or enable **Auto-Enhance on Generate** to automatically expand short prompts into rich, vivid descriptions with optimal scene details, lighting, and textures.
785
+ - **Fast Mode (Default, ~8s)**: Instantly rewrites the prompt directly.
786
+ - **Deep Thinking (~60s)**: Enables full 8-step Chain-of-Thought reasoning before outputting the rewritten prompt.
787
  2. **Text-to-Image (T2I)**:
788
  - Generates ultra-high quality 2K images with realistic lighting, textures, and accurate text rendering.
789
  3. **Multi-Image Reference (Up to 10 Images)**:
 
810
  42,
811
  False,
812
  False,
813
+ False,
814
  ],
815
  [
816
  "A cute 3D cartoon baby dragon sticker, vivid colors, smooth gradient shading, trending on ArtStation.",
 
822
  1234,
823
  True,
824
  False,
825
+ False,
826
  ],
827
  [
828
  "A gourmet cheeseburger on a rustic wooden board, melting cheddar cheese, fresh lettuce, sesame bun, dramatic studio food photography, shallow depth of field.",
 
834
  2026,
835
  False,
836
  False,
837
+ False,
838
  ],
839
  [
840
  "A breathtaking majestic waterfall in a fantasy bioluminescent jungle, towering ancient glowing trees, magical mist, hyper-detailed, masterpiece.",
 
846
  8888,
847
  False,
848
  False,
849
+ False,
850
  ],
851
  ],
852
  inputs=[
 
859
  seed,
860
  is_transparent,
861
  auto_enhance,
862
+ deep_thinking,
863
  ],
864
  )
865
 
866
  # Event handlers
867
  enhance_btn.click(
868
  fn=enhance_prompt_action,
869
+ inputs=[prompt, ref_files, deep_thinking],
870
  outputs=[prompt],
871
  )
872
 
 
885
  randomize_seed,
886
  is_transparent,
887
  auto_enhance,
888
+ deep_thinking,
889
  ],
890
  outputs=[result_image, seed, info_text, prompt],
891
  api_name="generate",