Spaces:
Running on Zero
Running on Zero
Upload 4 files
Browse files- __pycache__/app.cpython-314.pyc +0 -0
- app.py +43 -23
__pycache__/app.cpython-314.pyc
CHANGED
|
Binary files a/__pycache__/app.cpython-314.pyc and b/__pycache__/app.cpython-314.pyc differ
|
|
|
app.py
CHANGED
|
@@ -318,7 +318,7 @@ def _parse_rewrite(gen: str, fallback: str) -> str:
|
|
| 318 |
return lines[-1] if lines else fallback
|
| 319 |
|
| 320 |
|
| 321 |
-
def _enhance_prompt_t2i_core(prompt: str) -> str:
|
| 322 |
"""Execute T2I prompt expansion using Qwen-Image-2.1-PE-T2I."""
|
| 323 |
if not prompt or not prompt.strip():
|
| 324 |
return prompt
|
|
@@ -337,16 +337,19 @@ def _enhance_prompt_t2i_core(prompt: str) -> str:
|
|
| 337 |
],
|
| 338 |
tokenize=False,
|
| 339 |
add_generation_prompt=True,
|
| 340 |
-
enable_thinking=
|
| 341 |
)
|
| 342 |
inputs = tokenizer(text, return_tensors="pt").to(device)
|
|
|
|
|
|
|
|
|
|
| 343 |
with torch.no_grad():
|
| 344 |
out = model.generate(
|
| 345 |
**inputs,
|
| 346 |
-
max_new_tokens=
|
| 347 |
do_sample=True,
|
| 348 |
-
temperature=
|
| 349 |
-
top_p=
|
| 350 |
top_k=20,
|
| 351 |
)
|
| 352 |
gen = tokenizer.decode(out[0, inputs["input_ids"].shape[1]:], skip_special_tokens=True)
|
|
@@ -362,7 +365,7 @@ def _enhance_prompt_t2i_core(prompt: str) -> str:
|
|
| 362 |
return prompt.strip()
|
| 363 |
|
| 364 |
|
| 365 |
-
def _enhance_prompt_i2i_core(image: Image.Image, instruction: str) -> str:
|
| 366 |
"""Execute I2I prompt expansion using Qwen-Image-2.1-PE-I2I."""
|
| 367 |
if not instruction or not instruction.strip():
|
| 368 |
instruction = "Describe and enhance this image."
|
|
@@ -375,8 +378,8 @@ def _enhance_prompt_i2i_core(image: Image.Image, instruction: str) -> str:
|
|
| 375 |
model.to(device)
|
| 376 |
try:
|
| 377 |
pe_img = image.copy()
|
| 378 |
-
if max(pe_img.size) >
|
| 379 |
-
pe_img.thumbnail((
|
| 380 |
if pe_img.mode != "RGB":
|
| 381 |
pe_img = pe_img.convert("RGB")
|
| 382 |
|
|
@@ -396,15 +399,18 @@ def _enhance_prompt_i2i_core(image: Image.Image, instruction: str) -> str:
|
|
| 396 |
tokenize=True,
|
| 397 |
return_dict=True,
|
| 398 |
return_tensors="pt",
|
| 399 |
-
enable_thinking=
|
| 400 |
).to(device)
|
|
|
|
|
|
|
|
|
|
| 401 |
with torch.no_grad():
|
| 402 |
out = model.generate(
|
| 403 |
**inputs,
|
| 404 |
-
max_new_tokens=
|
| 405 |
do_sample=True,
|
| 406 |
-
temperature=
|
| 407 |
-
top_p=
|
| 408 |
top_k=20,
|
| 409 |
)
|
| 410 |
gen = processor.tokenizer.decode(
|
|
@@ -423,7 +429,7 @@ def _enhance_prompt_i2i_core(image: Image.Image, instruction: str) -> str:
|
|
| 423 |
|
| 424 |
|
| 425 |
@spaces.GPU(size="xlarge", duration=300)
|
| 426 |
-
def enhance_prompt_action(prompt: str, ref_files: list) -> str:
|
| 427 |
"""Standalone prompt enhancement triggered by the UI button."""
|
| 428 |
if not prompt or not prompt.strip():
|
| 429 |
gr.Warning("Please enter a prompt to enhance.")
|
|
@@ -434,12 +440,12 @@ def enhance_prompt_action(prompt: str, ref_files: list) -> str:
|
|
| 434 |
path = first_ref.name if hasattr(first_ref, "name") else first_ref
|
| 435 |
try:
|
| 436 |
img = Image.open(path)
|
| 437 |
-
return _enhance_prompt_i2i_core(img, prompt)
|
| 438 |
except Exception as e:
|
| 439 |
print(f"Failed to open reference image for PE: {e}")
|
| 440 |
-
return _enhance_prompt_t2i_core(prompt)
|
| 441 |
else:
|
| 442 |
-
return _enhance_prompt_t2i_core(prompt)
|
| 443 |
|
| 444 |
|
| 445 |
# ==========================================
|
|
@@ -483,6 +489,7 @@ def generate(
|
|
| 483 |
randomize_seed: bool,
|
| 484 |
is_transparent: bool,
|
| 485 |
auto_enhance: bool,
|
|
|
|
| 486 |
progress=gr.Progress(track_tqdm=True),
|
| 487 |
):
|
| 488 |
if not prompt or not prompt.strip():
|
|
@@ -519,11 +526,11 @@ def generate(
|
|
| 519 |
# Optional Auto-Enhance
|
| 520 |
effective_prompt = prompt.strip()
|
| 521 |
if auto_enhance:
|
| 522 |
-
print("Auto-enhancing prompt...")
|
| 523 |
if input_images:
|
| 524 |
-
effective_prompt = _enhance_prompt_i2i_core(input_images[0], effective_prompt)
|
| 525 |
else:
|
| 526 |
-
effective_prompt = _enhance_prompt_t2i_core(effective_prompt)
|
| 527 |
|
| 528 |
# Process prompt for transparency
|
| 529 |
final_prompt = effective_prompt
|
|
@@ -579,7 +586,8 @@ def generate(
|
|
| 579 |
f"**Ref Images**: {len(input_images)}"
|
| 580 |
)
|
| 581 |
if auto_enhance:
|
| 582 |
-
|
|
|
|
| 583 |
|
| 584 |
return result_image, seed, info_text, effective_prompt
|
| 585 |
|
|
@@ -647,7 +655,12 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
|
|
| 647 |
auto_enhance = gr.Checkbox(
|
| 648 |
label="Auto-Enhance on Generate",
|
| 649 |
value=False,
|
| 650 |
-
info="Automatically expands prompts
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 651 |
)
|
| 652 |
|
| 653 |
# Reference Images Section (Up to 10)
|
|
@@ -769,7 +782,8 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
|
|
| 769 |
### 📌 Key Capabilities:
|
| 770 |
1. **Prompt Enhancement (PE-T2I & PE-I2I)**:
|
| 771 |
- Click **✨ Enhance Prompt** or enable **Auto-Enhance on Generate** to automatically expand short prompts into rich, vivid descriptions with optimal scene details, lighting, and textures.
|
| 772 |
-
-
|
|
|
|
| 773 |
2. **Text-to-Image (T2I)**:
|
| 774 |
- Generates ultra-high quality 2K images with realistic lighting, textures, and accurate text rendering.
|
| 775 |
3. **Multi-Image Reference (Up to 10 Images)**:
|
|
@@ -796,6 +810,7 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
|
|
| 796 |
42,
|
| 797 |
False,
|
| 798 |
False,
|
|
|
|
| 799 |
],
|
| 800 |
[
|
| 801 |
"A cute 3D cartoon baby dragon sticker, vivid colors, smooth gradient shading, trending on ArtStation.",
|
|
@@ -807,6 +822,7 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
|
|
| 807 |
1234,
|
| 808 |
True,
|
| 809 |
False,
|
|
|
|
| 810 |
],
|
| 811 |
[
|
| 812 |
"A gourmet cheeseburger on a rustic wooden board, melting cheddar cheese, fresh lettuce, sesame bun, dramatic studio food photography, shallow depth of field.",
|
|
@@ -818,6 +834,7 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
|
|
| 818 |
2026,
|
| 819 |
False,
|
| 820 |
False,
|
|
|
|
| 821 |
],
|
| 822 |
[
|
| 823 |
"A breathtaking majestic waterfall in a fantasy bioluminescent jungle, towering ancient glowing trees, magical mist, hyper-detailed, masterpiece.",
|
|
@@ -829,6 +846,7 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
|
|
| 829 |
8888,
|
| 830 |
False,
|
| 831 |
False,
|
|
|
|
| 832 |
],
|
| 833 |
],
|
| 834 |
inputs=[
|
|
@@ -841,13 +859,14 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
|
|
| 841 |
seed,
|
| 842 |
is_transparent,
|
| 843 |
auto_enhance,
|
|
|
|
| 844 |
],
|
| 845 |
)
|
| 846 |
|
| 847 |
# Event handlers
|
| 848 |
enhance_btn.click(
|
| 849 |
fn=enhance_prompt_action,
|
| 850 |
-
inputs=[prompt, ref_files],
|
| 851 |
outputs=[prompt],
|
| 852 |
)
|
| 853 |
|
|
@@ -866,6 +885,7 @@ with gr.Blocks(title="Qwen-Image-2.1 ZeroGPU Space", css=custom_css, theme=gr.th
|
|
| 866 |
randomize_seed,
|
| 867 |
is_transparent,
|
| 868 |
auto_enhance,
|
|
|
|
| 869 |
],
|
| 870 |
outputs=[result_image, seed, info_text, prompt],
|
| 871 |
api_name="generate",
|
|
|
|
| 318 |
return lines[-1] if lines else fallback
|
| 319 |
|
| 320 |
|
| 321 |
+
def _enhance_prompt_t2i_core(prompt: str, enable_thinking: bool = False) -> str:
|
| 322 |
"""Execute T2I prompt expansion using Qwen-Image-2.1-PE-T2I."""
|
| 323 |
if not prompt or not prompt.strip():
|
| 324 |
return prompt
|
|
|
|
| 337 |
],
|
| 338 |
tokenize=False,
|
| 339 |
add_generation_prompt=True,
|
| 340 |
+
enable_thinking=enable_thinking,
|
| 341 |
)
|
| 342 |
inputs = tokenizer(text, return_tensors="pt").to(device)
|
| 343 |
+
max_tokens = 4096 if enable_thinking else 1024
|
| 344 |
+
temperature = 1.0 if enable_thinking else 0.7
|
| 345 |
+
top_p = 0.95 if enable_thinking else 0.9
|
| 346 |
with torch.no_grad():
|
| 347 |
out = model.generate(
|
| 348 |
**inputs,
|
| 349 |
+
max_new_tokens=max_tokens,
|
| 350 |
do_sample=True,
|
| 351 |
+
temperature=temperature,
|
| 352 |
+
top_p=top_p,
|
| 353 |
top_k=20,
|
| 354 |
)
|
| 355 |
gen = tokenizer.decode(out[0, inputs["input_ids"].shape[1]:], skip_special_tokens=True)
|
|
|
|
| 365 |
return prompt.strip()
|
| 366 |
|
| 367 |
|
| 368 |
+
def _enhance_prompt_i2i_core(image: Image.Image, instruction: str, enable_thinking: bool = False) -> str:
|
| 369 |
"""Execute I2I prompt expansion using Qwen-Image-2.1-PE-I2I."""
|
| 370 |
if not instruction or not instruction.strip():
|
| 371 |
instruction = "Describe and enhance this image."
|
|
|
|
| 378 |
model.to(device)
|
| 379 |
try:
|
| 380 |
pe_img = image.copy()
|
| 381 |
+
if max(pe_img.size) > 768:
|
| 382 |
+
pe_img.thumbnail((768, 768), Image.Resampling.LANCZOS)
|
| 383 |
if pe_img.mode != "RGB":
|
| 384 |
pe_img = pe_img.convert("RGB")
|
| 385 |
|
|
|
|
| 399 |
tokenize=True,
|
| 400 |
return_dict=True,
|
| 401 |
return_tensors="pt",
|
| 402 |
+
enable_thinking=enable_thinking,
|
| 403 |
).to(device)
|
| 404 |
+
max_tokens = 4096 if enable_thinking else 1024
|
| 405 |
+
temperature = 1.0 if enable_thinking else 0.7
|
| 406 |
+
top_p = 0.95 if enable_thinking else 0.9
|
| 407 |
with torch.no_grad():
|
| 408 |
out = model.generate(
|
| 409 |
**inputs,
|
| 410 |
+
max_new_tokens=max_tokens,
|
| 411 |
do_sample=True,
|
| 412 |
+
temperature=temperature,
|
| 413 |
+
top_p=top_p,
|
| 414 |
top_k=20,
|
| 415 |
)
|
| 416 |
gen = processor.tokenizer.decode(
|
|
|
|
| 429 |
|
| 430 |
|
| 431 |
@spaces.GPU(size="xlarge", duration=300)
|
| 432 |
+
def enhance_prompt_action(prompt: str, ref_files: list, deep_thinking: bool) -> str:
|
| 433 |
"""Standalone prompt enhancement triggered by the UI button."""
|
| 434 |
if not prompt or not prompt.strip():
|
| 435 |
gr.Warning("Please enter a prompt to enhance.")
|
|
|
|
| 440 |
path = first_ref.name if hasattr(first_ref, "name") else first_ref
|
| 441 |
try:
|
| 442 |
img = Image.open(path)
|
| 443 |
+
return _enhance_prompt_i2i_core(img, prompt, enable_thinking=deep_thinking)
|
| 444 |
except Exception as e:
|
| 445 |
print(f"Failed to open reference image for PE: {e}")
|
| 446 |
+
return _enhance_prompt_t2i_core(prompt, enable_thinking=deep_thinking)
|
| 447 |
else:
|
| 448 |
+
return _enhance_prompt_t2i_core(prompt, enable_thinking=deep_thinking)
|
| 449 |
|
| 450 |
|
| 451 |
# ==========================================
|
|
|
|
| 489 |
randomize_seed: bool,
|
| 490 |
is_transparent: bool,
|
| 491 |
auto_enhance: bool,
|
| 492 |
+
deep_thinking: bool,
|
| 493 |
progress=gr.Progress(track_tqdm=True),
|
| 494 |
):
|
| 495 |
if not prompt or not prompt.strip():
|
|
|
|
| 526 |
# Optional Auto-Enhance
|
| 527 |
effective_prompt = prompt.strip()
|
| 528 |
if auto_enhance:
|
| 529 |
+
print(f"Auto-enhancing prompt (deep_thinking={deep_thinking})...")
|
| 530 |
if input_images:
|
| 531 |
+
effective_prompt = _enhance_prompt_i2i_core(input_images[0], effective_prompt, enable_thinking=deep_thinking)
|
| 532 |
else:
|
| 533 |
+
effective_prompt = _enhance_prompt_t2i_core(effective_prompt, enable_thinking=deep_thinking)
|
| 534 |
|
| 535 |
# Process prompt for transparency
|
| 536 |
final_prompt = effective_prompt
|
|
|
|
| 586 |
f"**Ref Images**: {len(input_images)}"
|
| 587 |
)
|
| 588 |
if auto_enhance:
|
| 589 |
+
mode_label = "Deep Thinking" if deep_thinking else "Fast"
|
| 590 |
+
info_text += f" | **Prompt Enhanced**: ✨ Yes ({mode_label})"
|
| 591 |
|
| 592 |
return result_image, seed, info_text, effective_prompt
|
| 593 |
|
|
|
|
| 655 |
auto_enhance = gr.Checkbox(
|
| 656 |
label="Auto-Enhance on Generate",
|
| 657 |
value=False,
|
| 658 |
+
info="Automatically expands prompts before generation.",
|
| 659 |
+
)
|
| 660 |
+
deep_thinking = gr.Checkbox(
|
| 661 |
+
label="🧠 Deep Thinking (~60s)",
|
| 662 |
+
value=False,
|
| 663 |
+
info="Default off: Fast mode (~8s). On: Full 8-step CoT (~60s).",
|
| 664 |
)
|
| 665 |
|
| 666 |
# Reference Images Section (Up to 10)
|
|
|
|
| 782 |
### 📌 Key Capabilities:
|
| 783 |
1. **Prompt Enhancement (PE-T2I & PE-I2I)**:
|
| 784 |
- Click **✨ Enhance Prompt** or enable **Auto-Enhance on Generate** to automatically expand short prompts into rich, vivid descriptions with optimal scene details, lighting, and textures.
|
| 785 |
+
- **Fast Mode (Default, ~8s)**: Instantly rewrites the prompt directly.
|
| 786 |
+
- **Deep Thinking (~60s)**: Enables full 8-step Chain-of-Thought reasoning before outputting the rewritten prompt.
|
| 787 |
2. **Text-to-Image (T2I)**:
|
| 788 |
- Generates ultra-high quality 2K images with realistic lighting, textures, and accurate text rendering.
|
| 789 |
3. **Multi-Image Reference (Up to 10 Images)**:
|
|
|
|
| 810 |
42,
|
| 811 |
False,
|
| 812 |
False,
|
| 813 |
+
False,
|
| 814 |
],
|
| 815 |
[
|
| 816 |
"A cute 3D cartoon baby dragon sticker, vivid colors, smooth gradient shading, trending on ArtStation.",
|
|
|
|
| 822 |
1234,
|
| 823 |
True,
|
| 824 |
False,
|
| 825 |
+
False,
|
| 826 |
],
|
| 827 |
[
|
| 828 |
"A gourmet cheeseburger on a rustic wooden board, melting cheddar cheese, fresh lettuce, sesame bun, dramatic studio food photography, shallow depth of field.",
|
|
|
|
| 834 |
2026,
|
| 835 |
False,
|
| 836 |
False,
|
| 837 |
+
False,
|
| 838 |
],
|
| 839 |
[
|
| 840 |
"A breathtaking majestic waterfall in a fantasy bioluminescent jungle, towering ancient glowing trees, magical mist, hyper-detailed, masterpiece.",
|
|
|
|
| 846 |
8888,
|
| 847 |
False,
|
| 848 |
False,
|
| 849 |
+
False,
|
| 850 |
],
|
| 851 |
],
|
| 852 |
inputs=[
|
|
|
|
| 859 |
seed,
|
| 860 |
is_transparent,
|
| 861 |
auto_enhance,
|
| 862 |
+
deep_thinking,
|
| 863 |
],
|
| 864 |
)
|
| 865 |
|
| 866 |
# Event handlers
|
| 867 |
enhance_btn.click(
|
| 868 |
fn=enhance_prompt_action,
|
| 869 |
+
inputs=[prompt, ref_files, deep_thinking],
|
| 870 |
outputs=[prompt],
|
| 871 |
)
|
| 872 |
|
|
|
|
| 885 |
randomize_seed,
|
| 886 |
is_transparent,
|
| 887 |
auto_enhance,
|
| 888 |
+
deep_thinking,
|
| 889 |
],
|
| 890 |
outputs=[result_image, seed, info_text, prompt],
|
| 891 |
api_name="generate",
|