Spaces:
Running on Zero
Running on Zero
Advanced settings: pick v0.2.1 instead of v0.3 (both LoRAs loaded, switched per call)
Browse files
app.py
CHANGED
|
@@ -1,4 +1,5 @@
|
|
| 1 |
-
"""Gradio demo for the distilled Qwen-Image-2.1 turbo (v0.3: LoRA r256 sampled in 6 steps; 9 steps = 7 turbo + 2 base-model steps
|
|
|
|
| 2 |
|
| 3 |
# ZeroGPU patches torch at import time, so `spaces` must be imported before torch or any CUDA use.
|
| 4 |
# find_spec keeps this file runnable off-Spaces, where the package is absent.
|
|
@@ -38,13 +39,17 @@ from diffusers import FlowMatchEulerDiscreteScheduler, QwenImage21Pipeline
|
|
| 38 |
from diffusers.pipelines.qwenimage21.pipeline_qwenimage21 import calculate_dimensions
|
| 39 |
from huggingface_hub import CommitScheduler
|
| 40 |
|
| 41 |
-
# The
|
| 42 |
# (merging into bf16 keeps only ~47% of the adapter delta, the per-element deltas sit below the bf16 ULP of the base
|
| 43 |
-
# weights).
|
|
|
|
| 44 |
BASE_MODEL_ID = os.environ.get("BASE_MODEL_ID", "Qwen/Qwen-Image-2.1")
|
| 45 |
STUDENT_REPO = os.environ.get("STUDENT_REPO", os.environ.get("LORA_REPO", "Viggle/Qwen-Image-2.1-viggle-turbo"))
|
| 46 |
-
|
| 47 |
-
|
|
|
|
|
|
|
|
|
|
| 48 |
HF_TOKEN = os.environ.get("HF_TOKEN") # only needed while the model repo is private
|
| 49 |
# Usage counter. The Space autoscales to several replicas and its log view shows only one of them, so the [gen] log lines
|
| 50 |
# undercount. Every generate() call that reaches the GPU also appends one JSON line (metadata only: no prompt, no images)
|
|
@@ -134,7 +139,8 @@ CUSTOM_CAP = {False: 2048, True: 1536} # keyed by "has references"
|
|
| 134 |
MAX_SIDE = max(max(size) for size in SIZES.values() if size)
|
| 135 |
|
| 136 |
pipe = QwenImage21Pipeline.from_pretrained(BASE_MODEL_ID, dtype=torch.bfloat16, token=HF_TOKEN)
|
| 137 |
-
|
|
|
|
| 138 |
# The stock Qwen-Image-2.1 scheduler config carries shift_terminal=0.02, which stretches the last
|
| 139 |
# sigma node to 0.02 instead of 0 and silently costs the final step. The student was distilled
|
| 140 |
# against the unstretched schedule, so rebuild the scheduler with shift_terminal=None.
|
|
@@ -233,11 +239,12 @@ def base_tail(pipe, i, t, kwargs):
|
|
| 233 |
|
| 234 |
|
| 235 |
@gpu
|
| 236 |
-
def denoise(prompt, images, width, height, steps, seed):
|
| 237 |
# No VAE tiling anywhere (xlarge card): tiling the reference encode wrecks edits, and tiled decodes are not the
|
| 238 |
# validated pipeline either.
|
| 239 |
generator = torch.Generator(device="cuda").manual_seed(seed)
|
| 240 |
started = time.perf_counter()
|
|
|
|
| 241 |
pipe.enable_lora() # every call starts on the student, even if an earlier 9-step call died with the LoRA off
|
| 242 |
reextract[0] = False
|
| 243 |
result = pipe(
|
|
@@ -256,8 +263,10 @@ def denoise(prompt, images, width, height, steps, seed):
|
|
| 256 |
return result, time.perf_counter() - started
|
| 257 |
|
| 258 |
|
| 259 |
-
def generate(prompt, refs, size_label, seed, randomize_seed, enhance="Auto", steps=STEPS, custom_width=1024, custom_height=1024
|
|
|
|
| 260 |
t0 = time.perf_counter()
|
|
|
|
| 261 |
steps = min(max(int(steps), 3), 9) # 6 (RAW_NODES) is the default; 9 = 7 student steps + 2 base-model steps; fewer than 6 degrade quickly
|
| 262 |
# refs: the gallery's (filepath, caption) pairs in upload order, which is the order the prompt's "image 1, 2, ..." refer
|
| 263 |
# to. EXIF rotation + RGB is what the gr.Image slots used to do.
|
|
@@ -294,18 +303,18 @@ def generate(prompt, refs, size_label, seed, randomize_seed, enhance="Auto", ste
|
|
| 294 |
enhance_note = f" · enhanced {pe_time:.1f}s" if used_prompt != prompt else " · enhancement failed, prompt sent as written"
|
| 295 |
mode = "edit" if images else "t2i"
|
| 296 |
try:
|
| 297 |
-
result, elapsed = denoise(used_prompt, images, width, height, steps, seed)
|
| 298 |
except Exception as e: # ZeroGPU quota / GPU timeout errors land here: count them, then show them as before
|
| 299 |
-
record(ok=False, mode=mode, refs=len(images), steps=steps, error=f"{type(e).__name__}: {e}"[:200],
|
| 300 |
total=round(time.perf_counter() - t0, 2))
|
| 301 |
raise
|
| 302 |
# one line per generation, for counting usage from the Space logs (no prompt text)
|
| 303 |
print(f"[gen] {time.strftime('%Y-%m-%dT%H:%M:%SZ', time.gmtime())} mode={mode} refs={len(images)} "
|
| 304 |
-
f"steps={steps} size={result.width}x{result.height} enhance={int(used_prompt != prompt)} pe={pe_time:.1f}s gen={elapsed:.2f}s",
|
| 305 |
flush=True)
|
| 306 |
-
record(ok=True, mode=mode, refs=len(images), steps=steps, size=f"{result.width}x{result.height}", enhance=enhance,
|
| 307 |
enhanced=used_prompt != prompt, pe=round(pe_time, 2), gen=round(elapsed, 2), total=round(time.perf_counter() - t0, 2))
|
| 308 |
-
return result, f"seed `{seed}` · {result.width}×{result.height} · {steps} steps · {elapsed:.2f}s · {
|
| 309 |
|
| 310 |
|
| 311 |
def refresh_sizes(prompt, refs, current):
|
|
@@ -397,8 +406,8 @@ with gr.Blocks(title="Viggle Turbo v0.3 · Qwen-Image-2.1 6-step", theme=THEME,
|
|
| 397 |
"**v0.3 (2026-09-29):** at 6 steps, less grain than v0.2.1 and a little softer on fine texture. We think 6 steps "
|
| 398 |
"is close to its capacity: every further gain we found cost something elsewhere. **9 steps** runs 7 turbo "
|
| 399 |
"steps and lets the base model finish the last two: finer detail and small text right more often, at about 1.4–1.5× the "
|
| 400 |
-
"time of 6 steps. "
|
| 401 |
-
f"Weights: **
|
| 402 |
)
|
| 403 |
with gr.Tabs() as tabs:
|
| 404 |
with gr.Tab("Generate", id="generate"):
|
|
@@ -432,6 +441,9 @@ with gr.Blocks(title="Viggle Turbo v0.3 · Qwen-Image-2.1 6-step", theme=THEME,
|
|
| 432 |
with gr.Accordion("Advanced settings", open=False) as advanced:
|
| 433 |
steps = gr.Slider(label="Steps", minimum=3, maximum=9, step=1, value=STEPS,
|
| 434 |
info="6 is the default; 7-8 add steps at the high-noise end; 9 = 7 turbo + 2 base-model steps (finer detail; ~1.4–1.5× the time).")
|
|
|
|
|
|
|
|
|
|
| 435 |
with gr.Row():
|
| 436 |
seed = gr.Number(label="Seed", value=0, precision=0, interactive=False)
|
| 437 |
randomize_seed = gr.Checkbox(label="Randomize seed", value=True)
|
|
@@ -446,12 +458,13 @@ with gr.Blocks(title="Viggle Turbo v0.3 · Qwen-Image-2.1 6-step", theme=THEME,
|
|
| 446 |
def example_details(prompt, refs, result):
|
| 447 |
"""Runs on an example click (no GPU): fills the status line and the prompt stored with the row, and sets the
|
| 448 |
steps / size / enhancement / seed controls to what produced it, so Generate reproduces the result. It opens
|
| 449 |
-
Advanced settings to show that the seed is now fixed
|
| 450 |
-
text-to-image row loaded after another one leaves the gallery unchanged, so
|
|
|
|
| 451 |
row = next(row for row in EXAMPLES if row["prompt"] == prompt)
|
| 452 |
choices = EDIT_CHOICES if any(row["refs"]) else T2I_CHOICES
|
| 453 |
return (row["info"], row["used_prompt"], row["steps"], gr.update(choices=choices, value=row["size_label"]),
|
| 454 |
-
"On" if row["enhance"] else "Off", row["seed"], False, gr.update(open=True))
|
| 455 |
|
| 456 |
if EXAMPLES:
|
| 457 |
gr.Examples(
|
|
@@ -461,7 +474,7 @@ with gr.Blocks(title="Viggle Turbo v0.3 · Qwen-Image-2.1 6-step", theme=THEME,
|
|
| 461 |
],
|
| 462 |
inputs=[prompt, refs, output_image],
|
| 463 |
fn=example_details,
|
| 464 |
-
outputs=[info, used_prompt, steps, size_label, enhance, seed, randomize_seed, advanced],
|
| 465 |
run_on_click=True,
|
| 466 |
# On Spaces gr.Examples caches by default (lazily on ZeroGPU), and gradio 5.50 crashes caching an output that
|
| 467 |
# updates a Dropdown's choices ('Dropdown' object has no attribute 'proxy_url'). example_details needs no GPU.
|
|
@@ -491,7 +504,7 @@ with gr.Blocks(title="Viggle Turbo v0.3 · Qwen-Image-2.1 6-step", theme=THEME,
|
|
| 491 |
randomize_seed.change(lambda on: gr.update(interactive=not on), inputs=randomize_seed, outputs=seed)
|
| 492 |
run.click(
|
| 493 |
generate,
|
| 494 |
-
inputs=[prompt, refs, size_label, seed, randomize_seed, enhance, steps, custom_width, custom_height],
|
| 495 |
outputs=[output_image, info, used_prompt],
|
| 496 |
)
|
| 497 |
demo.load(open_tab, outputs=tabs, show_progress="hidden")
|
|
|
|
| 1 |
+
"""Gradio demo for the distilled Qwen-Image-2.1 turbo (v0.3: LoRA r256 sampled in 6 steps; 9 steps = 7 turbo + 2 base-model steps;
|
| 2 |
+
v0.2.1 selectable under Advanced settings)."""
|
| 3 |
|
| 4 |
# ZeroGPU patches torch at import time, so `spaces` must be imported before torch or any CUDA use.
|
| 5 |
# find_spec keeps this file runnable off-Spaces, where the package is absent.
|
|
|
|
| 39 |
from diffusers.pipelines.qwenimage21.pipeline_qwenimage21 import calculate_dimensions
|
| 40 |
from huggingface_hub import CommitScheduler
|
| 41 |
|
| 42 |
+
# The LoRAs (r256) at the model repo root, applied at runtime on top of the base transformer and never merged
|
| 43 |
# (merging into bf16 keeps only ~47% of the adapter delta, the per-element deltas sit below the bf16 ULP of the base
|
| 44 |
+
# weights). v0.3 (the default) and v0.2.1 are both loaded as named adapters (PEFT adapter names cannot contain dots)
|
| 45 |
+
# and each call activates the one picked under Advanced settings; both are sampled on the same nodes (raw_nodes below).
|
| 46 |
BASE_MODEL_ID = os.environ.get("BASE_MODEL_ID", "Qwen/Qwen-Image-2.1")
|
| 47 |
STUDENT_REPO = os.environ.get("STUDENT_REPO", os.environ.get("LORA_REPO", "Viggle/Qwen-Image-2.1-viggle-turbo"))
|
| 48 |
+
LORAS = { # version -> (adapter name, file)
|
| 49 |
+
"v0.3": ("v03", "Qwen-Image-2.1-viggle-turbo-v0.3-6step-lora-r256.safetensors"),
|
| 50 |
+
"v0.2.1": ("v021", "Qwen-Image-2.1-viggle-turbo-v0.2.1-6step-lora-r256.safetensors"),
|
| 51 |
+
}
|
| 52 |
+
VERSION = "v0.3"
|
| 53 |
HF_TOKEN = os.environ.get("HF_TOKEN") # only needed while the model repo is private
|
| 54 |
# Usage counter. The Space autoscales to several replicas and its log view shows only one of them, so the [gen] log lines
|
| 55 |
# undercount. Every generate() call that reaches the GPU also appends one JSON line (metadata only: no prompt, no images)
|
|
|
|
| 139 |
MAX_SIDE = max(max(size) for size in SIZES.values() if size)
|
| 140 |
|
| 141 |
pipe = QwenImage21Pipeline.from_pretrained(BASE_MODEL_ID, dtype=torch.bfloat16, token=HF_TOKEN)
|
| 142 |
+
for adapter, lora_file in LORAS.values():
|
| 143 |
+
pipe.load_lora_weights(STUDENT_REPO, weight_name=lora_file, adapter_name=adapter, token=HF_TOKEN)
|
| 144 |
# The stock Qwen-Image-2.1 scheduler config carries shift_terminal=0.02, which stretches the last
|
| 145 |
# sigma node to 0.02 instead of 0 and silently costs the final step. The student was distilled
|
| 146 |
# against the unstretched schedule, so rebuild the scheduler with shift_terminal=None.
|
|
|
|
| 239 |
|
| 240 |
|
| 241 |
@gpu
|
| 242 |
+
def denoise(prompt, images, width, height, steps, seed, version=VERSION):
|
| 243 |
# No VAE tiling anywhere (xlarge card): tiling the reference encode wrecks edits, and tiled decodes are not the
|
| 244 |
# validated pipeline either.
|
| 245 |
generator = torch.Generator(device="cuda").manual_seed(seed)
|
| 246 |
started = time.perf_counter()
|
| 247 |
+
pipe.set_adapters(LORAS[version][0])
|
| 248 |
pipe.enable_lora() # every call starts on the student, even if an earlier 9-step call died with the LoRA off
|
| 249 |
reextract[0] = False
|
| 250 |
result = pipe(
|
|
|
|
| 263 |
return result, time.perf_counter() - started
|
| 264 |
|
| 265 |
|
| 266 |
+
def generate(prompt, refs, size_label, seed, randomize_seed, enhance="Auto", steps=STEPS, custom_width=1024, custom_height=1024,
|
| 267 |
+
version=VERSION):
|
| 268 |
t0 = time.perf_counter()
|
| 269 |
+
version = version if version in LORAS else VERSION
|
| 270 |
steps = min(max(int(steps), 3), 9) # 6 (RAW_NODES) is the default; 9 = 7 student steps + 2 base-model steps; fewer than 6 degrade quickly
|
| 271 |
# refs: the gallery's (filepath, caption) pairs in upload order, which is the order the prompt's "image 1, 2, ..." refer
|
| 272 |
# to. EXIF rotation + RGB is what the gr.Image slots used to do.
|
|
|
|
| 303 |
enhance_note = f" · enhanced {pe_time:.1f}s" if used_prompt != prompt else " · enhancement failed, prompt sent as written"
|
| 304 |
mode = "edit" if images else "t2i"
|
| 305 |
try:
|
| 306 |
+
result, elapsed = denoise(used_prompt, images, width, height, steps, seed, version)
|
| 307 |
except Exception as e: # ZeroGPU quota / GPU timeout errors land here: count them, then show them as before
|
| 308 |
+
record(ok=False, mode=mode, refs=len(images), steps=steps, version=version, error=f"{type(e).__name__}: {e}"[:200],
|
| 309 |
total=round(time.perf_counter() - t0, 2))
|
| 310 |
raise
|
| 311 |
# one line per generation, for counting usage from the Space logs (no prompt text)
|
| 312 |
print(f"[gen] {time.strftime('%Y-%m-%dT%H:%M:%SZ', time.gmtime())} mode={mode} refs={len(images)} "
|
| 313 |
+
f"steps={steps} version={version} size={result.width}x{result.height} enhance={int(used_prompt != prompt)} pe={pe_time:.1f}s gen={elapsed:.2f}s",
|
| 314 |
flush=True)
|
| 315 |
+
record(ok=True, mode=mode, refs=len(images), steps=steps, version=version, size=f"{result.width}x{result.height}", enhance=enhance,
|
| 316 |
enhanced=used_prompt != prompt, pe=round(pe_time, 2), gen=round(elapsed, 2), total=round(time.perf_counter() - t0, 2))
|
| 317 |
+
return result, f"seed `{seed}` · {result.width}×{result.height} · {steps} steps · {elapsed:.2f}s · {version} LoRA r256{enhance_note}{clamped}", used_prompt
|
| 318 |
|
| 319 |
|
| 320 |
def refresh_sizes(prompt, refs, current):
|
|
|
|
| 406 |
"**v0.3 (2026-09-29):** at 6 steps, less grain than v0.2.1 and a little softer on fine texture. We think 6 steps "
|
| 407 |
"is close to its capacity: every further gain we found cost something elsewhere. **9 steps** runs 7 turbo "
|
| 408 |
"steps and lets the base model finish the last two: finer detail and small text right more often, at about 1.4–1.5× the "
|
| 409 |
+
"time of 6 steps. v0.2.1 can still be picked under **Advanced settings**. "
|
| 410 |
+
f"Weights: **v0.3 LoRA r256** (and v0.2.1) from [{STUDENT_REPO}](https://huggingface.co/{STUDENT_REPO})."
|
| 411 |
)
|
| 412 |
with gr.Tabs() as tabs:
|
| 413 |
with gr.Tab("Generate", id="generate"):
|
|
|
|
| 441 |
with gr.Accordion("Advanced settings", open=False) as advanced:
|
| 442 |
steps = gr.Slider(label="Steps", minimum=3, maximum=9, step=1, value=STEPS,
|
| 443 |
info="6 is the default; 7-8 add steps at the high-noise end; 9 = 7 turbo + 2 base-model steps (finer detail; ~1.4–1.5× the time).")
|
| 444 |
+
version = gr.Radio(list(LORAS), value=VERSION, label="Model version",
|
| 445 |
+
info="v0.3 is the default. v0.2.1 is the previous release: a little sharper on fine "
|
| 446 |
+
"texture, with more grain.")
|
| 447 |
with gr.Row():
|
| 448 |
seed = gr.Number(label="Seed", value=0, precision=0, interactive=False)
|
| 449 |
randomize_seed = gr.Checkbox(label="Randomize seed", value=True)
|
|
|
|
| 458 |
def example_details(prompt, refs, result):
|
| 459 |
"""Runs on an example click (no GPU): fills the status line and the prompt stored with the row, and sets the
|
| 460 |
steps / size / enhancement / seed controls to what produced it, so Generate reproduces the result. It opens
|
| 461 |
+
Advanced settings to show that the seed is now fixed, and picks v0.3, which rendered every example. It also sets
|
| 462 |
+
the size menu's mode itself: a text-to-image row loaded after another one leaves the gallery unchanged, so
|
| 463 |
+
refresh_sizes does not run."""
|
| 464 |
row = next(row for row in EXAMPLES if row["prompt"] == prompt)
|
| 465 |
choices = EDIT_CHOICES if any(row["refs"]) else T2I_CHOICES
|
| 466 |
return (row["info"], row["used_prompt"], row["steps"], gr.update(choices=choices, value=row["size_label"]),
|
| 467 |
+
"On" if row["enhance"] else "Off", row["seed"], False, VERSION, gr.update(open=True))
|
| 468 |
|
| 469 |
if EXAMPLES:
|
| 470 |
gr.Examples(
|
|
|
|
| 474 |
],
|
| 475 |
inputs=[prompt, refs, output_image],
|
| 476 |
fn=example_details,
|
| 477 |
+
outputs=[info, used_prompt, steps, size_label, enhance, seed, randomize_seed, version, advanced],
|
| 478 |
run_on_click=True,
|
| 479 |
# On Spaces gr.Examples caches by default (lazily on ZeroGPU), and gradio 5.50 crashes caching an output that
|
| 480 |
# updates a Dropdown's choices ('Dropdown' object has no attribute 'proxy_url'). example_details needs no GPU.
|
|
|
|
| 504 |
randomize_seed.change(lambda on: gr.update(interactive=not on), inputs=randomize_seed, outputs=seed)
|
| 505 |
run.click(
|
| 506 |
generate,
|
| 507 |
+
inputs=[prompt, refs, size_label, seed, randomize_seed, enhance, steps, custom_width, custom_height, version],
|
| 508 |
outputs=[output_image, info, used_prompt],
|
| 509 |
)
|
| 510 |
demo.load(open_tab, outputs=tabs, show_progress="hidden")
|