yycc commited on
Commit
628a7b6
·
verified ·
1 Parent(s): c38835c

Advanced settings: pick v0.2.1 instead of v0.3 (both LoRAs loaded, switched per call)

Browse files
Files changed (1) hide show
  1. app.py +33 -20
app.py CHANGED
@@ -1,4 +1,5 @@
1
- """Gradio demo for the distilled Qwen-Image-2.1 turbo (v0.3: LoRA r256 sampled in 6 steps; 9 steps = 7 turbo + 2 base-model steps)."""
 
2
 
3
  # ZeroGPU patches torch at import time, so `spaces` must be imported before torch or any CUDA use.
4
  # find_spec keeps this file runnable off-Spaces, where the package is absent.
@@ -38,13 +39,17 @@ from diffusers import FlowMatchEulerDiscreteScheduler, QwenImage21Pipeline
38
  from diffusers.pipelines.qwenimage21.pipeline_qwenimage21 import calculate_dimensions
39
  from huggingface_hub import CommitScheduler
40
 
41
- # The v0.3 LoRA (r256) at the model repo root, applied at runtime on top of the base transformer and never merged
42
  # (merging into bf16 keeps only ~47% of the adapter delta, the per-element deltas sit below the bf16 ULP of the base
43
- # weights). LORA_FILE selects the adapter file; the v0.2.1 and v0.2 adapters are still in the repo.
 
44
  BASE_MODEL_ID = os.environ.get("BASE_MODEL_ID", "Qwen/Qwen-Image-2.1")
45
  STUDENT_REPO = os.environ.get("STUDENT_REPO", os.environ.get("LORA_REPO", "Viggle/Qwen-Image-2.1-viggle-turbo"))
46
- LORA_FILE = os.environ.get("LORA_FILE", "Qwen-Image-2.1-viggle-turbo-v0.3-6step-lora-r256.safetensors")
47
- STUDENT_TAG = "v0.3 LoRA r256" if "v0.3" in LORA_FILE else f"LoRA {LORA_FILE}"
 
 
 
48
  HF_TOKEN = os.environ.get("HF_TOKEN") # only needed while the model repo is private
49
  # Usage counter. The Space autoscales to several replicas and its log view shows only one of them, so the [gen] log lines
50
  # undercount. Every generate() call that reaches the GPU also appends one JSON line (metadata only: no prompt, no images)
@@ -134,7 +139,8 @@ CUSTOM_CAP = {False: 2048, True: 1536} # keyed by "has references"
134
  MAX_SIDE = max(max(size) for size in SIZES.values() if size)
135
 
136
  pipe = QwenImage21Pipeline.from_pretrained(BASE_MODEL_ID, dtype=torch.bfloat16, token=HF_TOKEN)
137
- pipe.load_lora_weights(STUDENT_REPO, weight_name=LORA_FILE, token=HF_TOKEN)
 
138
  # The stock Qwen-Image-2.1 scheduler config carries shift_terminal=0.02, which stretches the last
139
  # sigma node to 0.02 instead of 0 and silently costs the final step. The student was distilled
140
  # against the unstretched schedule, so rebuild the scheduler with shift_terminal=None.
@@ -233,11 +239,12 @@ def base_tail(pipe, i, t, kwargs):
233
 
234
 
235
  @gpu
236
- def denoise(prompt, images, width, height, steps, seed):
237
  # No VAE tiling anywhere (xlarge card): tiling the reference encode wrecks edits, and tiled decodes are not the
238
  # validated pipeline either.
239
  generator = torch.Generator(device="cuda").manual_seed(seed)
240
  started = time.perf_counter()
 
241
  pipe.enable_lora() # every call starts on the student, even if an earlier 9-step call died with the LoRA off
242
  reextract[0] = False
243
  result = pipe(
@@ -256,8 +263,10 @@ def denoise(prompt, images, width, height, steps, seed):
256
  return result, time.perf_counter() - started
257
 
258
 
259
- def generate(prompt, refs, size_label, seed, randomize_seed, enhance="Auto", steps=STEPS, custom_width=1024, custom_height=1024):
 
260
  t0 = time.perf_counter()
 
261
  steps = min(max(int(steps), 3), 9) # 6 (RAW_NODES) is the default; 9 = 7 student steps + 2 base-model steps; fewer than 6 degrade quickly
262
  # refs: the gallery's (filepath, caption) pairs in upload order, which is the order the prompt's "image 1, 2, ..." refer
263
  # to. EXIF rotation + RGB is what the gr.Image slots used to do.
@@ -294,18 +303,18 @@ def generate(prompt, refs, size_label, seed, randomize_seed, enhance="Auto", ste
294
  enhance_note = f" · enhanced {pe_time:.1f}s" if used_prompt != prompt else " · enhancement failed, prompt sent as written"
295
  mode = "edit" if images else "t2i"
296
  try:
297
- result, elapsed = denoise(used_prompt, images, width, height, steps, seed)
298
  except Exception as e: # ZeroGPU quota / GPU timeout errors land here: count them, then show them as before
299
- record(ok=False, mode=mode, refs=len(images), steps=steps, error=f"{type(e).__name__}: {e}"[:200],
300
  total=round(time.perf_counter() - t0, 2))
301
  raise
302
  # one line per generation, for counting usage from the Space logs (no prompt text)
303
  print(f"[gen] {time.strftime('%Y-%m-%dT%H:%M:%SZ', time.gmtime())} mode={mode} refs={len(images)} "
304
- f"steps={steps} size={result.width}x{result.height} enhance={int(used_prompt != prompt)} pe={pe_time:.1f}s gen={elapsed:.2f}s",
305
  flush=True)
306
- record(ok=True, mode=mode, refs=len(images), steps=steps, size=f"{result.width}x{result.height}", enhance=enhance,
307
  enhanced=used_prompt != prompt, pe=round(pe_time, 2), gen=round(elapsed, 2), total=round(time.perf_counter() - t0, 2))
308
- return result, f"seed `{seed}` · {result.width}×{result.height} · {steps} steps · {elapsed:.2f}s · {STUDENT_TAG}{enhance_note}{clamped}", used_prompt
309
 
310
 
311
  def refresh_sizes(prompt, refs, current):
@@ -397,8 +406,8 @@ with gr.Blocks(title="Viggle Turbo v0.3 · Qwen-Image-2.1 6-step", theme=THEME,
397
  "**v0.3 (2026-09-29):** at 6 steps, less grain than v0.2.1 and a little softer on fine texture. We think 6 steps "
398
  "is close to its capacity: every further gain we found cost something elsewhere. **9 steps** runs 7 turbo "
399
  "steps and lets the base model finish the last two: finer detail and small text right more often, at about 1.4–1.5× the "
400
- "time of 6 steps. "
401
- f"Weights: **{STUDENT_TAG}** from [{STUDENT_REPO}](https://huggingface.co/{STUDENT_REPO})."
402
  )
403
  with gr.Tabs() as tabs:
404
  with gr.Tab("Generate", id="generate"):
@@ -432,6 +441,9 @@ with gr.Blocks(title="Viggle Turbo v0.3 · Qwen-Image-2.1 6-step", theme=THEME,
432
  with gr.Accordion("Advanced settings", open=False) as advanced:
433
  steps = gr.Slider(label="Steps", minimum=3, maximum=9, step=1, value=STEPS,
434
  info="6 is the default; 7-8 add steps at the high-noise end; 9 = 7 turbo + 2 base-model steps (finer detail; ~1.4–1.5× the time).")
 
 
 
435
  with gr.Row():
436
  seed = gr.Number(label="Seed", value=0, precision=0, interactive=False)
437
  randomize_seed = gr.Checkbox(label="Randomize seed", value=True)
@@ -446,12 +458,13 @@ with gr.Blocks(title="Viggle Turbo v0.3 · Qwen-Image-2.1 6-step", theme=THEME,
446
  def example_details(prompt, refs, result):
447
  """Runs on an example click (no GPU): fills the status line and the prompt stored with the row, and sets the
448
  steps / size / enhancement / seed controls to what produced it, so Generate reproduces the result. It opens
449
- Advanced settings to show that the seed is now fixed. It also sets the size menu's mode itself: a
450
- text-to-image row loaded after another one leaves the gallery unchanged, so refresh_sizes does not run."""
 
451
  row = next(row for row in EXAMPLES if row["prompt"] == prompt)
452
  choices = EDIT_CHOICES if any(row["refs"]) else T2I_CHOICES
453
  return (row["info"], row["used_prompt"], row["steps"], gr.update(choices=choices, value=row["size_label"]),
454
- "On" if row["enhance"] else "Off", row["seed"], False, gr.update(open=True))
455
 
456
  if EXAMPLES:
457
  gr.Examples(
@@ -461,7 +474,7 @@ with gr.Blocks(title="Viggle Turbo v0.3 · Qwen-Image-2.1 6-step", theme=THEME,
461
  ],
462
  inputs=[prompt, refs, output_image],
463
  fn=example_details,
464
- outputs=[info, used_prompt, steps, size_label, enhance, seed, randomize_seed, advanced],
465
  run_on_click=True,
466
  # On Spaces gr.Examples caches by default (lazily on ZeroGPU), and gradio 5.50 crashes caching an output that
467
  # updates a Dropdown's choices ('Dropdown' object has no attribute 'proxy_url'). example_details needs no GPU.
@@ -491,7 +504,7 @@ with gr.Blocks(title="Viggle Turbo v0.3 · Qwen-Image-2.1 6-step", theme=THEME,
491
  randomize_seed.change(lambda on: gr.update(interactive=not on), inputs=randomize_seed, outputs=seed)
492
  run.click(
493
  generate,
494
- inputs=[prompt, refs, size_label, seed, randomize_seed, enhance, steps, custom_width, custom_height],
495
  outputs=[output_image, info, used_prompt],
496
  )
497
  demo.load(open_tab, outputs=tabs, show_progress="hidden")
 
1
+ """Gradio demo for the distilled Qwen-Image-2.1 turbo (v0.3: LoRA r256 sampled in 6 steps; 9 steps = 7 turbo + 2 base-model steps;
2
+ v0.2.1 selectable under Advanced settings)."""
3
 
4
  # ZeroGPU patches torch at import time, so `spaces` must be imported before torch or any CUDA use.
5
  # find_spec keeps this file runnable off-Spaces, where the package is absent.
 
39
  from diffusers.pipelines.qwenimage21.pipeline_qwenimage21 import calculate_dimensions
40
  from huggingface_hub import CommitScheduler
41
 
42
+ # The LoRAs (r256) at the model repo root, applied at runtime on top of the base transformer and never merged
43
  # (merging into bf16 keeps only ~47% of the adapter delta, the per-element deltas sit below the bf16 ULP of the base
44
+ # weights). v0.3 (the default) and v0.2.1 are both loaded as named adapters (PEFT adapter names cannot contain dots)
45
+ # and each call activates the one picked under Advanced settings; both are sampled on the same nodes (raw_nodes below).
46
  BASE_MODEL_ID = os.environ.get("BASE_MODEL_ID", "Qwen/Qwen-Image-2.1")
47
  STUDENT_REPO = os.environ.get("STUDENT_REPO", os.environ.get("LORA_REPO", "Viggle/Qwen-Image-2.1-viggle-turbo"))
48
+ LORAS = { # version -> (adapter name, file)
49
+ "v0.3": ("v03", "Qwen-Image-2.1-viggle-turbo-v0.3-6step-lora-r256.safetensors"),
50
+ "v0.2.1": ("v021", "Qwen-Image-2.1-viggle-turbo-v0.2.1-6step-lora-r256.safetensors"),
51
+ }
52
+ VERSION = "v0.3"
53
  HF_TOKEN = os.environ.get("HF_TOKEN") # only needed while the model repo is private
54
  # Usage counter. The Space autoscales to several replicas and its log view shows only one of them, so the [gen] log lines
55
  # undercount. Every generate() call that reaches the GPU also appends one JSON line (metadata only: no prompt, no images)
 
139
  MAX_SIDE = max(max(size) for size in SIZES.values() if size)
140
 
141
  pipe = QwenImage21Pipeline.from_pretrained(BASE_MODEL_ID, dtype=torch.bfloat16, token=HF_TOKEN)
142
+ for adapter, lora_file in LORAS.values():
143
+ pipe.load_lora_weights(STUDENT_REPO, weight_name=lora_file, adapter_name=adapter, token=HF_TOKEN)
144
  # The stock Qwen-Image-2.1 scheduler config carries shift_terminal=0.02, which stretches the last
145
  # sigma node to 0.02 instead of 0 and silently costs the final step. The student was distilled
146
  # against the unstretched schedule, so rebuild the scheduler with shift_terminal=None.
 
239
 
240
 
241
  @gpu
242
+ def denoise(prompt, images, width, height, steps, seed, version=VERSION):
243
  # No VAE tiling anywhere (xlarge card): tiling the reference encode wrecks edits, and tiled decodes are not the
244
  # validated pipeline either.
245
  generator = torch.Generator(device="cuda").manual_seed(seed)
246
  started = time.perf_counter()
247
+ pipe.set_adapters(LORAS[version][0])
248
  pipe.enable_lora() # every call starts on the student, even if an earlier 9-step call died with the LoRA off
249
  reextract[0] = False
250
  result = pipe(
 
263
  return result, time.perf_counter() - started
264
 
265
 
266
+ def generate(prompt, refs, size_label, seed, randomize_seed, enhance="Auto", steps=STEPS, custom_width=1024, custom_height=1024,
267
+ version=VERSION):
268
  t0 = time.perf_counter()
269
+ version = version if version in LORAS else VERSION
270
  steps = min(max(int(steps), 3), 9) # 6 (RAW_NODES) is the default; 9 = 7 student steps + 2 base-model steps; fewer than 6 degrade quickly
271
  # refs: the gallery's (filepath, caption) pairs in upload order, which is the order the prompt's "image 1, 2, ..." refer
272
  # to. EXIF rotation + RGB is what the gr.Image slots used to do.
 
303
  enhance_note = f" · enhanced {pe_time:.1f}s" if used_prompt != prompt else " · enhancement failed, prompt sent as written"
304
  mode = "edit" if images else "t2i"
305
  try:
306
+ result, elapsed = denoise(used_prompt, images, width, height, steps, seed, version)
307
  except Exception as e: # ZeroGPU quota / GPU timeout errors land here: count them, then show them as before
308
+ record(ok=False, mode=mode, refs=len(images), steps=steps, version=version, error=f"{type(e).__name__}: {e}"[:200],
309
  total=round(time.perf_counter() - t0, 2))
310
  raise
311
  # one line per generation, for counting usage from the Space logs (no prompt text)
312
  print(f"[gen] {time.strftime('%Y-%m-%dT%H:%M:%SZ', time.gmtime())} mode={mode} refs={len(images)} "
313
+ f"steps={steps} version={version} size={result.width}x{result.height} enhance={int(used_prompt != prompt)} pe={pe_time:.1f}s gen={elapsed:.2f}s",
314
  flush=True)
315
+ record(ok=True, mode=mode, refs=len(images), steps=steps, version=version, size=f"{result.width}x{result.height}", enhance=enhance,
316
  enhanced=used_prompt != prompt, pe=round(pe_time, 2), gen=round(elapsed, 2), total=round(time.perf_counter() - t0, 2))
317
+ return result, f"seed `{seed}` · {result.width}×{result.height} · {steps} steps · {elapsed:.2f}s · {version} LoRA r256{enhance_note}{clamped}", used_prompt
318
 
319
 
320
  def refresh_sizes(prompt, refs, current):
 
406
  "**v0.3 (2026-09-29):** at 6 steps, less grain than v0.2.1 and a little softer on fine texture. We think 6 steps "
407
  "is close to its capacity: every further gain we found cost something elsewhere. **9 steps** runs 7 turbo "
408
  "steps and lets the base model finish the last two: finer detail and small text right more often, at about 1.4–1.5× the "
409
+ "time of 6 steps. v0.2.1 can still be picked under **Advanced settings**. "
410
+ f"Weights: **v0.3 LoRA r256** (and v0.2.1) from [{STUDENT_REPO}](https://huggingface.co/{STUDENT_REPO})."
411
  )
412
  with gr.Tabs() as tabs:
413
  with gr.Tab("Generate", id="generate"):
 
441
  with gr.Accordion("Advanced settings", open=False) as advanced:
442
  steps = gr.Slider(label="Steps", minimum=3, maximum=9, step=1, value=STEPS,
443
  info="6 is the default; 7-8 add steps at the high-noise end; 9 = 7 turbo + 2 base-model steps (finer detail; ~1.4–1.5× the time).")
444
+ version = gr.Radio(list(LORAS), value=VERSION, label="Model version",
445
+ info="v0.3 is the default. v0.2.1 is the previous release: a little sharper on fine "
446
+ "texture, with more grain.")
447
  with gr.Row():
448
  seed = gr.Number(label="Seed", value=0, precision=0, interactive=False)
449
  randomize_seed = gr.Checkbox(label="Randomize seed", value=True)
 
458
  def example_details(prompt, refs, result):
459
  """Runs on an example click (no GPU): fills the status line and the prompt stored with the row, and sets the
460
  steps / size / enhancement / seed controls to what produced it, so Generate reproduces the result. It opens
461
+ Advanced settings to show that the seed is now fixed, and picks v0.3, which rendered every example. It also sets
462
+ the size menu's mode itself: a text-to-image row loaded after another one leaves the gallery unchanged, so
463
+ refresh_sizes does not run."""
464
  row = next(row for row in EXAMPLES if row["prompt"] == prompt)
465
  choices = EDIT_CHOICES if any(row["refs"]) else T2I_CHOICES
466
  return (row["info"], row["used_prompt"], row["steps"], gr.update(choices=choices, value=row["size_label"]),
467
+ "On" if row["enhance"] else "Off", row["seed"], False, VERSION, gr.update(open=True))
468
 
469
  if EXAMPLES:
470
  gr.Examples(
 
474
  ],
475
  inputs=[prompt, refs, output_image],
476
  fn=example_details,
477
+ outputs=[info, used_prompt, steps, size_label, enhance, seed, randomize_seed, version, advanced],
478
  run_on_click=True,
479
  # On Spaces gr.Examples caches by default (lazily on ZeroGPU), and gradio 5.50 crashes caching an output that
480
  # updates a Dropdown's choices ('Dropdown' object has no attribute 'proxy_url'). example_details needs no GPU.
 
504
  randomize_seed.change(lambda on: gr.update(interactive=not on), inputs=randomize_seed, outputs=seed)
505
  run.click(
506
  generate,
507
+ inputs=[prompt, refs, size_label, seed, randomize_seed, enhance, steps, custom_width, custom_height, version],
508
  outputs=[output_image, info, used_prompt],
509
  )
510
  demo.load(open_tab, outputs=tabs, show_progress="hidden")