yycc's picture
Comparison tab (33 Qwen Space examples, turbo 6 steps vs base 40 steps, image slider); 5 examples from the Qwen Space; header: very competitive with the base (part 2)
a8e67a7 verified
Raw History Blame
9.48 kB
"""The demo's Comparison tab: 33 of the 37 examples of the official Qwen/Qwen-Image-2.1 Space, rendered with viggle-turbo
v0.2.1 in 6 steps (and in 8 steps on the 5 dense-text examples) and with the base model in 40 steps. Both get the same
prompt, input images and seed 42, with prompt enhancement off. No GPU: everything is pre-rendered. compare/ (WebP images
+ cases.json) is written by release/build_compare_space.py, and upload_space.sh builds it into the Space."""
import json
import statistics
from pathlib import Path
import gradio as gr
from PIL import Image
DIR = Path(__file__).parent / "compare"
CASES = json.loads((DIR / "cases.json").read_text(encoding="utf-8")) if (DIR / "cases.json").exists() else []
BY_ID = {case["id"]: case for case in CASES}
FIRST = "case01" # Character style infographic: a dense layout where we prefer the turbo's render
KINDS = {
"All": lambda case: True,
"Editing": lambda case: len(case["inputs"]) > 0,
"Multi-image editing": lambda case: len(case["inputs"]) > 1,
"Text-to-image": lambda case: not case["inputs"],
"Dense text (+8 steps)": lambda case: "turbo8" in case,
}
def arms(case):
"""(menu label, key) for every render of an example; the render time is in the label."""
menu = [(f"viggle-turbo · 6 steps · {case['turbo_s']:.1f} s", "turbo")]
if "turbo8" in case:
menu.append((f"viggle-turbo · 8 steps · {case['turbo8_s']:.1f} s", "turbo8"))
menu.append((f"Qwen-Image-2.1 base · 40 steps · {case['base_s']:.1f} s", "base"))
if case["reference"]:
menu.append(("Qwen API reference (bundled with Qwen's Space)", "reference"))
return menu
def image(case, key):
if key == "reference": # Qwen's API output has its own size; match the renders so the two slider halves line up
return Image.open(DIR / case["reference"]["src"]).resize(tuple(case["size"]), Image.LANCZOS)
return str(DIR / case[key]["src"])
def panels(case, left, right):
"""Slider pair, caption, input images and prompt for one example."""
names = dict((key, label) for label, key in arms(case))
width, height = case["size"]
title = case["title"] if case["title_zh"] == case["title"] else f"{case['title']} · {case['title_zh']}"
caption = (f"### {title}\n"
f"{width} × {height} · **left:** {names[left]} · **right:** {names[right]} · drag the handle to compare")
inputs = [(str(DIR / item["src"]), f"input {i + 1}") for i, item in enumerate(case["inputs"])]
return (image(case, left), image(case, right)), caption, inputs, case["prompt"]
def thumbs(ids):
return [(str(DIR / BY_ID[case_id]["turbo"]["thumb"]), BY_ID[case_id]["title"]) for case_id in ids]
def render():
"""Builds the tab inside the caller's gr.Blocks / gr.Tab context."""
edits = [case for case in CASES if case["inputs"]]
t2is = [case for case in CASES if not case["inputs"]]
median = lambda cases, key: statistics.median(case[key] for case in cases) # noqa: E731
speedup = sum(case["base_s"] for case in CASES) / sum(case["turbo_s"] for case in CASES)
gr.Markdown(
f"**viggle-turbo (6 steps) vs Qwen-Image-2.1 (40 steps)** on {len(CASES)} of the 37 examples of the official "
"[Qwen/Qwen-Image-2.1 Space](https://huggingface.co/spaces/Qwen/Qwen-Image-2.1). Both models get the same prompt, "
"input images and seed, with prompt enhancement off, one sample each and no seed picking. "
f"At **{speedup:.1f}× less time** the turbo is very competitive with the base model, and on some examples "
"(the character style infographic, shown first) we prefer its render; complicated edits can still fall short of the base. "
"Pick an example on the left, then drag the slider. The menus also offer the 8-step turbo on the dense-text examples "
"and Qwen's own API reference where the Space ships one.\n\n"
f"Median time, editing: **{median(edits, 'turbo_s'):.1f} s vs {median(edits, 'base_s'):.1f} s** · "
f"text-to-image (~4 MP): **{median(t2is, 'turbo_s'):.1f} s vs {median(t2is, 'base_s'):.1f} s** · "
f"all {len(CASES)} examples: **{speedup:.1f}× faster** (one NVIDIA B200, one pipeline call each)"
)
case = BY_ID[FIRST]
pair, caption, inputs, prompt = panels(case, "turbo", "base")
ids = gr.State([item["id"] for item in CASES])
with gr.Row():
with gr.Column(scale=1, min_width=280):
kind = gr.Radio([(f"{name} ({sum(map(test, CASES))})", name) for name, test in KINDS.items()], value="All", label="Show")
gallery = gr.Gallery(thumbs(ids.value), columns=3, height=760, object_fit="cover", allow_preview=False,
show_label=False, show_download_button=False, show_fullscreen_button=False)
with gr.Column(scale=3):
head = gr.Markdown(caption)
with gr.Row():
left = gr.Dropdown(arms(case), value="turbo", label="Left")
right = gr.Dropdown(arms(case), value="base", label="Right")
slider = gr.ImageSlider(pair, type="filepath", interactive=False, max_height=820, show_label=False)
with gr.Row():
shown_inputs = gr.Gallery(inputs, label="Input images", columns=3, height=220, visible=bool(inputs), scale=1)
shown_prompt = gr.Textbox(prompt, label="Prompt (sent as written, prompt enhancement off)", lines=8, max_lines=8,
interactive=False, show_copy_button=True, scale=2)
with gr.Accordion("How these were made", open=False):
gr.Markdown(
"- **Turbo:** Qwen-Image-2.1 + the viggle-turbo v0.2.1 LoRA (rank 256), 6 steps on sigmas "
"`[1, 0.9375, 0.875, 0.75, 0.5, 0.25]`, no classifier-free guidance, exactly as the Generate tab runs it.\n"
"- **Turbo, 8 steps** (the 5 dense-text examples): sigmas `[1, 0.9375, 0.875, 0.75, 0.625, 0.5, 0.25, 0.125]`. "
"At 6 steps small Latin text can print twice, like a double exposure (decided in the 0.75 → 0.5 step), and small "
"Chinese strokes can get colour blotches (the last 0.25 → 0 step); each added sigma splits one of the two. In our "
"OCR tests 8 steps raise word recall from 0.71 to 0.83 on the academic infographic's caption (16 seeds) and from "
"0.76 to 0.90 on the architecture board's Chinese labels (4 seeds); the base model scores 0.95 on both.\n"
"- **Base:** Qwen-Image-2.1, 40 steps, default scheduler, `true_cfg_scale=1.0` (the pipeline default).\n"
"- **Both:** diffusers `QwenImage21Pipeline` in bf16, seed 42. Input images are encoded at 1024² pixels; the output keeps "
"the aspect ratio of Qwen's reference at 2048² pixels for text-to-image and 1536² for editing, in multiples of 32.\n"
"- **Time:** one pipeline call on one NVIDIA B200 (text encoding, denoising and VAE decoding), after warm-up.\n"
"- **Left out (4 of 37):** three Chinese infographics and a subtitled storyboard whose short prompts leave the "
"on-image text to Qwen's prompt enhancement; without it both models fill them with made-up characters.\n"
"- **Qwen API reference:** the output bundled with the example in Qwen's Space, made with Qwen's API, which may "
"differ from the open weights. It is omitted for the 7 text-to-image examples Qwen generated with prompt enhancement on.\n\n"
"Prompts, input images and reference outputs come from the "
"[Qwen/Qwen-Image-2.1 Space](https://huggingface.co/spaces/Qwen/Qwen-Image-2.1) (Qwen Research License)."
)
def show(case_id, left_key, right_key, fixed="left"):
"""Keeps the chosen pair when the new example has it (turbo8 and reference are per-example), else turbo vs base.
If both menus land on the same render, the one the user just changed (`fixed`) wins and the other moves off it."""
case = BY_ID[case_id]
keys = [key for _, key in arms(case)]
left_key = left_key if left_key in keys else "turbo"
right_key = right_key if right_key in keys else "base"
if left_key == right_key:
other = next(key for key in ("base", "turbo") if key != left_key)
left_key, right_key = (left_key, other) if fixed == "left" else (other, right_key)
pair, caption, inputs, prompt = panels(case, left_key, right_key)
return (case_id, gr.update(choices=arms(case), value=left_key), gr.update(choices=arms(case), value=right_key),
pair, caption, gr.update(value=inputs, visible=bool(inputs)), prompt)
def pick(ids, left_key, right_key, evt: gr.SelectData):
return show(ids[evt.index], left_key, right_key)
def filter_cases(name):
ids = [case["id"] for case in CASES if KINDS[name](case)]
return ids, thumbs(ids)
current = gr.State(FIRST)
outputs = [current, left, right, slider, head, shown_inputs, shown_prompt]
kind.change(filter_cases, inputs=kind, outputs=[ids, gallery])
gallery.select(pick, inputs=[ids, left, right], outputs=outputs)
# .input, not .change: show() rewrites both menus, which must not fire another show()
left.input(show, inputs=[current, left, right], outputs=outputs)
right.input(lambda case_id, left_key, right_key: show(case_id, left_key, right_key, "right"), inputs=[current, left, right], outputs=outputs)