"""The demo's Comparison tab: 33 of the 37 examples of the official Qwen/Qwen-Image-2.1 Space, rendered with viggle-turbo v0.2.1 in 6 steps (and in 8 steps on the 5 dense-text examples) and with the base model in 40 steps. Both get the same prompt, input images and seed 42, with prompt enhancement off. No GPU: everything is pre-rendered. compare/ (WebP images + cases.json) is written by release/build_compare_space.py, and upload_space.sh builds it into the Space.""" import json import statistics from pathlib import Path import gradio as gr from PIL import Image DIR = Path(__file__).parent / "compare" CASES = json.loads((DIR / "cases.json").read_text(encoding="utf-8")) if (DIR / "cases.json").exists() else [] BY_ID = {case["id"]: case for case in CASES} FIRST = "case01" # Character style infographic: a dense layout where we prefer the turbo's render KINDS = { "All": lambda case: True, "Editing": lambda case: len(case["inputs"]) > 0, "Multi-image editing": lambda case: len(case["inputs"]) > 1, "Text-to-image": lambda case: not case["inputs"], "Dense text (+8 steps)": lambda case: "turbo8" in case, } def arms(case): """(menu label, key) for every render of an example; the render time is in the label.""" menu = [(f"viggle-turbo · 6 steps · {case['turbo_s']:.1f} s", "turbo")] if "turbo8" in case: menu.append((f"viggle-turbo · 8 steps · {case['turbo8_s']:.1f} s", "turbo8")) menu.append((f"Qwen-Image-2.1 base · 40 steps · {case['base_s']:.1f} s", "base")) if case["reference"]: menu.append(("Qwen API reference (bundled with Qwen's Space)", "reference")) return menu def image(case, key): if key == "reference": # Qwen's API output has its own size; match the renders so the two slider halves line up return Image.open(DIR / case["reference"]["src"]).resize(tuple(case["size"]), Image.LANCZOS) return str(DIR / case[key]["src"]) def panels(case, left, right): """Slider pair, caption, input images and prompt for one example.""" names = dict((key, label) for label, key in arms(case)) width, height = case["size"] title = case["title"] if case["title_zh"] == case["title"] else f"{case['title']} · {case['title_zh']}" caption = (f"### {title}\n" f"{width} × {height} · **left:** {names[left]} · **right:** {names[right]} · drag the handle to compare") inputs = [(str(DIR / item["src"]), f"input {i + 1}") for i, item in enumerate(case["inputs"])] return (image(case, left), image(case, right)), caption, inputs, case["prompt"] def thumbs(ids): return [(str(DIR / BY_ID[case_id]["turbo"]["thumb"]), BY_ID[case_id]["title"]) for case_id in ids] def render(): """Builds the tab inside the caller's gr.Blocks / gr.Tab context.""" edits = [case for case in CASES if case["inputs"]] t2is = [case for case in CASES if not case["inputs"]] median = lambda cases, key: statistics.median(case[key] for case in cases) # noqa: E731 speedup = sum(case["base_s"] for case in CASES) / sum(case["turbo_s"] for case in CASES) gr.Markdown( f"**viggle-turbo (6 steps) vs Qwen-Image-2.1 (40 steps)** on {len(CASES)} of the 37 examples of the official " "[Qwen/Qwen-Image-2.1 Space](https://huggingface.co/spaces/Qwen/Qwen-Image-2.1). Both models get the same prompt, " "input images and seed, with prompt enhancement off, one sample each and no seed picking. " f"At **{speedup:.1f}× less time** the turbo is very competitive with the base model, and on some examples " "(the character style infographic, shown first) we prefer its render; complicated edits can still fall short of the base. " "Pick an example on the left, then drag the slider. The menus also offer the 8-step turbo on the dense-text examples " "and Qwen's own API reference where the Space ships one.\n\n" f"Median time, editing: **{median(edits, 'turbo_s'):.1f} s vs {median(edits, 'base_s'):.1f} s** · " f"text-to-image (~4 MP): **{median(t2is, 'turbo_s'):.1f} s vs {median(t2is, 'base_s'):.1f} s** · " f"all {len(CASES)} examples: **{speedup:.1f}× faster** (one NVIDIA B200, one pipeline call each)" ) case = BY_ID[FIRST] pair, caption, inputs, prompt = panels(case, "turbo", "base") ids = gr.State([item["id"] for item in CASES]) with gr.Row(): with gr.Column(scale=1, min_width=280): kind = gr.Radio([(f"{name} ({sum(map(test, CASES))})", name) for name, test in KINDS.items()], value="All", label="Show") gallery = gr.Gallery(thumbs(ids.value), columns=3, height=760, object_fit="cover", allow_preview=False, show_label=False, show_download_button=False, show_fullscreen_button=False) with gr.Column(scale=3): head = gr.Markdown(caption) with gr.Row(): left = gr.Dropdown(arms(case), value="turbo", label="Left") right = gr.Dropdown(arms(case), value="base", label="Right") slider = gr.ImageSlider(pair, type="filepath", interactive=False, max_height=820, show_label=False) with gr.Row(): shown_inputs = gr.Gallery(inputs, label="Input images", columns=3, height=220, visible=bool(inputs), scale=1) shown_prompt = gr.Textbox(prompt, label="Prompt (sent as written, prompt enhancement off)", lines=8, max_lines=8, interactive=False, show_copy_button=True, scale=2) with gr.Accordion("How these were made", open=False): gr.Markdown( "- **Turbo:** Qwen-Image-2.1 + the viggle-turbo v0.2.1 LoRA (rank 256), 6 steps on sigmas " "`[1, 0.9375, 0.875, 0.75, 0.5, 0.25]`, no classifier-free guidance, exactly as the Generate tab runs it.\n" "- **Turbo, 8 steps** (the 5 dense-text examples): sigmas `[1, 0.9375, 0.875, 0.75, 0.625, 0.5, 0.25, 0.125]`. " "At 6 steps small Latin text can print twice, like a double exposure (decided in the 0.75 → 0.5 step), and small " "Chinese strokes can get colour blotches (the last 0.25 → 0 step); each added sigma splits one of the two. In our " "OCR tests 8 steps raise word recall from 0.71 to 0.83 on the academic infographic's caption (16 seeds) and from " "0.76 to 0.90 on the architecture board's Chinese labels (4 seeds); the base model scores 0.95 on both.\n" "- **Base:** Qwen-Image-2.1, 40 steps, default scheduler, `true_cfg_scale=1.0` (the pipeline default).\n" "- **Both:** diffusers `QwenImage21Pipeline` in bf16, seed 42. Input images are encoded at 1024² pixels; the output keeps " "the aspect ratio of Qwen's reference at 2048² pixels for text-to-image and 1536² for editing, in multiples of 32.\n" "- **Time:** one pipeline call on one NVIDIA B200 (text encoding, denoising and VAE decoding), after warm-up.\n" "- **Left out (4 of 37):** three Chinese infographics and a subtitled storyboard whose short prompts leave the " "on-image text to Qwen's prompt enhancement; without it both models fill them with made-up characters.\n" "- **Qwen API reference:** the output bundled with the example in Qwen's Space, made with Qwen's API, which may " "differ from the open weights. It is omitted for the 7 text-to-image examples Qwen generated with prompt enhancement on.\n\n" "Prompts, input images and reference outputs come from the " "[Qwen/Qwen-Image-2.1 Space](https://huggingface.co/spaces/Qwen/Qwen-Image-2.1) (Qwen Research License)." ) def show(case_id, left_key, right_key, fixed="left"): """Keeps the chosen pair when the new example has it (turbo8 and reference are per-example), else turbo vs base. If both menus land on the same render, the one the user just changed (`fixed`) wins and the other moves off it.""" case = BY_ID[case_id] keys = [key for _, key in arms(case)] left_key = left_key if left_key in keys else "turbo" right_key = right_key if right_key in keys else "base" if left_key == right_key: other = next(key for key in ("base", "turbo") if key != left_key) left_key, right_key = (left_key, other) if fixed == "left" else (other, right_key) pair, caption, inputs, prompt = panels(case, left_key, right_key) return (case_id, gr.update(choices=arms(case), value=left_key), gr.update(choices=arms(case), value=right_key), pair, caption, gr.update(value=inputs, visible=bool(inputs)), prompt) def pick(ids, left_key, right_key, evt: gr.SelectData): return show(ids[evt.index], left_key, right_key) def filter_cases(name): ids = [case["id"] for case in CASES if KINDS[name](case)] return ids, thumbs(ids) current = gr.State(FIRST) outputs = [current, left, right, slider, head, shown_inputs, shown_prompt] kind.change(filter_cases, inputs=kind, outputs=[ids, gallery]) gallery.select(pick, inputs=[ids, left, right], outputs=outputs) # .input, not .change: show() rewrites both menus, which must not fire another show() left.input(show, inputs=[current, left, right], outputs=outputs) right.input(lambda case_id, left_key, right_key: show(case_id, left_key, right_key, "right"), inputs=[current, left, right], outputs=outputs)