Spaces:
Running on Zero
Running on Zero
Download app.py from Qwen/Qwen-Image-2.1: direct link, hf CLI and curl.
- Browser
- Download file 39.6 kB
-
https://huggingface.co/spaces/Qwen/Qwen-Image-2.1/resolve/19e3cdce548c41c72161e1cc7790de4f7ab701f1/app.py
- Command line
-
hf download hf://spaces/Qwen/Qwen-Image-2.1@19e3cdce548c41c72161e1cc7790de4f7ab701f1/app.py
-
curl -L -o app.py https://huggingface.co/spaces/Qwen/Qwen-Image-2.1/resolve/19e3cdce548c41c72161e1cc7790de4f7ab701f1/app.py
39.6 kB
| import os | |
| os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True") | |
| import spaces | |
| import gradio as gr | |
| import numpy as np | |
| import random | |
| import io | |
| import time | |
| import uuid | |
| from datetime import datetime | |
| from PIL import Image | |
| import base64 | |
| import json | |
| import re | |
| import torch | |
| from diffusers import QwenImage21Pipeline | |
| import ncii_guard | |
| # ============== 配置参数 ============== | |
| # 日志目录,可通过环境变量 LOG_DIR 自定义 | |
| LOG_DIR = os.environ.get("LOG_DIR", "./generation_logs_paper_case") | |
| # ============== 模型配置 ============== | |
| MODEL_ID = os.environ.get("QWEN_IMAGE_MODEL", "Qwen/Qwen-Image-2.1") | |
| # Prompt rewriting (Qwen-Image-2.1-PE-T2I / PE-I2I) runs in a companion Space so this | |
| # Space only holds the diffusion pipeline and fits a `large` ZeroGPU slice. | |
| PE_SPACE_ID = os.environ.get("PE_SPACE_ID", "hugging-apps/qwen-image-2-1-prompt-enhancer") | |
| PE_MAX_NEW_TOKENS = {"t2i": 1536, "i2i": 2048} | |
| GUARD_THRESHOLD = 0.5 | |
| NUM_INFERENCE_STEPS = 28 | |
| QUALITY_RESOLUTIONS = {"speed": 1024, "quality": 2048} | |
| TRUE_CFG_SCALE = 4.0 | |
| # The prefix KV cache costs ~2 GB per 1K condition image; above this budget it is | |
| # switched off so many-image edits still fit next to the weights. | |
| KV_CACHE_BYTES_PER_TOKEN = 32 * 2 * 4096 * 2 | |
| KV_CACHE_BUDGET_GB = 10.0 | |
| MAX_INPUT_IMAGES = 10 | |
| DEFAULT_LANGUAGE = "en" | |
| pipe = QwenImage21Pipeline.from_pretrained(MODEL_ID, dtype=torch.bfloat16) | |
| pipe.to("cuda") | |
| # Decode large outputs in tiles so a 2K VAE decode fits next to the weights. | |
| pipe.vae.enable_tiling( | |
| tile_sample_min_height=1536, | |
| tile_sample_min_width=1536, | |
| tile_sample_stride_height=1152, | |
| tile_sample_stride_width=1152, | |
| ) | |
| # The NCII classifier runs in a CPU subprocess: a transformers forward in the main | |
| # process breaks the ZeroGPU worker fork. | |
| ncii_guard.start() | |
| # ============== 预设分辨率 ============== | |
| SIZE_PRESETS_2K = { | |
| "2688x1536 (16:9)": (2688, 1536), | |
| "1536x2688 (9:16)": (1536, 2688), | |
| "2048x2048 (1:1)": (2048, 2048), | |
| "2368x1728 (4:3)": (2368, 1728), | |
| "1728x2368 (3:4)": (1728, 2368), | |
| } | |
| SIZE_PRESETS_1K = { | |
| "1344x768 (16:9)": (1344, 768), | |
| "768x1344 (9:16)": (768, 1344), | |
| "1184x864 (4:3)": (1184, 864), | |
| "864x1184 (3:4)": (864, 1184), | |
| "1024x1024 (1:1)": (1024, 1024), | |
| } | |
| # 合并所有预设,添加自定义选项 | |
| ALL_SIZE_PRESETS = {"自定义 (Custom)": None} | |
| ALL_SIZE_PRESETS.update({f"2K - {k}": v for k, v in SIZE_PRESETS_2K.items()}) | |
| ALL_SIZE_PRESETS.update({f"1K - {k}": v for k, v in SIZE_PRESETS_1K.items()}) | |
| # ============== 日志记录类 ============== | |
| class GenerationLogger: | |
| """记录生成日志:JSON(prompt信息) + PNG(图片)""" | |
| def __init__(self, log_dir="./logs"): | |
| """ | |
| 初始化日志记录器 | |
| Args: | |
| log_dir: 日志存储目录 | |
| """ | |
| self.log_dir = log_dir | |
| self._ensure_dir() | |
| def _ensure_dir(self): | |
| """确保日志目录存在""" | |
| os.makedirs(self.log_dir, exist_ok=True) | |
| def set_log_dir(self, log_dir): | |
| """ | |
| 动态设置日志目录 | |
| Args: | |
| log_dir: 新的日志目录路径 | |
| """ | |
| self.log_dir = log_dir | |
| self._ensure_dir() | |
| def log_generation(self, original_prompt, enhanced_prompt, image, seed, params, gpu_id=None, input_images=None, username=None, api_response=None): | |
| """ | |
| 保存一次生成的完整记录 | |
| Args: | |
| original_prompt: 用户输入的原始prompt | |
| enhanced_prompt: LLM改写后的prompt | |
| image: 生成的PIL图片对象 | |
| seed: 使用的随机种子 | |
| params: 生成参数字典 | |
| gpu_id: 使用的GPU ID | |
| input_images: 输入图片列表(PIL Image对象) | |
| username: 触发生成的用户名 | |
| Returns: | |
| str: 日志文件的基础名称(不含扩展名) | |
| """ | |
| self._ensure_dir() | |
| timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") | |
| unique_id = str(uuid.uuid4())[:8] | |
| base_name = f"{timestamp}_{unique_id}" | |
| # 保存输入图片 | |
| input_image_paths = [] | |
| if input_images and len(input_images) > 0: | |
| for i, img in enumerate(input_images): | |
| input_path = os.path.join(self.log_dir, f"{base_name}.{i}.png") | |
| img.save(input_path) | |
| input_image_paths.append(os.path.abspath(input_path)) | |
| print(f"[Logger] 已保存 {len(input_images)} 张输入图片") | |
| # 保存输出图片 | |
| image_path = os.path.join(self.log_dir, f"{base_name}.png") | |
| image.save(image_path) | |
| # 保存JSON日志 | |
| json_path = os.path.join(self.log_dir, f"{base_name}.json") | |
| log_data = { | |
| "timestamp": timestamp, | |
| "username": username, | |
| "original_prompt": original_prompt, | |
| "enhanced_prompt": enhanced_prompt, | |
| "seed": seed, | |
| "gpu_id": gpu_id, | |
| "parameters": params, | |
| "input_images": input_image_paths, | |
| "image_path": os.path.abspath(image_path), | |
| "api_response": api_response | |
| } | |
| with open(json_path, 'w', encoding='utf-8') as f: | |
| json.dump(log_data, f, ensure_ascii=False, indent=2) | |
| print(f"[Logger] 日志已保存: {base_name}.json / {base_name}.png") | |
| return base_name | |
| def validate_image_count(images): | |
| if images is not None and len(images) > MAX_INPUT_IMAGES: | |
| raise gr.Error("最多支持 10 张输入图片。 / Up to 10 input images are supported.") | |
| def check_prompt_guard(prompt): | |
| """Reject image-editing prompts the NCII classifier flags. The error is | |
| deliberately generic and does not say which classifier fired.""" | |
| if ncii_guard.score(prompt or "") >= GUARD_THRESHOLD: | |
| raise gr.Error("prompt invalid based on our classifiers, try again") | |
| def ratio_to_size(ratio, area=1024 * 1024, multiple=32): | |
| """Turn a rewriter aspect ratio like "16:9" into a height/width pair of about `area` pixels.""" | |
| match = re.fullmatch(r"\s*(\d+(?:\.\d+)?)\s*:\s*(\d+(?:\.\d+)?)\s*", str(ratio or "")) | |
| if not match or float(match.group(2)) == 0: | |
| return None, None | |
| aspect = float(match.group(1)) / float(match.group(2)) | |
| if not 1 / 4 <= aspect <= 4: | |
| return None, None | |
| return aspect_to_size(aspect, area, multiple) | |
| def aspect_to_size(aspect, area, multiple=32): | |
| width = round((area * aspect) ** 0.5 / multiple) * multiple | |
| height = round((area / aspect) ** 0.5 / multiple) * multiple | |
| return height, width | |
| _pe_client = None | |
| def enhance_prompt(prompt, image_paths): | |
| """Rewrite the prompt with the official PE models in the companion Space. | |
| Returns (rewritten_prompt, wh_ratio).""" | |
| global _pe_client | |
| from gradio_client import Client, handle_file | |
| if _pe_client is None: | |
| _pe_client = Client(PE_SPACE_ID, token=os.environ.get("HF_TOKEN"), httpx_kwargs={"timeout": 900}, verbose=False) | |
| rewritten, wh_ratio, *_ = _pe_client.predict( | |
| prompt=prompt, | |
| image_paths=[handle_file(p) for p in image_paths], | |
| max_new_tokens=PE_MAX_NEW_TOKENS["i2i" if image_paths else "t2i"], | |
| enable_thinking=False, | |
| seed=0, | |
| randomize_seed=True, | |
| api_name="/enhance", | |
| ) | |
| return str(rewritten or "").strip(), str(wh_ratio or "").strip() | |
| def kv_cache_fits(n_images): | |
| return n_images * (1024 // 16) ** 2 * KV_CACHE_BYTES_PER_TOKEN <= KV_CACHE_BUDGET_GB * 1e9 | |
| def generation_duration(prompt, image_paths, height, width, negative_prompt, seed): | |
| """GPU seconds, fitted on this pipeline (eager, large ZeroGPU): step cost scales as | |
| latent_tokens ** 1.363 and each cached condition image adds a fifth of its tokens.""" | |
| n_images = len(image_paths or []) | |
| pixels = (int(width) * int(height)) if (width and height) else 1024 * 1024 | |
| prefix = n_images * (1024 // 16) ** 2 | |
| cached = kv_cache_fits(n_images) | |
| tokens = pixels / 256 + (0.2 * prefix if cached else prefix) | |
| per_step = 3.4e-6 * tokens ** 1.363 * 1.3 | |
| if negative_prompt: | |
| per_step *= 2 | |
| fixed = 4 + 2 * n_images + 1.5e-6 * pixels | |
| return int(min(300, (fixed + NUM_INFERENCE_STEPS * per_step) * 1.25)) | |
| def run_pipeline(prompt, image_paths, height, width, negative_prompt, seed): | |
| images = [Image.open(p) for p in image_paths] or None | |
| kwargs = {"use_kv_cache": kv_cache_fits(len(image_paths))} | |
| tile, stride = (1024, 768) if max(height or 0, width or 0) > 1536 else (1536, 1152) | |
| pipe.vae.enable_tiling( | |
| tile_sample_min_height=tile, | |
| tile_sample_min_width=tile, | |
| tile_sample_stride_height=stride, | |
| tile_sample_stride_width=stride, | |
| ) | |
| if negative_prompt: | |
| kwargs.update(negative_prompt=negative_prompt, true_cfg_scale=TRUE_CFG_SCALE) | |
| return pipe( | |
| prompt, | |
| image=images, | |
| height=height, | |
| width=width, | |
| num_inference_steps=NUM_INFERENCE_STEPS, | |
| generator=torch.Generator("cuda").manual_seed(int(seed)), | |
| **kwargs, | |
| ).images[0] | |
| def gallery_paths(input_images): | |
| """Save gallery images to PNG files so they can be sent to the rewriter Space and | |
| reopened in the GPU worker without loss (RGBA inputs keep their alpha).""" | |
| import tempfile | |
| paths = [] | |
| for item in input_images or []: | |
| try: | |
| if isinstance(item, (tuple, list)) and item: | |
| item = item[0] | |
| if isinstance(item, str): | |
| item = Image.open(item) | |
| elif hasattr(item, "name") and not isinstance(item, Image.Image): | |
| item = Image.open(item.name) | |
| if not isinstance(item, Image.Image): | |
| continue | |
| if item.mode not in ("RGB", "RGBA"): | |
| item = item.convert("RGBA") | |
| path = tempfile.NamedTemporaryFile(suffix=".png", delete=False).name | |
| item.save(path) | |
| paths.append(path) | |
| except Exception as e: | |
| print(f"[Warning] Failed to load input image: {e}") | |
| return paths | |
| # 初始化日志记录器 | |
| print(f"Initializing logger with directory: {LOG_DIR}") | |
| logger = GenerationLogger(log_dir=LOG_DIR) | |
| # --- UI Constants and Helpers --- | |
| MAX_SEED = np.iinfo(np.int32).max | |
| # --- Stage 1: screen and rewrite the prompt (CPU, no GPU quota) --- | |
| def prepare_stage(input_images, original_prompt, enable_extend, custom_size, quality, seed, randomize_seed): | |
| """ | |
| 校验输入、NCII 检查(仅在有输入图片时)、可选的提示词改写,全部在 GPU 之外完成。 | |
| Returns (image_paths, final_prompt, rewritten_prompt, seed, auto_height, auto_width). | |
| """ | |
| validate_image_count(input_images) | |
| if not original_prompt or not original_prompt.strip(): | |
| raise gr.Error("请输入提示词。 / Please enter a prompt.") | |
| image_paths = gallery_paths(input_images) | |
| if image_paths: | |
| check_prompt_guard(original_prompt) | |
| if randomize_seed: | |
| seed = random.randint(0, MAX_SEED) | |
| rewritten_prompt, wh_ratio = "", "" | |
| if enable_extend: | |
| try: | |
| rewritten_prompt, wh_ratio = enhance_prompt(original_prompt, image_paths) | |
| except Exception as e: | |
| print(f"[Warning] Prompt enhancement failed: {e!r}") | |
| gr.Warning("提示词改写暂不可用,已使用原始提示词。 / Prompt enhancement is unavailable; using the original prompt.") | |
| auto_height, auto_width = (None, None) | |
| if not custom_size: | |
| area = QUALITY_RESOLUTIONS.get(quality, 1024) ** 2 | |
| if image_paths: | |
| last_width, last_height = Image.open(image_paths[-1]).size | |
| auto_height, auto_width = aspect_to_size(last_width / last_height, area) | |
| else: | |
| auto_height, auto_width = ratio_to_size(wh_ratio, area) | |
| if auto_height is None: | |
| auto_height, auto_width = aspect_to_size(1.0, area) | |
| return image_paths, rewritten_prompt or original_prompt, rewritten_prompt, seed, auto_height, auto_width | |
| # --- Stage 2: Image Generation Function --- | |
| def generate_image_stage( | |
| image_paths, | |
| original_prompt, | |
| final_prompt, | |
| custom_size, | |
| log_dir, | |
| seed, | |
| height, | |
| width, | |
| auto_height, | |
| auto_width, | |
| negative_prompt=" ", | |
| prompt_extend=True, | |
| username=None, | |
| ): | |
| """ | |
| 根据是否有输入图片自动判断模式:有图片 -> Edit,无图片 -> T2I。 | |
| """ | |
| # 更新日志目录(如果用户修改了) | |
| if log_dir and log_dir.strip(): | |
| logger.set_log_dir(log_dir.strip()) | |
| negative_prompt = (negative_prompt or "").strip() | |
| is_edit_mode = len(image_paths) > 0 | |
| mode = "edit" if is_edit_mode else "t2i" | |
| # 构造 size:开启自定义尺寸时使用设置值,否则由改写模型推荐的比例或模型自动决定 | |
| if custom_size: | |
| out_height, out_width = int(height), int(width) | |
| else: | |
| out_height, out_width = auto_height, auto_width | |
| print(f"Mode: {mode}, prompt extend: {prompt_extend}, input images: {len(image_paths)}, " | |
| f"seed: {seed}, size: {f'{out_width}x{out_height}' if out_width else 'auto'}") | |
| image = run_pipeline(final_prompt, image_paths, out_height, out_width, negative_prompt, seed) | |
| # 记录生成日志 | |
| params = { | |
| "mode": mode, | |
| "height": height, | |
| "width": width, | |
| "negative_prompt": negative_prompt, | |
| "prompt_extend": prompt_extend, | |
| "input_images_count": len(image_paths), | |
| "model": MODEL_ID, | |
| "num_inference_steps": NUM_INFERENCE_STEPS, | |
| } | |
| log_name = logger.log_generation( | |
| original_prompt=original_prompt, | |
| enhanced_prompt=final_prompt, | |
| image=image, | |
| seed=seed, | |
| params=params, | |
| gpu_id=None, | |
| input_images=[Image.open(p) for p in image_paths] or None, | |
| username=username, | |
| ) | |
| print(f"Generation complete, logged as: {log_name}") | |
| return image | |
| def make_placeholder_image(text, width=512, height=320, bg_color=(30, 30, 30), text_color=(200, 200, 200)): | |
| """生成一张带有居中文字的占位图片""" | |
| from PIL import ImageDraw, ImageFont | |
| img = Image.new("RGBA", (width, height), (*bg_color, 255)) | |
| draw = ImageDraw.Draw(img) | |
| try: | |
| font = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", 28) | |
| except Exception: | |
| font = ImageFont.load_default() | |
| bbox = draw.textbbox((0, 0), text, font=font) | |
| tw, th = bbox[2] - bbox[0], bbox[3] - bbox[1] | |
| x = (width - tw) // 2 | |
| y = (height - th) // 2 | |
| draw.text((x, y), text, fill=text_color, font=font) | |
| return img | |
| # --- Two-step flow: prepare (CPU) then generate (GPU) --- | |
| def prepare_request(input_images, original_prompt, enable_extend, custom_size, quality, seed, randomize_seed): | |
| image_paths, final_prompt, rewritten_prompt, seed, auto_height, auto_width = prepare_stage( | |
| input_images, original_prompt, enable_extend, custom_size, quality, seed, randomize_seed, | |
| ) | |
| request_state = { | |
| "image_paths": image_paths, | |
| "final_prompt": final_prompt, | |
| "auto_height": auto_height, | |
| "auto_width": auto_width, | |
| } | |
| return make_placeholder_image("Generating image..."), seed, rewritten_prompt, request_state | |
| def generate_request( | |
| request_state, | |
| original_prompt, | |
| enable_extend, | |
| custom_size, | |
| log_dir, | |
| seed, | |
| height, | |
| width, | |
| negative_prompt, | |
| request: gr.Request, | |
| progress=gr.Progress(track_tqdm=True), | |
| ): | |
| """ | |
| 生成图片。提示词改写已在上一步完成(开启时)。 | |
| """ | |
| if not request_state: | |
| raise gr.Error("请重新点击生成。 / Please click generate again.") | |
| username = request.username if request else None | |
| print(f"Request from user: {username}") | |
| return generate_image_stage( | |
| request_state["image_paths"], original_prompt, request_state["final_prompt"], | |
| custom_size, log_dir, seed, height, width, | |
| request_state["auto_height"], request_state["auto_width"], | |
| negative_prompt=negative_prompt, | |
| prompt_extend=enable_extend, | |
| username=username, | |
| ) | |
| def make_example_loader(images, text, extend): | |
| """Bind each example to a zero-argument callback without late binding.""" | |
| def load_example(): | |
| return list(images), text, extend | |
| return load_example | |
| # UI translations are registered once; callbacks update only the current session. | |
| localized_components = [] | |
| localized_properties = [] | |
| def localize(component, **properties): | |
| localized_components.append(component) | |
| localized_properties.append(properties) | |
| for name, values in properties.items(): | |
| setattr(component, name, values[1 if DEFAULT_LANGUAGE == "en" else 0]) | |
| return component | |
| def switch_language(language): | |
| index = 1 if language == "en" else 0 | |
| return [gr.update(**{name: values[index] for name, values in props.items()}) | |
| for props in localized_properties] | |
| EXAMPLE_TITLES_EN = { | |
| "电影角色三视图": "Film character turnaround", | |
| "小羊肖恩生日故事": "Shaun the Sheep birthday story", | |
| "民国漫剧分镜": "Period drama storyboard", | |
| "祈年殿拆解科普卡": "Temple architecture infographic", | |
| "AI 进化时间轴": "AI evolution timeline", | |
| "唇膏广告分镜": "Lipstick commercial storyboard", | |
| "巨兽对决故事板": "Giant creature battle storyboard", | |
| } | |
| ENGLISH_EXAMPLES = [ | |
| ("Editorial portrait", "Create an editorial portrait of a botanist in a sunlit greenhouse, surrounded by ferns and delicate orchids. Natural skin texture, linen clothing, soft morning backlight, subtle film grain, medium-format photography, calm expression, no text or watermark."), | |
| ("Typography poster", 'Design a refined travel poster for a fictional night train. Render the headline exactly as "THE MIDNIGHT EXPRESS" and the subtitle "A journey under the stars". A silver train curves through dark blue mountains beneath a crescent moon. Art Deco geometry, ivory and gold lettering, clear typographic hierarchy, generous margins, print-ready composition.'), | |
| ("Six-panel storyboard", "Create a six-panel cinematic storyboard about a small robot restoring an abandoned rooftop garden. Show: arrival at dawn, discovery of a dried seedling, repairing an irrigation pipe, planting new seeds, the first rain, and a lush garden at sunset. Keep the robot's round yellow body and blue eyes consistent in every panel. Clear panel borders, expressive visual storytelling, detailed environments, no captions."), | |
| ("Product photography", "Photograph a translucent emerald perfume bottle on pale limestone beside a shallow pool. Rippling sunlight reflects through the glass onto the stone. A single olive branch frames the upper left corner. Luxury product photography, realistic refraction, crisp bottle edges, soft shadows, uncluttered composition, no logo or text."), | |
| ] | |
| # --- Examples and UI Layout --- | |
| EXAMPLE_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "examples") | |
| with open(os.path.join(EXAMPLE_DIR, "cases.json"), encoding="utf-8") as case_file: | |
| DEMO_CASES = json.load(case_file) | |
| with open(os.path.join(EXAMPLE_DIR, "generated_references.json"), encoding="utf-8") as reference_file: | |
| GENERATED_REFERENCES = {case["prompt"]: case for case in json.load(reference_file)} | |
| def add_generated_reference(prompt): | |
| reference = GENERATED_REFERENCES[prompt] | |
| localize(gr.Gallery( | |
| value=[os.path.join(EXAMPLE_DIR, name) for name in reference["outputs"]], | |
| columns=1, height=280, interactive=False, | |
| ), label=("参考结果", "Reference output")) | |
| def add_case_examples(cases, input_images, prompt, enable_extend): | |
| """Show bundled reference images and load full prompts in their original order.""" | |
| for case in cases: | |
| paths = [os.path.join(EXAMPLE_DIR, name) for name in case["inputs"]] | |
| references = [os.path.join(EXAMPLE_DIR, name) for name in case["outputs"]] | |
| with gr.Accordion(case["title"], open=False) as panel: | |
| localize(panel, label=(case["title"], case["title_en"])) | |
| with gr.Row(): | |
| if paths: | |
| localize(gr.Gallery( | |
| value=[(path, str(index)) for index, path in enumerate(paths, 1)], | |
| columns=min(len(paths), 5), height=280, interactive=False, | |
| ), label=("输入参考图(按编号顺序)", "Input references (in numbered order)")) | |
| if references: | |
| localize(gr.Gallery(value=references, columns=1, height=280, interactive=False), | |
| label=("参考结果", "Reference output")) | |
| localize(gr.Textbox(value=case["prompt"], lines=4, max_lines=20, interactive=False), | |
| label=("完整提示词", "Full prompt")) | |
| button = localize(gr.Button(), value=("使用此示例", "Use this example")) | |
| button.click( | |
| fn=make_example_loader(paths, case["prompt"], case["prompt_extend"]), | |
| inputs=[], outputs=[input_images, prompt, enable_extend], queue=False, | |
| ) | |
| css = """ | |
| #col-container { | |
| margin: 0 auto; | |
| max-width: 1024px; | |
| } | |
| #edit_text{margin-top: -62px !important} | |
| """ | |
| with gr.Blocks(title="Qwen Image 2.1 Demo") as demo: | |
| with gr.Column(elem_id="col-container"): | |
| gr.HTML('<a href="https://huggingface.co/Qwen/Qwen-Image-2.1" target="_blank"><img src="https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen-Image/image2.1/logo.png" alt="Qwen-Image Logo" width="400" style="display: block; margin: 0 auto;"></a>') | |
| language = gr.Radio(choices=[("中文", "zh"), ("English", "en")], value=DEFAULT_LANGUAGE, label="语言 / Language") | |
| instructions = gr.Markdown(""" | |
| ## Qwen Image 2.1 Demo 使用说明 | |
| 1. 如果不输入图片,默认进入文生图模式;如果输入图片,则进入图像编辑模式(支持1-10张图片)。 | |
| 2. 默认"Enable Prompt Extend"为开启状态,会自动进行提示词智能改写。如果希望直接使用原始提示词,可以关闭该选项。 | |
| 3. 下方提供了包括文生图,图生图的测试样例,可以作为模型基础能力的参考。 | |
| 4. 生成透明图时,建议提示词遵循以下格式,将中间的省略号替换为具体画面描述: | |
| `这是一张带有透明度的RGBA图像。.... 该图像具有alpha通道,背景是透明的。` | |
| """) | |
| with gr.Row(): | |
| with gr.Column(): | |
| input_images = gr.Gallery( | |
| label="Input Images (for editing)", | |
| show_label=True, | |
| type="pil", | |
| interactive=True, | |
| columns=5, | |
| height="auto" | |
| ) | |
| with gr.Column(): | |
| result = gr.Image(label="Result", show_label=False, type="pil", image_mode="RGBA", format='png') | |
| # Prompt Input | |
| with gr.Row(): | |
| prompt = gr.Text( | |
| label="Prompt", | |
| lines=6, | |
| max_lines=24, | |
| show_label=True, | |
| placeholder="Describe what you want to generate or edit...", | |
| container=True, | |
| ) | |
| with gr.Row(): | |
| enable_extend = gr.Checkbox( | |
| label="Enable Prompt Extend (提示词智能改写)", | |
| value=True, | |
| ) | |
| quality = gr.Radio( | |
| choices=[("Speed (1024px)", "speed"), ("Quality (2048px)", "quality")], | |
| value="speed", | |
| show_label=False, | |
| ) | |
| generate_button = gr.Button("Generate Image", variant="primary") | |
| rewritten_prompt_output = localize( | |
| gr.Textbox(value="", lines=4, max_lines=20, interactive=False), | |
| label=("改写结果", "Rewritten prompt"), | |
| placeholder=("开启智能改写并生成图片后,改写后的提示词会显示在这里。", | |
| "After generation with prompt enhancement enabled, the rewritten prompt appears here."), | |
| ) | |
| enable_extend.change(fn=lambda: "", inputs=[], outputs=[rewritten_prompt_output], queue=False) | |
| with gr.Accordion("Advanced Settings", open=False) as advanced: | |
| # 日志目录配置 | |
| log_dir_input = gr.Textbox( | |
| label="Log Directory (日志保存目录)", | |
| show_label=True, | |
| value=LOG_DIR, | |
| placeholder="输入日志保存目录路径,例如: ./generation_logs", | |
| interactive=True, | |
| visible=False, | |
| ) | |
| seed = gr.Slider( | |
| label="Seed", | |
| minimum=0, | |
| maximum=MAX_SEED, | |
| step=1, | |
| value=0, | |
| ) | |
| randomize_seed = gr.Checkbox(label="Randomize seed", value=True) | |
| negative_prompt_input = gr.Textbox( | |
| label="Negative Prompt (负向提示词)", | |
| value=" ", | |
| placeholder="输入不希望出现的内容", | |
| lines=2, | |
| interactive=True, | |
| ) | |
| # 自定义输出尺寸 | |
| custom_size = gr.Checkbox( | |
| label="自定义输出尺寸 (Customize output size)", | |
| value=False, | |
| info="关闭时由模型自动决定输出尺寸;开启后使用下方分辨率设置", | |
| ) | |
| # 分辨率预设选择 | |
| resolution_help = gr.Markdown("**分辨率设置 (Resolution Settings) — 需开启上方自定义尺寸选项**") | |
| with gr.Row(): | |
| resolution_preset = gr.Dropdown( | |
| choices=list(ALL_SIZE_PRESETS.keys()), | |
| value="2K - 2688x1536 (16:9)", | |
| label="预设分辨率 (Resolution Preset)", | |
| info="选择预设分辨率或自定义", | |
| ) | |
| with gr.Row(): | |
| width = gr.Slider( | |
| label="Width", | |
| minimum=256, | |
| maximum=2688, | |
| step=8, | |
| value=2688, | |
| ) | |
| height = gr.Slider( | |
| label="Height", | |
| minimum=256, | |
| maximum=2688, | |
| step=8, | |
| value=1536, | |
| ) | |
| # 分辨率预设回调函数 | |
| def update_resolution(preset_name): | |
| if preset_name == "自定义 (Custom)" or ALL_SIZE_PRESETS.get(preset_name) is None: | |
| # 自定义模式,不改变当前值 | |
| return gr.update(), gr.update() | |
| w, h = ALL_SIZE_PRESETS[preset_name] | |
| return w, h | |
| resolution_preset.change( | |
| fn=update_resolution, | |
| inputs=[resolution_preset], | |
| outputs=[width, height], | |
| ) | |
| # --- Examples --- | |
| examples_heading = gr.Markdown("### 示例") | |
| localize(gr.Markdown(), value=("**精选文生图示例**", "**Featured text-to-image examples**")) | |
| add_case_examples([case for case in DEMO_CASES if not case["inputs"]], | |
| input_images, prompt, enable_extend) | |
| edit_heading = gr.Markdown("**图像编辑示例**") | |
| add_case_examples([case for case in DEMO_CASES if case["inputs"]], | |
| input_images, prompt, enable_extend) | |
| # T2I examples (no input images) | |
| t2i_heading = gr.Markdown("**中文文生图示例**") | |
| t2i_examples = [ | |
| ["""生成一张真人电影质感的单角色三视图设定参考图。干净灰色背景,左侧为一张较大的完整头肩半身细节图,必须是同一角色从头部到肩部/胸口的连续完整特写,清楚展示面部、发型、上身服装、配饰和材质细节;左侧不要拆成多个局部小图、不要拼贴多个细节框、不要只给眼睛/衣料/配饰等碎片特写。右侧展示同一角色正面、侧面、背面全身三视图,全身可见。角色设定如下:丹尼尔·斯通,纯白背景,无其他人物和场景;丹尼尔·斯通的单人全身立绘,正对镜头站立,表情自然;30岁青年男性,2020年代现代美国,欧洲裔白人外貌,浅肤色,身高约185cm,9头身,魁梧健壮体型,肌肉发达,2020年代棕色短寸头,发质粗硬,眼窝深陷,瞳孔呈灰蓝色,鼻梁高挺且鼻头宽大,唇形厚实且唇色苍白,方下颌轮廓分明,面部皮肤粗糙并带有污垢痕迹,穿着2020年代脏污灰色连帽卫衣配同色系工装裤,脚穿2020年代黑色防滑工装靴,双手佩戴破旧皮革手套,双手自然下垂;光照均匀,高画质,手部完美,无文字水印。 三视图中面部特征、身形比例、发型、服装、配饰、鞋履、姿态和关键视觉元素保持一致;表情自然中性,站姿稳定,比例统一,皮肤和布料金属等材质细节清楚,电影级写实光照,高画质,无文字、无logo、无水印。""", True], | |
| ["""3D粘土漫画,生成小羊肖恩\n镜头一:清晨的青苔底农场暖意融融,小羊肖恩发现农夫正在精心装饰巧克力生日蛋糕。镜头二:随即心生主意,独自带着小羊提米靠近厨房。镜头三:它悄悄避开熟睡的牧羊犬,带着提米溜进屋内。镜头四:不料提米意外打翻面粉、蹭损蛋糕,场面十分狼狈。镜头五:关键时刻肖恩灵机一动,伸出手指,在厚厚的白色粉末上流畅地画了起来。\n首先是一个巨大的、歪歪扭拙却充满爱意的爱心;接着是"HAPPY BIRTHDAY"的字样。随后,他指挥提米将散落的鲜红草莓摆放在爱心周围,又将从窗外采摘的野花(雏菊、矢车菊)插在蛋糕受损的边缘,巧妙地遮盖了瑕疵。原本狼藉的桌面,瞬间变成了一幅质朴而温馨的田园画作。肖恩退后一步,满意地拍了拍手上的面粉,脸上露出自豪的微笑。镜头六:农夫归来后,看见这份质朴又温馨的布置开怀大笑,这场小小的意外,最终变成了农场治愈又暖心的生日惊喜。""", True], | |
| ["""一组连贯短视频故事板分镜图,竖版漫剧短剧叙事节奏,2.5D 拟真人细腻质感,民国新旧交替时代背景,主角是清冷民国女学生,配儒雅留洋青年配角,镜头遵循全景 - 中景 - 近景 - 特写顺序;镜头内容: 老街全景雨雾、 二人撑伞对视中景、女生垂眸特写、青年递信手部特写、 巷口转身远景、灯下拆信近景、含泪侧脸特写、留白空镜;每格底部标注简短剧情字幕,画面低饱和复古色调,柔和分层柔光,人物五官写实精致,布料纹理清晰,无杂乱水印,分镜边框规整,镜头衔接流畅,叙事紧凑适配短视频,整体光影层次丰富,全程统一 2.5D 拟真人画风,人物形象全程不崩,场景道具贴合民国时代设定""", True], | |
| ["""竖版祈年殿拆解科普卡,纯白色宣纸底色,左竖排书法"祈年殿"、小字天坛与红印章,附线稿和简介;主体从上到下分层拆解全建筑构件,右侧虚线序号标注名称,中式手绘,配色棕、深蓝、朱红、米白。""", True], | |
| ["""生成一张计算机/技术主题的"AI进化历程"时间演化卡,采用线性顺序的时间轴/编年结构,风格为手绘/涂鸦。背景用大地棕褐色系单色渐变,搭配手绘涂鸦元素与统一科普插图。版式采用卡片网格、上图下文,图文留白清晰。标题用粗黑标题体,正文用中文手写体,重点文字采用渐变色填充字与双色调文字,文字信息整体规整易读,完整呈现AI发展节点。""", True], | |
| ["""生成一张原创高端美妆广告分镜展示板,主题为"LUNE 丝绒唇膏 15 秒广告脚本"。整体像真实品牌团队制作的商业广告脚本视觉板,包含 5 个连续分镜画面 + 时间轴 + 字幕 / 台词 + 镜头说明 + 转场提示。排版清晰、统一、精致、高级,具有女性向、轻奢、电影感、美妆广告质感。主色调为 奶油白、黑色、酒红、玫瑰豆沙、暖金色,灯光柔和,肤质细腻,材质高级。""", True], | |
| ["""核心大纲 1. 场景:3DCG质感,废弃都市遗迹(黄昏),尘土漫天,绯红色夕阳,地面有巨大脚印。 2. 人物:炎狱巨蜥(熔岩鳞片、喷烈焰)、霜牙巨象(雪白、喷寒气)、幸存人类(远景蜷缩)。 3. 剧情脉络: (1)对峙(0-20秒):两只巨兽在废墟对峙,人类蜷缩避险,氛围压迫。 (2)开战(21-50秒):炎狱巨蜥喷烈焰,霜牙巨象喷寒气对抗,引发冲击,高楼坍塌。 (3)激战(51-90秒):双方互相攻击,均受重伤,嘶吼震彻废墟。 (4)落幕(91-110秒):两只巨兽重伤对峙,镜头拉远,定格巨兽与废墟,留下悬念""", True], | |
| ] | |
| for example_title, example in zip(['电影角色三视图', '小羊肖恩生日故事', '民国漫剧分镜', '祈年殿拆解科普卡', 'AI 进化时间轴', '唇膏广告分镜', '巨兽对决故事板'], t2i_examples): | |
| with gr.Accordion(example_title, open=False) as example_panel: | |
| localize(example_panel, label=(example_title, EXAMPLE_TITLES_EN[example_title])) | |
| add_generated_reference(example[0]) | |
| gr.Markdown(example[0]) | |
| example_button = localize(gr.Button(), value=("使用此提示词", "Use this prompt")) | |
| example_button.click( | |
| fn=make_example_loader([], example[0], example[1]), | |
| inputs=[], | |
| outputs=[input_images, prompt, enable_extend], | |
| queue=False, | |
| ) | |
| english_heading = localize(gr.Markdown(), value=("**英文文生图示例**", "**English text-to-image examples**")) | |
| for title, text in ENGLISH_EXAMPLES: | |
| with gr.Accordion(title, open=False): | |
| add_generated_reference(text) | |
| gr.Markdown(text) | |
| button = localize(gr.Button(), value=("使用此提示词", "Use this prompt")) | |
| button.click(fn=make_example_loader([], text, True), inputs=[], | |
| outputs=[input_images, prompt, enable_extend], queue=False) | |
| localize(instructions, value=(instructions.value, """## Qwen Image 2.1 Demo | |
| 1. Generate from text without an input image, or upload 1–10 images to edit or combine them. Refer to images by their upload order in your prompt. | |
| 2. Prompt enhancement is enabled by default. Turn it off to use your original prompt directly. | |
| 3. Expand an example to read its full prompt, then click its button to fill the input. Chinese and English prompts are both supported. | |
| 4. For transparent image generation, use the following prompt format and replace `xxxxx` with your image description: | |
| `This is an RGBA image with transparency. xxxxx The image has alpha channel and the background is transparent.` | |
| """)) | |
| localize(input_images, label=("输入图片(编辑用,最多 10 张)", "Input images (editing, up to 10)")) | |
| localize(result, label=("生成结果", "Result")) | |
| localize(prompt, label=("提示词", "Prompt"), placeholder=("描述想生成或编辑的内容,可按上传顺序引用第 1–10 张图…", "Describe what to generate or edit; refer to images 1–10 in upload order…")) | |
| localize(enable_extend, label=("智能改写提示词", "Enhance prompt")) | |
| localize(quality, choices=( | |
| [("速度 (1024px)", "speed"), ("质量 (2048px)", "quality")], | |
| [("Speed (1024px)", "speed"), ("Quality (2048px)", "quality")])) | |
| localize(generate_button, value=("生成图片", "Generate image")) | |
| localize(advanced, label=("高级设置", "Advanced settings")) | |
| localize(log_dir_input, label=("日志保存目录", "Log directory"), placeholder=("输入日志保存目录", "Enter a log directory")) | |
| localize(seed, label=("随机种子", "Seed")) | |
| localize(randomize_seed, label=("随机种子自动变化", "Randomize seed")) | |
| localize(negative_prompt_input, label=("负向提示词", "Negative prompt"), placeholder=("输入不希望出现的内容", "Describe what to exclude")) | |
| localize(custom_size, label=("自定义输出尺寸", "Customize output size"), info=("关闭时自动选择尺寸;开启后使用下方设置", "When off, size is chosen automatically. Enable to use the settings below.")) | |
| localize(resolution_help, value=("**分辨率设置 — 需开启自定义尺寸**", "**Resolution settings — enable custom output size first**")) | |
| preset_values = list(ALL_SIZE_PRESETS) | |
| localize(resolution_preset, label=("预设分辨率", "Resolution preset"), info=("选择预设分辨率或自定义", "Choose a preset or custom dimensions"), choices=( | |
| [("自定义" if ALL_SIZE_PRESETS[v] is None else v, v) for v in preset_values], | |
| [("Custom" if ALL_SIZE_PRESETS[v] is None else v, v) for v in preset_values])) | |
| localize(width, label=("宽度", "Width")) | |
| localize(height, label=("高度", "Height")) | |
| localize(examples_heading, value=("### 示例", "### Examples")) | |
| localize(t2i_heading, value=("**中文文生图示例**", "**Chinese text-to-image examples**")) | |
| localize(edit_heading, value=("**图像编辑示例**", "**Image-editing examples**")) | |
| language.change(fn=switch_language, inputs=[language], outputs=localized_components, queue=False) | |
| input_images.upload(fn=validate_image_count, inputs=[input_images], outputs=[], queue=False) | |
| # Generate Image button event: screen/rewrite the prompt off-GPU, then generate. | |
| # Two events so the GPU step is scheduled with a fresh ZeroGPU token after a long rewrite. | |
| request_state = gr.State(None) | |
| generate_button.click( | |
| fn=prepare_request, | |
| inputs=[ | |
| input_images, # input_images (有图片->Edit模式, 无图片->T2I模式) | |
| prompt, # original_prompt | |
| enable_extend, # enable_extend (是否开启提示词改写) | |
| custom_size, # custom_size (是否自定义输出尺寸) | |
| quality, | |
| seed, | |
| randomize_seed, | |
| ], | |
| outputs=[result, seed, rewritten_prompt_output, request_state], | |
| concurrency_limit=2, | |
| ).success( | |
| fn=generate_request, | |
| inputs=[ | |
| request_state, | |
| prompt, | |
| enable_extend, | |
| custom_size, | |
| log_dir_input, # log_dir | |
| seed, | |
| height, | |
| width, | |
| negative_prompt_input, # negative_prompt (负向提示词) | |
| ], | |
| outputs=[result], | |
| concurrency_limit=2, | |
| ) | |
| if __name__ == "__main__": | |
| # 使用 os.path.realpath 解析真实路径,避免 NAS 软链接导致 Gradio 文件权限检查失败(404) | |
| _script_dir = os.path.realpath(os.path.dirname(os.path.abspath(__file__))) | |
| _allowed = [ | |
| os.path.join(_script_dir, "examples"), | |
| os.path.realpath(EXAMPLE_DIR), | |
| ] | |
| demo.launch( | |
| server_name="0.0.0.0", | |
| allowed_paths=_allowed, | |
| css=css, | |
| ) | |