#!/usr/bin/env python3 """zen-image-edit inference: no native text encoder (Qwen3-VL-8B, 17.5 GB) anywhere. On the GPU: Qwen3.5-0.8B (~1.7 GB), the DiT with the adapter inside (~14.5 GB) and the VAE (~1.4 GB, fp32). # text-to-image python example.py --prompt "a red fox in a snowy forest at dusk, cinematic, 85mm" --out fox.png # editing: 1..N condition images, referenced in the prompt by TAG , , ... python example.py --image ref.png scene.png \ --prompt "Replace the woman in with the woman from ; keep pose, \\ clothing and background unchanged." --out swap.png """ import argparse import os import sys import torch from PIL import Image as PILImage sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from pipeline import ZenImageEditPipeline # noqa: E402 HERE = os.path.dirname(os.path.abspath(__file__)) def main(): ap = argparse.ArgumentParser(description="Qwen-Image-2.1 with Qwen3.5-0.8B and the adapter inside the DiT") ap.add_argument("--prompt", required=True) ap.add_argument("--image", nargs="*", default=[], help="condition images, order = , , ...") ap.add_argument("--out", default="out.png") ap.add_argument("--model", default=HERE, help="model folder (the layout shipped in this repo)") ap.add_argument("--size", type=int, default=1024, help="output_resolution (frame side)") ap.add_argument("--steps", type=int, default=30) ap.add_argument("--seed", type=int, default=1234) ap.add_argument("--device", default="cuda") ap.add_argument("--no-offload", action="store_true", help="keep every component on the device (needs a large GPU)") args = ap.parse_args() pipe = ZenImageEditPipeline.from_pretrained(args.model, dtype=torch.float16) pipe.set_progress_bar_config(disable=True) # Phase-by-phase offload by default: the 14.5 GB fp16 DiT and the fp32 VAE decoder do not fit # an 32 GB card at the same time. Keeping everything resident needs roughly 40 GB. if args.device.startswith("cuda") and not args.no_offload: pipe.enable_model_cpu_offload(device=args.device) else: pipe.to(args.device) generator = torch.Generator(args.device).manual_seed(args.seed) condition = [PILImage.open(path) for path in args.image] or None if condition and len(condition) > 1 and "1 the prompt must reference , , ...", flush=True) image = pipe(prompt=args.prompt, image=condition, output_resolution=args.size, num_inference_steps=args.steps, true_cfg_scale=1.0, generator=generator, output_type="pil").images[0] image.save(args.out) print(f"{args.size}px, {args.steps} steps, seed {args.seed} -> {args.out} {image.size}", flush=True) if __name__ == "__main__": main()