"""Generate 5 fixed prompts with Qwen-Image-2.1 OpenVINO INT4. Config follows official model card (QwenImage21Pipeline example): num_inference_steps=40, true_cfg_scale=1.0 (default, no guidance), height=1024, width=1024 (task requirement; example uses 2048). Callback via callback_on_step_end records per-step time + psutil RSS. Outputs ./outputs/{name}_1024.png + {name}_512.png, plus benchmark.json + prompts.txt with CPU/OpenVINO info. """ import os import time import json import platform import subprocess import psutil import torch from pathlib import Path from datetime import datetime from optimum.intel import OVDiffusionPipeline MODEL_ID = "HelloSun/Qwen-Image-2.1-OpenVINO-INT4" OUTPUT_DIR = "./outputs" NUM_STEPS = 40 HEIGHT = 1024 WIDTH = 1024 class StepCallback: """Per-step timer via diffusers callback_on_step_end API.""" def __init__(self): self.step_times = [] self.memory_usage = [] self.step_start_time = None def __call__(self, pipe, step: int, timestep: int, callback_kwargs: dict): now = time.time() if self.step_start_time is not None: self.step_times.append(now - self.step_start_time) proc = psutil.Process(os.getpid()) self.memory_usage.append(proc.memory_info().rss / 1024 / 1024) self.step_start_time = now return callback_kwargs def get_cpu_info(): try: lscpu = subprocess.run(["lscpu"], capture_output=True, text=True, timeout=10).stdout except Exception as e: lscpu = f"lscpu unavailable: {e}" try: import openvino ov_ver = openvino.__version__ except Exception as e: ov_ver = f"unknown ({e})" return { "lscpu": lscpu, "cpu_count_logical": os.cpu_count(), "cpu_count_psutil": psutil.cpu_count(logical=True), "cpu_count_physical": psutil.cpu_count(logical=False), "platform": platform.platform(), "processor": platform.processor(), "openvino_version": ov_ver, } def generate_images(): os.makedirs(OUTPUT_DIR, exist_ok=True) cpu_info = get_cpu_info() print(f"CPU logical: {cpu_info['cpu_count_logical']}, OpenVINO: {cpu_info['openvino_version']}") print(f"Loading INT4 model from {MODEL_ID}...") load_start = time.time() pipeline = OVDiffusionPipeline.from_pretrained(MODEL_ID, compile=True) load_time = time.time() - load_start print(f"Model loaded and compiled in {load_time:.2f}s") prompts = { "01_hanfu": { "prompt": "Young Chinese woman in red Hanfu, intricate embroidery, impeccable makeup, red floral forehead pattern, elaborate high bun, golden phoenix headdress, soft-lit outdoor night background, silhouetted tiered pagoda, blurred colorful distant lights, photorealistic, ultra detailed, 8k", "seed": 42, }, "02_astronaut": { "prompt": "Astronaut in a jungle, cold color palette, muted colors, detailed, 8k, photorealistic, cinematic lighting", "seed": 43, }, "03_taipei": { "prompt": "Cyberpunk street in Taipei at night, heavy rain, neon signs with text 'TAIPEI' and Chinese characters '台北', reflections on wet asphalt, crowded night market, cinematic, ultra detailed", "seed": 44, }, "04_shiba": { "prompt": "Cute Shiba Inu wearing a tiny astronaut helmet, sitting in a field of sunflowers under a starry sky, dreamy illustration, vibrant colors, high quality", "seed": 45, }, "05_ink": { "prompt": "Traditional Chinese ink wash landscape, misty mountains, a small pagoda on a cliff, cranes flying, minimalist, elegant, high aesthetic quality", "seed": 46, }, } results = [] for name, cfg in prompts.items(): print(f"\nGenerating {name} (seed={cfg['seed']})...") callback = StepCallback() generator = torch.Generator().manual_seed(cfg["seed"]) gen_start = time.time() image = pipeline( prompt=cfg["prompt"], num_inference_steps=NUM_STEPS, height=HEIGHT, width=WIDTH, generator=generator, callback_on_step_end=callback, callback_on_step_end_tensor_inputs=["latents"], ).images[0] gen_time = time.time() - gen_start out1024 = Path(OUTPUT_DIR) / f"{name}_1024.png" image.save(out1024) out512 = Path(OUTPUT_DIR) / f"{name}_512.png" image.resize((512, 512), resample=3).save(out512) print(f"Saved {out1024} + {out512}") st = callback.step_times mu = callback.memory_usage r = { "name": name, "prompt": cfg["prompt"], "seed": cfg["seed"], "resolution": f"{HEIGHT}x{WIDTH}", "num_inference_steps": NUM_STEPS, "true_cfg_scale": 1.0, "total_generation_time": gen_time, "load_compile_time": load_time, "step_times": st, "avg_step_time": sum(st) / len(st) if st else 0, "memory_usage_mb": mu, "peak_memory_mb": max(mu) if mu else 0, "avg_memory_mb": sum(mu) / len(mu) if mu else 0, "output_1024": str(out1024), "output_512": str(out512), } results.append(r) print(f" gen {gen_time:.2f}s avg_step {r['avg_step_time']:.4f}s peakRSS {r['peak_memory_mb']:.1f}MB") proc = psutil.Process(os.getpid()) rss_mb = proc.memory_info().rss / 1024 / 1024 bench = { "model": "Qwen/Qwen-Image-2.1", "ov_model_dir": MODEL_ID, "quantization": "INT4 transformer+text_encoder(+text_encoder_i2i) (bits=4,sym=False,group_size=128,fallback=adjust,ratio=1.0), rest INT8", "openvino_version": cpu_info["openvino_version"], "cpu": { "logical": cpu_info["cpu_count_logical"], "psutil_logical": cpu_info["cpu_count_psutil"], "physical": cpu_info["cpu_count_physical"], "lscpu": cpu_info["lscpu"], }, "generation_config": { "num_inference_steps": NUM_STEPS, "true_cfg_scale": 1.0, "height": HEIGHT, "width": WIDTH, }, "load_compile_time_seconds": load_time, "final_process_rss_mb": rss_mb, "results": results, "timestamp": datetime.now().isoformat(), } with open(f"{OUTPUT_DIR}/benchmark.json", "w") as f: json.dump(bench, f, indent=2) with open(f"{OUTPUT_DIR}/prompts.txt", "w") as f: for r in results: f.write(f"{r['name']} (seed={r['seed']}, steps={r['num_inference_steps']}): {r['prompt']}\n\n") print("\nDone. benchmark.json + prompts.txt written.") return bench if __name__ == "__main__": generate_images()