"""Generate the 5 fixed-seed 1024x1024 examples plus the 512px control set, recording per-step latency and process RSS via a diffusers callback. For FLUX.1-schnell (1-4 steps, CFG 0.0, max_sequence_length=256). Outputs -> /home/user/app/outputs_flux/ and a machine readable log in /home/user/app/outputs_flux/benchmark.json """ import argparse import json import os import platform import statistics import time from pathlib import Path import psutil import torch from optimum.intel import OVFluxPipeline MODEL_PATH = "/home/user/app/flux-schnell-ov-int4" OUTPUT_DIR = Path("/home/user/app/outputs_flux") NEGATIVE_PROMPT = "" # name, seed, prompt PROMPTS = [ ( "01_hanfu", 42, "Young Chinese woman in red Hanfu, intricate embroidery, impeccable makeup, " "red floral forehead pattern, elaborate high bun, golden phoenix headdress, " "soft-lit outdoor night background, silhouetted tiered pagoda, blurred colorful " "distant lights, photorealistic, ultra detailed, 8k", ), ( "02_astronaut", 43, "Astronaut in a jungle, cold color palette, muted colors, detailed, 8k, " "photorealistic, cinematic lighting", ), ( "03_taipei", 44, "Cyberpunk street in Taipei at night, heavy rain, neon signs with text 'TAIPEI' " "and Chinese characters '台北', reflections on wet asphalt, crowded night market, " "cinematic, ultra detailed", ), ( "04_shiba", 45, "Cute Shiba Inu wearing a tiny astronaut helmet, sitting in a field of sunflowers " "under a starry sky, dreamy illustration, vibrant colors, high quality", ), ( "05_ink", 46, "Traditional Chinese ink wash landscape, misty mountains, a small pagoda on a " "cliff, cranes flying, minimalist, elegant, high aesthetic quality", ), ] def rss_mb() -> float: return psutil.Process(os.getpid()).memory_info().rss / 1024**2 class StepProfiler: """diffusers callback: records wall time and RSS after every denoising step.""" def __init__(self, total_steps: int): self.total_steps = total_steps self.times: list[float] = [] self.rss: list[float] = [] self._last = time.perf_counter() self.t_start = self._last def __call__(self, pipe, step_index, timestep, callback_kwargs): now = time.perf_counter() self.times.append(now - self._last) self._last = now self.rss.append(rss_mb()) print( f" step {step_index + 1}/{self.total_steps}: " f"{self.times[-1]:.3f}s rss={self.rss[-1]:.0f}MB", flush=True, ) return callback_kwargs def system_info() -> dict: info = { "platform": platform.platform(), "python": platform.python_version(), "logical_cpus_os_cpu_count": os.cpu_count(), "psutil_physical_cores": psutil.cpu_count(logical=False), "psutil_logical_cores": psutil.cpu_count(logical=True), "total_ram_gb": round(psutil.virtual_memory().total / 1024**3, 1), } try: import openvino info["openvino_version"] = openvino.__version__ except Exception: pass import importlib.metadata as md for pkg in [ "optimum", "optimum-intel", "diffusers", "transformers", "tokenizers", "huggingface-hub", "nncf", "torch", "pillow", "psutil", ]: try: info[f"pkg_{pkg}"] = md.version(pkg) except Exception: pass try: out = os.popen("lscpu").read() for line in out.splitlines(): if line.startswith("Model name"): info["cpu_model"] = line.split(":", 1)[1].strip() except Exception: pass return info def run_one(pipe, name, seed, prompt, size, steps, guidance, max_seq_len, out_dir): print(f"[{name}] {size[0]}x{size[1]} seed={seed} steps={steps} cfg={guidance}", flush=True) rss_before = rss_mb() profiler = StepProfiler(steps) generator = torch.Generator(device="cpu").manual_seed(seed) t0 = time.perf_counter() result = pipe( prompt=prompt, negative_prompt=NEGATIVE_PROMPT, width=size[0], height=size[1], num_inference_steps=steps, guidance_scale=guidance, max_sequence_length=max_seq_len, generator=generator, callback_on_step_end=profiler, ) total = time.perf_counter() - t0 rss_peak = max(profiler.rss) if profiler.rss else rss_mb() out_path = out_dir / f"{name}.png" result.images[0].save(out_path) rec = { "name": name, "prompt": prompt, "negative_prompt": NEGATIVE_PROMPT, "seed": seed, "width": size[0], "height": size[1], "num_inference_steps": steps, "guidance_scale": guidance, "max_sequence_length": max_seq_len, "steps": len(profiler.times), "total_time_s": round(total, 3), "step_time_mean_s": round(statistics.mean(profiler.times), 3) if profiler.times else None, "step_time_median_s": round(statistics.median(profiler.times), 3) if profiler.times else None, "step_time_min_s": round(min(profiler.times), 3) if profiler.times else None, "step_time_max_s": round(max(profiler.times), 3) if profiler.times else None, "rss_before_mb": round(rss_before, 1), "rss_peak_mb": round(rss_peak, 1), "rss_after_mb": round(rss_mb(), 1), "per_step_time_s": [round(t, 4) for t in profiler.times], "per_step_rss_mb": [round(r, 1) for r in profiler.rss], "image": str(out_path), } print( f"[{name}] done {total:.1f}s mean {rec['step_time_mean_s']}s/step peak RSS {rss_peak:.0f}MB", flush=True, ) return rec def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--model_path", type=str, default=MODEL_PATH) parser.add_argument("--steps", type=int, default=4, help="FLUX.1-schnell: 1-4 steps") parser.add_argument("--guidance_scale", type=float, default=0.0, help="FLUX uses CFG 0.0") parser.add_argument("--max_sequence_length", type=int, default=256) parser.add_argument("--small_steps", type=int, default=4) parser.add_argument("--skip_small", action="store_true") args = parser.parse_args() out_dir = OUTPUT_DIR out_dir.mkdir(parents=True, exist_ok=True) print("loading + compiling pipeline ...", flush=True) t0 = time.perf_counter() pipe = OVFluxPipeline.from_pretrained(args.model_path, compile=True, device="CPU") load_s = time.perf_counter() - t0 rss_after_load = rss_mb() print(f"load+compile {load_s:.1f}s rss={rss_after_load:.0f}MB", flush=True) records = [] for name, seed, prompt in PROMPTS: records.append( run_one(pipe, name, seed, prompt, (1024, 1024), args.steps, args.guidance_scale, args.max_sequence_length, out_dir) ) if not args.skip_small: for name, seed, prompt in PROMPTS: records.append( run_one( pipe, f"{name}_512", seed, prompt, (512, 512), args.small_steps, args.guidance_scale, args.max_sequence_length, out_dir, ) ) fp16_dir = Path("/home/user/app/flux-schnell-ov-fp16") benchmark = { "system": system_info(), "load_compile_time_s": round(load_s, 3), "rss_after_load_mb": round(rss_after_load, 1), "pipeline_dir_size_mb": None, "fp16_dir_size_mb": None, "images": records, } int4_dir = Path(args.model_path) if int4_dir.exists(): benchmark["pipeline_dir_size_mb"] = round( sum(f.stat().st_size for f in int4_dir.rglob("*") if f.is_file()) / 1024**2, 1 ) if fp16_dir.exists(): benchmark["fp16_dir_size_mb"] = round( sum(f.stat().st_size for f in fp16_dir.rglob("*") if f.is_file()) / 1024**2, 1 ) with open(out_dir / "benchmark.json", "w") as f: json.dump(benchmark, f, indent=2, ensure_ascii=False) with open(out_dir / "prompts.txt", "w") as f: for name, seed, prompt in PROMPTS: f.write(f"{name} | seed={seed} | 1024x1024\n") f.write(f" prompt: {prompt}\n") f.write(f" negative: {NEGATIVE_PROMPT}\n\n") print("benchmark.json + prompts.txt written") if __name__ == "__main__": main()