FLUX.1-schnell-OpenVINO-INT4 / generate5_flux.py
HelloSun's picture
Add generate5_flux.py
a26636f verified
Raw History Blame
8.7 kB
"""Generate the 5 fixed-seed 1024x1024 examples plus the 512px control set,
recording per-step latency and process RSS via a diffusers callback.
For FLUX.1-schnell (1-4 steps, CFG 0.0, max_sequence_length=256).
Outputs -> /home/user/app/outputs_flux/ and a machine readable log in
/home/user/app/outputs_flux/benchmark.json
"""
import argparse
import json
import os
import platform
import statistics
import time
from pathlib import Path
import psutil
import torch
from optimum.intel import OVFluxPipeline
MODEL_PATH = "/home/user/app/flux-schnell-ov-int4"
OUTPUT_DIR = Path("/home/user/app/outputs_flux")
NEGATIVE_PROMPT = ""
# name, seed, prompt
PROMPTS = [
(
"01_hanfu",
42,
"Young Chinese woman in red Hanfu, intricate embroidery, impeccable makeup, "
"red floral forehead pattern, elaborate high bun, golden phoenix headdress, "
"soft-lit outdoor night background, silhouetted tiered pagoda, blurred colorful "
"distant lights, photorealistic, ultra detailed, 8k",
),
(
"02_astronaut",
43,
"Astronaut in a jungle, cold color palette, muted colors, detailed, 8k, "
"photorealistic, cinematic lighting",
),
(
"03_taipei",
44,
"Cyberpunk street in Taipei at night, heavy rain, neon signs with text 'TAIPEI' "
"and Chinese characters '台北', reflections on wet asphalt, crowded night market, "
"cinematic, ultra detailed",
),
(
"04_shiba",
45,
"Cute Shiba Inu wearing a tiny astronaut helmet, sitting in a field of sunflowers "
"under a starry sky, dreamy illustration, vibrant colors, high quality",
),
(
"05_ink",
46,
"Traditional Chinese ink wash landscape, misty mountains, a small pagoda on a "
"cliff, cranes flying, minimalist, elegant, high aesthetic quality",
),
]
def rss_mb() -> float:
return psutil.Process(os.getpid()).memory_info().rss / 1024**2
class StepProfiler:
"""diffusers callback: records wall time and RSS after every denoising step."""
def __init__(self, total_steps: int):
self.total_steps = total_steps
self.times: list[float] = []
self.rss: list[float] = []
self._last = time.perf_counter()
self.t_start = self._last
def __call__(self, pipe, step_index, timestep, callback_kwargs):
now = time.perf_counter()
self.times.append(now - self._last)
self._last = now
self.rss.append(rss_mb())
print(
f" step {step_index + 1}/{self.total_steps}: "
f"{self.times[-1]:.3f}s rss={self.rss[-1]:.0f}MB",
flush=True,
)
return callback_kwargs
def system_info() -> dict:
info = {
"platform": platform.platform(),
"python": platform.python_version(),
"logical_cpus_os_cpu_count": os.cpu_count(),
"psutil_physical_cores": psutil.cpu_count(logical=False),
"psutil_logical_cores": psutil.cpu_count(logical=True),
"total_ram_gb": round(psutil.virtual_memory().total / 1024**3, 1),
}
try:
import openvino
info["openvino_version"] = openvino.__version__
except Exception:
pass
import importlib.metadata as md
for pkg in [
"optimum",
"optimum-intel",
"diffusers",
"transformers",
"tokenizers",
"huggingface-hub",
"nncf",
"torch",
"pillow",
"psutil",
]:
try:
info[f"pkg_{pkg}"] = md.version(pkg)
except Exception:
pass
try:
out = os.popen("lscpu").read()
for line in out.splitlines():
if line.startswith("Model name"):
info["cpu_model"] = line.split(":", 1)[1].strip()
except Exception:
pass
return info
def run_one(pipe, name, seed, prompt, size, steps, guidance, max_seq_len, out_dir):
print(f"[{name}] {size[0]}x{size[1]} seed={seed} steps={steps} cfg={guidance}", flush=True)
rss_before = rss_mb()
profiler = StepProfiler(steps)
generator = torch.Generator(device="cpu").manual_seed(seed)
t0 = time.perf_counter()
result = pipe(
prompt=prompt,
negative_prompt=NEGATIVE_PROMPT,
width=size[0],
height=size[1],
num_inference_steps=steps,
guidance_scale=guidance,
max_sequence_length=max_seq_len,
generator=generator,
callback_on_step_end=profiler,
)
total = time.perf_counter() - t0
rss_peak = max(profiler.rss) if profiler.rss else rss_mb()
out_path = out_dir / f"{name}.png"
result.images[0].save(out_path)
rec = {
"name": name,
"prompt": prompt,
"negative_prompt": NEGATIVE_PROMPT,
"seed": seed,
"width": size[0],
"height": size[1],
"num_inference_steps": steps,
"guidance_scale": guidance,
"max_sequence_length": max_seq_len,
"steps": len(profiler.times),
"total_time_s": round(total, 3),
"step_time_mean_s": round(statistics.mean(profiler.times), 3) if profiler.times else None,
"step_time_median_s": round(statistics.median(profiler.times), 3) if profiler.times else None,
"step_time_min_s": round(min(profiler.times), 3) if profiler.times else None,
"step_time_max_s": round(max(profiler.times), 3) if profiler.times else None,
"rss_before_mb": round(rss_before, 1),
"rss_peak_mb": round(rss_peak, 1),
"rss_after_mb": round(rss_mb(), 1),
"per_step_time_s": [round(t, 4) for t in profiler.times],
"per_step_rss_mb": [round(r, 1) for r in profiler.rss],
"image": str(out_path),
}
print(
f"[{name}] done {total:.1f}s mean {rec['step_time_mean_s']}s/step peak RSS {rss_peak:.0f}MB",
flush=True,
)
return rec
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--model_path", type=str, default=MODEL_PATH)
parser.add_argument("--steps", type=int, default=4, help="FLUX.1-schnell: 1-4 steps")
parser.add_argument("--guidance_scale", type=float, default=0.0, help="FLUX uses CFG 0.0")
parser.add_argument("--max_sequence_length", type=int, default=256)
parser.add_argument("--small_steps", type=int, default=4)
parser.add_argument("--skip_small", action="store_true")
args = parser.parse_args()
out_dir = OUTPUT_DIR
out_dir.mkdir(parents=True, exist_ok=True)
print("loading + compiling pipeline ...", flush=True)
t0 = time.perf_counter()
pipe = OVFluxPipeline.from_pretrained(args.model_path, compile=True, device="CPU")
load_s = time.perf_counter() - t0
rss_after_load = rss_mb()
print(f"load+compile {load_s:.1f}s rss={rss_after_load:.0f}MB", flush=True)
records = []
for name, seed, prompt in PROMPTS:
records.append(
run_one(pipe, name, seed, prompt, (1024, 1024), args.steps, args.guidance_scale, args.max_sequence_length, out_dir)
)
if not args.skip_small:
for name, seed, prompt in PROMPTS:
records.append(
run_one(
pipe,
f"{name}_512",
seed,
prompt,
(512, 512),
args.small_steps,
args.guidance_scale,
args.max_sequence_length,
out_dir,
)
)
fp16_dir = Path("/home/user/app/flux-schnell-ov-fp16")
benchmark = {
"system": system_info(),
"load_compile_time_s": round(load_s, 3),
"rss_after_load_mb": round(rss_after_load, 1),
"pipeline_dir_size_mb": None,
"fp16_dir_size_mb": None,
"images": records,
}
int4_dir = Path(args.model_path)
if int4_dir.exists():
benchmark["pipeline_dir_size_mb"] = round(
sum(f.stat().st_size for f in int4_dir.rglob("*") if f.is_file()) / 1024**2, 1
)
if fp16_dir.exists():
benchmark["fp16_dir_size_mb"] = round(
sum(f.stat().st_size for f in fp16_dir.rglob("*") if f.is_file()) / 1024**2, 1
)
with open(out_dir / "benchmark.json", "w") as f:
json.dump(benchmark, f, indent=2, ensure_ascii=False)
with open(out_dir / "prompts.txt", "w") as f:
for name, seed, prompt in PROMPTS:
f.write(f"{name} | seed={seed} | 1024x1024\n")
f.write(f" prompt: {prompt}\n")
f.write(f" negative: {NEGATIVE_PROMPT}\n\n")
print("benchmark.json + prompts.txt written")
if __name__ == "__main__":
main()