#!/usr/bin/env python3 """Qwen/Qwen-Image-2.1 text-to-image on the halo box (gfx1151 / ROCm 7.13). Hard requirement: runs on GPU or not at all. Prints machine-readable receipts. """ import argparse, os, sys, time def main(): ap = argparse.ArgumentParser() ap.add_argument("--model", default="/home/kingjones/models/qwen-image-2.1") ap.add_argument("--out", required=True) ap.add_argument("--prompt", required=True) ap.add_argument("--negative", default=None) ap.add_argument("--steps", type=int, default=40) ap.add_argument("--width", type=int, default=2048) ap.add_argument("--height", type=int, default=2048) ap.add_argument("--seed", type=int, default=42) ap.add_argument("--cfg", type=float, default=None, help="true_cfg_scale, if the pipeline takes it") a = ap.parse_args() import torch print(f"python {sys.version.split()[0]}", flush=True) print(f"torch {torch.__version__} hip {torch.version.hip}", flush=True) if not torch.cuda.is_available(): print("FATAL: CUDA/HIP not available — refusing to run on CPU", flush=True) sys.exit(2) print(f"device {torch.cuda.get_device_name(0)}", flush=True) import diffusers, transformers print(f"diffusers {diffusers.__version__} transformers {transformers.__version__}", flush=True) from diffusers import QwenImage21Pipeline t0 = time.time() pipe = QwenImage21Pipeline.from_pretrained(a.model, torch_dtype=torch.bfloat16) pipe.to("cuda") print(f"LOAD_TIME_S {time.time() - t0:.1f}", flush=True) print(f"GPU_MEM_AFTER_LOAD_GB {torch.cuda.memory_allocated() / 2**30:.2f}", flush=True) state = {"t_first": None} def cb(_pipe, step_index, _timestep, cb_kwargs): now = time.time() if state["t_first"] is None: state["t_first"] = now print(f"step {step_index + 1}/{a.steps} @ {now - state['t_first']:.1f}s", flush=True) return cb_kwargs kwargs = dict( prompt=a.prompt, num_inference_steps=a.steps, width=a.width, height=a.height, generator=torch.Generator("cuda").manual_seed(a.seed), callback_on_step_end=cb, ) if a.negative is not None: kwargs["negative_prompt"] = a.negative if a.cfg is not None: kwargs["true_cfg_scale"] = a.cfg t1 = time.time() img = pipe(**kwargs).images[0] t_gen = time.time() - t1 os.makedirs(os.path.dirname(a.out), exist_ok=True) img.save(a.out) print(f"GEN_TIME_S {t_gen:.1f}", flush=True) print(f"SAVED {a.out}", flush=True) print(f"BYTES {os.path.getsize(a.out)}", flush=True) print(f"DIMS {img.size[0]}x{img.size[1]}", flush=True) print(f"PEAK_GPU_ALLOC_GB {torch.cuda.max_memory_allocated() / 2**30:.2f}", flush=True) print(f"PEAK_GPU_RESERVED_GB {torch.cuda.max_memory_reserved() / 2**30:.2f}", flush=True) print("GEN-DONE", flush=True) if __name__ == "__main__": main()