kingjones777 commited on
Commit
49808e1
·
verified ·
1 Parent(s): f18707b

Qwen-Image-2.1 on gfx1151: cold vs warm, MIOpen kernel-db finding

Browse files
Files changed (1) hide show
  1. gen.py +78 -0
gen.py ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Qwen/Qwen-Image-2.1 text-to-image on the halo box (gfx1151 / ROCm 7.13).
3
+
4
+ Hard requirement: runs on GPU or not at all. Prints machine-readable receipts.
5
+ """
6
+ import argparse, os, sys, time
7
+
8
+
9
+ def main():
10
+ ap = argparse.ArgumentParser()
11
+ ap.add_argument("--model", default="/home/kingjones/models/qwen-image-2.1")
12
+ ap.add_argument("--out", required=True)
13
+ ap.add_argument("--prompt", required=True)
14
+ ap.add_argument("--negative", default=None)
15
+ ap.add_argument("--steps", type=int, default=40)
16
+ ap.add_argument("--width", type=int, default=2048)
17
+ ap.add_argument("--height", type=int, default=2048)
18
+ ap.add_argument("--seed", type=int, default=42)
19
+ ap.add_argument("--cfg", type=float, default=None, help="true_cfg_scale, if the pipeline takes it")
20
+ a = ap.parse_args()
21
+
22
+ import torch
23
+ print(f"python {sys.version.split()[0]}", flush=True)
24
+ print(f"torch {torch.__version__} hip {torch.version.hip}", flush=True)
25
+ if not torch.cuda.is_available():
26
+ print("FATAL: CUDA/HIP not available — refusing to run on CPU", flush=True)
27
+ sys.exit(2)
28
+ print(f"device {torch.cuda.get_device_name(0)}", flush=True)
29
+
30
+ import diffusers, transformers
31
+ print(f"diffusers {diffusers.__version__} transformers {transformers.__version__}", flush=True)
32
+ from diffusers import QwenImage21Pipeline
33
+
34
+ t0 = time.time()
35
+ pipe = QwenImage21Pipeline.from_pretrained(a.model, torch_dtype=torch.bfloat16)
36
+ pipe.to("cuda")
37
+ print(f"LOAD_TIME_S {time.time() - t0:.1f}", flush=True)
38
+ print(f"GPU_MEM_AFTER_LOAD_GB {torch.cuda.memory_allocated() / 2**30:.2f}", flush=True)
39
+
40
+ state = {"t_first": None}
41
+
42
+ def cb(_pipe, step_index, _timestep, cb_kwargs):
43
+ now = time.time()
44
+ if state["t_first"] is None:
45
+ state["t_first"] = now
46
+ print(f"step {step_index + 1}/{a.steps} @ {now - state['t_first']:.1f}s", flush=True)
47
+ return cb_kwargs
48
+
49
+ kwargs = dict(
50
+ prompt=a.prompt,
51
+ num_inference_steps=a.steps,
52
+ width=a.width,
53
+ height=a.height,
54
+ generator=torch.Generator("cuda").manual_seed(a.seed),
55
+ callback_on_step_end=cb,
56
+ )
57
+ if a.negative is not None:
58
+ kwargs["negative_prompt"] = a.negative
59
+ if a.cfg is not None:
60
+ kwargs["true_cfg_scale"] = a.cfg
61
+
62
+ t1 = time.time()
63
+ img = pipe(**kwargs).images[0]
64
+ t_gen = time.time() - t1
65
+
66
+ os.makedirs(os.path.dirname(a.out), exist_ok=True)
67
+ img.save(a.out)
68
+ print(f"GEN_TIME_S {t_gen:.1f}", flush=True)
69
+ print(f"SAVED {a.out}", flush=True)
70
+ print(f"BYTES {os.path.getsize(a.out)}", flush=True)
71
+ print(f"DIMS {img.size[0]}x{img.size[1]}", flush=True)
72
+ print(f"PEAK_GPU_ALLOC_GB {torch.cuda.max_memory_allocated() / 2**30:.2f}", flush=True)
73
+ print(f"PEAK_GPU_RESERVED_GB {torch.cuda.max_memory_reserved() / 2**30:.2f}", flush=True)
74
+ print("GEN-DONE", flush=True)
75
+
76
+
77
+ if __name__ == "__main__":
78
+ main()