Instructions to use kingjones777/Qwen-Image-2.1-ROCm-gfx1151 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Diffusers
How to use kingjones777/Qwen-Image-2.1-ROCm-gfx1151 with Diffusers:
pip install -U diffusers transformers accelerate
import torch from diffusers import DiffusionPipeline # switch to "mps" for apple devices pipe = DiffusionPipeline.from_pretrained("kingjones777/Qwen-Image-2.1-ROCm-gfx1151", dtype=torch.bfloat16, device_map="cuda") prompt = "Astronaut in a jungle, cold color palette, muted colors, detailed, 8k" image = pipe(prompt).images[0] - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Draw Things
- DiffusionBee
Qwen-Image-2.1 on gfx1151: cold vs warm, MIOpen kernel-db finding
Browse files
gen.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Qwen/Qwen-Image-2.1 text-to-image on the halo box (gfx1151 / ROCm 7.13).
|
| 3 |
+
|
| 4 |
+
Hard requirement: runs on GPU or not at all. Prints machine-readable receipts.
|
| 5 |
+
"""
|
| 6 |
+
import argparse, os, sys, time
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
def main():
|
| 10 |
+
ap = argparse.ArgumentParser()
|
| 11 |
+
ap.add_argument("--model", default="/home/kingjones/models/qwen-image-2.1")
|
| 12 |
+
ap.add_argument("--out", required=True)
|
| 13 |
+
ap.add_argument("--prompt", required=True)
|
| 14 |
+
ap.add_argument("--negative", default=None)
|
| 15 |
+
ap.add_argument("--steps", type=int, default=40)
|
| 16 |
+
ap.add_argument("--width", type=int, default=2048)
|
| 17 |
+
ap.add_argument("--height", type=int, default=2048)
|
| 18 |
+
ap.add_argument("--seed", type=int, default=42)
|
| 19 |
+
ap.add_argument("--cfg", type=float, default=None, help="true_cfg_scale, if the pipeline takes it")
|
| 20 |
+
a = ap.parse_args()
|
| 21 |
+
|
| 22 |
+
import torch
|
| 23 |
+
print(f"python {sys.version.split()[0]}", flush=True)
|
| 24 |
+
print(f"torch {torch.__version__} hip {torch.version.hip}", flush=True)
|
| 25 |
+
if not torch.cuda.is_available():
|
| 26 |
+
print("FATAL: CUDA/HIP not available — refusing to run on CPU", flush=True)
|
| 27 |
+
sys.exit(2)
|
| 28 |
+
print(f"device {torch.cuda.get_device_name(0)}", flush=True)
|
| 29 |
+
|
| 30 |
+
import diffusers, transformers
|
| 31 |
+
print(f"diffusers {diffusers.__version__} transformers {transformers.__version__}", flush=True)
|
| 32 |
+
from diffusers import QwenImage21Pipeline
|
| 33 |
+
|
| 34 |
+
t0 = time.time()
|
| 35 |
+
pipe = QwenImage21Pipeline.from_pretrained(a.model, torch_dtype=torch.bfloat16)
|
| 36 |
+
pipe.to("cuda")
|
| 37 |
+
print(f"LOAD_TIME_S {time.time() - t0:.1f}", flush=True)
|
| 38 |
+
print(f"GPU_MEM_AFTER_LOAD_GB {torch.cuda.memory_allocated() / 2**30:.2f}", flush=True)
|
| 39 |
+
|
| 40 |
+
state = {"t_first": None}
|
| 41 |
+
|
| 42 |
+
def cb(_pipe, step_index, _timestep, cb_kwargs):
|
| 43 |
+
now = time.time()
|
| 44 |
+
if state["t_first"] is None:
|
| 45 |
+
state["t_first"] = now
|
| 46 |
+
print(f"step {step_index + 1}/{a.steps} @ {now - state['t_first']:.1f}s", flush=True)
|
| 47 |
+
return cb_kwargs
|
| 48 |
+
|
| 49 |
+
kwargs = dict(
|
| 50 |
+
prompt=a.prompt,
|
| 51 |
+
num_inference_steps=a.steps,
|
| 52 |
+
width=a.width,
|
| 53 |
+
height=a.height,
|
| 54 |
+
generator=torch.Generator("cuda").manual_seed(a.seed),
|
| 55 |
+
callback_on_step_end=cb,
|
| 56 |
+
)
|
| 57 |
+
if a.negative is not None:
|
| 58 |
+
kwargs["negative_prompt"] = a.negative
|
| 59 |
+
if a.cfg is not None:
|
| 60 |
+
kwargs["true_cfg_scale"] = a.cfg
|
| 61 |
+
|
| 62 |
+
t1 = time.time()
|
| 63 |
+
img = pipe(**kwargs).images[0]
|
| 64 |
+
t_gen = time.time() - t1
|
| 65 |
+
|
| 66 |
+
os.makedirs(os.path.dirname(a.out), exist_ok=True)
|
| 67 |
+
img.save(a.out)
|
| 68 |
+
print(f"GEN_TIME_S {t_gen:.1f}", flush=True)
|
| 69 |
+
print(f"SAVED {a.out}", flush=True)
|
| 70 |
+
print(f"BYTES {os.path.getsize(a.out)}", flush=True)
|
| 71 |
+
print(f"DIMS {img.size[0]}x{img.size[1]}", flush=True)
|
| 72 |
+
print(f"PEAK_GPU_ALLOC_GB {torch.cuda.max_memory_allocated() / 2**30:.2f}", flush=True)
|
| 73 |
+
print(f"PEAK_GPU_RESERVED_GB {torch.cuda.max_memory_reserved() / 2**30:.2f}", flush=True)
|
| 74 |
+
print("GEN-DONE", flush=True)
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
if __name__ == "__main__":
|
| 78 |
+
main()
|