"""Z-Image CMF demo: text-to-image with the cortiq engine (Rust, no Python ML stack). The same file runs on a Hugging Face Space and locally. Paths are set by env: CORTIQ_BIN an existing cortiq binary (default: download the Linux x86-64 release) ZIMAGE_TURBO_CMF an existing z-image-turbo.cmf (default: download from the Hub on first use) ZIMAGE_BASE_CMF an existing z-image.cmf (default: download from the Hub on first use) CACHE_DIR where downloads go (default: /data if writable, else ./cache) CORTIQ_DEVICE auto | cpu | gpu (auto: GPU only when nvidia-smi or macOS is present) MAX_JOB_MINUTES refuse jobs whose estimate exceeds this on the CPU (default 45) """ import glob import json import os import re import shutil import subprocess import sys import tarfile import threading import time import urllib.request import uuid import gradio as gr from huggingface_hub import hf_hub_download from PIL import Image CORTIQ_VERSION = "0.7.5" CORTIQ_URL = os.environ.get( "CORTIQ_URL", f"https://github.com/infosave2007/cmf/releases/download/v{CORTIQ_VERSION}/" "cortiq-x86_64-unknown-linux-gnu.tar.gz", ) GITHUB = "https://github.com/infosave2007/cmf" def _pick_cache(): if os.environ.get("CACHE_DIR"): return os.environ["CACHE_DIR"] if os.path.isdir("/data") and os.access("/data", os.W_OK): return "/data/zimage-cache" return os.path.join(os.path.dirname(os.path.abspath(__file__)), "cache") CACHE = _pick_cache() os.makedirs(CACHE, exist_ok=True) OUT_DIR = os.path.join(CACHE, "out") os.makedirs(OUT_DIR, exist_ok=True) MODELS = { "Z-Image-Turbo (8 steps, no CFG)": { "key": "turbo", "repo": "infosave/Z-Image-Turbo-cmf", "file": "z-image-turbo.cmf", "env": "ZIMAGE_TURBO_CMF", "steps": 8, "cfg": 0.0, "size_gb": 10.46, }, "Z-Image base (28 steps, CFG 4)": { "key": "base", "repo": "infosave/Z-Image-cmf", "file": "z-image.cmf", "env": "ZIMAGE_BASE_CMF", "steps": 28, "cfg": 4.0, "size_gb": 10.46, }, } # ZIMAGE_ONLY=turbo|base pins a Space to one model (the dedicated Turbo Space) ONLY = os.environ.get("ZIMAGE_ONLY", "").strip().lower() if ONLY in ("turbo", "base"): MODELS = {k: v for k, v in MODELS.items() if v["key"] == ONLY} MODEL_NAMES = list(MODELS) SAMPLE_PROMPTS = [ "A cat on a windowsill at sunset, photorealistic", "A fisherman's hands tying a rope", 'A "CORTIQ" neon sign on a brick wall', "An aerial view of a river in autumn", ] # ── hardware ───────────────────────────────────────────────────────────────── def effective_cpus(): """vCPUs this container may use: the cgroup quota, not the host's core count.""" n = os.cpu_count() or 1 try: n = len(os.sched_getaffinity(0)) except AttributeError: pass try: # cgroup v2 quota, period = open("/sys/fs/cgroup/cpu.max").read().split()[:2] if quota != "max": n = min(n, max(1, round(int(quota) / int(period)))) except (OSError, ValueError): try: # cgroup v1 q = int(open("/sys/fs/cgroup/cpu/cpu.cfs_quota_us").read()) p = int(open("/sys/fs/cgroup/cpu/cpu.cfs_period_us").read()) if q > 0: n = min(n, max(1, round(q / p))) except (OSError, ValueError): pass return n def ram_gb(): try: for f in ("/sys/fs/cgroup/memory.max", "/sys/fs/cgroup/memory/memory.limit_in_bytes"): if os.path.exists(f): v = open(f).read().strip() if v != "max" and int(v) < 1 << 50: return int(v) / 1e9 return os.sysconf("SC_PAGE_SIZE") * os.sysconf("SC_PHYS_PAGES") / 1e9 except (OSError, ValueError, AttributeError): return 0.0 NCPU = effective_cpus() RAM_GB = ram_gb() def detect_device(): want = os.environ.get("CORTIQ_DEVICE", "auto").lower() if want == "cpu": return "cpu" if want == "gpu": return "metal" if sys.platform == "darwin" else "vulkan" if sys.platform == "darwin": return "metal" if shutil.which("nvidia-smi"): try: r = subprocess.run(["nvidia-smi", "-L"], capture_output=True, text=True, timeout=10) if r.returncode == 0 and "GPU" in r.stdout: return "vulkan" except Exception: pass return "cpu" DEVICE = detect_device() DEVICE_LABEL = { "cpu": f"CPU only, {NCPU} vCPU{'s' if NCPU != 1 else ''}", "metal": "Apple silicon GPU (Metal)", "vulkan": "NVIDIA GPU (Vulkan)", }[DEVICE] # ── cost model for the ETA ─────────────────────────────────────────────────── # One DiT forward is ~2 * 6.15e9 FLOP per token row (image patches + caption) # plus attention; 256² ≈ 3.4 TFLOP, 512² ≈ 13.6 TFLOP. CFG doubles it. DIT_PARAMS = 6.15e9 CAPTION_ROWS = 64 def step_tflop(w, h, cfg): rows = (w // 16) * (h // 16) + CAPTION_ROWS f = 2 * DIT_PARAMS * rows + 4 * rows * rows * 3840 * 34 return f * (2 if cfg > 0 else 1) / 1e12 def vae_tflop(w, h): return 2.5 * (w * h) / (1024 * 1024) # Default effective throughputs (TFLOP/s) and fixed overhead (s), all measured: # the CPU row on this Space's free hardware (2 vCPUs: 69.5 s per Turbo step at # 256², 10 min per image), the others on an M4 and an RTX 3090. Each finished # image refines them for the running process. DEFAULT_SPEED = { "cpu": {"dit": 0.0287 * NCPU, "vae": 0.03 * NCPU, "overhead": 5 + 40 / max(NCPU, 1)}, "metal": {"dit": 3.7, "vae": 2.5, "overhead": 6.0}, "vulkan": {"dit": 50.0, "vae": 20.0, "overhead": 3.0}, } SPEED_FILE = os.path.join(CACHE, f"speed-{DEVICE}-{NCPU}.json") def load_speed(): s = dict(DEFAULT_SPEED[DEVICE]) s["measured"] = False try: s.update(json.load(open(SPEED_FILE))) except (OSError, ValueError): pass return s SPEED = load_speed() def save_speed(dit_tflops, overhead): SPEED["dit"] = dit_tflops if not SPEED.get("measured") else 0.5 * SPEED["dit"] + 0.5 * dit_tflops SPEED["overhead"] = overhead if not SPEED.get("measured") else 0.5 * SPEED["overhead"] + 0.5 * overhead SPEED["measured"] = True try: json.dump(SPEED, open(SPEED_FILE, "w")) except OSError: pass def estimate_seconds(w, h, steps, cfg): return ( SPEED["overhead"] + steps * step_tflop(w, h, cfg) / SPEED["dit"] + vae_tflop(w, h) / SPEED["vae"] ) def fmt_dur(s): s = max(0, int(round(s))) if s < 90: return f"{s} s" if s < 3600: return f"{s // 60} min {s % 60:02d} s" return f"{s // 3600} h {(s % 3600) // 60:02d} min" MAX_JOB_S = float(os.environ.get("MAX_JOB_MINUTES", "45")) * 60 if DEVICE == "cpu" else float("inf") # ── downloads ──────────────────────────────────────────────────────────────── _bin_lock = threading.Lock() _dl_lock = threading.Lock() _downloads = {} # model key -> {"thread", "error", "path"} def cortiq_bin(): env = os.environ.get("CORTIQ_BIN") if env: return env path = os.path.join(CACHE, "bin", "cortiq") with _bin_lock: if not os.path.exists(path): os.makedirs(os.path.dirname(path), exist_ok=True) tgz = path + ".tar.gz" urllib.request.urlretrieve(CORTIQ_URL, tgz) with tarfile.open(tgz) as t: member = next(m for m in t.getmembers() if os.path.basename(m.name) == "cortiq") member.name = "cortiq" t.extract(member, os.path.dirname(path)) os.remove(tgz) os.chmod(path, 0o755) return path def model_local(name): m = MODELS[name] env = os.environ.get(m["env"]) if env and os.path.exists(env): return env p = os.path.join(CACHE, "models", m["key"], m["file"]) return p if os.path.exists(p) else None def _download(name): m = MODELS[name] d = _downloads[name] try: d["path"] = hf_hub_download( m["repo"], m["file"], local_dir=os.path.join(CACHE, "models", m["key"]) ) except Exception as e: # surfaced to the user by the generator d["error"] = f"{type(e).__name__}: {e}" def start_download(name): with _dl_lock: d = _downloads.get(name) if d and (d["thread"].is_alive() or d.get("path")): return d d = {"thread": None, "error": None, "path": None, "t0": time.time()} _downloads[name] = d d["thread"] = threading.Thread(target=_download, args=(name,), daemon=True) d["thread"].start() return d def download_bytes(name): root = os.path.join(CACHE, "models", MODELS[name]["key"]) return sum( os.path.getsize(f) for f in glob.glob(os.path.join(root, "**", "*.incomplete"), recursive=True) + glob.glob(os.path.join(root, ".cache", "**", "*.incomplete"), recursive=True) if os.path.exists(f) ) # ── samples for the gallery ────────────────────────────────────────────────── def load_samples(): items = [] pairs = (("Z-Image-Turbo", "infosave/Z-Image-Turbo-cmf"), ("Z-Image", "infosave/Z-Image-cmf")) if ONLY == "turbo": pairs = pairs[:1] elif ONLY == "base": pairs = pairs[1:] for label, repo in pairs: for i, prompt in enumerate(SAMPLE_PROMPTS, 1): cap = f"{label}: {prompt} (1024², seed 7)" try: p = hf_hub_download(repo, f"samples/{i}.png", local_dir=os.path.join(CACHE, "samples", label)) items.append((p, cap)) except Exception: items.append((f"https://huggingface.co/{repo}/resolve/main/samples/{i}.png", cap)) return items # ── generation ─────────────────────────────────────────────────────────────── ANSI = re.compile(r"\x1b\[[0-9;]*[A-Za-z]") RE_STEP = re.compile(r"^zimage: image (\d+) step (\d+)/(\d+) ([\d.]+)s") RE_TE = re.compile(r"^zimage: text-encode ([\d.]+)s") RE_STAGES = re.compile(r"^zimage stages: (.*)") def _reader(stream, sink): buf = b"" while True: ch = stream.read1(4096) if hasattr(stream, "read1") else stream.read(4096) if not ch: break buf += ch parts = re.split(rb"[\r\n]", buf) buf = parts.pop() for p in parts: line = ANSI.sub("", p.decode("utf-8", "replace")).strip() if line: sink.append(line) if buf.strip(): sink.append(ANSI.sub("", buf.decode("utf-8", "replace")).strip()) def _status(title, lines): return f"**{title}**\n\n" + "\n".join(f"- {l}" for l in lines) def on_model_change(name): m = MODELS[name] base = m["key"] == "base" return ( gr.update(value=m["steps"]), gr.update(value=m["cfg"], visible=base), gr.update(visible=base), ) def estimate_text(name, size, steps, cfg): m = MODELS[name] w = h = int(size) cfg = float(cfg) if m["key"] == "base" else 0.0 est = estimate_seconds(w, h, int(steps), cfg) src = "measured on this machine" if SPEED.get("measured") else "from measured runs" dl = "" if not model_local(name): dl = f" + a one-time {m['size_gb']:.1f} GB model download" warn = "" if est > MAX_JOB_S: warn = f" — over this Space's {fmt_dur(MAX_JOB_S)} limit: use Turbo, 256², or fewer steps" return f"Estimated time on {DEVICE_LABEL}: **~{fmt_dur(est)}**{dl} ({src}){warn}" def generate(prompt, negative, name, size, steps, seed, cfg, progress=gr.Progress()): prompt = (prompt or "").strip() if not prompt: raise gr.Error("Enter a prompt.") m = MODELS[name] w = h = int(size) steps = int(steps) seed = int(seed) base = m["key"] == "base" cfg = float(cfg) if base else 0.0 est = estimate_seconds(w, h, steps, cfg) if est > MAX_JOB_S: raise gr.Error( f"This job is estimated at {fmt_dur(est)} on {DEVICE_LABEL}, over the " f"{fmt_dur(MAX_JOB_S)} limit. Use Z-Image-Turbo, 256², or fewer steps." ) t_start = time.time() yield None, _status("Preparing", ["fetching the cortiq engine"]) try: exe = cortiq_bin() except Exception as e: raise gr.Error(f"Could not fetch the cortiq binary: {e}") path = model_local(name) if not path: d = start_download(name) total = m["size_gb"] * 1e9 while d["thread"].is_alive(): got = download_bytes(name) el = time.time() - d["t0"] rate = got / el if el > 5 and got else 0 eta = f", ~{fmt_dur((total - got) / rate)} left" if rate > 0 else "" progress(min(got / total, 0.99), desc="downloading model") yield None, _status( f"Downloading {m['file']} (one time, {m['size_gb']:.1f} GB)", [f"{got / 1e9:.2f} GB so far{eta}", "the image starts right after"], ) time.sleep(2) if d.get("error"): raise gr.Error(f"Model download failed: {d['error']}") path = model_local(name) or d["path"] out = os.path.join(OUT_DIR, f"{uuid.uuid4().hex}.png") cmd = [exe, "imagine", path, "--prompt", prompt, "--width", str(w), "--height", str(h), "--steps", str(steps), "--seed", str(seed), "--out", out] if base: cmd += ["--cfg", f"{cfg:g}"] if cfg > 0: cmd += ["--negative-prompt", (negative or "").strip()] env = dict(os.environ, CMF_ZIMAGE_PROF="1", XDG_RUNTIME_DIR=os.environ.get("XDG_RUNTIME_DIR", "/tmp")) if DEVICE == "cpu": env["CMF_GPU"] = "0" lines = [] proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, env=env) reader = threading.Thread(target=_reader, args=(proc.stdout, lines), daemon=True) reader.start() t0 = time.time() hard_limit = max(MAX_JOB_S * 1.5, 600) if MAX_JOB_S != float("inf") else None try: while proc.poll() is None: el = time.time() - t0 if hard_limit and el > hard_limit: proc.kill() raise gr.Error(f"Stopped after {fmt_dur(el)}: over this Space's time limit.") steps_done, step_times, te = 0, [], None for l in lines: s = RE_STEP.match(l) if s: steps_done = int(s.group(2)) step_times.append(float(s.group(4))) t = RE_TE.match(l) if t: te = float(t.group(1)) per_step = step_tflop(w, h, cfg) / SPEED["dit"] if step_times: per_step = sorted(step_times)[len(step_times) // 2] vae_s = vae_tflop(w, h) / SPEED["vae"] if steps_done == 0: phase = "loading the model and encoding the prompt" if te is None else "preparing the transformer" remaining = max(SPEED["overhead"] - el, 5) + steps * per_step + vae_s elif steps_done < steps: phase = f"denoising: step {steps_done}/{steps} ({per_step:.1f} s/step)" remaining = (steps - steps_done) * per_step + vae_s else: phase = "decoding the image (VAE)" remaining = vae_s progress((steps_done + 0.5 * (te is not None)) / (steps + 1), desc=phase) info = [phase, f"elapsed {fmt_dur(el)}, about {fmt_dur(remaining)} left", f"{w}×{h}, {steps} steps" + (f", CFG {cfg:g}" if cfg > 0 else "") + f", seed {seed}, {DEVICE_LABEL}"] yield None, _status("Generating", info) time.sleep(1.0) finally: if proc.poll() is None: proc.kill() proc.wait() reader.join(timeout=5) total = time.time() - t_start if proc.returncode != 0 or not os.path.exists(out): tail = "\n".join(lines[-12:]) raise gr.Error(f"cortiq exited with code {proc.returncode}:\n{tail}") # refine the ETA model from what this machine actually did step_times = [float(s.group(4)) for s in (RE_STEP.match(l) for l in lines) if s] stages = next((s.group(1) for s in (RE_STAGES.match(l) for l in lines) if s), "") if step_times: med = sorted(step_times)[len(step_times) // 2] run_s = time.time() - t0 overhead = max(run_s - sum(step_times) - vae_tflop(w, h) / SPEED["vae"], 1.0) save_speed(step_tflop(w, h, cfg) / med, overhead) img = Image.open(out) img.load() os.remove(out) summary = [f"{w}×{h}, {steps} steps" + (f", CFG {cfg:g}" if cfg > 0 else "") + f", seed {seed}", f"total {fmt_dur(total)} on {DEVICE_LABEL}"] if stages: summary.append("engine stages: " + stages) yield img, _status("Done", summary) # ── UI ─────────────────────────────────────────────────────────────────────── RUN_LOCALLY = f""" The same files run on your own machine with `cortiq` {CORTIQ_VERSION}, a single binary with no Python: NVIDIA GPUs through Vulkan (tensor cores), Apple silicon through Metal, and a CPU fallback everywhere. **1. Get cortiq** — prebuilt binaries on the [releases page]({GITHUB}/releases), or build it: ```bash # Linux x86-64 curl -L {CORTIQ_URL} | tar xz # or, with Rust installed (macOS, Linux, Windows) cargo install cortiq-cli ``` **2. Download a model** (10.5 GB each): ```bash hf download infosave/Z-Image-Turbo-cmf z-image-turbo.cmf --local-dir . hf download infosave/Z-Image-cmf z-image.cmf --local-dir . ``` **3. Generate:** ```bash ./cortiq imagine z-image-turbo.cmf --prompt "A cat sitting on a windowsill at sunset, photorealistic" ./cortiq imagine z-image.cmf --prompt "..." --negative-prompt "blurry, low quality" --steps 28 --cfg 4 ``` No flags are needed: each file stores its recipe (Turbo: 1024², 8 steps, no CFG; base: 1024², 28 steps, CFG 4). Useful options: `--width`/`--height` (multiples of 16), `--steps`, `--seed`, `--num-images N`, `--out file.png`. `CMF_ZIMAGE_PROF=1` prints stage times; `CMF_GPU=0` forces the CPU. On a headless Linux box set `XDG_RUNTIME_DIR=/tmp`. **Measured speed** (cortiq 0.7.5, one image per process, no flags): | hardware | model | 512×512 | 1024×1024 | |---|---|---:|---:| | RTX 3090 (Vulkan) | Turbo, 8 steps | 4.6 s | 11.4 s | | RTX 3090 (Vulkan) | base, 28 steps + CFG | 16.2 s | 60 s | | Mac mini M4 24 GB (Metal) | Turbo, 8 steps | 30 s | 160 s | | Mac mini M4 24 GB (Metal) | base, 28 steps + CFG | 3.9 min | 21 min | This Space's free CPU tier (2 vCPUs, 16 GB): Turbo 256², 8 steps in about 10 minutes (69.5 s per step). About 8 GB of memory on the Mac; 14–15 GB of VRAM peak on the 3090 (16 GB cards fit). On a CPU the transformer costs about 3.4 TFLOP per step at 256² and 13.6 at 512², so expect minutes per image. Models: [Z-Image-Turbo-cmf](https://huggingface.co/infosave/Z-Image-Turbo-cmf) · [Z-Image-cmf](https://huggingface.co/infosave/Z-Image-cmf) · engine: [{GITHUB}]({GITHUB}) """ CSS = """ #title h1 {margin-bottom: 0} .gradio-container {max-width: 1180px !important; margin: 0 auto} """ def build(): samples = load_samples() default_model = MODEL_NAMES[0] default_size = "256" if DEVICE == "cpu" else "512" title = "Z-Image-Turbo · CMF" if ONLY == "turbo" else ("Z-Image · CMF") header = ( f"# {title}\n" + ("Text-to-image with Tongyi-MAI's Z-Image-Turbo (6B DiT + Qwen3-4B text encoder, 8 steps, no CFG), " if ONLY == "turbo" else "Text-to-image with Tongyi-MAI's Z-Image and Z-Image-Turbo (6B DiT + Qwen3-4B text encoder), ") + f"run by [cortiq]({GITHUB}), a Rust engine with no Python ML stack. " "Models: [Z-Image-Turbo-cmf](https://huggingface.co/infosave/Z-Image-Turbo-cmf) · " "[Z-Image-cmf](https://huggingface.co/infosave/Z-Image-cmf).\n\n" f"This Space runs on **{DEVICE_LABEL}**" + (f", {RAM_GB:.0f} GB RAM" if RAM_GB else "") + ". " + ( "The CPU is slow for a 6B image model: a 256² Turbo image takes about " f"{fmt_dur(estimate_seconds(256, 256, 8, 0))}, and jobs run one at a time in a queue. " "The Gallery tab shows what the models do at 1024² on a GPU; the same file makes a 512² image " "in 4.6 s on an RTX 3090." if DEVICE == "cpu" else "Jobs run one at a time in a queue." ) ) with gr.Blocks(title=title) as demo: gr.Markdown(header, elem_id="title") with gr.Tabs(): with gr.Tab("Generate"): with gr.Row(): with gr.Column(scale=5): prompt = gr.Textbox(label="Prompt", lines=3, value=SAMPLE_PROMPTS[0]) negative = gr.Textbox(label="Negative prompt (base model only)", value="blurry, low quality", visible=False) model = gr.Dropdown(MODEL_NAMES, value=default_model, label="Model") size = gr.Radio(["256", "384", "512"], value=default_size, label="Size (square, pixels)") with gr.Row(): steps = gr.Slider(1, 50, value=8, step=1, label="Steps") seed = gr.Number(value=7, precision=0, label="Seed") cfg = gr.Slider(0, 10, value=4.0, step=0.5, label="CFG (base model)", visible=False) eta = gr.Markdown(estimate_text(default_model, default_size, 8, 0)) base_note = gr.Markdown( "The base model runs 28 steps with CFG (two transformer passes per step): " "about 7× the Turbo cost, which is very slow on a CPU. Lower the steps for a draft.", visible=False, ) btn = gr.Button("Generate", variant="primary") with gr.Column(scale=6): image = gr.Image(label="Result", type="pil", format="png", height=560) status = gr.Markdown("Ready. " + ( "The first run also downloads the model (10.5 GB)." if not model_local(default_model) else "")) gr.Examples([[p] for p in SAMPLE_PROMPTS], inputs=[prompt], label="Sample prompts") model.change(on_model_change, [model], [steps, cfg, negative]).then( lambda n: gr.update(visible=MODELS[n]["key"] == "base"), [model], [base_note]) for c in (model, size, steps, cfg): c.change(estimate_text, [model, size, steps, cfg], [eta], queue=False) btn.click(generate, [prompt, negative, model, size, steps, seed, cfg], [image, status], concurrency_limit=1, show_progress_on=[image]) with gr.Tab("Gallery"): gr.Markdown("1024×1024, seed 7, each file's default recipe (Turbo: 8 steps; base: 28 steps, CFG 4), " "generated by cortiq on a GPU.") gr.Gallery(samples, columns=4, rows=2, height=640, object_fit="contain", label="Samples", show_label=False, preview=False) with gr.Tab("Run it locally"): gr.Markdown(RUN_LOCALLY) return demo demo = build() demo.queue(default_concurrency_limit=1, max_size=16) if __name__ == "__main__": demo.launch( server_name=os.environ.get("GRADIO_SERVER_NAME", "0.0.0.0"), server_port=int(os.environ.get("PORT", os.environ.get("GRADIO_SERVER_PORT", "7860"))), allowed_paths=[CACHE], css=CSS, theme=gr.themes.Soft(), )