"""MiniMax-H3 · Third-Person View LoRA — game cutscenes from design reference sheets. [`WarmBloodAban/Minimax_H3_LoRAs`](https://huggingface.co/WarmBloodAban/Minimax_H3_LoRAs) ships `Minimax-h3_Third_person_view.safetensors`, a rank-64 ai-toolkit LoRA whose own metadata names its base as `minimax_h3_ref2va` — the **reference** partition of MiniMax-H3, the one that conditions on an ordered list of reference images rather than on a first frame. That is why this Space is a `ref2va` deployment: the adapter goes onto `transformer_ref`, and the user's uploads arrive as `` in the order they are given. What the LoRA does, from its card: cinematic game cutscenes — third-person over-the-shoulder / spring-arm camera tracking, first-person POV, whip-pan view transitions, and native game HUD overlays (target-lock reticles, Boss health bars, QTE prompts, floating damage text). Its card is explicit that it wants MiniMax-H3's **structured** prompt format (`subject_definitions` / `summary` / `retention_analysis` / `detailed_description` / `overall_soundscape` / `non_diegetic_music`) with `` / `` / `` cross-references, so this demo is a composer for exactly that document rather than a prompt box: you label each reference sheet, write one line of action, pick a camera and whether the HUD is on, and the Space assembles the card's format around the LoRA's own trigger vocabulary. The assembled document is shown next to the result, and an override box takes a hand-written one. Recipe, all from the card: LoRA weight **0.6–0.85, 0.7 recommended** (a live slider here, which is why the adapter is attached with PEFT rather than folded into the weights). Deployment follows the other MiniMax-H3 Spaces. H3 is 195.9 GiB in bfloat16 and a ZeroGPU Space is evicted at 150 GB of storage, so `MiniMaxH3Ref2VAGeneratorBlocks` is cut at its `text_encoder` step: the 62 GiB Qwen3-VL conditioner runs in [`multimodalart/qwen3vl-conditioner`](https://huggingface.co/spaces/multimodalart/qwen3vl-conditioner) and is called over the gradio API, while this Space holds the DiT and the two autoencoders. The DiT is [`multimodalart/MiniMax-H3-Pruned`](https://huggingface.co/multimodalart/MiniMax-H3-Pruned)'s `transformer_ref` (37.5 GiB — the AdaLN input projections folded onto their reachable rank-8 subspace, 1.5e-5 relative, ~250x below one bfloat16 step), which this LoRA does not touch: it adapts `attn.qkv_proj`, `attn.out_proj` and `mlp.fc1`/`fc2` only. """ from __future__ import annotations import os import tempfile import time import traceback from functools import cache # Before anything that could initialize CUDA: `import spaces` patches `torch.cuda` so the weights can be loaded at # startup rather than on GPU time. import spaces # noqa: F401 import gradio as gr import torch VERSION = "tpv-lora" MODEL_REPO = os.environ.get("H3_MODEL_REPO", "multimodalart/MiniMax-H3-Pruned") LORA_REPO = os.environ.get("H3_LORA_REPO", "WarmBloodAban/Minimax_H3_LoRAs") LORA_FILE = os.environ.get("H3_LORA_FILE", "Minimax-h3_Third_person_view.safetensors") ADAPTER = "third_person_view" CONDITIONER_SPACE = os.environ.get("H3_CONDITIONER", "multimodalart/qwen3vl-conditioner") # `lazy` moves the weights onto the card inside the first GPU call and leaves them there. Startup placement is not an # option: `spaces`' startup `torch.pack()` writes a second on-disk copy of every startup-resident CUDA tensor, and # 48 GB of weights plus its pack runs at the 150 GB Space storage quota. PLACEMENT = os.environ.get("H3_PLACEMENT", "lazy").lower() # cuDNN's fused attention is 10-20% faster than the SDPA default on this pool and needs nothing installed. # flash-attention 3 is sm90-only and this card is sm120 (the `zero-a10g` flavour name is legacy). ATTENTION = os.environ.get("H3_ATTENTION", "_native_cudnn").lower() GPU_SIZE = os.environ.get("H3_GPU_SIZE", "xlarge") MIN_GPU_DURATION = int(os.environ.get("H3_GPU_DURATION_MIN", "120")) MAX_GPU_DURATION = int(os.environ.get("H3_GPU_DURATION_MAX", "1500")) # The card's recommended weight window, verbatim: "LoRA Weight: 0.6 - 0.85 (0.7 is recommended as a starting point)". WEIGHT_MIN, WEIGHT_MAX, DEFAULT_WEIGHT = 0.6, 0.85, 0.7 DEFAULT_STEPS = 20 DEFAULT_SEED = 42 # Must stay identical to the conditioner's table: the *label* goes over the wire, so a canvas that half does not know # is rejected there and surfaces as a failure here. CANVASES = { # 16:9 "960x544 · 16:9 fast": (544, 960), "1024x576 · 16:9 fast": (576, 1024), "1152x640 · 16:9": (640, 1152), "1280x704 · 16:9": (704, 1280), "1344x768 · 16:9 full": (768, 1344), # 21:9 "1152x512 · 21:9 fast": (512, 1152), "1536x672 · 21:9 full": (672, 1536), # 9:16 "544x960 · 9:16 fast": (960, 544), "640x1152 · 9:16": (1152, 640), # 4:3 "768x576 · 4:3 fast": (576, 768), "1024x768 · 4:3 full": (768, 1024), } # Game cutscenes are widescreen; the cheapest 16:9 bucket keeps a default request inside a few minutes of GPU. DEFAULT_CANVAS = "960x544 · 16:9 fast" FPS, FRAMES_PER_CHUNK, LATENTS_PER_CHUNK = 24, 17, 5 MIN_DURATION, MAX_UI_DURATION = 2, 14 MAX_IMAGE_SLOTS, OPEN_IMAGE_SLOTS = 3, 2 # Seconds of GPU one request needs, from the packed sequence it is about to denoise: linear in the rows for the # matmuls, quadratic for the attention. Fit shared with the other MiniMax-H3 Spaces on this pool. STEP_LINEAR, STEP_QUADRATIC, SAFETY = 1.1745e-4, 3.8396e-9, 1.15 # The fit above was measured with the first-block cache off. With it on — the default — roughly a third of the # forwards skip the trunk, and the measured request below came in well under the uncached prediction. Both numbers # are calibrated against this Space's own smoke test: S=40696 at 20 steps reserved 445s under the old 1.3/90/no-FBC # constants and actually used 250s of GPU including placement, so a visitor's quota was paying for 195 idle seconds. FBC_SPEEDUP = float(os.environ.get("H3_FBC_SPEEDUP", "1.45")) PLACEMENT_ALLOWANCE = int(os.environ.get("H3_PLACEMENT_ALLOWANCE", "60")) AUDIO_LATENTS_PER_SECOND, AUDIO_CHANNELS = 40, 2 REFERENCE_IMAGE_SHORT_EDGE, CANVAS_MULTIPLE = 2048, 32 DECODE_BASE, DECODE_PER_DEFAULT_CANVAS, DEFAULT_CANVAS_PIXELS = 15, 25, 960 * 544 * 124 # A 2048-short-edge reference also becomes Qwen3-VL vision tokens in the text stream. Only the *estimate* label needs # a number for them; the booking itself reads the tags the conditioner actually returned. ESTIMATED_TEXT_ROWS, ESTIMATED_VISION_ROWS_PER_IMAGE = 900, 1800 # ── The LoRA's prompt contract ──────────────────────────────────────────────── # Roles, camera phrasings and the HUD clause are built out of the card's own "Key Trigger Words & Recommended Tags" # (camera views: third-person perspective, over-the-shoulder camera, first-person POV, FPV HUD; game rendering: # rendered in Unreal Engine, gameplay sequence, combat stance; UI elements: transparent HUD elements, target-lock UI # reticle, QTE UI prompt, floating damage text UI) and out of the structured example prompt it publishes. ROLES = { "Character": ("subject", "character design reference sheet"), "Creature / Boss": ("subject", "monster design reference sheet"), "Prop / Weapon": ("subject", "prop design reference sheet"), "Environment": ("environment", "environment design reference sheet"), } CAMERAS = { "Third-person · over-the-shoulder": { "genre": "third-person", "summary": "tracked continuously by an over-the-shoulder spring-arm camera", "aesthetic": "third-person over-the-shoulder spring-arm camera tracking", "opening": "an over-the-shoulder camera locked 3.0 meters directly behind", "continues": ( "The camera holds its over-the-back angle and adjusts its spring-arm distance with every movement, " "with hit-impulse micro-shakes on each impact." ), }, "Third-person · orbiting combat camera": { "genre": "third-person", "summary": "tracked by an orbiting third-person combat camera", "aesthetic": "orbiting third-person combat camera with a dynamic spring-arm distance", "opening": "a third-person combat camera orbiting 4.0 meters around", "continues": ( "The camera orbits around the action and tightens its distance on every impact, keeping the combat " "stance centred in frame." ), }, "First-person POV (FPV)": { "genre": "first-person", "summary": "shown entirely in first-person POV with FPV framing", "aesthetic": "first-person POV (FPV) camera with weapon-in-hand framing", "opening": "a first-person POV camera looking out through the eyes of", "continues": ( "The camera stays in first-person POV throughout, with FPV head-bob, fast view whip-pans and " "weapon-in-hand framing at the bottom of frame." ), }, "Whip-pan: third-person → first-person": { "genre": "third-person", "summary": "starting over-the-shoulder and whip-panning into first-person POV", "aesthetic": "third-person to first-person perspective transition driven by a fast camera whip-pan", "opening": "an over-the-shoulder camera locked 3.0 meters behind", "continues": ( "Mid-sequence the camera whip-pans forward into a first-person POV view and holds it to the end of " "the shot." ), }, } DEFAULT_CAMERA = "Third-person · over-the-shoulder" HUD_CLAUSE = ( ", and transparent combat HUD elements in the top-right corner: a target-lock UI reticle, a Boss health bar, " "floating damage text UI and QTE UI prompts" ) NO_HUD_CLAUSE = ", and a clean cinematic frame with no UI overlays" DEFAULT_SOUNDSCAPE = ( "Footsteps on wet stone, weapon impacts and metallic blade clashes, monstrous roars, thruster and dash bursts, " "and crisp action RPG combat UI audio effects." ) DEFAULT_MUSIC = ( "An intense, high-tempo epic battle track, orchestral-electronic, with heavy industrial percussion and a low " "pulsing synth bass that swells through the sequence." ) def snap_frames(seconds: float) -> int: """The frame count MiniMax-H3's video VAE can decode: the next `17 * n + 5` at 24 fps.""" frames = max(1, round(float(seconds) * FPS)) while frames % FRAMES_PER_CHUNK != LATENTS_PER_CHUNK: frames += 1 return frames def lower_duration_floor(seconds: float = MIN_DURATION) -> None: """Let the pipeline generate below its 5 s floor. 56 frames (2.33 s) is fine on this checkpoint.""" from diffusers.modular_pipelines.minimax_h3.modular_pipeline import MiniMaxH3ModularPipeline MiniMaxH3ModularPipeline.min_duration = property(lambda self: float(seconds)) def video_latent_frames(num_frames: int) -> int: """`17 * n + 5` frames become `5 * n + 2` video latents.""" return 5 * ((num_frames - LATENTS_PER_CHUNK) // FRAMES_PER_CHUNK) + 2 def target_rows(height: int, width: int, num_frames: int) -> int: """The generated rows of the packed sequence: video patched `(1, 2, 2)`, plus two audio rows per latent.""" video = video_latent_frames(num_frames) * (height // CANVAS_MULTIPLE) * (width // CANVAS_MULTIPLE) return video + round(num_frames / FPS * AUDIO_LATENTS_PER_SECOND) * AUDIO_CHANNELS def reference_rows(image_paths: list[str]) -> int: """The rows the image reference blocks add, from metadata alone — no decode. A reference image is resized to a 2048-pixel short edge and encoded as a single frame, so a squarer reference is a cheaper one. """ from PIL import Image rows = 0 for path in image_paths: if not path: continue try: width, height = Image.open(path).size except Exception: width, height = 1024, 1024 scale = REFERENCE_IMAGE_SHORT_EDGE / min(width, height) resolved = [ max(CANVAS_MULTIPLE, round(edge * scale / CANVAS_MULTIPLE) * CANVAS_MULTIPLE) for edge in (height, width) ] rows += (resolved[0] // CANVAS_MULTIPLE) * (resolved[1] // CANVAS_MULTIPLE) return rows def denoise_seconds(sequence: int, steps: int) -> float: import h3_fbc uncached = int(steps) * (STEP_LINEAR * sequence + STEP_QUADRATIC * sequence**2) * SAFETY return uncached / (FBC_SPEEDUP if h3_fbc.ENABLED else 1.0) def get_duration(prompt_embeds, text_token_tags, image_paths, height, width, num_frames, steps, weight, seed, **_): """Seconds of GPU to reserve for one request, from the packed sequence it is about to denoise.""" sequence = int(text_token_tags.shape[0]) + reference_rows(image_paths) + target_rows(height, width, num_frames) denoise = denoise_seconds(sequence, steps) encode = 5 + reference_rows(image_paths) * 1e-3 decode = DECODE_BASE + DECODE_PER_DEFAULT_CANVAS * (height * width * num_frames) / DEFAULT_CANVAS_PIXELS total = PLACEMENT_ALLOWANCE + encode + denoise + decode + 10 duration = max(MIN_GPU_DURATION, min(MAX_GPU_DURATION, int(total))) print(f"[{VERSION}] S={sequence} -> reserving {duration}s ({denoise:.0f}s of denoise at {steps} steps)", flush=True) return duration def estimate_label(canvas, duration, steps, *image_paths) -> str: """The same estimate, phrased for the UI, so the cost of a canvas / duration / steps choice is visible up front.""" height, width = CANVASES.get(canvas, CANVASES[DEFAULT_CANVAS]) num_frames = snap_frames(duration) paths = [path for path in image_paths if path] refs = reference_rows(paths) sequence = ESTIMATED_TEXT_ROWS + ESTIMATED_VISION_ROWS_PER_IMAGE * len(paths) + refs sequence += target_rows(height, width, num_frames) seconds = int(PLACEMENT_ALLOWANCE + 5 + denoise_seconds(sequence, steps) + DECODE_BASE) return ( f"{width}x{height} · {num_frames} frames ({num_frames / FPS:.2f} s) · {int(steps)} steps · " f"{len(paths)} reference{'' if len(paths) == 1 else 's'} → roughly " f"**{seconds // 60}m {seconds % 60:02d}s** of GPU time" ) # ── The LoRA, onto the diffusers port ──────────────────────────────────────── # # `Minimax-h3_Third_person_view.safetensors` is an ai-toolkit export: 416 tensors, rank 64, no `.alpha`, keys # `diffusion_model.blocks.N.{attn.qkv_proj,attn.out_proj,mlp.fc1,mlp.fc2}.lora_{A,B}.weight` plus the same four under # `token_refiner.blocks.N`, and `ss_base_model_version: minimax_h3_ref2va` in its metadata. diffusers serves the # *converted* port, so every name — and, for two of them, the row layout of `lora_B` — has to be pushed through the # same transforms `scripts/convert_minimax_h3_to_diffusers.py` applied to the base weights. This reproduces # `_convert_non_diffusers_minimax_h3_lora_to_diffusers`: # # * `blocks.` -> `transformer_blocks.`, `token_refiner.blocks.` -> `token_refiner.refiner_blocks.`, # `attn.out_proj` -> `attn.to_out.0`, `mlp.fc2` -> `ff.net.2` (pure renames), # * `mlp.fc1` -> `ff.net.0.proj` with its two fused halves swapped: the reference computes `fc2(silu(gate) * value)` # from a fused `[gate; value]` while diffusers' `SwiGLU` computes `value * silu(gate)` from a fused # `[value; gate]`, so the halves trade places, # * `attn.qkv_proj` -> `to_q` / `to_k` / `to_v`: split `lora_B`'s 21504 rows into **contiguous** thirds of 7168 # (`num_attention_heads * attention_head_dim`). # # That last split is where MiniMax-H3 LoRAs diverge. DiffSynth-Studio exports run the raw checkpoint's *per-head # interleaved* fused QKV and have to be de-interleaved before the split; they are identified by peft's `.default.` # infix over these names. ai-toolkit exports under `diffusion_model.` are already `[q_all; k_all; v_all]` and must # **not** be reordered — de-interleaving this file anyway would scatter each head's q/k/v across all three # projections and turn the adapter into structured noise on all 50 blocks. # # Row transforms only ever touch `lora_B`, so the three attention projections share one `lora_A`: the rows of `B @ A` # are the rows of `B`, which makes the split exact rather than an approximation. def _lora_target_name(source_name: str) -> str: """Map an original-checkpoint module path to the diffusers one (no `.lora_A/B.*` suffix).""" if source_name.startswith("token_refiner.blocks."): return source_name.replace("token_refiner.blocks.", "token_refiner.refiner_blocks.", 1) if source_name.startswith("blocks."): return source_name.replace("blocks.", "transformer_blocks.", 1) return source_name def _lora_modules(name: str, a_weight, b_weight, inner_dim: int): """Yield `(diffusers_module_path, lora_A, lora_B)` for one original module path.""" target = _lora_target_name(name) if target.endswith(".attn.qkv_proj"): prefix = target.removesuffix("qkv_proj") for kind, part in zip(("q", "k", "v"), b_weight.split(inner_dim, dim=0)): yield f"{prefix}to_{kind}", a_weight, part.contiguous() elif target.endswith(".mlp.fc1"): gate, value = b_weight.chunk(2, dim=0) yield target.replace(".mlp.fc1", ".ff.net.0.proj"), a_weight, torch.cat([value, gate]).contiguous() elif target.endswith(".mlp.fc2"): yield target.replace(".mlp.fc2", ".ff.net.2"), a_weight, b_weight elif target.endswith(".attn.out_proj"): yield target.replace(".attn.out_proj", ".attn.to_out.0"), a_weight, b_weight else: raise ValueError( f"unexpected LoRA target `{name}`: this adapter is documented as attention + feed-forward only, so a " "new module type means the checkpoint changed" ) def build_lora_state_dict(transformer) -> tuple[dict, int, int]: """Download the LoRA and remap it into a PEFT-format state dict for the diffusers transformer. Every target is validated against the transformer's own parameter shapes, and an unresolved one is fatal: a silently dropped target means the name mapping is wrong and the Space would serve a half-applied adapter that still *looks* like it worked. """ from huggingface_hub import hf_hub_download from safetensors.torch import load_file raw = load_file(hf_hub_download(LORA_REPO, LORA_FILE)) prefix, suffix_a, suffix_b = "diffusion_model.", ".lora_A.weight", ".lora_B.weight" unexpected = [key for key in raw if not (key.startswith(prefix) and key.endswith((suffix_a, suffix_b)))] if unexpected: raise ValueError(f"{LORA_FILE} holds {len(unexpected)} unexpected tensors, e.g. {unexpected[:5]}") bases = sorted({key[len(prefix) : -len(suffix_a)] for key in raw if key.endswith(suffix_a)}) if not bases: raise ValueError(f"No `{prefix}*{suffix_a}` / `{suffix_b}` pairs found in {LORA_FILE}") ranks = set() for name in bases: if f"{prefix}{name}{suffix_b}" not in raw: raise ValueError(f"LoRA is missing the lora_B twin of {prefix}{name}{suffix_a}") ranks.add(raw[f"{prefix}{name}{suffix_a}"].shape[0]) if len(ranks) != 1: raise ValueError(f"LoRA mixes ranks {sorted(ranks)}; this loader assumes a single rank") rank = ranks.pop() inner_dim = transformer.config.num_attention_heads * transformer.config.attention_head_dim base_shapes = {key: tuple(value.shape) for key, value in transformer.state_dict().items()} state_dict: dict[str, torch.Tensor] = {} missed: list[str] = [] for name in bases: a_weight = raw[f"{prefix}{name}{suffix_a}"] b_weight = raw[f"{prefix}{name}{suffix_b}"] for module, a_part, b_part in _lora_modules(name, a_weight, b_weight, inner_dim): base = base_shapes.get(f"{module}.weight") if base is None: missed.append(module) continue # `W` is [out, in]; the adapter must be `lora_B` [out, r] @ `lora_A` [r, in]. if (b_part.shape[0], a_part.shape[1]) != base: raise ValueError( f"LoRA delta for `{module}` would be {(b_part.shape[0], a_part.shape[1])}, " f"base weight is {base}" ) state_dict[f"{module}.lora_A.weight"] = a_part state_dict[f"{module}.lora_B.weight"] = b_part if missed: raise ValueError( f"{len(missed)} LoRA targets matched no transformer weight, e.g. {missed[:5]}. " "The LoRA and the diffusers transformer disagree on module naming." ) return state_dict, rank, len(bases) def load_and_apply_lora(transformer) -> str: """Attach the LoRA as a PEFT adapter so its strength stays a per-request knob.""" state_dict, rank, targets = build_lora_state_dict(transformer) # `prefix=None` because the keys are already the transformer's own module paths, and no `network_alphas` because # the file carries no `.alpha` tensors — PEFT then sets alpha == rank, i.e. the adapter's intrinsic scale is 1.0 # and the `set_adapters` weight *is* the strength the card talks about. transformer.load_lora_adapter(state_dict, prefix=None, adapter_name=ADAPTER) transformer.set_adapters([ADAPTER], [DEFAULT_WEIGHT]) return ( f"LoRA attached · {targets} checkpoint targets -> {len(state_dict) // 2} diffusers modules, rank {rank}, " f"alpha == rank · default weight {DEFAULT_WEIGHT:g} · {LORA_FILE}" ) # ── Model loading ──────────────────────────────────────────────────────────── PIPE = None LOAD_ERROR: str | None = None LORA_STATUS: str | None = None def load_models() -> str | None: """Load the denoising half at startup, but *not* onto the card — see `PLACEMENT`. `MiniMaxH3Ref2VAGeneratorBlocks` declares `transformer_ref`, `vae`, `audio_vae`, the two schedulers and `video_processor`, so `load_components` fetches exactly those subfolders — `text_encoder/` and the `transformer/` partition are never touched. Both autoencoders carry `_keep_in_fp32_modules` over every module and stay float32: a bfloat16 audio VAE decodes the soundtrack roughly 20 dB too quiet. """ global PIPE, LOAD_ERROR, LORA_STATUS if PIPE is not None or LOAD_ERROR is not None: return LOAD_ERROR started = time.time() try: from diffusers import ComponentsManager from h3_split_blocks import MiniMaxH3Ref2VAGeneratorBlocks lower_duration_floor() blocks = MiniMaxH3Ref2VAGeneratorBlocks() print(f"[{VERSION}] loading {[c.name for c in blocks.expected_components]} from {MODEL_REPO} ...", flush=True) pipe = blocks.init_pipeline(MODEL_REPO, components_manager=ComponentsManager(), collection="h3") # The pruned DiT is served as remote code (`transformer_ref/modeling_minimax_h3_pruned.py`, reached through # the `AutoModel` type hint in `modular_model_index.json`). `load_components` forwards `trust_remote_code` # only to components that live in the pipeline's own repo, so the two VAEs and the schedulers — which # `modular_model_index.json` still points at `MiniMaxAI/MiniMax-H3` — never see it. pipe.load_components(dtype=torch.bfloat16, trust_remote_code=True) # Both VAEs explicitly, and before the transformer: `set_attention_backend` also sets the registry's *global* # backend, and the float32 audio VAE has no cuDNN kernel. pipe.vae.set_attention_backend("native") pipe.audio_vae.set_attention_backend("native") pipe.transformer_ref.set_attention_backend(ATTENTION) # Before any GPU placement, so the adapter's parameters travel with the transformer. No AoTI package on this # Space on purpose: a compiled block graph is captured around the base linears and would silently bypass the # LoRA layers PEFT injects. LORA_STATUS = load_and_apply_lora(pipe.transformer_ref) print(f"[{VERSION}] {LORA_STATUS}", flush=True) import h3_fbc print(f"[h3-fbc] {h3_fbc.status()}", flush=True) PIPE = pipe print(f"[{VERSION}] ready in {time.time() - started:.0f}s", flush=True) except Exception as error: traceback.print_exc() LOAD_ERROR = ( f"**Loading `{MODEL_REPO}` failed** after {time.time() - started:.0f}s: " f"`{type(error).__name__}: {error}`" ) return LOAD_ERROR # ── Prompt composition ─────────────────────────────────────────────────────── def collect_slots(slots) -> list[tuple[str, str, str]]: """The `(path, role, description)` references of a request, in the order the model reads them. That order numbers `` and advances the shared audio/video rotary clock, so the same references in a different order are a different request. """ return [(path, role, (detail or "").strip()) for path, role, detail in slots if path] def compose_prompt(references, action, camera, hud, soundscape, music, seconds) -> str: """Assemble MiniMax-H3's structured document in the layout the LoRA's card publishes.""" action = (action or "").strip().rstrip(".") spec = CAMERAS.get(camera, CAMERAS[DEFAULT_CAMERA]) entries, subjects, environments = [], 0, 0 for index, (_, role, detail) in enumerate(references, start=1): kind, sheet = ROLES.get(role, ROLES["Character"]) if kind == "subject": subjects += 1 label = f"" else: environments += 1 label = f"" entries.append((label, index, detail, sheet)) subject = next((label for label, _, _, sheet in entries if "environment" not in sheet), "the player character") environment = next((label for label, _, _, sheet in entries if "environment" in sheet), None) place = f" in {environment}" if environment else "" definitions = [ f"{label} is {detail} from ." if detail else f"{label} is the subject shown in ." for label, index, detail, _ in entries ] definitions += [f" is the {sheet} for {label}." for label, index, _, sheet in entries] retention = [ f"{label} (appears in [Shot 1]): fully_preserved - the design, colours and silhouette from " f" are fully preserved." for label, index, _, _ in entries ] sections = [ "subject_definitions:\n" + "\n".join(definitions), ( "summary:\n" f"[reference generation] A {float(seconds):.1f}-second photorealistic {spec['genre']} action RPG " f"gameplay sequence{place}, where {subject} {action}, {spec['summary']}." ), "retention_analysis:\n" + "\n".join(retention), ( "detailed_description:\n" f"The target video features a photorealistic 3D {spec['genre']} action RPG gameplay aesthetic rendered " f"in Unreal Engine with real-time game mechanics, {spec['aesthetic']}, deep depth of field" f"{HUD_CLAUSE if hud else NO_HUD_CLAUSE}.\n" f"[Shot 1] The video opens with {spec['opening']} {subject}{place}. {action.capitalize()}. " f"{spec['continues']}" ), ] if (soundscape or "").strip(): sections.append("overall_soundscape:\n" + soundscape.strip()) if (music or "").strip(): sections.append("non_diegetic_music:\n" + music.strip()) return "\n\n".join(sections) # ── Guard, conditioner ─────────────────────────────────────────────────────── def check_prompt(prompt: str) -> None: """The NCII guard. Every request here carries an uploaded image, so every request is checked; it runs before the conditioner call and the denoise booking, so a refused prompt costs no GPU time on either half.""" import ncii_guard try: flag = ncii_guard.classify(prompt) except Exception as error: traceback.print_exc() raise gr.Error(f"The content filter is unavailable (`{type(error).__name__}`), so nothing was run.") if flag["label"] == "ncii": print(f"[guard] prompt refused (ncii {flag['score']:.2f})", flush=True) raise gr.Error("This prompt was flagged by a content filter and wasn't run.") @cache def conditioner(): """The other half, over the gradio API.""" from gradio_client import Client return Client(CONDITIONER_SPACE) def conditioner_client(ip_token): """A client carrying *this* caller's ZeroGPU token, so the conditioner's own booking is billed to whoever asked for the video rather than to this Space.""" if not ip_token: return conditioner() from gradio_client import Client return Client(CONDITIONER_SPACE, headers={"x-ip-token": ip_token}) def caller_ip_token() -> str | None: from gradio.context import LocalContext request = LocalContext.request.get() return request.headers.get("x-ip-token") if request is not None else None def encode_remote(prompt, image_paths, canvas, num_frames, ip_token=None): """`/encode_ref2va` on the conditioner Space: a safetensors file holding `prompt_embeds` + `text_token_tags`, with the resolved `height` / `width` / `num_frames` in its metadata, plus the plan. `canvas` is the label. `media` and `kinds` are parallel and ordered; the references go over because `ref2va`'s presentation puts a vision block in front of the prompt for every image. """ from gradio_client import handle_file from safetensors import safe_open def call(): return conditioner_client(ip_token).predict( prompt=prompt, media=[handle_file(image) for image in image_paths], kinds=",".join("image" for _ in image_paths), canvas=canvas, num_frames=num_frames, # Never rewritten: H3's own LLM expansion would paraphrase away the card's structured document — the # `` bindings and the trigger vocabulary this adapter was trained against. rewrite_prompt=False, api_name="/encode_ref2va", ) try: path, plan = call() except Exception as first: # A cached client whose upstream Space restarted fails once; a fresh handshake usually succeeds. print(f"[conditioner] retrying with a fresh client after: {first}", flush=True) conditioner.cache_clear() path, plan = call() with safe_open(path, framework="pt") as handle: return handle.get_tensor("prompt_embeds"), handle.get_tensor("text_token_tags"), handle.metadata(), plan # ── Inference ──────────────────────────────────────────────────────────────── @spaces.GPU(duration=get_duration, size=GPU_SIZE) def _generate(prompt_embeds, text_token_tags, image_paths, height, width, num_frames, steps, weight, seed): """The only thing on GPU time: the reference encoder, the packed-sequence denoise loop and the decoders. References cross as paths and are decoded here; only the generated outputs come back. A `@spaces.GPU` argument crosses a process boundary by pickling, and expanded reference frames are large. """ from diffusers.modular_pipelines.minimax_h3 import MiniMaxH3ImageReference import h3_fbc if PLACEMENT == "lazy": PIPE.to("cuda") # The whole reason the adapter is not folded into the weights: strength is a per-request scale on the LoRA layers. PIPE.transformer_ref.set_adapters([ADAPTER], [float(weight)]) with h3_fbc.enabled(PIPE.transformer_ref, steps=int(steps)): state = PIPE( prompt_embeds=prompt_embeds.to("cuda"), text_token_tags=text_token_tags, references=[MiniMaxH3ImageReference.from_file(path) for path in image_paths], height=height, width=width, num_frames=num_frames, num_inference_steps=int(steps), generator=torch.Generator("cpu").manual_seed(int(seed)), ) return state.get("videos")[0], state.get("audio")[0].cpu(), state.get("sampling_rate") def generate( # The first nine are the columns `gr.Examples` varies, and they lead the signature for that reason: an example row # is applied to `inputs` positionally. Every parameter has a default, which is what lets a nine-column row call # this at all. action="", picture_1=None, role_1="Character", detail_1="", picture_2=None, role_2="Environment", detail_2="", camera=DEFAULT_CAMERA, hud=True, picture_3=None, role_3="Creature / Boss", detail_3="", canvas=DEFAULT_CANVAS, duration=5, steps=DEFAULT_STEPS, weight=DEFAULT_WEIGHT, seed=DEFAULT_SEED, soundscape=DEFAULT_SOUNDSCAPE, music=DEFAULT_MUSIC, override="", progress=gr.Progress(track_tqdm=True), ): """One request: compose the structured prompt, condition it remotely, denoise here.""" if LOAD_ERROR: raise gr.Error(LOAD_ERROR) if PIPE is None: raise gr.Error("The denoiser is still loading.") from diffusers.utils import encode_video references = collect_slots( [(picture_1, role_1, detail_1), (picture_2, role_2, detail_2), (picture_3, role_3, detail_3)] ) if not references: raise gr.Error("Add at least one reference sheet — a character, an environment or a creature.") if not (override or "").strip() and not (action or "").strip(): raise gr.Error("Write one line of action, or paste a full structured prompt in the advanced options.") num_frames = snap_frames(duration) prompt = (override or "").strip() or compose_prompt( references, action, camera, hud, soundscape, music, num_frames / FPS ) check_prompt(prompt) image_paths = [path for path, _, _ in references] progress(0.0, desc="Reading the prompt and the reference sheets ...") conditioned = time.time() try: prompt_embeds, text_token_tags, metadata, plan = encode_remote( prompt, image_paths, canvas, num_frames, ip_token=caller_ip_token() ) except gr.Error: raise except Exception as error: # gradio only puts the exception *type* on the wire, so the useful half of a conditioner-side failure is in # that Space's logs. traceback.print_exc() raise gr.Error( f"The conditioner ({CONDITIONER_SPACE}) failed with `{type(error).__name__}: {error}`. " "Its logs carry the full traceback." ) from error condition_seconds = time.time() - conditioned height, width, num_frames = (int(metadata[key]) for key in ("height", "width", "num_frames")) progress(0.1, desc=f"Rendering {num_frames / FPS:.1f} s at {width}x{height} ...") started = time.time() frames, audio, sampling_rate = _generate( prompt_embeds, text_token_tags, image_paths, height, width, num_frames, steps, weight, seed ) generate_seconds = time.time() - started directory = os.path.join(tempfile.gettempdir(), "h3-outputs") os.makedirs(directory, exist_ok=True) path = os.path.join(directory, f"h3-tpv-{int(time.time() * 1000)}.mp4") encode_video(frames, fps=FPS, output_path=path, audio=audio, audio_sample_rate=sampling_rate) print( f"[{VERSION}] {len(image_paths)} references · {width}x{height}, {num_frames} frames " f"({num_frames / FPS:.3f} s), {int(steps)} steps, weight {float(weight):g} · conditioner " f"{condition_seconds:.0f}s ({plan['num_text_tokens']} tokens) · denoise + decode {generate_seconds:.0f}s " f"({generate_seconds / int(steps):.1f} s/step) · seed {int(seed)}", flush=True, ) return path, prompt # ── UI ─────────────────────────────────────────────────────────────────────── try: import ncii_guard ncii_guard.start() except Exception as guard_error: # the guard revives itself per request; a cold cache should not fail the boot print(f"[guard] start failed ({type(guard_error).__name__}: {guard_error}); it will be retried per request") load_models() INTRO = """# MiniMax-H3 · Third-Person View LoRA **Game cutscenes out of design reference sheets.** Label each sheet you upload, write one line of action, pick a camera, and this composes MiniMax-H3's structured prompt around the LoRA's own vocabulary — third-person over-the-shoulder / spring-arm tracking, first-person POV, whip-pan transitions, target-lock reticles and Boss health bars — then renders the clip **with its synchronized soundtrack** in a single denoising pass. """ CSS = """ .main.fillable { max-width: 1250px !important; } .dark .gradio-container { color: var(--body-text-color); } """ with gr.Blocks(title="MiniMax-H3 · Third-Person View LoRA") as demo: gr.Markdown(INTRO) if LOAD_ERROR: gr.Markdown(LOAD_ERROR) with gr.Row(): with gr.Column(): gr.Markdown("### 1 · Reference sheets — they become ``, in this order") pictures, roles, details, slot_columns = [], [], [], [] with gr.Row(): for index in range(MAX_IMAGE_SLOTS): with gr.Column(min_width=180, visible=index < OPEN_IMAGE_SLOTS) as column: pictures.append( gr.Image(label=f"Picture {index + 1}", type="filepath", height=190, show_label=True) ) roles.append( gr.Dropdown( label="Role", choices=list(ROLES), value="Character" if index == 0 else ("Environment" if index == 1 else "Creature / Boss"), container=True, ) ) details.append( gr.Textbox( label="What's in it", lines=2, placeholder="the mint-green cat mecha with a tattered beige cape", ) ) slot_columns.append(column) add_slot = gr.Button("+ Add a third reference sheet", size="sm", variant="secondary") action = gr.Textbox( label="2 · What happens in the shot", lines=2, placeholder="dodges a leaping wyvern attack, counterattacks with an energy-blade combo, then parries " "a wing-claw swipe", ) with gr.Row(): camera = gr.Dropdown(label="3 · Camera", choices=list(CAMERAS), value=DEFAULT_CAMERA, scale=3) hud = gr.Checkbox(label="Combat HUD overlay", value=True, scale=1) run = gr.Button("Render the cutscene", variant="primary") estimate = gr.Markdown(estimate_label(DEFAULT_CANVAS, 5, DEFAULT_STEPS, None)) with gr.Accordion("Advanced options", open=False): weight = gr.Slider( label="LoRA weight (the card recommends 0.6–0.85)", minimum=WEIGHT_MIN, maximum=WEIGHT_MAX, step=0.05, value=DEFAULT_WEIGHT, ) canvas = gr.Dropdown(label="Canvas", choices=list(CANVASES), value=DEFAULT_CANVAS) duration = gr.Slider( label="Duration (s)", minimum=MIN_DURATION, maximum=MAX_UI_DURATION, step=1, value=5 ) steps = gr.Slider(label="Steps", minimum=10, maximum=40, step=1, value=DEFAULT_STEPS) seed = gr.Number(label="Seed", value=DEFAULT_SEED, precision=0) soundscape = gr.Textbox(label="Diegetic soundscape", lines=2, value=DEFAULT_SOUNDSCAPE) music = gr.Textbox(label="Non-diegetic music", lines=2, value=DEFAULT_MUSIC) override = gr.Textbox( label="Write the structured prompt myself (overrides everything above)", lines=6, placeholder="subject_definitions:\n is ... from .\n\nsummary:\n" "[reference generation] A 10-second photorealistic third-person ...", ) with gr.Column(): result = gr.Video(label="Cutscene + soundtrack") with gr.Accordion("Structured prompt sent to MiniMax-H3", open=False): prompt_view = gr.Textbox(show_label=False, lines=18, interactive=False) open_slots = gr.State(OPEN_IMAGE_SLOTS) def reveal_slot(open_count): open_count = min(open_count + 1, MAX_IMAGE_SLOTS) return [ open_count, *[gr.update(visible=index < open_count) for index in range(MAX_IMAGE_SLOTS)], gr.update(visible=open_count < MAX_IMAGE_SLOTS), ] add_slot.click(reveal_slot, open_slots, [open_slots, *slot_columns, add_slot], api_name=False) for control in (canvas, duration, steps, *pictures): control.change( estimate_label, [canvas, duration, steps, *pictures], estimate, show_progress="hidden", api_name=False, ) request = [ action, pictures[0], roles[0], details[0], pictures[1], roles[1], details[1], camera, hud, pictures[2], roles[2], details[2], canvas, duration, steps, weight, seed, soundscape, music, override, ] exampled = request[:9] gr.Examples( examples=[ [ "dodges a leaping wyvern attack, counterattacks with a glowing energy-blade combo, then parries a " "wing-claw swipe in a shower of sparks", "examples/katana_character.jpg", "Character", "the young woman with long black hair in two buns, blue eyes, a black mouth mask, an orange utility " "jacket and two white-corded katana hilts rising behind her shoulders", "examples/ruined_village.jpg", "Environment", "the foggy ruined village street with wooden houses, a glowing paper lantern, blood-red spider " "lilies and wet stone paving", "Third-person · over-the-shoulder", True, ], [ "plants her feet into a combat stance and charges an ultimate finisher as the pale wyvern staggers " "and opens its ring-like tooth-lined maw", "examples/katana_character.jpg", "Character", "the young woman with long black hair in two buns, an orange utility jacket and two katana hilts " "rising behind her shoulders", "examples/wyvern_boss.jpg", "Creature / Boss", "the pale eyeless wyvern with a circular ring-like maw lined with rows of sharp teeth, a long " "muscular neck, clawed wings and heavy bipedal limbs", "Whip-pan: third-person → first-person", True, ], [ "sprints across a rain-slick rooftop, vaults a neon billboard gap and lands into a slide while " "tracer fire streaks past", "examples/doorway_character.jpg", "Character", "the young girl with brown hair, a white shirt and skirt and a guitar case slung across her back", "examples/night_city.jpg", "Environment", "the night city of lit high-rise towers and a faceted glass skyscraper reflected in the dark river " "below", "First-person POV (FPV)", True, ], ], inputs=exampled, outputs=[result, prompt_view], fn=generate, cache_examples=True, cache_mode="lazy", ) run.click(generate, request, [result, prompt_view], api_name="generate") if __name__ == "__main__": # Gradio 6 moved `theme` / `css` off the Blocks constructor and onto `launch`. demo.launch(theme=gr.themes.Citrus(), css=CSS, show_error=True, max_threads=1000)