Spaces:
Running on Zero
Running on Zero
Download app.py from hugging-apps/minimax-h3-third-person-view-lora: direct link, hf CLI and curl.
- Browser
- Download file 45.7 kB
-
https://huggingface.co/spaces/hugging-apps/minimax-h3-third-person-view-lora/resolve/1110d757ef2b61aebf7e162dfd7c54993e1aee72/app.py
- Command line
-
hf download hf://spaces/hugging-apps/minimax-h3-third-person-view-lora@1110d757ef2b61aebf7e162dfd7c54993e1aee72/app.py
-
curl -L -o app.py https://huggingface.co/spaces/hugging-apps/minimax-h3-third-person-view-lora/resolve/1110d757ef2b61aebf7e162dfd7c54993e1aee72/app.py
45.7 kB
| """MiniMax-H3 · Third-Person View LoRA — game cutscenes from design reference sheets. | |
| [`WarmBloodAban/Minimax_H3_LoRAs`](https://huggingface.co/WarmBloodAban/Minimax_H3_LoRAs) ships | |
| `Minimax-h3_Third_person_view.safetensors`, a rank-64 ai-toolkit LoRA whose own metadata names its base as | |
| `minimax_h3_ref2va` — the **reference** partition of MiniMax-H3, the one that conditions on an ordered list of | |
| reference images rather than on a first frame. That is why this Space is a `ref2va` deployment: the adapter goes | |
| onto `transformer_ref`, and the user's uploads arrive as `<Picture 1..3>` in the order they are given. | |
| What the LoRA does, from its card: cinematic game cutscenes — third-person over-the-shoulder / spring-arm camera | |
| tracking, first-person POV, whip-pan view transitions, and native game HUD overlays (target-lock reticles, Boss | |
| health bars, QTE prompts, floating damage text). Its card is explicit that it wants MiniMax-H3's **structured** | |
| prompt format (`subject_definitions` / `summary` / `retention_analysis` / `detailed_description` / | |
| `overall_soundscape` / `non_diegetic_music`) with `<Subject N>` / `<Environment N>` / `<Picture N>` cross-references, | |
| so this demo is a composer for exactly that document rather than a prompt box: you label each reference sheet, write | |
| one line of action, pick a camera and whether the HUD is on, and the Space assembles the card's format around the | |
| LoRA's own trigger vocabulary. The assembled document is shown next to the result, and an override box takes a | |
| hand-written one. | |
| Recipe, all from the card: LoRA weight **0.6–0.85, 0.7 recommended** (a live slider here, which is why the adapter | |
| is attached with PEFT rather than folded into the weights). | |
| Deployment follows the other MiniMax-H3 Spaces. H3 is 195.9 GiB in bfloat16 and a ZeroGPU Space is evicted at 150 GB | |
| of storage, so `MiniMaxH3Ref2VAGeneratorBlocks` is cut at its `text_encoder` step: the 62 GiB Qwen3-VL conditioner | |
| runs in [`multimodalart/qwen3vl-conditioner`](https://huggingface.co/spaces/multimodalart/qwen3vl-conditioner) and is | |
| called over the gradio API, while this Space holds the DiT and the two autoencoders. The DiT is | |
| [`multimodalart/MiniMax-H3-Pruned`](https://huggingface.co/multimodalart/MiniMax-H3-Pruned)'s `transformer_ref` | |
| (37.5 GiB — the AdaLN input projections folded onto their reachable rank-8 subspace, 1.5e-5 relative, ~250x below one | |
| bfloat16 step), which this LoRA does not touch: it adapts `attn.qkv_proj`, `attn.out_proj` and `mlp.fc1`/`fc2` only. | |
| """ | |
| from __future__ import annotations | |
| import os | |
| import tempfile | |
| import time | |
| import traceback | |
| from functools import cache | |
| # Before anything that could initialize CUDA: `import spaces` patches `torch.cuda` so the weights can be loaded at | |
| # startup rather than on GPU time. | |
| import spaces # noqa: F401 | |
| import gradio as gr | |
| import torch | |
| VERSION = "tpv-lora" | |
| MODEL_REPO = os.environ.get("H3_MODEL_REPO", "multimodalart/MiniMax-H3-Pruned") | |
| LORA_REPO = os.environ.get("H3_LORA_REPO", "WarmBloodAban/Minimax_H3_LoRAs") | |
| LORA_FILE = os.environ.get("H3_LORA_FILE", "Minimax-h3_Third_person_view.safetensors") | |
| ADAPTER = "third_person_view" | |
| CONDITIONER_SPACE = os.environ.get("H3_CONDITIONER", "multimodalart/qwen3vl-conditioner") | |
| # `lazy` moves the weights onto the card inside the first GPU call and leaves them there. Startup placement is not an | |
| # option: `spaces`' startup `torch.pack()` writes a second on-disk copy of every startup-resident CUDA tensor, and | |
| # 48 GB of weights plus its pack runs at the 150 GB Space storage quota. | |
| PLACEMENT = os.environ.get("H3_PLACEMENT", "lazy").lower() | |
| # cuDNN's fused attention is 10-20% faster than the SDPA default on this pool and needs nothing installed. | |
| # flash-attention 3 is sm90-only and this card is sm120 (the `zero-a10g` flavour name is legacy). | |
| ATTENTION = os.environ.get("H3_ATTENTION", "_native_cudnn").lower() | |
| GPU_SIZE = os.environ.get("H3_GPU_SIZE", "xlarge") | |
| MIN_GPU_DURATION = int(os.environ.get("H3_GPU_DURATION_MIN", "120")) | |
| MAX_GPU_DURATION = int(os.environ.get("H3_GPU_DURATION_MAX", "1500")) | |
| # The card's recommended weight window, verbatim: "LoRA Weight: 0.6 - 0.85 (0.7 is recommended as a starting point)". | |
| WEIGHT_MIN, WEIGHT_MAX, DEFAULT_WEIGHT = 0.6, 0.85, 0.7 | |
| DEFAULT_STEPS = 20 | |
| DEFAULT_SEED = 42 | |
| # Must stay identical to the conditioner's table: the *label* goes over the wire, so a canvas that half does not know | |
| # is rejected there and surfaces as a failure here. | |
| CANVASES = { | |
| # 16:9 | |
| "960x544 · 16:9 fast": (544, 960), | |
| "1024x576 · 16:9 fast": (576, 1024), | |
| "1152x640 · 16:9": (640, 1152), | |
| "1280x704 · 16:9": (704, 1280), | |
| "1344x768 · 16:9 full": (768, 1344), | |
| # 21:9 | |
| "1152x512 · 21:9 fast": (512, 1152), | |
| "1536x672 · 21:9 full": (672, 1536), | |
| # 9:16 | |
| "544x960 · 9:16 fast": (960, 544), | |
| "640x1152 · 9:16": (1152, 640), | |
| # 4:3 | |
| "768x576 · 4:3 fast": (576, 768), | |
| "1024x768 · 4:3 full": (768, 1024), | |
| } | |
| # Game cutscenes are widescreen; the cheapest 16:9 bucket keeps a default request inside a few minutes of GPU. | |
| DEFAULT_CANVAS = "960x544 · 16:9 fast" | |
| FPS, FRAMES_PER_CHUNK, LATENTS_PER_CHUNK = 24, 17, 5 | |
| MIN_DURATION, MAX_UI_DURATION = 2, 14 | |
| MAX_IMAGE_SLOTS, OPEN_IMAGE_SLOTS = 3, 2 | |
| # Seconds of GPU one request needs, from the packed sequence it is about to denoise: linear in the rows for the | |
| # matmuls, quadratic for the attention. Fit shared with the other MiniMax-H3 Spaces on this pool. | |
| STEP_LINEAR, STEP_QUADRATIC, SAFETY = 1.1745e-4, 3.8396e-9, 1.15 | |
| # The fit above was measured with the first-block cache off. With it on — the default — roughly a third of the | |
| # forwards skip the trunk, and the measured request below came in well under the uncached prediction. Both numbers | |
| # are calibrated against this Space's own smoke test: S=40696 at 20 steps reserved 445s under the old 1.3/90/no-FBC | |
| # constants and actually used 250s of GPU including placement, so a visitor's quota was paying for 195 idle seconds. | |
| FBC_SPEEDUP = float(os.environ.get("H3_FBC_SPEEDUP", "1.45")) | |
| PLACEMENT_ALLOWANCE = int(os.environ.get("H3_PLACEMENT_ALLOWANCE", "60")) | |
| AUDIO_LATENTS_PER_SECOND, AUDIO_CHANNELS = 40, 2 | |
| REFERENCE_IMAGE_SHORT_EDGE, CANVAS_MULTIPLE = 2048, 32 | |
| DECODE_BASE, DECODE_PER_DEFAULT_CANVAS, DEFAULT_CANVAS_PIXELS = 15, 25, 960 * 544 * 124 | |
| # A 2048-short-edge reference also becomes Qwen3-VL vision tokens in the text stream. Only the *estimate* label needs | |
| # a number for them; the booking itself reads the tags the conditioner actually returned. | |
| ESTIMATED_TEXT_ROWS, ESTIMATED_VISION_ROWS_PER_IMAGE = 900, 1800 | |
| # ── The LoRA's prompt contract ──────────────────────────────────────────────── | |
| # Roles, camera phrasings and the HUD clause are built out of the card's own "Key Trigger Words & Recommended Tags" | |
| # (camera views: third-person perspective, over-the-shoulder camera, first-person POV, FPV HUD; game rendering: | |
| # rendered in Unreal Engine, gameplay sequence, combat stance; UI elements: transparent HUD elements, target-lock UI | |
| # reticle, QTE UI prompt, floating damage text UI) and out of the structured example prompt it publishes. | |
| ROLES = { | |
| "Character": ("subject", "character design reference sheet"), | |
| "Creature / Boss": ("subject", "monster design reference sheet"), | |
| "Prop / Weapon": ("subject", "prop design reference sheet"), | |
| "Environment": ("environment", "environment design reference sheet"), | |
| } | |
| CAMERAS = { | |
| "Third-person · over-the-shoulder": { | |
| "genre": "third-person", | |
| "summary": "tracked continuously by an over-the-shoulder spring-arm camera", | |
| "aesthetic": "third-person over-the-shoulder spring-arm camera tracking", | |
| "opening": "an over-the-shoulder camera locked 3.0 meters directly behind", | |
| "continues": ( | |
| "The camera holds its over-the-back angle and adjusts its spring-arm distance with every movement, " | |
| "with hit-impulse micro-shakes on each impact." | |
| ), | |
| }, | |
| "Third-person · orbiting combat camera": { | |
| "genre": "third-person", | |
| "summary": "tracked by an orbiting third-person combat camera", | |
| "aesthetic": "orbiting third-person combat camera with a dynamic spring-arm distance", | |
| "opening": "a third-person combat camera orbiting 4.0 meters around", | |
| "continues": ( | |
| "The camera orbits around the action and tightens its distance on every impact, keeping the combat " | |
| "stance centred in frame." | |
| ), | |
| }, | |
| "First-person POV (FPV)": { | |
| "genre": "first-person", | |
| "summary": "shown entirely in first-person POV with FPV framing", | |
| "aesthetic": "first-person POV (FPV) camera with weapon-in-hand framing", | |
| "opening": "a first-person POV camera looking out through the eyes of", | |
| "continues": ( | |
| "The camera stays in first-person POV throughout, with FPV head-bob, fast view whip-pans and " | |
| "weapon-in-hand framing at the bottom of frame." | |
| ), | |
| }, | |
| "Whip-pan: third-person → first-person": { | |
| "genre": "third-person", | |
| "summary": "starting over-the-shoulder and whip-panning into first-person POV", | |
| "aesthetic": "third-person to first-person perspective transition driven by a fast camera whip-pan", | |
| "opening": "an over-the-shoulder camera locked 3.0 meters behind", | |
| "continues": ( | |
| "Mid-sequence the camera whip-pans forward into a first-person POV view and holds it to the end of " | |
| "the shot." | |
| ), | |
| }, | |
| } | |
| DEFAULT_CAMERA = "Third-person · over-the-shoulder" | |
| HUD_CLAUSE = ( | |
| ", and transparent combat HUD elements in the top-right corner: a target-lock UI reticle, a Boss health bar, " | |
| "floating damage text UI and QTE UI prompts" | |
| ) | |
| NO_HUD_CLAUSE = ", and a clean cinematic frame with no UI overlays" | |
| DEFAULT_SOUNDSCAPE = ( | |
| "Footsteps on wet stone, weapon impacts and metallic blade clashes, monstrous roars, thruster and dash bursts, " | |
| "and crisp action RPG combat UI audio effects." | |
| ) | |
| DEFAULT_MUSIC = ( | |
| "An intense, high-tempo epic battle track, orchestral-electronic, with heavy industrial percussion and a low " | |
| "pulsing synth bass that swells through the sequence." | |
| ) | |
| def snap_frames(seconds: float) -> int: | |
| """The frame count MiniMax-H3's video VAE can decode: the next `17 * n + 5` at 24 fps.""" | |
| frames = max(1, round(float(seconds) * FPS)) | |
| while frames % FRAMES_PER_CHUNK != LATENTS_PER_CHUNK: | |
| frames += 1 | |
| return frames | |
| def lower_duration_floor(seconds: float = MIN_DURATION) -> None: | |
| """Let the pipeline generate below its 5 s floor. 56 frames (2.33 s) is fine on this checkpoint.""" | |
| from diffusers.modular_pipelines.minimax_h3.modular_pipeline import MiniMaxH3ModularPipeline | |
| MiniMaxH3ModularPipeline.min_duration = property(lambda self: float(seconds)) | |
| def video_latent_frames(num_frames: int) -> int: | |
| """`17 * n + 5` frames become `5 * n + 2` video latents.""" | |
| return 5 * ((num_frames - LATENTS_PER_CHUNK) // FRAMES_PER_CHUNK) + 2 | |
| def target_rows(height: int, width: int, num_frames: int) -> int: | |
| """The generated rows of the packed sequence: video patched `(1, 2, 2)`, plus two audio rows per latent.""" | |
| video = video_latent_frames(num_frames) * (height // CANVAS_MULTIPLE) * (width // CANVAS_MULTIPLE) | |
| return video + round(num_frames / FPS * AUDIO_LATENTS_PER_SECOND) * AUDIO_CHANNELS | |
| def reference_rows(image_paths: list[str]) -> int: | |
| """The rows the image reference blocks add, from metadata alone — no decode. | |
| A reference image is resized to a 2048-pixel short edge and encoded as a single frame, so a squarer reference is | |
| a cheaper one. | |
| """ | |
| from PIL import Image | |
| rows = 0 | |
| for path in image_paths: | |
| if not path: | |
| continue | |
| try: | |
| width, height = Image.open(path).size | |
| except Exception: | |
| width, height = 1024, 1024 | |
| scale = REFERENCE_IMAGE_SHORT_EDGE / min(width, height) | |
| resolved = [ | |
| max(CANVAS_MULTIPLE, round(edge * scale / CANVAS_MULTIPLE) * CANVAS_MULTIPLE) for edge in (height, width) | |
| ] | |
| rows += (resolved[0] // CANVAS_MULTIPLE) * (resolved[1] // CANVAS_MULTIPLE) | |
| return rows | |
| def denoise_seconds(sequence: int, steps: int) -> float: | |
| import h3_fbc | |
| uncached = int(steps) * (STEP_LINEAR * sequence + STEP_QUADRATIC * sequence**2) * SAFETY | |
| return uncached / (FBC_SPEEDUP if h3_fbc.ENABLED else 1.0) | |
| def get_duration(prompt_embeds, text_token_tags, image_paths, height, width, num_frames, steps, weight, seed, **_): | |
| """Seconds of GPU to reserve for one request, from the packed sequence it is about to denoise.""" | |
| sequence = int(text_token_tags.shape[0]) + reference_rows(image_paths) + target_rows(height, width, num_frames) | |
| denoise = denoise_seconds(sequence, steps) | |
| encode = 5 + reference_rows(image_paths) * 1e-3 | |
| decode = DECODE_BASE + DECODE_PER_DEFAULT_CANVAS * (height * width * num_frames) / DEFAULT_CANVAS_PIXELS | |
| total = PLACEMENT_ALLOWANCE + encode + denoise + decode + 10 | |
| duration = max(MIN_GPU_DURATION, min(MAX_GPU_DURATION, int(total))) | |
| print(f"[{VERSION}] S={sequence} -> reserving {duration}s ({denoise:.0f}s of denoise at {steps} steps)", flush=True) | |
| return duration | |
| def estimate_label(canvas, duration, steps, *image_paths) -> str: | |
| """The same estimate, phrased for the UI, so the cost of a canvas / duration / steps choice is visible up front.""" | |
| height, width = CANVASES.get(canvas, CANVASES[DEFAULT_CANVAS]) | |
| num_frames = snap_frames(duration) | |
| paths = [path for path in image_paths if path] | |
| refs = reference_rows(paths) | |
| sequence = ESTIMATED_TEXT_ROWS + ESTIMATED_VISION_ROWS_PER_IMAGE * len(paths) + refs | |
| sequence += target_rows(height, width, num_frames) | |
| seconds = int(PLACEMENT_ALLOWANCE + 5 + denoise_seconds(sequence, steps) + DECODE_BASE) | |
| return ( | |
| f"{width}x{height} · {num_frames} frames ({num_frames / FPS:.2f} s) · {int(steps)} steps · " | |
| f"{len(paths)} reference{'' if len(paths) == 1 else 's'} → roughly " | |
| f"**{seconds // 60}m {seconds % 60:02d}s** of GPU time" | |
| ) | |
| # ── The LoRA, onto the diffusers port ──────────────────────────────────────── | |
| # | |
| # `Minimax-h3_Third_person_view.safetensors` is an ai-toolkit export: 416 tensors, rank 64, no `.alpha`, keys | |
| # `diffusion_model.blocks.N.{attn.qkv_proj,attn.out_proj,mlp.fc1,mlp.fc2}.lora_{A,B}.weight` plus the same four under | |
| # `token_refiner.blocks.N`, and `ss_base_model_version: minimax_h3_ref2va` in its metadata. diffusers serves the | |
| # *converted* port, so every name — and, for two of them, the row layout of `lora_B` — has to be pushed through the | |
| # same transforms `scripts/convert_minimax_h3_to_diffusers.py` applied to the base weights. This reproduces | |
| # `_convert_non_diffusers_minimax_h3_lora_to_diffusers`: | |
| # | |
| # * `blocks.` -> `transformer_blocks.`, `token_refiner.blocks.` -> `token_refiner.refiner_blocks.`, | |
| # `attn.out_proj` -> `attn.to_out.0`, `mlp.fc2` -> `ff.net.2` (pure renames), | |
| # * `mlp.fc1` -> `ff.net.0.proj` with its two fused halves swapped: the reference computes `fc2(silu(gate) * value)` | |
| # from a fused `[gate; value]` while diffusers' `SwiGLU` computes `value * silu(gate)` from a fused | |
| # `[value; gate]`, so the halves trade places, | |
| # * `attn.qkv_proj` -> `to_q` / `to_k` / `to_v`: split `lora_B`'s 21504 rows into **contiguous** thirds of 7168 | |
| # (`num_attention_heads * attention_head_dim`). | |
| # | |
| # That last split is where MiniMax-H3 LoRAs diverge. DiffSynth-Studio exports run the raw checkpoint's *per-head | |
| # interleaved* fused QKV and have to be de-interleaved before the split; they are identified by peft's `.default.` | |
| # infix over these names. ai-toolkit exports under `diffusion_model.` are already `[q_all; k_all; v_all]` and must | |
| # **not** be reordered — de-interleaving this file anyway would scatter each head's q/k/v across all three | |
| # projections and turn the adapter into structured noise on all 50 blocks. | |
| # | |
| # Row transforms only ever touch `lora_B`, so the three attention projections share one `lora_A`: the rows of `B @ A` | |
| # are the rows of `B`, which makes the split exact rather than an approximation. | |
| def _lora_target_name(source_name: str) -> str: | |
| """Map an original-checkpoint module path to the diffusers one (no `.lora_A/B.*` suffix).""" | |
| if source_name.startswith("token_refiner.blocks."): | |
| return source_name.replace("token_refiner.blocks.", "token_refiner.refiner_blocks.", 1) | |
| if source_name.startswith("blocks."): | |
| return source_name.replace("blocks.", "transformer_blocks.", 1) | |
| return source_name | |
| def _lora_modules(name: str, a_weight, b_weight, inner_dim: int): | |
| """Yield `(diffusers_module_path, lora_A, lora_B)` for one original module path.""" | |
| target = _lora_target_name(name) | |
| if target.endswith(".attn.qkv_proj"): | |
| prefix = target.removesuffix("qkv_proj") | |
| for kind, part in zip(("q", "k", "v"), b_weight.split(inner_dim, dim=0)): | |
| yield f"{prefix}to_{kind}", a_weight, part.contiguous() | |
| elif target.endswith(".mlp.fc1"): | |
| gate, value = b_weight.chunk(2, dim=0) | |
| yield target.replace(".mlp.fc1", ".ff.net.0.proj"), a_weight, torch.cat([value, gate]).contiguous() | |
| elif target.endswith(".mlp.fc2"): | |
| yield target.replace(".mlp.fc2", ".ff.net.2"), a_weight, b_weight | |
| elif target.endswith(".attn.out_proj"): | |
| yield target.replace(".attn.out_proj", ".attn.to_out.0"), a_weight, b_weight | |
| else: | |
| raise ValueError( | |
| f"unexpected LoRA target `{name}`: this adapter is documented as attention + feed-forward only, so a " | |
| "new module type means the checkpoint changed" | |
| ) | |
| def build_lora_state_dict(transformer) -> tuple[dict, int, int]: | |
| """Download the LoRA and remap it into a PEFT-format state dict for the diffusers transformer. | |
| Every target is validated against the transformer's own parameter shapes, and an unresolved one is fatal: a | |
| silently dropped target means the name mapping is wrong and the Space would serve a half-applied adapter that | |
| still *looks* like it worked. | |
| """ | |
| from huggingface_hub import hf_hub_download | |
| from safetensors.torch import load_file | |
| raw = load_file(hf_hub_download(LORA_REPO, LORA_FILE)) | |
| prefix, suffix_a, suffix_b = "diffusion_model.", ".lora_A.weight", ".lora_B.weight" | |
| unexpected = [key for key in raw if not (key.startswith(prefix) and key.endswith((suffix_a, suffix_b)))] | |
| if unexpected: | |
| raise ValueError(f"{LORA_FILE} holds {len(unexpected)} unexpected tensors, e.g. {unexpected[:5]}") | |
| bases = sorted({key[len(prefix) : -len(suffix_a)] for key in raw if key.endswith(suffix_a)}) | |
| if not bases: | |
| raise ValueError(f"No `{prefix}*{suffix_a}` / `{suffix_b}` pairs found in {LORA_FILE}") | |
| ranks = set() | |
| for name in bases: | |
| if f"{prefix}{name}{suffix_b}" not in raw: | |
| raise ValueError(f"LoRA is missing the lora_B twin of {prefix}{name}{suffix_a}") | |
| ranks.add(raw[f"{prefix}{name}{suffix_a}"].shape[0]) | |
| if len(ranks) != 1: | |
| raise ValueError(f"LoRA mixes ranks {sorted(ranks)}; this loader assumes a single rank") | |
| rank = ranks.pop() | |
| inner_dim = transformer.config.num_attention_heads * transformer.config.attention_head_dim | |
| base_shapes = {key: tuple(value.shape) for key, value in transformer.state_dict().items()} | |
| state_dict: dict[str, torch.Tensor] = {} | |
| missed: list[str] = [] | |
| for name in bases: | |
| a_weight = raw[f"{prefix}{name}{suffix_a}"] | |
| b_weight = raw[f"{prefix}{name}{suffix_b}"] | |
| for module, a_part, b_part in _lora_modules(name, a_weight, b_weight, inner_dim): | |
| base = base_shapes.get(f"{module}.weight") | |
| if base is None: | |
| missed.append(module) | |
| continue | |
| # `W` is [out, in]; the adapter must be `lora_B` [out, r] @ `lora_A` [r, in]. | |
| if (b_part.shape[0], a_part.shape[1]) != base: | |
| raise ValueError( | |
| f"LoRA delta for `{module}` would be {(b_part.shape[0], a_part.shape[1])}, " | |
| f"base weight is {base}" | |
| ) | |
| state_dict[f"{module}.lora_A.weight"] = a_part | |
| state_dict[f"{module}.lora_B.weight"] = b_part | |
| if missed: | |
| raise ValueError( | |
| f"{len(missed)} LoRA targets matched no transformer weight, e.g. {missed[:5]}. " | |
| "The LoRA and the diffusers transformer disagree on module naming." | |
| ) | |
| return state_dict, rank, len(bases) | |
| def load_and_apply_lora(transformer) -> str: | |
| """Attach the LoRA as a PEFT adapter so its strength stays a per-request knob.""" | |
| state_dict, rank, targets = build_lora_state_dict(transformer) | |
| # `prefix=None` because the keys are already the transformer's own module paths, and no `network_alphas` because | |
| # the file carries no `.alpha` tensors — PEFT then sets alpha == rank, i.e. the adapter's intrinsic scale is 1.0 | |
| # and the `set_adapters` weight *is* the strength the card talks about. | |
| transformer.load_lora_adapter(state_dict, prefix=None, adapter_name=ADAPTER) | |
| transformer.set_adapters([ADAPTER], [DEFAULT_WEIGHT]) | |
| return ( | |
| f"LoRA attached · {targets} checkpoint targets -> {len(state_dict) // 2} diffusers modules, rank {rank}, " | |
| f"alpha == rank · default weight {DEFAULT_WEIGHT:g} · {LORA_FILE}" | |
| ) | |
| # ── Model loading ──────────────────────────────────────────────────────────── | |
| PIPE = None | |
| LOAD_ERROR: str | None = None | |
| LORA_STATUS: str | None = None | |
| def load_models() -> str | None: | |
| """Load the denoising half at startup, but *not* onto the card — see `PLACEMENT`. | |
| `MiniMaxH3Ref2VAGeneratorBlocks` declares `transformer_ref`, `vae`, `audio_vae`, the two schedulers and | |
| `video_processor`, so `load_components` fetches exactly those subfolders — `text_encoder/` and the `transformer/` | |
| partition are never touched. Both autoencoders carry `_keep_in_fp32_modules` over every module and stay float32: | |
| a bfloat16 audio VAE decodes the soundtrack roughly 20 dB too quiet. | |
| """ | |
| global PIPE, LOAD_ERROR, LORA_STATUS | |
| if PIPE is not None or LOAD_ERROR is not None: | |
| return LOAD_ERROR | |
| started = time.time() | |
| try: | |
| from diffusers import ComponentsManager | |
| from h3_split_blocks import MiniMaxH3Ref2VAGeneratorBlocks | |
| lower_duration_floor() | |
| blocks = MiniMaxH3Ref2VAGeneratorBlocks() | |
| print(f"[{VERSION}] loading {[c.name for c in blocks.expected_components]} from {MODEL_REPO} ...", flush=True) | |
| pipe = blocks.init_pipeline(MODEL_REPO, components_manager=ComponentsManager(), collection="h3") | |
| # The pruned DiT is served as remote code (`transformer_ref/modeling_minimax_h3_pruned.py`, reached through | |
| # the `AutoModel` type hint in `modular_model_index.json`). `load_components` forwards `trust_remote_code` | |
| # only to components that live in the pipeline's own repo, so the two VAEs and the schedulers — which | |
| # `modular_model_index.json` still points at `MiniMaxAI/MiniMax-H3` — never see it. | |
| pipe.load_components(dtype=torch.bfloat16, trust_remote_code=True) | |
| # Both VAEs explicitly, and before the transformer: `set_attention_backend` also sets the registry's *global* | |
| # backend, and the float32 audio VAE has no cuDNN kernel. | |
| pipe.vae.set_attention_backend("native") | |
| pipe.audio_vae.set_attention_backend("native") | |
| pipe.transformer_ref.set_attention_backend(ATTENTION) | |
| # Before any GPU placement, so the adapter's parameters travel with the transformer. No AoTI package on this | |
| # Space on purpose: a compiled block graph is captured around the base linears and would silently bypass the | |
| # LoRA layers PEFT injects. | |
| LORA_STATUS = load_and_apply_lora(pipe.transformer_ref) | |
| print(f"[{VERSION}] {LORA_STATUS}", flush=True) | |
| import h3_fbc | |
| print(f"[h3-fbc] {h3_fbc.status()}", flush=True) | |
| PIPE = pipe | |
| print(f"[{VERSION}] ready in {time.time() - started:.0f}s", flush=True) | |
| except Exception as error: | |
| traceback.print_exc() | |
| LOAD_ERROR = ( | |
| f"**Loading `{MODEL_REPO}` failed** after {time.time() - started:.0f}s: " | |
| f"`{type(error).__name__}: {error}`" | |
| ) | |
| return LOAD_ERROR | |
| # ── Prompt composition ─────────────────────────────────────────────────────── | |
| def collect_slots(slots) -> list[tuple[str, str, str]]: | |
| """The `(path, role, description)` references of a request, in the order the model reads them. | |
| That order numbers `<Picture N>` and advances the shared audio/video rotary clock, so the same references in a | |
| different order are a different request. | |
| """ | |
| return [(path, role, (detail or "").strip()) for path, role, detail in slots if path] | |
| def compose_prompt(references, action, camera, hud, soundscape, music, seconds) -> str: | |
| """Assemble MiniMax-H3's structured document in the layout the LoRA's card publishes.""" | |
| action = (action or "").strip().rstrip(".") | |
| spec = CAMERAS.get(camera, CAMERAS[DEFAULT_CAMERA]) | |
| entries, subjects, environments = [], 0, 0 | |
| for index, (_, role, detail) in enumerate(references, start=1): | |
| kind, sheet = ROLES.get(role, ROLES["Character"]) | |
| if kind == "subject": | |
| subjects += 1 | |
| label = f"<Subject {subjects}>" | |
| else: | |
| environments += 1 | |
| label = f"<Environment {environments}>" | |
| entries.append((label, index, detail, sheet)) | |
| subject = next((label for label, _, _, sheet in entries if "environment" not in sheet), "the player character") | |
| environment = next((label for label, _, _, sheet in entries if "environment" in sheet), None) | |
| place = f" in {environment}" if environment else "" | |
| definitions = [ | |
| f"{label} is {detail} from <Picture {index}>." | |
| if detail | |
| else f"{label} is the subject shown in <Picture {index}>." | |
| for label, index, detail, _ in entries | |
| ] | |
| definitions += [f"<Picture {index}> is the {sheet} for {label}." for label, index, _, sheet in entries] | |
| retention = [ | |
| f"{label} (appears in [Shot 1]): fully_preserved - the design, colours and silhouette from " | |
| f"<Picture {index}> are fully preserved." | |
| for label, index, _, _ in entries | |
| ] | |
| sections = [ | |
| "subject_definitions:\n" + "\n".join(definitions), | |
| ( | |
| "summary:\n" | |
| f"[reference generation] A {float(seconds):.1f}-second photorealistic {spec['genre']} action RPG " | |
| f"gameplay sequence{place}, where {subject} {action}, {spec['summary']}." | |
| ), | |
| "retention_analysis:\n" + "\n".join(retention), | |
| ( | |
| "detailed_description:\n" | |
| f"The target video features a photorealistic 3D {spec['genre']} action RPG gameplay aesthetic rendered " | |
| f"in Unreal Engine with real-time game mechanics, {spec['aesthetic']}, deep depth of field" | |
| f"{HUD_CLAUSE if hud else NO_HUD_CLAUSE}.\n" | |
| f"[Shot 1] The video opens with {spec['opening']} {subject}{place}. {action.capitalize()}. " | |
| f"{spec['continues']}" | |
| ), | |
| ] | |
| if (soundscape or "").strip(): | |
| sections.append("overall_soundscape:\n" + soundscape.strip()) | |
| if (music or "").strip(): | |
| sections.append("non_diegetic_music:\n" + music.strip()) | |
| return "\n\n".join(sections) | |
| # ── Guard, conditioner ─────────────────────────────────────────────────────── | |
| def check_prompt(prompt: str) -> None: | |
| """The NCII guard. Every request here carries an uploaded image, so every request is checked; it runs before the | |
| conditioner call and the denoise booking, so a refused prompt costs no GPU time on either half.""" | |
| import ncii_guard | |
| try: | |
| flag = ncii_guard.classify(prompt) | |
| except Exception as error: | |
| traceback.print_exc() | |
| raise gr.Error(f"The content filter is unavailable (`{type(error).__name__}`), so nothing was run.") | |
| if flag["label"] == "ncii": | |
| print(f"[guard] prompt refused (ncii {flag['score']:.2f})", flush=True) | |
| raise gr.Error("This prompt was flagged by a content filter and wasn't run.") | |
| def conditioner(): | |
| """The other half, over the gradio API.""" | |
| from gradio_client import Client | |
| return Client(CONDITIONER_SPACE) | |
| def conditioner_client(ip_token): | |
| """A client carrying *this* caller's ZeroGPU token, so the conditioner's own booking is billed to whoever asked | |
| for the video rather than to this Space.""" | |
| if not ip_token: | |
| return conditioner() | |
| from gradio_client import Client | |
| return Client(CONDITIONER_SPACE, headers={"x-ip-token": ip_token}) | |
| def caller_ip_token() -> str | None: | |
| from gradio.context import LocalContext | |
| request = LocalContext.request.get() | |
| return request.headers.get("x-ip-token") if request is not None else None | |
| def encode_remote(prompt, image_paths, canvas, num_frames, ip_token=None): | |
| """`/encode_ref2va` on the conditioner Space: a safetensors file holding `prompt_embeds` + `text_token_tags`, with | |
| the resolved `height` / `width` / `num_frames` in its metadata, plus the plan. | |
| `canvas` is the label. `media` and `kinds` are parallel and ordered; the references go over because `ref2va`'s | |
| presentation puts a vision block in front of the prompt for every image. | |
| """ | |
| from gradio_client import handle_file | |
| from safetensors import safe_open | |
| def call(): | |
| return conditioner_client(ip_token).predict( | |
| prompt=prompt, | |
| media=[handle_file(image) for image in image_paths], | |
| kinds=",".join("image" for _ in image_paths), | |
| canvas=canvas, | |
| num_frames=num_frames, | |
| # Never rewritten: H3's own LLM expansion would paraphrase away the card's structured document — the | |
| # `<Picture N>` bindings and the trigger vocabulary this adapter was trained against. | |
| rewrite_prompt=False, | |
| api_name="/encode_ref2va", | |
| ) | |
| try: | |
| path, plan = call() | |
| except Exception as first: | |
| # A cached client whose upstream Space restarted fails once; a fresh handshake usually succeeds. | |
| print(f"[conditioner] retrying with a fresh client after: {first}", flush=True) | |
| conditioner.cache_clear() | |
| path, plan = call() | |
| with safe_open(path, framework="pt") as handle: | |
| return handle.get_tensor("prompt_embeds"), handle.get_tensor("text_token_tags"), handle.metadata(), plan | |
| # ── Inference ──────────────────────────────────────────────────────────────── | |
| def _generate(prompt_embeds, text_token_tags, image_paths, height, width, num_frames, steps, weight, seed): | |
| """The only thing on GPU time: the reference encoder, the packed-sequence denoise loop and the decoders. | |
| References cross as paths and are decoded here; only the generated outputs come back. A `@spaces.GPU` argument | |
| crosses a process boundary by pickling, and expanded reference frames are large. | |
| """ | |
| from diffusers.modular_pipelines.minimax_h3 import MiniMaxH3ImageReference | |
| import h3_fbc | |
| if PLACEMENT == "lazy": | |
| PIPE.to("cuda") | |
| # The whole reason the adapter is not folded into the weights: strength is a per-request scale on the LoRA layers. | |
| PIPE.transformer_ref.set_adapters([ADAPTER], [float(weight)]) | |
| with h3_fbc.enabled(PIPE.transformer_ref, steps=int(steps)): | |
| state = PIPE( | |
| prompt_embeds=prompt_embeds.to("cuda"), | |
| text_token_tags=text_token_tags, | |
| references=[MiniMaxH3ImageReference.from_file(path) for path in image_paths], | |
| height=height, | |
| width=width, | |
| num_frames=num_frames, | |
| num_inference_steps=int(steps), | |
| generator=torch.Generator("cpu").manual_seed(int(seed)), | |
| ) | |
| return state.get("videos")[0], state.get("audio")[0].cpu(), state.get("sampling_rate") | |
| def generate( | |
| # The first nine are the columns `gr.Examples` varies, and they lead the signature for that reason: an example row | |
| # is applied to `inputs` positionally. Every parameter has a default, which is what lets a nine-column row call | |
| # this at all. | |
| action="", | |
| picture_1=None, | |
| role_1="Character", | |
| detail_1="", | |
| picture_2=None, | |
| role_2="Environment", | |
| detail_2="", | |
| camera=DEFAULT_CAMERA, | |
| hud=True, | |
| picture_3=None, | |
| role_3="Creature / Boss", | |
| detail_3="", | |
| canvas=DEFAULT_CANVAS, | |
| duration=5, | |
| steps=DEFAULT_STEPS, | |
| weight=DEFAULT_WEIGHT, | |
| seed=DEFAULT_SEED, | |
| soundscape=DEFAULT_SOUNDSCAPE, | |
| music=DEFAULT_MUSIC, | |
| override="", | |
| progress=gr.Progress(track_tqdm=True), | |
| ): | |
| """One request: compose the structured prompt, condition it remotely, denoise here.""" | |
| if LOAD_ERROR: | |
| raise gr.Error(LOAD_ERROR) | |
| if PIPE is None: | |
| raise gr.Error("The denoiser is still loading.") | |
| from diffusers.utils import encode_video | |
| references = collect_slots( | |
| [(picture_1, role_1, detail_1), (picture_2, role_2, detail_2), (picture_3, role_3, detail_3)] | |
| ) | |
| if not references: | |
| raise gr.Error("Add at least one reference sheet — a character, an environment or a creature.") | |
| if not (override or "").strip() and not (action or "").strip(): | |
| raise gr.Error("Write one line of action, or paste a full structured prompt in the advanced options.") | |
| num_frames = snap_frames(duration) | |
| prompt = (override or "").strip() or compose_prompt( | |
| references, action, camera, hud, soundscape, music, num_frames / FPS | |
| ) | |
| check_prompt(prompt) | |
| image_paths = [path for path, _, _ in references] | |
| progress(0.0, desc="Reading the prompt and the reference sheets ...") | |
| conditioned = time.time() | |
| try: | |
| prompt_embeds, text_token_tags, metadata, plan = encode_remote( | |
| prompt, image_paths, canvas, num_frames, ip_token=caller_ip_token() | |
| ) | |
| except gr.Error: | |
| raise | |
| except Exception as error: | |
| # gradio only puts the exception *type* on the wire, so the useful half of a conditioner-side failure is in | |
| # that Space's logs. | |
| traceback.print_exc() | |
| raise gr.Error( | |
| f"The conditioner ({CONDITIONER_SPACE}) failed with `{type(error).__name__}: {error}`. " | |
| "Its logs carry the full traceback." | |
| ) from error | |
| condition_seconds = time.time() - conditioned | |
| height, width, num_frames = (int(metadata[key]) for key in ("height", "width", "num_frames")) | |
| progress(0.1, desc=f"Rendering {num_frames / FPS:.1f} s at {width}x{height} ...") | |
| started = time.time() | |
| frames, audio, sampling_rate = _generate( | |
| prompt_embeds, text_token_tags, image_paths, height, width, num_frames, steps, weight, seed | |
| ) | |
| generate_seconds = time.time() - started | |
| directory = os.path.join(tempfile.gettempdir(), "h3-outputs") | |
| os.makedirs(directory, exist_ok=True) | |
| path = os.path.join(directory, f"h3-tpv-{int(time.time() * 1000)}.mp4") | |
| encode_video(frames, fps=FPS, output_path=path, audio=audio, audio_sample_rate=sampling_rate) | |
| print( | |
| f"[{VERSION}] {len(image_paths)} references · {width}x{height}, {num_frames} frames " | |
| f"({num_frames / FPS:.3f} s), {int(steps)} steps, weight {float(weight):g} · conditioner " | |
| f"{condition_seconds:.0f}s ({plan['num_text_tokens']} tokens) · denoise + decode {generate_seconds:.0f}s " | |
| f"({generate_seconds / int(steps):.1f} s/step) · seed {int(seed)}", | |
| flush=True, | |
| ) | |
| return path, prompt | |
| # ── UI ─────────────────────────────────────────────────────────────────────── | |
| try: | |
| import ncii_guard | |
| ncii_guard.start() | |
| except Exception as guard_error: # the guard revives itself per request; a cold cache should not fail the boot | |
| print(f"[guard] start failed ({type(guard_error).__name__}: {guard_error}); it will be retried per request") | |
| load_models() | |
| INTRO = """# MiniMax-H3 · Third-Person View LoRA | |
| <div align="center"> | |
| <a href="https://huggingface.co/WarmBloodAban/Minimax_H3_LoRAs" target="_blank" rel="noopener"><strong>[ LoRA ]</strong></a> | |
| <a href="https://huggingface.co/MiniMaxAI/MiniMax-H3" target="_blank" rel="noopener"><strong>[ base model ]</strong></a> | |
| <a href="https://huggingface.co/spaces/multimodalart/minimax-h3-reference" target="_blank" rel="noopener"><strong>[ base demo ]</strong></a> | |
| </div> | |
| **Game cutscenes out of design reference sheets.** Label each sheet you upload, write one line of action, pick a | |
| camera, and this composes MiniMax-H3's structured prompt around the LoRA's own vocabulary — third-person | |
| over-the-shoulder / spring-arm tracking, first-person POV, whip-pan transitions, target-lock reticles and Boss health | |
| bars — then renders the clip **with its synchronized soundtrack** in a single denoising pass. | |
| """ | |
| CSS = """ | |
| .main.fillable { max-width: 1250px !important; } | |
| .dark .gradio-container { color: var(--body-text-color); } | |
| """ | |
| with gr.Blocks(title="MiniMax-H3 · Third-Person View LoRA") as demo: | |
| gr.Markdown(INTRO) | |
| if LOAD_ERROR: | |
| gr.Markdown(LOAD_ERROR) | |
| with gr.Row(): | |
| with gr.Column(): | |
| gr.Markdown("### 1 · Reference sheets — they become `<Picture 1..3>`, in this order") | |
| pictures, roles, details, slot_columns = [], [], [], [] | |
| with gr.Row(): | |
| for index in range(MAX_IMAGE_SLOTS): | |
| with gr.Column(min_width=180, visible=index < OPEN_IMAGE_SLOTS) as column: | |
| pictures.append( | |
| gr.Image(label=f"Picture {index + 1}", type="filepath", height=190, show_label=True) | |
| ) | |
| roles.append( | |
| gr.Dropdown( | |
| label="Role", | |
| choices=list(ROLES), | |
| value="Character" if index == 0 else ("Environment" if index == 1 else "Creature / Boss"), | |
| container=True, | |
| ) | |
| ) | |
| details.append( | |
| gr.Textbox( | |
| label="What's in it", | |
| lines=2, | |
| placeholder="the mint-green cat mecha with a tattered beige cape", | |
| ) | |
| ) | |
| slot_columns.append(column) | |
| add_slot = gr.Button("+ Add a third reference sheet", size="sm", variant="secondary") | |
| action = gr.Textbox( | |
| label="2 · What happens in the shot", | |
| lines=2, | |
| placeholder="dodges a leaping wyvern attack, counterattacks with an energy-blade combo, then parries " | |
| "a wing-claw swipe", | |
| ) | |
| with gr.Row(): | |
| camera = gr.Dropdown(label="3 · Camera", choices=list(CAMERAS), value=DEFAULT_CAMERA, scale=3) | |
| hud = gr.Checkbox(label="Combat HUD overlay", value=True, scale=1) | |
| run = gr.Button("Render the cutscene", variant="primary") | |
| estimate = gr.Markdown(estimate_label(DEFAULT_CANVAS, 5, DEFAULT_STEPS, None)) | |
| with gr.Accordion("Advanced options", open=False): | |
| weight = gr.Slider( | |
| label="LoRA weight (the card recommends 0.6–0.85)", | |
| minimum=WEIGHT_MIN, | |
| maximum=WEIGHT_MAX, | |
| step=0.05, | |
| value=DEFAULT_WEIGHT, | |
| ) | |
| canvas = gr.Dropdown(label="Canvas", choices=list(CANVASES), value=DEFAULT_CANVAS) | |
| duration = gr.Slider( | |
| label="Duration (s)", minimum=MIN_DURATION, maximum=MAX_UI_DURATION, step=1, value=5 | |
| ) | |
| steps = gr.Slider(label="Steps", minimum=10, maximum=40, step=1, value=DEFAULT_STEPS) | |
| seed = gr.Number(label="Seed", value=DEFAULT_SEED, precision=0) | |
| soundscape = gr.Textbox(label="Diegetic soundscape", lines=2, value=DEFAULT_SOUNDSCAPE) | |
| music = gr.Textbox(label="Non-diegetic music", lines=2, value=DEFAULT_MUSIC) | |
| override = gr.Textbox( | |
| label="Write the structured prompt myself (overrides everything above)", | |
| lines=6, | |
| placeholder="subject_definitions:\n<Subject 1> is ... from <Picture 1>.\n\nsummary:\n" | |
| "[reference generation] A 10-second photorealistic third-person ...", | |
| ) | |
| with gr.Column(): | |
| result = gr.Video(label="Cutscene + soundtrack") | |
| with gr.Accordion("Structured prompt sent to MiniMax-H3", open=False): | |
| prompt_view = gr.Textbox(show_label=False, lines=18, interactive=False) | |
| open_slots = gr.State(OPEN_IMAGE_SLOTS) | |
| def reveal_slot(open_count): | |
| open_count = min(open_count + 1, MAX_IMAGE_SLOTS) | |
| return [ | |
| open_count, | |
| *[gr.update(visible=index < open_count) for index in range(MAX_IMAGE_SLOTS)], | |
| gr.update(visible=open_count < MAX_IMAGE_SLOTS), | |
| ] | |
| add_slot.click(reveal_slot, open_slots, [open_slots, *slot_columns, add_slot], api_name=False) | |
| for control in (canvas, duration, steps, *pictures): | |
| control.change( | |
| estimate_label, | |
| [canvas, duration, steps, *pictures], | |
| estimate, | |
| show_progress="hidden", | |
| api_name=False, | |
| ) | |
| request = [ | |
| action, | |
| pictures[0], | |
| roles[0], | |
| details[0], | |
| pictures[1], | |
| roles[1], | |
| details[1], | |
| camera, | |
| hud, | |
| pictures[2], | |
| roles[2], | |
| details[2], | |
| canvas, | |
| duration, | |
| steps, | |
| weight, | |
| seed, | |
| soundscape, | |
| music, | |
| override, | |
| ] | |
| exampled = request[:9] | |
| gr.Examples( | |
| examples=[ | |
| [ | |
| "dodges a leaping wyvern attack, counterattacks with a glowing energy-blade combo, then parries a " | |
| "wing-claw swipe in a shower of sparks", | |
| "examples/katana_character.jpg", | |
| "Character", | |
| "the young woman with long black hair in two buns, blue eyes, a black mouth mask, an orange utility " | |
| "jacket and two white-corded katana hilts rising behind her shoulders", | |
| "examples/ruined_village.jpg", | |
| "Environment", | |
| "the foggy ruined village street with wooden houses, a glowing paper lantern, blood-red spider " | |
| "lilies and wet stone paving", | |
| "Third-person · over-the-shoulder", | |
| True, | |
| ], | |
| [ | |
| "plants her feet into a combat stance and charges an ultimate finisher as the pale wyvern staggers " | |
| "and opens its ring-like tooth-lined maw", | |
| "examples/katana_character.jpg", | |
| "Character", | |
| "the young woman with long black hair in two buns, an orange utility jacket and two katana hilts " | |
| "rising behind her shoulders", | |
| "examples/wyvern_boss.jpg", | |
| "Creature / Boss", | |
| "the pale eyeless wyvern with a circular ring-like maw lined with rows of sharp teeth, a long " | |
| "muscular neck, clawed wings and heavy bipedal limbs", | |
| "Whip-pan: third-person → first-person", | |
| True, | |
| ], | |
| [ | |
| "sprints across a rain-slick rooftop, vaults a neon billboard gap and lands into a slide while " | |
| "tracer fire streaks past", | |
| "examples/doorway_character.jpg", | |
| "Character", | |
| "the young girl with brown hair, a white shirt and skirt and a guitar case slung across her back", | |
| "examples/night_city.jpg", | |
| "Environment", | |
| "the night city of lit high-rise towers and a faceted glass skyscraper reflected in the dark river " | |
| "below", | |
| "First-person POV (FPV)", | |
| True, | |
| ], | |
| ], | |
| inputs=exampled, | |
| outputs=[result, prompt_view], | |
| fn=generate, | |
| cache_examples=True, | |
| cache_mode="lazy", | |
| ) | |
| run.click(generate, request, [result, prompt_view], api_name="generate") | |
| if __name__ == "__main__": | |
| # Gradio 6 moved `theme` / `css` off the Blocks constructor and onto `launch`. | |
| demo.launch(theme=gr.themes.Citrus(), css=CSS, show_error=True, max_threads=1000) | |