# Copyright 2026 Viggle AI. Licensed under the Apache License, Version 2.0 (see LICENSE-CODE). # SPDX-License-Identifier: Apache-2.0 """Rewrite a Meridian adapter from diffusers PEFT layout into ComfyUI's generic LoRA layout. python -m recam.to_comfyui teacher_lora/pytorch_lora_weights.safetensors comfyui/meridian_teacher.safetensors Same delta, different packing. ComfyUI loads MiniMax-H3 from its own repack (`Comfy-Org/MiniMax-H3`), which keeps the reference implementation's module names and two of its fusions, so three things have to change and nothing else does: names `transformer_blocks.N` -> `diffusion_model.blocks.N`; `proj_in` -> `video_patch_proj`, `proj_out` -> `final_layer.video_out`, `attn.to_out.0` -> `attn.out_proj`, `ff.net.0.proj` -> `mlp.fc1`, `ff.net.2` -> `mlp.fc2`. qkv diffusers keeps `to_q/to_k/to_v` separate, ComfyUI fuses them into one `attn.qkv_proj` whose rows are `[q; k; v]` in that order. Concatenating A down the rank axis and putting the three B's on the block diagonal keeps each output slab reading only its own A, so the fused delta equals the three separate ones stacked. Rank triples, and alpha with it. SwiGLU both fuse the gate and value projections into one `fc1`, in opposite order: the reference computes `fc2(silu(gate) * value)` from `[gate; value]`, diffusers' `SwiGLU` computes `value * silu(gate)` from `[value; gate]`. B's two row halves swap; A is untouched. Neither the row order nor the half swap is a guess: `blocks.0` of ComfyUI's `minimax_h3_fl2va_bf16.safetensors` is bit-identical to this repo's base transformer once both are applied, and the official ComfyUI H3 LoRA documents the same recipe in its own metadata. Note the base: Meridian trains on MiniMax-H3's **fl2va** partition, so the ComfyUI file to load is `minimax_h3_fl2va_bf16.safetensors`, not the ref2va one. `alpha` is written equal to rank so ComfyUI's `alpha / rank` scale is 1.0, which is what diffusers applies here (`lora_alpha` 128, `r` 128). Load the teacher and the turbo adapter together at strength 1.0, in that order; the turbo adapter was distilled against the teacher and does nothing sensible without it. """ import sys import torch from safetensors.torch import load_file, save_file BLOCKS, DIM, QKV, FFN = 50, 5376, 7168, 14336 def convert(src, dtype=torch.float16): w = load_file(src) r = w["proj_in.lora_A.weight"].shape[0] out = {} def put(name, A, B, rank): out[f"diffusion_model.{name}.lora_A.weight"] = A.to(dtype).contiguous() out[f"diffusion_model.{name}.lora_B.weight"] = B.to(dtype).contiguous() out[f"diffusion_model.{name}.alpha"] = torch.tensor(float(rank)) put("video_patch_proj", w["proj_in.lora_A.weight"], w["proj_in.lora_B.weight"], r) put("final_layer.video_out", w["proj_out.lora_A.weight"], w["proj_out.lora_B.weight"], r) for i in range(BLOCKS): p, q = f"transformer_blocks.{i}", f"blocks.{i}" A = torch.cat([w[f"{p}.attn.to_{x}.lora_A.weight"] for x in "qkv"]) B = A.new_zeros(3 * QKV, 3 * r) for j, x in enumerate("qkv"): B[j * QKV:(j + 1) * QKV, j * r:(j + 1) * r] = w[f"{p}.attn.to_{x}.lora_B.weight"] put(f"{q}.attn.qkv_proj", A, B, 3 * r) put(f"{q}.attn.out_proj", w[f"{p}.attn.to_out.0.lora_A.weight"], w[f"{p}.attn.to_out.0.lora_B.weight"], r) value, gate = w[f"{p}.ff.net.0.proj.lora_B.weight"].chunk(2) put(f"{q}.mlp.fc1", w[f"{p}.ff.net.0.proj.lora_A.weight"], torch.cat([gate, value]), r) put(f"{q}.mlp.fc2", w[f"{p}.ff.net.2.lora_A.weight"], w[f"{p}.ff.net.2.lora_B.weight"], r) return out if __name__ == "__main__": src, dst = sys.argv[1], sys.argv[2] out = convert(src) save_file(out, dst, metadata={ "format": "pt", "source_format": "Diffusers PEFT LoRA", "target_format": "ComfyUI generic LoRA", "source_file": src, "qkv_fusion": "concat A; block diagonal B; alpha multiplied by 3", "swi_glu_mapping": "Diffusers [value;gate] -> ComfyUI [gate;value]", }) print(f"{len(out)} tensors -> {dst}")