File size: 1,803 Bytes
75487f8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
#!/usr/bin/env bash
# Reproduce the demo on the model card, and in doing so check your install.
#
# This repository bundles no driving footage. It builds the driving clip from a demo
# video that ships with the base model, using the exact crop documented below, and
# pairs it with the repainted first frame in media/reference.png.
#
# The result should match media/output.mp4. On the same GPU model we get it back
# bit-identical; on different hardware bf16 kernel scheduling shifts things, and a mean
# absolute difference around 1.5/255 is normal. What matters is that it is the same
# elderly woman holding the same black lamb. If it comes back as the young man from the
# source video instead, the LoRA did not load. If it comes back as noise, the weights
# are wrong.
#
#   ./examples/demo.sh /path/to/MiniMax-H3
#
# About a minute on a B200, most of it loading weights.

set -euo pipefail

MODEL_DIR="${1:?usage: demo.sh /path/to/MiniMax-H3}"
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
OUT="${OUT:-$HERE/demo_out}"
mkdir -p "$OUT"

SRC="$MODEL_DIR/assets/ref2va.mp4"
[ -f "$SRC" ] || { echo "missing $SRC -- see the download command on the model card"; exit 1; }

# The source is 1344x768 and exactly 124 frames, which is the sampler's window. Crop a
# 512x768 portrait window around the figure; it stays in frame for the whole push-in, so
# no scaling and no padding are needed. media/reference.png is this crop's first frame,
# repainted.
ffmpeg -y -loglevel error -i "$SRC" \
  -vf "crop=512:768:389:0" -frames:v 124 -an "$OUT/driving.mp4"

python "$HERE/../inference/sample.py" \
  --model-dir "$MODEL_DIR" \
  --cond "$OUT/driving.mp4" \
  --ref  "$HERE/media/reference.png" \
  --out  "$OUT/output.mp4"

echo
echo "wrote $OUT/output.mp4 -- compare against $HERE/media/output.mp4"