Spaces:
Running on Zero
Running on Zero
Port to transformers 5.x API (hub 1.x / gradio 6 compatible)
Browse files- README.md +3 -1
- app.py +31 -16
- requirements.txt +4 -5
README.md
CHANGED
|
@@ -47,7 +47,9 @@ repo, and message construction, the duration grid, reference labelling /
|
|
| 47 |
validation and the output schema check are ported 1:1 from the adapter repo's
|
| 48 |
`infer.py`. Encoding uses the same `ProcessorMixin.apply_chat_template` path
|
| 49 |
with `max_pixels` 301056 (images) / 100352 (video), `load_audio_from_video=False`
|
| 50 |
-
and `use_audio_in_video=False`.
|
|
|
|
|
|
|
| 51 |
|
| 52 |
Runs on ZeroGPU: the Thinker is loaded in bf16 with SDPA attention and the LoRA
|
| 53 |
is attached with PEFT at module scope.
|
|
|
|
| 47 |
validation and the output schema check are ported 1:1 from the adapter repo's
|
| 48 |
`infer.py`. Encoding uses the same `ProcessorMixin.apply_chat_template` path
|
| 49 |
with `max_pixels` 301056 (images) / 100352 (video), `load_audio_from_video=False`
|
| 50 |
+
and `use_audio_in_video=False`. Those processing kwargs are passed through
|
| 51 |
+
transformers 5.x's `processor_kwargs` dict (transformers 4.x, which the
|
| 52 |
+
reference pins, is incompatible with `huggingface_hub` 1.x / Gradio 6).
|
| 53 |
|
| 54 |
Runs on ZeroGPU: the Thinker is loaded in bf16 with SDPA attention and the LoRA
|
| 55 |
is attached with PEFT at module scope.
|
app.py
CHANGED
|
@@ -33,6 +33,7 @@ import gradio as gr
|
|
| 33 |
from huggingface_hub import hf_hub_download
|
| 34 |
from safetensors import safe_open
|
| 35 |
from transformers import (
|
|
|
|
| 36 |
Qwen2_5OmniProcessor,
|
| 37 |
Qwen2_5OmniThinkerForConditionalGeneration,
|
| 38 |
TextIteratorStreamer,
|
|
@@ -46,6 +47,9 @@ ADAPTER_REPO = "lightx2v/MiniMax-H3-Prompt-Rewriter-LoRA-Omni"
|
|
| 46 |
# Defaults taken from the adapter repo's infer.py CLI.
|
| 47 |
IMAGE_MAX_PIXELS = 301056
|
| 48 |
VIDEO_MAX_PIXELS = 100352
|
|
|
|
|
|
|
|
|
|
| 49 |
|
| 50 |
TASKS = ["T2AV", "I2AV", "L2AV", "FL2AV", "Ref2AV"]
|
| 51 |
RATIOS = ["adaptive", "21:9", "16:9", "4:3", "1:1", "3:4", "9:16"]
|
|
@@ -265,22 +269,25 @@ def output_schema_ok(task: str, text: str) -> bool:
|
|
| 265 |
|
| 266 |
print(f"[boot] Loading Qwen2.5-Omni processor from {BASE_MODEL} …", flush=True)
|
| 267 |
processor = Qwen2_5OmniProcessor.from_pretrained(BASE_MODEL, trust_remote_code=False)
|
| 268 |
-
processor.image_processor.max_pixels = IMAGE_MAX_PIXELS
|
| 269 |
-
try:
|
| 270 |
-
processor.video_processor.max_pixels = VIDEO_MAX_PIXELS
|
| 271 |
-
except Exception as error: # pragma: no cover - attribute layout varies by version
|
| 272 |
-
print(f"[boot] video_processor.max_pixels not settable ({error!r})", flush=True)
|
| 273 |
-
processor.tokenizer.padding_side = "right"
|
| 274 |
if processor.tokenizer.pad_token_id is None:
|
| 275 |
processor.tokenizer.pad_token = processor.tokenizer.eos_token
|
| 276 |
|
| 277 |
print(f"[boot] Loading Thinker weights from {BASE_MODEL} (bf16, sdpa) …", flush=True)
|
|
|
|
|
|
|
|
|
|
| 278 |
model = Qwen2_5OmniThinkerForConditionalGeneration.from_pretrained(
|
| 279 |
BASE_MODEL,
|
|
|
|
| 280 |
dtype=torch.bfloat16,
|
| 281 |
attn_implementation="sdpa",
|
| 282 |
trust_remote_code=False,
|
| 283 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 284 |
)
|
| 285 |
|
| 286 |
print(f"[boot] Applying LoRA adapter {ADAPTER_REPO} …", flush=True)
|
|
@@ -291,7 +298,7 @@ from peft import LoraConfig, PeftModel, set_peft_model_state_dict # noqa: E402
|
|
| 291 |
|
| 292 |
_adapter_file = hf_hub_download(ADAPTER_REPO, "adapter_model.safetensors")
|
| 293 |
_adapter_state_dict: dict[str, torch.Tensor] = {}
|
| 294 |
-
with safe_open(_adapter_file, framework="numpy") as handle:
|
| 295 |
for key in handle.keys():
|
| 296 |
_adapter_state_dict[key] = torch.from_numpy(handle.get_tensor(key))
|
| 297 |
|
|
@@ -340,15 +347,23 @@ def _encode(messages: list[dict[str, Any]], video_fps: float) -> dict[str, Any]:
|
|
| 340 |
add_generation_prompt=True,
|
| 341 |
return_dict=True,
|
| 342 |
return_tensors="pt",
|
| 343 |
-
text_kwargs={"padding": False},
|
| 344 |
-
images_kwargs={"max_pixels": IMAGE_MAX_PIXELS},
|
| 345 |
-
videos_kwargs={
|
| 346 |
-
"max_pixels": VIDEO_MAX_PIXELS,
|
| 347 |
-
"fps": video_fps,
|
| 348 |
-
"use_audio_in_video": False,
|
| 349 |
-
},
|
| 350 |
-
video_fps=video_fps,
|
| 351 |
load_audio_from_video=False,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 352 |
)
|
| 353 |
encoded = dict(encoded)
|
| 354 |
encoded["use_audio_in_video"] = False
|
|
|
|
| 33 |
from huggingface_hub import hf_hub_download
|
| 34 |
from safetensors import safe_open
|
| 35 |
from transformers import (
|
| 36 |
+
Qwen2_5OmniConfig,
|
| 37 |
Qwen2_5OmniProcessor,
|
| 38 |
Qwen2_5OmniThinkerForConditionalGeneration,
|
| 39 |
TextIteratorStreamer,
|
|
|
|
| 47 |
# Defaults taken from the adapter repo's infer.py CLI.
|
| 48 |
IMAGE_MAX_PIXELS = 301056
|
| 49 |
VIDEO_MAX_PIXELS = 100352
|
| 50 |
+
# Qwen2-VL image-processor default lower bound, kept explicit because
|
| 51 |
+
# transformers 5.x only honours max_pixels when min_pixels is given too.
|
| 52 |
+
IMAGE_MIN_PIXELS = 56 * 56
|
| 53 |
|
| 54 |
TASKS = ["T2AV", "I2AV", "L2AV", "FL2AV", "Ref2AV"]
|
| 55 |
RATIOS = ["adaptive", "21:9", "16:9", "4:3", "1:1", "3:4", "9:16"]
|
|
|
|
| 269 |
|
| 270 |
print(f"[boot] Loading Qwen2.5-Omni processor from {BASE_MODEL} …", flush=True)
|
| 271 |
processor = Qwen2_5OmniProcessor.from_pretrained(BASE_MODEL, trust_remote_code=False)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 272 |
if processor.tokenizer.pad_token_id is None:
|
| 273 |
processor.tokenizer.pad_token = processor.tokenizer.eos_token
|
| 274 |
|
| 275 |
print(f"[boot] Loading Thinker weights from {BASE_MODEL} (bf16, sdpa) …", flush=True)
|
| 276 |
+
# The checkpoint is the full Omni model; take the thinker sub-config explicitly
|
| 277 |
+
# and let `base_model_prefix = "thinker"` strip the `thinker.` weight prefix.
|
| 278 |
+
_thinker_config = Qwen2_5OmniConfig.from_pretrained(BASE_MODEL).thinker_config
|
| 279 |
model = Qwen2_5OmniThinkerForConditionalGeneration.from_pretrained(
|
| 280 |
BASE_MODEL,
|
| 281 |
+
config=_thinker_config,
|
| 282 |
dtype=torch.bfloat16,
|
| 283 |
attn_implementation="sdpa",
|
| 284 |
trust_remote_code=False,
|
| 285 |
+
)
|
| 286 |
+
print(
|
| 287 |
+
"[boot] Thinker loaded: hidden_size="
|
| 288 |
+
f"{model.config.text_config.hidden_size}, "
|
| 289 |
+
f"vocab_size={model.config.text_config.vocab_size}",
|
| 290 |
+
flush=True,
|
| 291 |
)
|
| 292 |
|
| 293 |
print(f"[boot] Applying LoRA adapter {ADAPTER_REPO} …", flush=True)
|
|
|
|
| 298 |
|
| 299 |
_adapter_file = hf_hub_download(ADAPTER_REPO, "adapter_model.safetensors")
|
| 300 |
_adapter_state_dict: dict[str, torch.Tensor] = {}
|
| 301 |
+
with safe_open(_adapter_file, framework="numpy", device="cpu") as handle:
|
| 302 |
for key in handle.keys():
|
| 303 |
_adapter_state_dict[key] = torch.from_numpy(handle.get_tensor(key))
|
| 304 |
|
|
|
|
| 347 |
add_generation_prompt=True,
|
| 348 |
return_dict=True,
|
| 349 |
return_tensors="pt",
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 350 |
load_audio_from_video=False,
|
| 351 |
+
# transformers 5.x: processing kwargs must live in `processor_kwargs`.
|
| 352 |
+
# Passing any other flat kwarg here silently REPLACES this dict.
|
| 353 |
+
processor_kwargs={
|
| 354 |
+
"text_kwargs": {"padding": False},
|
| 355 |
+
"images_kwargs": {
|
| 356 |
+
"min_pixels": IMAGE_MIN_PIXELS,
|
| 357 |
+
"max_pixels": IMAGE_MAX_PIXELS,
|
| 358 |
+
},
|
| 359 |
+
"videos_kwargs": {
|
| 360 |
+
"min_pixels": VIDEO_MAX_PIXELS,
|
| 361 |
+
"max_pixels": VIDEO_MAX_PIXELS,
|
| 362 |
+
"fps": video_fps,
|
| 363 |
+
"do_sample_frames": True,
|
| 364 |
+
"use_audio_in_video": False,
|
| 365 |
+
},
|
| 366 |
+
},
|
| 367 |
)
|
| 368 |
encoded = dict(encoded)
|
| 369 |
encoded["use_audio_in_video"] = False
|
requirements.txt
CHANGED
|
@@ -1,9 +1,8 @@
|
|
| 1 |
-
#
|
| 2 |
-
#
|
| 3 |
-
|
| 4 |
-
transformers==4.57.6
|
| 5 |
accelerate>=1.10
|
| 6 |
-
peft>=0.
|
| 7 |
safetensors>=0.5.0
|
| 8 |
torchvision
|
| 9 |
numpy
|
|
|
|
| 1 |
+
# transformers 4.x pins huggingface-hub<1.0, which conflicts with gradio 6,
|
| 2 |
+
# so this Space runs the 5.x API (processing kwargs go in `processor_kwargs`).
|
| 3 |
+
transformers==5.16.1
|
|
|
|
| 4 |
accelerate>=1.10
|
| 5 |
+
peft>=0.18
|
| 6 |
safetensors>=0.5.0
|
| 7 |
torchvision
|
| 8 |
numpy
|