multimodalart HF Staff commited on
Commit
94cd091
·
verified ·
1 Parent(s): ed15f63

Port to transformers 5.x API (hub 1.x / gradio 6 compatible)

Browse files
Files changed (3) hide show
  1. README.md +3 -1
  2. app.py +31 -16
  3. requirements.txt +4 -5
README.md CHANGED
@@ -47,7 +47,9 @@ repo, and message construction, the duration grid, reference labelling /
47
  validation and the output schema check are ported 1:1 from the adapter repo's
48
  `infer.py`. Encoding uses the same `ProcessorMixin.apply_chat_template` path
49
  with `max_pixels` 301056 (images) / 100352 (video), `load_audio_from_video=False`
50
- and `use_audio_in_video=False`.
 
 
51
 
52
  Runs on ZeroGPU: the Thinker is loaded in bf16 with SDPA attention and the LoRA
53
  is attached with PEFT at module scope.
 
47
  validation and the output schema check are ported 1:1 from the adapter repo's
48
  `infer.py`. Encoding uses the same `ProcessorMixin.apply_chat_template` path
49
  with `max_pixels` 301056 (images) / 100352 (video), `load_audio_from_video=False`
50
+ and `use_audio_in_video=False`. Those processing kwargs are passed through
51
+ transformers 5.x's `processor_kwargs` dict (transformers 4.x, which the
52
+ reference pins, is incompatible with `huggingface_hub` 1.x / Gradio 6).
53
 
54
  Runs on ZeroGPU: the Thinker is loaded in bf16 with SDPA attention and the LoRA
55
  is attached with PEFT at module scope.
app.py CHANGED
@@ -33,6 +33,7 @@ import gradio as gr
33
  from huggingface_hub import hf_hub_download
34
  from safetensors import safe_open
35
  from transformers import (
 
36
  Qwen2_5OmniProcessor,
37
  Qwen2_5OmniThinkerForConditionalGeneration,
38
  TextIteratorStreamer,
@@ -46,6 +47,9 @@ ADAPTER_REPO = "lightx2v/MiniMax-H3-Prompt-Rewriter-LoRA-Omni"
46
  # Defaults taken from the adapter repo's infer.py CLI.
47
  IMAGE_MAX_PIXELS = 301056
48
  VIDEO_MAX_PIXELS = 100352
 
 
 
49
 
50
  TASKS = ["T2AV", "I2AV", "L2AV", "FL2AV", "Ref2AV"]
51
  RATIOS = ["adaptive", "21:9", "16:9", "4:3", "1:1", "3:4", "9:16"]
@@ -265,22 +269,25 @@ def output_schema_ok(task: str, text: str) -> bool:
265
 
266
  print(f"[boot] Loading Qwen2.5-Omni processor from {BASE_MODEL} …", flush=True)
267
  processor = Qwen2_5OmniProcessor.from_pretrained(BASE_MODEL, trust_remote_code=False)
268
- processor.image_processor.max_pixels = IMAGE_MAX_PIXELS
269
- try:
270
- processor.video_processor.max_pixels = VIDEO_MAX_PIXELS
271
- except Exception as error: # pragma: no cover - attribute layout varies by version
272
- print(f"[boot] video_processor.max_pixels not settable ({error!r})", flush=True)
273
- processor.tokenizer.padding_side = "right"
274
  if processor.tokenizer.pad_token_id is None:
275
  processor.tokenizer.pad_token = processor.tokenizer.eos_token
276
 
277
  print(f"[boot] Loading Thinker weights from {BASE_MODEL} (bf16, sdpa) …", flush=True)
 
 
 
278
  model = Qwen2_5OmniThinkerForConditionalGeneration.from_pretrained(
279
  BASE_MODEL,
 
280
  dtype=torch.bfloat16,
281
  attn_implementation="sdpa",
282
  trust_remote_code=False,
283
- low_cpu_mem_usage=True,
 
 
 
 
 
284
  )
285
 
286
  print(f"[boot] Applying LoRA adapter {ADAPTER_REPO} …", flush=True)
@@ -291,7 +298,7 @@ from peft import LoraConfig, PeftModel, set_peft_model_state_dict # noqa: E402
291
 
292
  _adapter_file = hf_hub_download(ADAPTER_REPO, "adapter_model.safetensors")
293
  _adapter_state_dict: dict[str, torch.Tensor] = {}
294
- with safe_open(_adapter_file, framework="numpy") as handle:
295
  for key in handle.keys():
296
  _adapter_state_dict[key] = torch.from_numpy(handle.get_tensor(key))
297
 
@@ -340,15 +347,23 @@ def _encode(messages: list[dict[str, Any]], video_fps: float) -> dict[str, Any]:
340
  add_generation_prompt=True,
341
  return_dict=True,
342
  return_tensors="pt",
343
- text_kwargs={"padding": False},
344
- images_kwargs={"max_pixels": IMAGE_MAX_PIXELS},
345
- videos_kwargs={
346
- "max_pixels": VIDEO_MAX_PIXELS,
347
- "fps": video_fps,
348
- "use_audio_in_video": False,
349
- },
350
- video_fps=video_fps,
351
  load_audio_from_video=False,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
352
  )
353
  encoded = dict(encoded)
354
  encoded["use_audio_in_video"] = False
 
33
  from huggingface_hub import hf_hub_download
34
  from safetensors import safe_open
35
  from transformers import (
36
+ Qwen2_5OmniConfig,
37
  Qwen2_5OmniProcessor,
38
  Qwen2_5OmniThinkerForConditionalGeneration,
39
  TextIteratorStreamer,
 
47
  # Defaults taken from the adapter repo's infer.py CLI.
48
  IMAGE_MAX_PIXELS = 301056
49
  VIDEO_MAX_PIXELS = 100352
50
+ # Qwen2-VL image-processor default lower bound, kept explicit because
51
+ # transformers 5.x only honours max_pixels when min_pixels is given too.
52
+ IMAGE_MIN_PIXELS = 56 * 56
53
 
54
  TASKS = ["T2AV", "I2AV", "L2AV", "FL2AV", "Ref2AV"]
55
  RATIOS = ["adaptive", "21:9", "16:9", "4:3", "1:1", "3:4", "9:16"]
 
269
 
270
  print(f"[boot] Loading Qwen2.5-Omni processor from {BASE_MODEL} …", flush=True)
271
  processor = Qwen2_5OmniProcessor.from_pretrained(BASE_MODEL, trust_remote_code=False)
 
 
 
 
 
 
272
  if processor.tokenizer.pad_token_id is None:
273
  processor.tokenizer.pad_token = processor.tokenizer.eos_token
274
 
275
  print(f"[boot] Loading Thinker weights from {BASE_MODEL} (bf16, sdpa) …", flush=True)
276
+ # The checkpoint is the full Omni model; take the thinker sub-config explicitly
277
+ # and let `base_model_prefix = "thinker"` strip the `thinker.` weight prefix.
278
+ _thinker_config = Qwen2_5OmniConfig.from_pretrained(BASE_MODEL).thinker_config
279
  model = Qwen2_5OmniThinkerForConditionalGeneration.from_pretrained(
280
  BASE_MODEL,
281
+ config=_thinker_config,
282
  dtype=torch.bfloat16,
283
  attn_implementation="sdpa",
284
  trust_remote_code=False,
285
+ )
286
+ print(
287
+ "[boot] Thinker loaded: hidden_size="
288
+ f"{model.config.text_config.hidden_size}, "
289
+ f"vocab_size={model.config.text_config.vocab_size}",
290
+ flush=True,
291
  )
292
 
293
  print(f"[boot] Applying LoRA adapter {ADAPTER_REPO} …", flush=True)
 
298
 
299
  _adapter_file = hf_hub_download(ADAPTER_REPO, "adapter_model.safetensors")
300
  _adapter_state_dict: dict[str, torch.Tensor] = {}
301
+ with safe_open(_adapter_file, framework="numpy", device="cpu") as handle:
302
  for key in handle.keys():
303
  _adapter_state_dict[key] = torch.from_numpy(handle.get_tensor(key))
304
 
 
347
  add_generation_prompt=True,
348
  return_dict=True,
349
  return_tensors="pt",
 
 
 
 
 
 
 
 
350
  load_audio_from_video=False,
351
+ # transformers 5.x: processing kwargs must live in `processor_kwargs`.
352
+ # Passing any other flat kwarg here silently REPLACES this dict.
353
+ processor_kwargs={
354
+ "text_kwargs": {"padding": False},
355
+ "images_kwargs": {
356
+ "min_pixels": IMAGE_MIN_PIXELS,
357
+ "max_pixels": IMAGE_MAX_PIXELS,
358
+ },
359
+ "videos_kwargs": {
360
+ "min_pixels": VIDEO_MAX_PIXELS,
361
+ "max_pixels": VIDEO_MAX_PIXELS,
362
+ "fps": video_fps,
363
+ "do_sample_frames": True,
364
+ "use_audio_in_video": False,
365
+ },
366
+ },
367
  )
368
  encoded = dict(encoded)
369
  encoded["use_audio_in_video"] = False
requirements.txt CHANGED
@@ -1,9 +1,8 @@
1
- # Pinned to the last 4.x: transformers 5.x changed
2
- # ProcessorMixin.apply_chat_template to require a `processor_kwargs` dict,
3
- # which is incompatible with the adapter's reference encoding path.
4
- transformers==4.57.6
5
  accelerate>=1.10
6
- peft>=0.15.2
7
  safetensors>=0.5.0
8
  torchvision
9
  numpy
 
1
+ # transformers 4.x pins huggingface-hub<1.0, which conflicts with gradio 6,
2
+ # so this Space runs the 5.x API (processing kwargs go in `processor_kwargs`).
3
+ transformers==5.16.1
 
4
  accelerate>=1.10
5
+ peft>=0.18
6
  safetensors>=0.5.0
7
  torchvision
8
  numpy