multimodalart HF Staff commited on
Commit
903d38e
·
verified ·
1 Parent(s): aaabf82

Upload app.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. app.py +2 -1
app.py CHANGED
@@ -211,10 +211,11 @@ model = Qwen2_5OmniThinkerForConditionalGeneration.from_pretrained(
211
  attn_implementation="sdpa",
212
  trust_remote_code=False,
213
  low_cpu_mem_usage=True,
214
- ).to("cuda")
215
 
216
  print("Loading LoRA adapter...", flush=True)
217
  model = PeftModel.from_pretrained(model, ADAPTER, is_trainable=False)
 
218
  model.eval()
219
 
220
  # Determine input device and context limit
 
211
  attn_implementation="sdpa",
212
  trust_remote_code=False,
213
  low_cpu_mem_usage=True,
214
+ )
215
 
216
  print("Loading LoRA adapter...", flush=True)
217
  model = PeftModel.from_pretrained(model, ADAPTER, is_trainable=False)
218
+ model = model.to("cuda")
219
  model.eval()
220
 
221
  # Determine input device and context limit