Spaces:
Running on Zero
Running on Zero
Upload app.py with huggingface_hub
Browse files
app.py
CHANGED
|
@@ -211,10 +211,11 @@ model = Qwen2_5OmniThinkerForConditionalGeneration.from_pretrained(
|
|
| 211 |
attn_implementation="sdpa",
|
| 212 |
trust_remote_code=False,
|
| 213 |
low_cpu_mem_usage=True,
|
| 214 |
-
)
|
| 215 |
|
| 216 |
print("Loading LoRA adapter...", flush=True)
|
| 217 |
model = PeftModel.from_pretrained(model, ADAPTER, is_trainable=False)
|
|
|
|
| 218 |
model.eval()
|
| 219 |
|
| 220 |
# Determine input device and context limit
|
|
|
|
| 211 |
attn_implementation="sdpa",
|
| 212 |
trust_remote_code=False,
|
| 213 |
low_cpu_mem_usage=True,
|
| 214 |
+
)
|
| 215 |
|
| 216 |
print("Loading LoRA adapter...", flush=True)
|
| 217 |
model = PeftModel.from_pretrained(model, ADAPTER, is_trainable=False)
|
| 218 |
+
model = model.to("cuda")
|
| 219 |
model.eval()
|
| 220 |
|
| 221 |
# Determine input device and context limit
|