Jev-Omni demo: text, image, audio and video decisions on ZeroGPU from the unified bf16 checkpoint
Browse files- README.md +15 -7
- app.py +142 -0
- packages.txt +1 -0
- requirements.txt +11 -0
README.md
CHANGED
|
@@ -1,13 +1,21 @@
|
|
| 1 |
---
|
| 2 |
-
title: Jev
|
| 3 |
-
emoji:
|
| 4 |
-
colorFrom:
|
| 5 |
-
colorTo:
|
| 6 |
sdk: gradio
|
| 7 |
-
sdk_version:
|
| 8 |
-
python_version: '3.12'
|
| 9 |
app_file: app.py
|
| 10 |
pinned: false
|
|
|
|
|
|
|
|
|
|
|
|
|
| 11 |
---
|
| 12 |
|
| 13 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
+
title: Jev-Omni
|
| 3 |
+
emoji: 🎯
|
| 4 |
+
colorFrom: blue
|
| 5 |
+
colorTo: gray
|
| 6 |
sdk: gradio
|
| 7 |
+
sdk_version: 5.49.1
|
|
|
|
| 8 |
app_file: app.py
|
| 9 |
pinned: false
|
| 10 |
+
license: apache-2.0
|
| 11 |
+
short_description: Multimodal decision classifier — text, image, audio, video
|
| 12 |
+
models:
|
| 13 |
+
- akhilaaa3/Jev-Omni
|
| 14 |
---
|
| 15 |
|
| 16 |
+
# Jev-Omni
|
| 17 |
+
|
| 18 |
+
Give it a situation and a question with options. It returns a probability for each option from one
|
| 19 |
+
forward pass — no generated text. Built on Gemma 4 12B IT with a 30,000-question fine-tune.
|
| 20 |
+
|
| 21 |
+
Model card and results: [akhilaaa3/Jev-Omni](https://huggingface.co/akhilaaa3/Jev-Omni).
|
app.py
ADDED
|
@@ -0,0 +1,142 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Jev-Omni on ZeroGPU: text, image, audio or video in; a probability per option out.
|
| 2 |
+
|
| 3 |
+
The model is placed on `cuda` at module level on purpose. ZeroGPU emulates CUDA outside `@spaces.GPU`
|
| 4 |
+
and swaps in a real GPU inside it, and its docs say startup placement is the efficient path.
|
| 5 |
+
Weights are the single unified bf16 checkpoint under `unified/` in the model repo, so one ~24 GB
|
| 6 |
+
download covers text, vision and audio.
|
| 7 |
+
"""
|
| 8 |
+
import os, subprocess, tempfile, time
|
| 9 |
+
import gradio as gr
|
| 10 |
+
import numpy as np
|
| 11 |
+
import spaces
|
| 12 |
+
import torch
|
| 13 |
+
torch.backends.cuda.enable_cudnn_sdp(False) # cuDNN attention hit a version mismatch on one CUDA image; the other SDPA backends are fine
|
| 14 |
+
from huggingface_hub import hf_hub_download, snapshot_download
|
| 15 |
+
from PIL import Image
|
| 16 |
+
from transformers import AutoConfig, AutoProcessor
|
| 17 |
+
|
| 18 |
+
REPO = "akhilaaa3/Jev-Omni"
|
| 19 |
+
DEVICE = "cuda"
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
class Head256(torch.nn.Module):
|
| 23 |
+
def __init__(self, hidden):
|
| 24 |
+
super().__init__()
|
| 25 |
+
self.register_buffer("mu", torch.zeros(1, hidden))
|
| 26 |
+
self.register_buffer("sd", torch.ones(1, hidden))
|
| 27 |
+
self.linear = torch.nn.Linear(hidden, 256, dtype=torch.float32)
|
| 28 |
+
|
| 29 |
+
def forward(self, features, counts):
|
| 30 |
+
z = self.linear((features.float() - self.mu) / self.sd)
|
| 31 |
+
return z.masked_fill(torch.arange(256, device=z.device)[None] >= counts[:, None], -1e30)
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def _prompt(state, question, options):
|
| 35 |
+
choices = "\n".join(f"{i + 1}. {v}" for i, v in enumerate(options))
|
| 36 |
+
return (f"{state}\n\n---\n\nQUESTION: {question}\n\nOPTIONS:\n{choices}\n\n"
|
| 37 |
+
f"Reply with only the number of the correct option (1-{len(options)}).\n"
|
| 38 |
+
"Output a single number and nothing else.")
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
def _video_frames(path, count=16):
|
| 42 |
+
import cv2
|
| 43 |
+
cap = cv2.VideoCapture(str(path)); total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
|
| 44 |
+
wanted = sorted({int(round((total - 1) * (k + .5) / count)) for k in range(count)})
|
| 45 |
+
frames, cur = [], 0
|
| 46 |
+
for idx in wanted:
|
| 47 |
+
while cur < idx and cap.grab(): cur += 1
|
| 48 |
+
ok, fr = cap.read(); cur += 1
|
| 49 |
+
if not ok: break
|
| 50 |
+
frames.append(Image.fromarray(cv2.cvtColor(fr, cv2.COLOR_BGR2RGB)))
|
| 51 |
+
cap.release()
|
| 52 |
+
if not frames: raise gr.Error("Could not decode that video.")
|
| 53 |
+
return frames
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
# ---- load once, at startup, onto (emulated) cuda
|
| 57 |
+
t0 = time.time()
|
| 58 |
+
local = snapshot_download(REPO, allow_patterns=["unified/*", "head.pt", "decision_config.json"])
|
| 59 |
+
import transformers, json
|
| 60 |
+
cfg = AutoConfig.from_pretrained(local, subfolder="unified")
|
| 61 |
+
MODEL = getattr(transformers, cfg.architectures[0]).from_pretrained(
|
| 62 |
+
local, subfolder="unified", dtype=torch.bfloat16, device_map=DEVICE).eval()
|
| 63 |
+
PROC = AutoProcessor.from_pretrained(local, subfolder="unified")
|
| 64 |
+
dc = json.load(open(os.path.join(local, "decision_config.json")))
|
| 65 |
+
HEAD = Head256(dc["hidden_size"]).to(DEVICE).eval()
|
| 66 |
+
HEAD.load_state_dict(torch.load(os.path.join(local, "head.pt"), map_location=DEVICE, weights_only=True))
|
| 67 |
+
CAP = {}
|
| 68 |
+
for path in ("model.language_model", "language_model.model", "model.text_model", "model"):
|
| 69 |
+
node = MODEL
|
| 70 |
+
for part in path.split("."):
|
| 71 |
+
node = getattr(node, part, None)
|
| 72 |
+
if node is None: break
|
| 73 |
+
if node is not None and hasattr(node, "layers"):
|
| 74 |
+
node.register_forward_hook(lambda _m, _a, out: CAP.__setitem__(
|
| 75 |
+
"h", (out.last_hidden_state if hasattr(out, "last_hidden_state") else out[0])[:, -1].float()))
|
| 76 |
+
break
|
| 77 |
+
else:
|
| 78 |
+
raise RuntimeError("text backbone not found")
|
| 79 |
+
print(f"loaded in {time.time() - t0:.0f}s", flush=True)
|
| 80 |
+
|
| 81 |
+
|
| 82 |
+
def _duration(state, question, options_text, media, modality):
|
| 83 |
+
return 90 if modality == "video" else 45
|
| 84 |
+
|
| 85 |
+
|
| 86 |
+
@spaces.GPU(duration=_duration)
|
| 87 |
+
@torch.inference_mode()
|
| 88 |
+
def decide(state, question, options_text, media, modality):
|
| 89 |
+
options = [o.strip() for o in options_text.splitlines() if o.strip()]
|
| 90 |
+
if not 2 <= len(options) <= 256: raise gr.Error("Give between 2 and 256 options, one per line.")
|
| 91 |
+
if modality != "text" and not media: raise gr.Error(f"Upload a file for {modality} input.")
|
| 92 |
+
content, tmp = [], None
|
| 93 |
+
if modality == "image": content.append({"type": "image", "image": Image.open(media).convert("RGB")})
|
| 94 |
+
elif modality == "video": content.extend({"type": "image", "image": f} for f in _video_frames(media))
|
| 95 |
+
elif modality == "audio":
|
| 96 |
+
tmp = tempfile.NamedTemporaryFile(suffix=".wav", delete=False); tmp.close()
|
| 97 |
+
subprocess.run(["ffmpeg", "-v", "error", "-y", "-i", str(media), "-t", "30", "-ac", "1", "-ar", "16000", tmp.name], check=True)
|
| 98 |
+
content.append({"type": "audio", "audio": tmp.name})
|
| 99 |
+
content.append({"type": "text", "text": _prompt(state, question, options)})
|
| 100 |
+
try:
|
| 101 |
+
inputs = PROC.apply_chat_template([{"role": "user", "content": content}], add_generation_prompt=True,
|
| 102 |
+
tokenize=True, return_dict=True, return_tensors="pt", enable_thinking=False)
|
| 103 |
+
finally:
|
| 104 |
+
if tmp: os.unlink(tmp.name)
|
| 105 |
+
inputs = {k: (v.to(DEVICE, dtype=torch.bfloat16) if torch.is_floating_point(v) else v.to(DEVICE)) for k, v in inputs.items()}
|
| 106 |
+
CAP.clear(); t = time.time()
|
| 107 |
+
with torch.autocast("cuda", dtype=torch.bfloat16):
|
| 108 |
+
MODEL(**inputs, use_cache=False, logits_to_keep=1)
|
| 109 |
+
probs = HEAD(CAP["h"], torch.tensor([len(options)], device=DEVICE))[0, :len(options)].softmax(-1)
|
| 110 |
+
ms = (time.time() - t) * 1000
|
| 111 |
+
p = probs.float().cpu().tolist()
|
| 112 |
+
best = int(np.argmax(p))
|
| 113 |
+
return {o: v for o, v in zip(options, p)}, f"**{options[best]}** · {p[best]*100:.1f}% · {ms:.0f} ms"
|
| 114 |
+
|
| 115 |
+
|
| 116 |
+
EXAMPLES = [
|
| 117 |
+
["The meeting starts at 10 AM. It is now 9 AM.", "Has the meeting started?", "Yes\nNo", None, "text"],
|
| 118 |
+
["Customer: I was charged twice.\nAgent: I've refunded $29 to your card.\nCustomer: Got it. All sorted, thanks!",
|
| 119 |
+
"Was the issue actually resolved?", "Yes\nNo", None, "text"],
|
| 120 |
+
]
|
| 121 |
+
|
| 122 |
+
with gr.Blocks(title="Jev-Omni") as demo:
|
| 123 |
+
gr.Markdown("# Jev-Omni\nA decision classifier for text, images, audio and video. "
|
| 124 |
+
"Give it a situation and a question with options; it returns a probability for each option — "
|
| 125 |
+
"no generated text, one forward pass.")
|
| 126 |
+
with gr.Row():
|
| 127 |
+
with gr.Column():
|
| 128 |
+
modality = gr.Radio(["text", "image", "audio", "video"], value="text", label="Input type")
|
| 129 |
+
media = gr.File(label="Image / audio / video file (not needed for text)", file_types=["image", "audio", "video"])
|
| 130 |
+
state = gr.Textbox(label="Situation", lines=6, placeholder="What the model should know. For media, this can be short.")
|
| 131 |
+
question = gr.Textbox(label="Question")
|
| 132 |
+
options = gr.Textbox(label="Options, one per line", lines=4, value="Yes\nNo")
|
| 133 |
+
go = gr.Button("Decide", variant="primary")
|
| 134 |
+
with gr.Column():
|
| 135 |
+
verdict = gr.Markdown()
|
| 136 |
+
probs = gr.Label(label="Probabilities", num_top_classes=10)
|
| 137 |
+
go.click(decide, [state, question, options, media, modality], [probs, verdict])
|
| 138 |
+
gr.Examples(EXAMPLES, [state, question, options, media, modality])
|
| 139 |
+
gr.Markdown("Best supported at ≤20 options; audio is capped at 30 s; video is sampled to 16 frames. "
|
| 140 |
+
"Weights: [akhilaaa3/Jev-Omni](https://huggingface.co/akhilaaa3/Jev-Omni), Apache-2.0.")
|
| 141 |
+
|
| 142 |
+
demo.queue().launch()
|
packages.txt
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
ffmpeg
|
requirements.txt
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
torch>=2.10
|
| 2 |
+
torchvision
|
| 3 |
+
transformers==5.17.0
|
| 4 |
+
accelerate
|
| 5 |
+
huggingface_hub
|
| 6 |
+
safetensors
|
| 7 |
+
numpy
|
| 8 |
+
Pillow
|
| 9 |
+
opencv-python-headless
|
| 10 |
+
soundfile
|
| 11 |
+
librosa
|