akhilaaa3 commited on
Commit
316c44f
·
verified ·
1 Parent(s): 3ef96ac

Jev-Omni demo: text, image, audio and video decisions on ZeroGPU from the unified bf16 checkpoint

Browse files
Files changed (4) hide show
  1. README.md +15 -7
  2. app.py +142 -0
  3. packages.txt +1 -0
  4. requirements.txt +11 -0
README.md CHANGED
@@ -1,13 +1,21 @@
1
  ---
2
- title: Jev Omni
3
- emoji: 📈
4
- colorFrom: indigo
5
- colorTo: red
6
  sdk: gradio
7
- sdk_version: 6.28.0
8
- python_version: '3.12'
9
  app_file: app.py
10
  pinned: false
 
 
 
 
11
  ---
12
 
13
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
1
  ---
2
+ title: Jev-Omni
3
+ emoji: 🎯
4
+ colorFrom: blue
5
+ colorTo: gray
6
  sdk: gradio
7
+ sdk_version: 5.49.1
 
8
  app_file: app.py
9
  pinned: false
10
+ license: apache-2.0
11
+ short_description: Multimodal decision classifier — text, image, audio, video
12
+ models:
13
+ - akhilaaa3/Jev-Omni
14
  ---
15
 
16
+ # Jev-Omni
17
+
18
+ Give it a situation and a question with options. It returns a probability for each option from one
19
+ forward pass — no generated text. Built on Gemma 4 12B IT with a 30,000-question fine-tune.
20
+
21
+ Model card and results: [akhilaaa3/Jev-Omni](https://huggingface.co/akhilaaa3/Jev-Omni).
app.py ADDED
@@ -0,0 +1,142 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Jev-Omni on ZeroGPU: text, image, audio or video in; a probability per option out.
2
+
3
+ The model is placed on `cuda` at module level on purpose. ZeroGPU emulates CUDA outside `@spaces.GPU`
4
+ and swaps in a real GPU inside it, and its docs say startup placement is the efficient path.
5
+ Weights are the single unified bf16 checkpoint under `unified/` in the model repo, so one ~24 GB
6
+ download covers text, vision and audio.
7
+ """
8
+ import os, subprocess, tempfile, time
9
+ import gradio as gr
10
+ import numpy as np
11
+ import spaces
12
+ import torch
13
+ torch.backends.cuda.enable_cudnn_sdp(False) # cuDNN attention hit a version mismatch on one CUDA image; the other SDPA backends are fine
14
+ from huggingface_hub import hf_hub_download, snapshot_download
15
+ from PIL import Image
16
+ from transformers import AutoConfig, AutoProcessor
17
+
18
+ REPO = "akhilaaa3/Jev-Omni"
19
+ DEVICE = "cuda"
20
+
21
+
22
+ class Head256(torch.nn.Module):
23
+ def __init__(self, hidden):
24
+ super().__init__()
25
+ self.register_buffer("mu", torch.zeros(1, hidden))
26
+ self.register_buffer("sd", torch.ones(1, hidden))
27
+ self.linear = torch.nn.Linear(hidden, 256, dtype=torch.float32)
28
+
29
+ def forward(self, features, counts):
30
+ z = self.linear((features.float() - self.mu) / self.sd)
31
+ return z.masked_fill(torch.arange(256, device=z.device)[None] >= counts[:, None], -1e30)
32
+
33
+
34
+ def _prompt(state, question, options):
35
+ choices = "\n".join(f"{i + 1}. {v}" for i, v in enumerate(options))
36
+ return (f"{state}\n\n---\n\nQUESTION: {question}\n\nOPTIONS:\n{choices}\n\n"
37
+ f"Reply with only the number of the correct option (1-{len(options)}).\n"
38
+ "Output a single number and nothing else.")
39
+
40
+
41
+ def _video_frames(path, count=16):
42
+ import cv2
43
+ cap = cv2.VideoCapture(str(path)); total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
44
+ wanted = sorted({int(round((total - 1) * (k + .5) / count)) for k in range(count)})
45
+ frames, cur = [], 0
46
+ for idx in wanted:
47
+ while cur < idx and cap.grab(): cur += 1
48
+ ok, fr = cap.read(); cur += 1
49
+ if not ok: break
50
+ frames.append(Image.fromarray(cv2.cvtColor(fr, cv2.COLOR_BGR2RGB)))
51
+ cap.release()
52
+ if not frames: raise gr.Error("Could not decode that video.")
53
+ return frames
54
+
55
+
56
+ # ---- load once, at startup, onto (emulated) cuda
57
+ t0 = time.time()
58
+ local = snapshot_download(REPO, allow_patterns=["unified/*", "head.pt", "decision_config.json"])
59
+ import transformers, json
60
+ cfg = AutoConfig.from_pretrained(local, subfolder="unified")
61
+ MODEL = getattr(transformers, cfg.architectures[0]).from_pretrained(
62
+ local, subfolder="unified", dtype=torch.bfloat16, device_map=DEVICE).eval()
63
+ PROC = AutoProcessor.from_pretrained(local, subfolder="unified")
64
+ dc = json.load(open(os.path.join(local, "decision_config.json")))
65
+ HEAD = Head256(dc["hidden_size"]).to(DEVICE).eval()
66
+ HEAD.load_state_dict(torch.load(os.path.join(local, "head.pt"), map_location=DEVICE, weights_only=True))
67
+ CAP = {}
68
+ for path in ("model.language_model", "language_model.model", "model.text_model", "model"):
69
+ node = MODEL
70
+ for part in path.split("."):
71
+ node = getattr(node, part, None)
72
+ if node is None: break
73
+ if node is not None and hasattr(node, "layers"):
74
+ node.register_forward_hook(lambda _m, _a, out: CAP.__setitem__(
75
+ "h", (out.last_hidden_state if hasattr(out, "last_hidden_state") else out[0])[:, -1].float()))
76
+ break
77
+ else:
78
+ raise RuntimeError("text backbone not found")
79
+ print(f"loaded in {time.time() - t0:.0f}s", flush=True)
80
+
81
+
82
+ def _duration(state, question, options_text, media, modality):
83
+ return 90 if modality == "video" else 45
84
+
85
+
86
+ @spaces.GPU(duration=_duration)
87
+ @torch.inference_mode()
88
+ def decide(state, question, options_text, media, modality):
89
+ options = [o.strip() for o in options_text.splitlines() if o.strip()]
90
+ if not 2 <= len(options) <= 256: raise gr.Error("Give between 2 and 256 options, one per line.")
91
+ if modality != "text" and not media: raise gr.Error(f"Upload a file for {modality} input.")
92
+ content, tmp = [], None
93
+ if modality == "image": content.append({"type": "image", "image": Image.open(media).convert("RGB")})
94
+ elif modality == "video": content.extend({"type": "image", "image": f} for f in _video_frames(media))
95
+ elif modality == "audio":
96
+ tmp = tempfile.NamedTemporaryFile(suffix=".wav", delete=False); tmp.close()
97
+ subprocess.run(["ffmpeg", "-v", "error", "-y", "-i", str(media), "-t", "30", "-ac", "1", "-ar", "16000", tmp.name], check=True)
98
+ content.append({"type": "audio", "audio": tmp.name})
99
+ content.append({"type": "text", "text": _prompt(state, question, options)})
100
+ try:
101
+ inputs = PROC.apply_chat_template([{"role": "user", "content": content}], add_generation_prompt=True,
102
+ tokenize=True, return_dict=True, return_tensors="pt", enable_thinking=False)
103
+ finally:
104
+ if tmp: os.unlink(tmp.name)
105
+ inputs = {k: (v.to(DEVICE, dtype=torch.bfloat16) if torch.is_floating_point(v) else v.to(DEVICE)) for k, v in inputs.items()}
106
+ CAP.clear(); t = time.time()
107
+ with torch.autocast("cuda", dtype=torch.bfloat16):
108
+ MODEL(**inputs, use_cache=False, logits_to_keep=1)
109
+ probs = HEAD(CAP["h"], torch.tensor([len(options)], device=DEVICE))[0, :len(options)].softmax(-1)
110
+ ms = (time.time() - t) * 1000
111
+ p = probs.float().cpu().tolist()
112
+ best = int(np.argmax(p))
113
+ return {o: v for o, v in zip(options, p)}, f"**{options[best]}** · {p[best]*100:.1f}% · {ms:.0f} ms"
114
+
115
+
116
+ EXAMPLES = [
117
+ ["The meeting starts at 10 AM. It is now 9 AM.", "Has the meeting started?", "Yes\nNo", None, "text"],
118
+ ["Customer: I was charged twice.\nAgent: I've refunded $29 to your card.\nCustomer: Got it. All sorted, thanks!",
119
+ "Was the issue actually resolved?", "Yes\nNo", None, "text"],
120
+ ]
121
+
122
+ with gr.Blocks(title="Jev-Omni") as demo:
123
+ gr.Markdown("# Jev-Omni\nA decision classifier for text, images, audio and video. "
124
+ "Give it a situation and a question with options; it returns a probability for each option — "
125
+ "no generated text, one forward pass.")
126
+ with gr.Row():
127
+ with gr.Column():
128
+ modality = gr.Radio(["text", "image", "audio", "video"], value="text", label="Input type")
129
+ media = gr.File(label="Image / audio / video file (not needed for text)", file_types=["image", "audio", "video"])
130
+ state = gr.Textbox(label="Situation", lines=6, placeholder="What the model should know. For media, this can be short.")
131
+ question = gr.Textbox(label="Question")
132
+ options = gr.Textbox(label="Options, one per line", lines=4, value="Yes\nNo")
133
+ go = gr.Button("Decide", variant="primary")
134
+ with gr.Column():
135
+ verdict = gr.Markdown()
136
+ probs = gr.Label(label="Probabilities", num_top_classes=10)
137
+ go.click(decide, [state, question, options, media, modality], [probs, verdict])
138
+ gr.Examples(EXAMPLES, [state, question, options, media, modality])
139
+ gr.Markdown("Best supported at ≤20 options; audio is capped at 30 s; video is sampled to 16 frames. "
140
+ "Weights: [akhilaaa3/Jev-Omni](https://huggingface.co/akhilaaa3/Jev-Omni), Apache-2.0.")
141
+
142
+ demo.queue().launch()
packages.txt ADDED
@@ -0,0 +1 @@
 
 
1
+ ffmpeg
requirements.txt ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ torch>=2.10
2
+ torchvision
3
+ transformers==5.17.0
4
+ accelerate
5
+ huggingface_hub
6
+ safetensors
7
+ numpy
8
+ Pillow
9
+ opencv-python-headless
10
+ soundfile
11
+ librosa