Spaces:
Running on Zero
Running on Zero
Oscar-1 demo: gradio app over the published 17m/32m deciders
Browse files- README.md +20 -8
- app.py +142 -0
- requirements.txt +3 -0
README.md
CHANGED
|
@@ -1,13 +1,25 @@
|
|
| 1 |
---
|
| 2 |
-
title: Oscar
|
| 3 |
-
emoji:
|
| 4 |
-
colorFrom:
|
| 5 |
-
colorTo:
|
| 6 |
sdk: gradio
|
| 7 |
-
sdk_version: 6.28.0
|
| 8 |
-
python_version: '3.13'
|
| 9 |
app_file: app.py
|
| 10 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 11 |
---
|
| 12 |
|
| 13 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
+
title: Oscar-1 Decision Demo
|
| 3 |
+
emoji: 🎯
|
| 4 |
+
colorFrom: indigo
|
| 5 |
+
colorTo: purple
|
| 6 |
sdk: gradio
|
|
|
|
|
|
|
| 7 |
app_file: app.py
|
| 8 |
+
license: apache-2.0
|
| 9 |
+
short_description: RLCD-calibrated deciders 00b7 choice 00b7 score 00b7 noul
|
| 10 |
+
tags:
|
| 11 |
+
- laya
|
| 12 |
+
- rlcd
|
| 13 |
+
- calibrated-classification
|
| 14 |
+
- ettin
|
| 15 |
---
|
| 16 |
|
| 17 |
+
Oscar-1 demo — loads [mgoeckel/oscar-1-17m](https://huggingface.co/mgoeckel/oscar-1-17m) and
|
| 18 |
+
[mgoeckel/oscar-1-32m](https://huggingface.co/mgoeckel/oscar-1-32m) (RLCD-calibrated
|
| 19 |
+
Laya-compatible decision encoders on Ettin backbones) and serves the full calibrated
|
| 20 |
+
decision surface: **choice** label probabilities, **score** expected level 0–4 with the full
|
| 21 |
+
level distribution, and **noul** calibrated binary confidence — all in one forward pass
|
| 22 |
+
through the stock `laya.Agent` seam.
|
| 23 |
+
|
| 24 |
+
Benchmark (typed-decisions, 400 cases): 32M **0.701** acc / Brier 0.087; 17M 0.678 / 0.104 —
|
| 25 |
+
at 2.5 ms vs hosted jev's ~710 ms at 0.727.
|
app.py
ADDED
|
@@ -0,0 +1,142 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
import warnings
|
| 3 |
+
|
| 4 |
+
warnings.filterwarnings("ignore")
|
| 5 |
+
|
| 6 |
+
import gradio as gr # noqa: E402
|
| 7 |
+
from huggingface_hub import snapshot_download # noqa: E402
|
| 8 |
+
from laya import Agent # noqa: E402
|
| 9 |
+
|
| 10 |
+
REPOS = {
|
| 11 |
+
"Oscar-1 17M": "mgoeckel/oscar-1-17m",
|
| 12 |
+
"Oscar-1 32M": "mgoeckel/oscar-1-32m",
|
| 13 |
+
}
|
| 14 |
+
DEFAULT_OPTIONS = "refund, cancel, information, other"
|
| 15 |
+
URGENCY_LEVELS = ["routine", "low", "moderate", "high", "critical"]
|
| 16 |
+
|
| 17 |
+
_agents = {}
|
| 18 |
+
_cache_root = os.environ.get("OSCAR_CACHE", "oscar_models")
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
def get_agent(model_key: str) -> Agent:
|
| 22 |
+
if model_key not in _agents:
|
| 23 |
+
local = os.path.join(_cache_root, model_key.split()[-1].lower())
|
| 24 |
+
snapshot_download(REPOS[model_key], local_dir=local)
|
| 25 |
+
_agents[model_key] = Agent(local)
|
| 26 |
+
return _agents[model_key]
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
def decide(model_key, text, options_raw, urgency_instructions, sensitive_instructions):
|
| 30 |
+
text = (text or "").strip()
|
| 31 |
+
if not text:
|
| 32 |
+
raise gr.Error("Enter a message to classify.")
|
| 33 |
+
|
| 34 |
+
questions = {}
|
| 35 |
+
options = [o.strip() for o in (options_raw or "").split(",") if o.strip()]
|
| 36 |
+
if len(options) >= 2:
|
| 37 |
+
questions["intent"] = {
|
| 38 |
+
"type": "choice",
|
| 39 |
+
"instructions": "What does the customer want?",
|
| 40 |
+
"criteria": options,
|
| 41 |
+
}
|
| 42 |
+
if (urgency_instructions or "").strip():
|
| 43 |
+
questions["urgency"] = {
|
| 44 |
+
"type": "score",
|
| 45 |
+
"instructions": urgency_instructions.strip(),
|
| 46 |
+
"criteria": list(URGENCY_LEVELS),
|
| 47 |
+
}
|
| 48 |
+
if (sensitive_instructions or "").strip():
|
| 49 |
+
questions["sensitive"] = {
|
| 50 |
+
"type": "noul",
|
| 51 |
+
"instructions": sensitive_instructions.strip(),
|
| 52 |
+
"criteria": {"true": "sensitive", "false": "not sensitive"},
|
| 53 |
+
}
|
| 54 |
+
if not questions:
|
| 55 |
+
raise gr.Error("Enable at least one question (add intent options or fill a prompt).")
|
| 56 |
+
|
| 57 |
+
answers = get_agent(model_key).predict({"text": text}, questions)["answers"]
|
| 58 |
+
|
| 59 |
+
intent_label = urgency_label = noul_label = None
|
| 60 |
+
lines = [f"**Model:** {model_key}", ""]
|
| 61 |
+
if "intent" in answers:
|
| 62 |
+
a = answers["intent"]
|
| 63 |
+
probs = {k: float(v) for k, v in a["probabilities"].items()}
|
| 64 |
+
intent_label = probs
|
| 65 |
+
lines += [
|
| 66 |
+
f"- **Intent:** `{a['choice']}` at **{a['answer_confidence']:.1%}** confidence",
|
| 67 |
+
]
|
| 68 |
+
if "urgency" in answers:
|
| 69 |
+
a = answers["urgency"]
|
| 70 |
+
levels = {f"{int(k)} · {URGENCY_LEVELS[int(k)]}": float(v) for k, v in a["probabilities"].items()}
|
| 71 |
+
urgency_label = levels
|
| 72 |
+
lines += [
|
| 73 |
+
f"- **Urgency score:** **{a['score']:.2f}** (expected level, 0–4) — top level "
|
| 74 |
+
f"`{max(levels, key=levels.get)}` at {a['answer_confidence']:.1%}",
|
| 75 |
+
]
|
| 76 |
+
if "sensitive" in answers:
|
| 77 |
+
a = answers["sensitive"]
|
| 78 |
+
p_true = float(a["noul"])
|
| 79 |
+
conf = float(a["confidence"])
|
| 80 |
+
verdict = "SENSITIVE" if p_true > 0.5 else "not sensitive"
|
| 81 |
+
noul_label = {"true (sensitive)": p_true, "false (not sensitive)": 1.0 - p_true}
|
| 82 |
+
lines += [f"- **Sensitive:** {verdict} — **{conf:.1%}** calibrated confidence"]
|
| 83 |
+
lines += ["", f"*(one forward pass, CPU — p50 latency on RTX 5060 Ti was 2.5–2.6 ms)*"]
|
| 84 |
+
return intent_label, urgency_label, noul_label, "\n".join(lines)
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
INTRO = """# Oscar-1 — RLCD-calibrated decision models
|
| 88 |
+
|
| 89 |
+
**Oscar-1** is a series of [Laya](https://huggingface.co/convaiinnovations/laya-typed)-compatible,
|
| 90 |
+
calibrated **decision encoders** trained with RLCD (proper scoring rules: log + spherical + RPS,
|
| 91 |
+
GRPO-style noisy-logit REINFORCE) on [JHU-CLSP Ettin](https://huggingface.co/jhu-clsp) backbones.
|
| 92 |
+
One forward pass answers **choice** (label probabilities), **score** (expected level + full
|
| 93 |
+
distribution) and **noul** (calibrated binary confidence) — post-hoc temperature-calibrated.
|
| 94 |
+
|
| 95 |
+
| typed-decisions test | acc | Brier | latency p50 | vs published |
|
| 96 |
+
|---|---:|---:|---:|---|
|
| 97 |
+
| **Oscar-1 32M** | **0.701** | 0.087 | **2.5 ms** | laya-typed 0.766 · jev 0.727 @ ~710 ms |
|
| 98 |
+
| **Oscar-1 17M** | 0.678 | 0.104 | **2.6 ms** | ~1/250th of jev's latency |
|
| 99 |
+
|
| 100 |
+
Weights: [oscar-1-32m](https://huggingface.co/mgoeckel/oscar-1-32m) ·
|
| 101 |
+
[oscar-1-17m](https://huggingface.co/mgoeckel/oscar-1-17m) (Apache-2.0; base encoders MIT).
|
| 102 |
+
Runs the stock `laya.Agent` seam — try editing the intent options and prompts below.
|
| 103 |
+
"""
|
| 104 |
+
|
| 105 |
+
with gr.Blocks(title="Oscar-1 decision demo") as demo:
|
| 106 |
+
gr.Markdown(INTRO)
|
| 107 |
+
with gr.Row():
|
| 108 |
+
with gr.Column(scale=1):
|
| 109 |
+
model = gr.Radio(list(REPOS), value="Oscar-1 32M", label="Model")
|
| 110 |
+
text = gr.Textbox(label="Message",
|
| 111 |
+
placeholder="e.g. I was charged twice, please return the money.")
|
| 112 |
+
options = gr.Textbox(label="Intent options (comma-separated; blank to disable intent question)",
|
| 113 |
+
value=DEFAULT_OPTIONS)
|
| 114 |
+
urg = gr.Textbox(label="Score prompt (blank to disable score question)",
|
| 115 |
+
value="Rate the urgency of this message on a 0 to 4 scale.")
|
| 116 |
+
sens = gr.Textbox(label="Binary prompt (blank to disable noul question)",
|
| 117 |
+
value="Is this request fraud or a security issue?")
|
| 118 |
+
btn = gr.Button("Decide", variant="primary")
|
| 119 |
+
with gr.Column(scale=1):
|
| 120 |
+
out_intent = gr.Label(label="Intent (choice)", num_top_classes=5)
|
| 121 |
+
out_urgency = gr.Label(label="Urgency level (score 0–4)", num_top_classes=5)
|
| 122 |
+
out_noul = gr.Label(label="Sensitivity (noul)")
|
| 123 |
+
out_summary = gr.Markdown()
|
| 124 |
+
|
| 125 |
+
btn.click(decide, [model, text, options, urg, sens],
|
| 126 |
+
[out_intent, out_urgency, out_noul, out_summary])
|
| 127 |
+
for w in (text, options, urg, sens, model):
|
| 128 |
+
w.change(decide, [model, text, options, urg, sens],
|
| 129 |
+
[out_intent, out_urgency, out_noul, out_summary], show_progress="hidden")
|
| 130 |
+
|
| 131 |
+
gr.Examples(
|
| 132 |
+
examples=[
|
| 133 |
+
["Oscar-1 32M", "I was charged twice, please return the money.", DEFAULT_OPTIONS, urg.value, sens.value],
|
| 134 |
+
["Oscar-1 32M", "The API has been down since Tuesday and it is blocking our launch.", DEFAULT_OPTIONS, urg.value, sens.value],
|
| 135 |
+
["Oscar-1 32M", "Someone tried to take over my account and changed the email.", DEFAULT_OPTIONS, urg.value, sens.value],
|
| 136 |
+
["Oscar-1 17M", "How do I reset my password?", DEFAULT_OPTIONS, urg.value, sens.value],
|
| 137 |
+
],
|
| 138 |
+
inputs=[model, text, options, urg, sens],
|
| 139 |
+
)
|
| 140 |
+
|
| 141 |
+
if __name__ == "__main__":
|
| 142 |
+
demo.launch()
|
requirements.txt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
laya==0.3.20
|
| 2 |
+
transformers>=4.44,<5
|
| 3 |
+
torch
|