mgoeckel commited on
Commit
3a2c7b8
·
verified ·
1 Parent(s): a229bdf

Oscar-1 demo: gradio app over the published 17m/32m deciders

Browse files
Files changed (3) hide show
  1. README.md +20 -8
  2. app.py +142 -0
  3. requirements.txt +3 -0
README.md CHANGED
@@ -1,13 +1,25 @@
1
  ---
2
- title: Oscar 1 Demo
3
- emoji: 📉
4
- colorFrom: pink
5
- colorTo: pink
6
  sdk: gradio
7
- sdk_version: 6.28.0
8
- python_version: '3.13'
9
  app_file: app.py
10
- pinned: false
 
 
 
 
 
 
11
  ---
12
 
13
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
 
 
 
1
  ---
2
+ title: Oscar-1 Decision Demo
3
+ emoji: 🎯
4
+ colorFrom: indigo
5
+ colorTo: purple
6
  sdk: gradio
 
 
7
  app_file: app.py
8
+ license: apache-2.0
9
+ short_description: RLCD-calibrated deciders 00b7 choice 00b7 score 00b7 noul
10
+ tags:
11
+ - laya
12
+ - rlcd
13
+ - calibrated-classification
14
+ - ettin
15
  ---
16
 
17
+ Oscar-1 demo — loads [mgoeckel/oscar-1-17m](https://huggingface.co/mgoeckel/oscar-1-17m) and
18
+ [mgoeckel/oscar-1-32m](https://huggingface.co/mgoeckel/oscar-1-32m) (RLCD-calibrated
19
+ Laya-compatible decision encoders on Ettin backbones) and serves the full calibrated
20
+ decision surface: **choice** label probabilities, **score** expected level 0–4 with the full
21
+ level distribution, and **noul** calibrated binary confidence — all in one forward pass
22
+ through the stock `laya.Agent` seam.
23
+
24
+ Benchmark (typed-decisions, 400 cases): 32M **0.701** acc / Brier 0.087; 17M 0.678 / 0.104 —
25
+ at 2.5 ms vs hosted jev's ~710 ms at 0.727.
app.py ADDED
@@ -0,0 +1,142 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import warnings
3
+
4
+ warnings.filterwarnings("ignore")
5
+
6
+ import gradio as gr # noqa: E402
7
+ from huggingface_hub import snapshot_download # noqa: E402
8
+ from laya import Agent # noqa: E402
9
+
10
+ REPOS = {
11
+ "Oscar-1 17M": "mgoeckel/oscar-1-17m",
12
+ "Oscar-1 32M": "mgoeckel/oscar-1-32m",
13
+ }
14
+ DEFAULT_OPTIONS = "refund, cancel, information, other"
15
+ URGENCY_LEVELS = ["routine", "low", "moderate", "high", "critical"]
16
+
17
+ _agents = {}
18
+ _cache_root = os.environ.get("OSCAR_CACHE", "oscar_models")
19
+
20
+
21
+ def get_agent(model_key: str) -> Agent:
22
+ if model_key not in _agents:
23
+ local = os.path.join(_cache_root, model_key.split()[-1].lower())
24
+ snapshot_download(REPOS[model_key], local_dir=local)
25
+ _agents[model_key] = Agent(local)
26
+ return _agents[model_key]
27
+
28
+
29
+ def decide(model_key, text, options_raw, urgency_instructions, sensitive_instructions):
30
+ text = (text or "").strip()
31
+ if not text:
32
+ raise gr.Error("Enter a message to classify.")
33
+
34
+ questions = {}
35
+ options = [o.strip() for o in (options_raw or "").split(",") if o.strip()]
36
+ if len(options) >= 2:
37
+ questions["intent"] = {
38
+ "type": "choice",
39
+ "instructions": "What does the customer want?",
40
+ "criteria": options,
41
+ }
42
+ if (urgency_instructions or "").strip():
43
+ questions["urgency"] = {
44
+ "type": "score",
45
+ "instructions": urgency_instructions.strip(),
46
+ "criteria": list(URGENCY_LEVELS),
47
+ }
48
+ if (sensitive_instructions or "").strip():
49
+ questions["sensitive"] = {
50
+ "type": "noul",
51
+ "instructions": sensitive_instructions.strip(),
52
+ "criteria": {"true": "sensitive", "false": "not sensitive"},
53
+ }
54
+ if not questions:
55
+ raise gr.Error("Enable at least one question (add intent options or fill a prompt).")
56
+
57
+ answers = get_agent(model_key).predict({"text": text}, questions)["answers"]
58
+
59
+ intent_label = urgency_label = noul_label = None
60
+ lines = [f"**Model:** {model_key}", ""]
61
+ if "intent" in answers:
62
+ a = answers["intent"]
63
+ probs = {k: float(v) for k, v in a["probabilities"].items()}
64
+ intent_label = probs
65
+ lines += [
66
+ f"- **Intent:** `{a['choice']}` at **{a['answer_confidence']:.1%}** confidence",
67
+ ]
68
+ if "urgency" in answers:
69
+ a = answers["urgency"]
70
+ levels = {f"{int(k)} · {URGENCY_LEVELS[int(k)]}": float(v) for k, v in a["probabilities"].items()}
71
+ urgency_label = levels
72
+ lines += [
73
+ f"- **Urgency score:** **{a['score']:.2f}** (expected level, 0–4) — top level "
74
+ f"`{max(levels, key=levels.get)}` at {a['answer_confidence']:.1%}",
75
+ ]
76
+ if "sensitive" in answers:
77
+ a = answers["sensitive"]
78
+ p_true = float(a["noul"])
79
+ conf = float(a["confidence"])
80
+ verdict = "SENSITIVE" if p_true > 0.5 else "not sensitive"
81
+ noul_label = {"true (sensitive)": p_true, "false (not sensitive)": 1.0 - p_true}
82
+ lines += [f"- **Sensitive:** {verdict} — **{conf:.1%}** calibrated confidence"]
83
+ lines += ["", f"*(one forward pass, CPU — p50 latency on RTX 5060 Ti was 2.5–2.6 ms)*"]
84
+ return intent_label, urgency_label, noul_label, "\n".join(lines)
85
+
86
+
87
+ INTRO = """# Oscar-1 — RLCD-calibrated decision models
88
+
89
+ **Oscar-1** is a series of [Laya](https://huggingface.co/convaiinnovations/laya-typed)-compatible,
90
+ calibrated **decision encoders** trained with RLCD (proper scoring rules: log + spherical + RPS,
91
+ GRPO-style noisy-logit REINFORCE) on [JHU-CLSP Ettin](https://huggingface.co/jhu-clsp) backbones.
92
+ One forward pass answers **choice** (label probabilities), **score** (expected level + full
93
+ distribution) and **noul** (calibrated binary confidence) — post-hoc temperature-calibrated.
94
+
95
+ | typed-decisions test | acc | Brier | latency p50 | vs published |
96
+ |---|---:|---:|---:|---|
97
+ | **Oscar-1 32M** | **0.701** | 0.087 | **2.5 ms** | laya-typed 0.766 · jev 0.727 @ ~710 ms |
98
+ | **Oscar-1 17M** | 0.678 | 0.104 | **2.6 ms** | ~1/250th of jev's latency |
99
+
100
+ Weights: [oscar-1-32m](https://huggingface.co/mgoeckel/oscar-1-32m) ·
101
+ [oscar-1-17m](https://huggingface.co/mgoeckel/oscar-1-17m) (Apache-2.0; base encoders MIT).
102
+ Runs the stock `laya.Agent` seam — try editing the intent options and prompts below.
103
+ """
104
+
105
+ with gr.Blocks(title="Oscar-1 decision demo") as demo:
106
+ gr.Markdown(INTRO)
107
+ with gr.Row():
108
+ with gr.Column(scale=1):
109
+ model = gr.Radio(list(REPOS), value="Oscar-1 32M", label="Model")
110
+ text = gr.Textbox(label="Message",
111
+ placeholder="e.g. I was charged twice, please return the money.")
112
+ options = gr.Textbox(label="Intent options (comma-separated; blank to disable intent question)",
113
+ value=DEFAULT_OPTIONS)
114
+ urg = gr.Textbox(label="Score prompt (blank to disable score question)",
115
+ value="Rate the urgency of this message on a 0 to 4 scale.")
116
+ sens = gr.Textbox(label="Binary prompt (blank to disable noul question)",
117
+ value="Is this request fraud or a security issue?")
118
+ btn = gr.Button("Decide", variant="primary")
119
+ with gr.Column(scale=1):
120
+ out_intent = gr.Label(label="Intent (choice)", num_top_classes=5)
121
+ out_urgency = gr.Label(label="Urgency level (score 0–4)", num_top_classes=5)
122
+ out_noul = gr.Label(label="Sensitivity (noul)")
123
+ out_summary = gr.Markdown()
124
+
125
+ btn.click(decide, [model, text, options, urg, sens],
126
+ [out_intent, out_urgency, out_noul, out_summary])
127
+ for w in (text, options, urg, sens, model):
128
+ w.change(decide, [model, text, options, urg, sens],
129
+ [out_intent, out_urgency, out_noul, out_summary], show_progress="hidden")
130
+
131
+ gr.Examples(
132
+ examples=[
133
+ ["Oscar-1 32M", "I was charged twice, please return the money.", DEFAULT_OPTIONS, urg.value, sens.value],
134
+ ["Oscar-1 32M", "The API has been down since Tuesday and it is blocking our launch.", DEFAULT_OPTIONS, urg.value, sens.value],
135
+ ["Oscar-1 32M", "Someone tried to take over my account and changed the email.", DEFAULT_OPTIONS, urg.value, sens.value],
136
+ ["Oscar-1 17M", "How do I reset my password?", DEFAULT_OPTIONS, urg.value, sens.value],
137
+ ],
138
+ inputs=[model, text, options, urg, sens],
139
+ )
140
+
141
+ if __name__ == "__main__":
142
+ demo.launch()
requirements.txt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ laya==0.3.20
2
+ transformers>=4.44,<5
3
+ torch