multimodalart HF Staff commited on
Commit
43ec77e
·
verified ·
1 Parent(s): fa66274

Upload folder using huggingface_hub

Browse files
Files changed (3) hide show
  1. README.md +10 -7
  2. app.py +295 -0
  3. requirements.txt +4 -0
README.md CHANGED
@@ -1,13 +1,16 @@
1
  ---
2
- title: K2 Type 0 9b
3
- emoji: 🚀
4
- colorFrom: blue
5
- colorTo: pink
6
  sdk: gradio
7
  sdk_version: 6.29.1
8
- python_version: '3.12'
9
  app_file: app.py
10
- pinned: false
 
 
 
 
11
  ---
12
 
13
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
1
  ---
2
+ title: K2-Type-0.9B
3
+ emoji: ⚖️
4
+ colorFrom: gray
5
+ colorTo: green
6
  sdk: gradio
7
  sdk_version: 6.29.1
 
8
  app_file: app.py
9
+ short_description: Typed decisions (yes/no, choice, score) in one pass
10
+ python_version: "3.12"
11
+ startup_duration_timeout: 30m
12
+ models:
13
+ - IFM/K2-Type-0.9B
14
  ---
15
 
16
+ Demo of [IFM/K2-Type-0.9B](https://huggingface.co/IFM/K2-Type-0.9B), a 0.9B decision model: one state + typed questions → calibrated probabilities from a single forward pass, using the model repo's own `jev/` inference code.
app.py ADDED
@@ -0,0 +1,295 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import spaces # must come before torch
2
+
3
+ import json
4
+ import sys
5
+ import time
6
+
7
+ import gradio as gr
8
+ import torch
9
+ from huggingface_hub import snapshot_download
10
+ from safetensors.torch import load_file
11
+
12
+ MODEL_ID = "IFM/K2-Type-0.9B"
13
+ MODEL_DIR = snapshot_download(MODEL_ID)
14
+ sys.path.insert(0, MODEL_DIR) # the repo ships its own inference package `jev/`
15
+
16
+ from transformers import AutoTokenizer # noqa: E402
17
+
18
+ from jev.encode import Encoder, collate # noqa: E402
19
+ from jev.model import DecisionModel # noqa: E402
20
+ from jev.serve import answer, to_record # noqa: E402
21
+
22
+ CFG = json.load(open(f"{MODEL_DIR}/decision_config.json"))
23
+ MAX_LEN = 8192
24
+
25
+ tokenizer = AutoTokenizer.from_pretrained(MODEL_DIR, trust_remote_code=True)
26
+ model = DecisionModel(MODEL_DIR, head_dim=CFG.get("head_dim", 256))
27
+ model.head.load_state_dict(load_file(f"{MODEL_DIR}/pointer_head.safetensors"))
28
+ model.temperature.fill_(CFG["temperature"])
29
+ model = model.eval().to("cuda")
30
+
31
+ encoder = Encoder(tokenizer, MAX_LEN, MAX_LEN - 1024)
32
+ PAD_ID = tokenizer.pad_token_id if tokenizer.pad_token_id is not None else tokenizer.eos_token_id
33
+
34
+ TYPE_LABELS = {"Yes / No": "noul", "Choice": "choice", "Score (ordered)": "score"}
35
+ TYPE_NAMES = {v: k for k, v in TYPE_LABELS.items()}
36
+
37
+
38
+ def _run_request(req: dict) -> dict:
39
+ """Run one /v1/systemone-style request through the model (single forward pass)."""
40
+ t0 = time.perf_counter()
41
+ try:
42
+ rec = to_record(req)
43
+ except Exception as e: # jev raises fastapi HTTPException
44
+ raise gr.Error(getattr(e, "detail", str(e)))
45
+ e = encoder.encode(rec)
46
+ if e is None or len(e["decide"]) != len(rec["questions"]):
47
+ raise gr.Error(f"Request does not fit in {MAX_LEN} tokens.")
48
+ with torch.no_grad(), torch.autocast("cuda", dtype=torch.bfloat16):
49
+ scores = model(collate([e], PAD_ID))
50
+ answers = {}
51
+ for k, s in zip(e["qkeys"], scores):
52
+ p = torch.softmax(s.float(), -1).tolist()
53
+ answers[k] = answer(rec["questions"][k], p)
54
+ torch.cuda.synchronize()
55
+ return {"answers": answers, "model": CFG["name"], "input_tokens": len(e["ids"]),
56
+ "latency_ms": round((time.perf_counter() - t0) * 1000, 1)}
57
+
58
+
59
+ def _parse_state(state: str):
60
+ s = (state or "").strip()
61
+ if not s:
62
+ raise gr.Error("Please enter a state (text or JSON).")
63
+ if s[:1] in "{[":
64
+ try:
65
+ return json.loads(s)
66
+ except json.JSONDecodeError:
67
+ pass
68
+ return s
69
+
70
+
71
+ def _kv_lines(text: str) -> dict:
72
+ out = {}
73
+ for line in (text or "").splitlines():
74
+ line = line.strip()
75
+ if not line:
76
+ continue
77
+ if ":" in line:
78
+ k, v = line.split(":", 1)
79
+ out[k.strip()] = v.strip() or None
80
+ else:
81
+ out[line] = None
82
+ return out
83
+
84
+
85
+ def _build_question(qtype: str, instructions: str, options: str):
86
+ instructions = (instructions or "").strip()
87
+ if not instructions:
88
+ return None
89
+ t = TYPE_LABELS.get(qtype, qtype)
90
+ if t == "choice":
91
+ crit = _kv_lines(options)
92
+ if not crit:
93
+ raise gr.Error(f"Choice question “{instructions}” needs at least one option (one per line).")
94
+ return {"type": "choice", "instructions": instructions, "criteria": crit}
95
+ if t == "score":
96
+ levels = [l.strip() for l in (options or "").splitlines() if l.strip()]
97
+ if len(levels) < 2:
98
+ raise gr.Error(f"Score question “{instructions}” needs at least two levels (one per line, low → high).")
99
+ return {"type": "score", "instructions": instructions, "criteria": levels}
100
+ crit = {k.lower(): v for k, v in _kv_lines(options).items() if k.lower() in ("true", "false") and v}
101
+ q = {"type": "noul", "instructions": instructions}
102
+ if crit:
103
+ q["criteria"] = crit
104
+ return q
105
+
106
+
107
+ def _label_for(ans: dict) -> dict:
108
+ if ans["type"] == "noul":
109
+ return {"true": ans["noul"], "false": round(1 - ans["noul"], 4)}
110
+ if ans["type"] == "choice":
111
+ return ans["probabilities"]
112
+ return {f"{i}: {ans['legend'][i]}": p for i, p in ans["probabilities"].items()}
113
+
114
+
115
+ def _summary(qid: str, q: dict, ans: dict) -> str:
116
+ if ans["type"] == "noul":
117
+ p = ans["noul"]
118
+ return f"**{qid}** — {q['instructions']} → **{'YES' if p >= 0.5 else 'NO'}** (P(true) = {p:.3f})"
119
+ if ans["type"] == "choice":
120
+ return (f"**{qid}** — {q['instructions']} → **{ans['choice']}** "
121
+ f"(p = {ans['probabilities'][ans['choice']]:.3f}, confidence {ans['confidence']:.2f})")
122
+ lvl = ans["legend"][str(round(ans["score"]))]
123
+ return (f"**{qid}** — {q['instructions']} → expected level **{ans['score']:.2f}** "
124
+ f"(≈ {lvl}; confidence {ans['confidence']:.2f})")
125
+
126
+
127
+ @spaces.GPU(duration=15)
128
+ def decide(
129
+ state: str,
130
+ q1_type: str = "Choice",
131
+ q1_instructions: str = "",
132
+ q1_options: str = "",
133
+ q2_type: str = "Yes / No",
134
+ q2_instructions: str = "",
135
+ q2_options: str = "",
136
+ q3_type: str = "Score (ordered)",
137
+ q3_instructions: str = "",
138
+ q3_options: str = "",
139
+ ):
140
+ """Answer up to three typed questions about a state with K2-Type-0.9B in one forward pass.
141
+
142
+ Args:
143
+ state: The situation to decide about, as plain text or a JSON object.
144
+ q1_type: "Yes / No", "Choice" or "Score (ordered)".
145
+ q1_instructions: The question / statement. Leave empty to skip this question.
146
+ q1_options: Choice: one option per line ("name: description"). Score: one level per line, low to high.
147
+ Yes / No: optional "true: ..." and "false: ..." definitions.
148
+ q2_type: Type of question 2.
149
+ q2_instructions: Question 2 text (empty to skip).
150
+ q2_options: Question 2 options.
151
+ q3_type: Type of question 3.
152
+ q3_instructions: Question 3 text (empty to skip).
153
+ q3_options: Question 3 options.
154
+
155
+ Returns:
156
+ A markdown summary, one probability label per question, and the raw /v1/systemone response.
157
+ """
158
+ slots = [(q1_type, q1_instructions, q1_options), (q2_type, q2_instructions, q2_options),
159
+ (q3_type, q3_instructions, q3_options)]
160
+ questions, keys = {}, []
161
+ for i, slot in enumerate(slots, 1):
162
+ q = _build_question(*slot)
163
+ keys.append(f"q{i}" if q else None)
164
+ if q:
165
+ questions[f"q{i}"] = q
166
+ if not questions:
167
+ raise gr.Error("Fill in at least one question.")
168
+ req = {"state": _parse_state(state), "questions": questions}
169
+ res = _run_request(req)
170
+ lines = [_summary(k, questions[k], res["answers"][k]) for k in questions]
171
+ lines.append(f"\n<sub>{res['input_tokens']} input tokens · {res['latency_ms']} ms on GPU</sub>")
172
+ labels = [gr.update(value=_label_for(res["answers"][k]), visible=True) if k else gr.update(value=None, visible=False)
173
+ for k in keys]
174
+ return "\n\n".join(lines), *labels, {"request": req, "response": res}
175
+
176
+
177
+ @spaces.GPU(duration=15)
178
+ def systemone(request_json: str) -> dict:
179
+ """Raw TypeSafe /v1/systemone call: {"state": ..., "questions": {id: {"type", "instructions", "criteria"}}}.
180
+
181
+ Args:
182
+ request_json: The request body as a JSON string.
183
+
184
+ Returns:
185
+ The /v1/systemone response with per-question answers and probabilities.
186
+ """
187
+ try:
188
+ req = json.loads(request_json)
189
+ except json.JSONDecodeError as e:
190
+ raise gr.Error(f"Invalid JSON: {e}")
191
+ if not isinstance(req, dict) or "state" not in req or not req.get("questions"):
192
+ raise gr.Error('Request must be an object with "state" and "questions".')
193
+ return _run_request(req)
194
+
195
+
196
+ TICKET = json.dumps({"subject": "Charged twice",
197
+ "body": "You billed my card twice for March. Refund one or I cancel."}, indent=2)
198
+
199
+ EXAMPLES = [
200
+ [TICKET,
201
+ "Choice", "Which queue handles this?",
202
+ "billing: Payments and refunds\ntechnical: Bugs and login\ngeneral: Anything else",
203
+ "Yes / No", "The customer sounds angry.", "",
204
+ "Score (ordered)", "How urgent is it?", "Low\nNormal\nHigh\nCritical"],
205
+ ["The app crashes every time I try to log in with Google on my Android phone since yesterday's update. "
206
+ "I have a client demo in two hours.",
207
+ "Choice", "Which queue handles this?",
208
+ "billing: Payments and refunds\ntechnical: Bugs and login\ngeneral: Anything else",
209
+ "Yes / No", "The user mentions a time constraint.", "",
210
+ "Score (ordered)", "How urgent is it?", "Low\nNormal\nHigh\nCritical"],
211
+ ["Review: The hotel room was spotless and the staff were lovely, but the walls were paper-thin "
212
+ "and we barely slept because of the party next door.",
213
+ "Choice", "What is the overall sentiment of the review?",
214
+ "positive\nnegative\nmixed",
215
+ "Yes / No", "The reviewer would recommend this hotel to a light sleeper.",
216
+ "true: they would recommend it\nfalse: they would not recommend it",
217
+ "Score (ordered)", "Star rating the reviewer most likely gave.", "1 star\n2 stars\n3 stars\n4 stars\n5 stars"],
218
+ ["Premise: A man is playing a guitar on a crowded street corner while people drop coins in his case.\n"
219
+ "Hypothesis: A musician is performing in public.",
220
+ "Choice", "Does the premise entail the hypothesis?",
221
+ "entailment\nneutral\ncontradiction",
222
+ "Yes / No", "The man is being paid for his music.", "",
223
+ "Score (ordered)", "How confident can we be that the man is a professional musician?",
224
+ "Not at all\nSlightly\nModerately\nVery"],
225
+ ]
226
+
227
+ RAW_EXAMPLE = json.dumps({
228
+ "state": {"subject": "Charged twice", "body": "You billed my card twice for March. Refund one or I cancel."},
229
+ "questions": {
230
+ "queue": {"type": "choice", "instructions": "Which queue handles this?",
231
+ "criteria": {"billing": "Payments and refunds", "technical": "Bugs and login",
232
+ "general": "Anything else"}},
233
+ "angry": {"type": "noul", "instructions": "The customer sounds angry."},
234
+ "urgency": {"type": "score", "instructions": "How urgent is it?",
235
+ "criteria": ["Low", "Normal", "High", "Critical"]},
236
+ }}, indent=2)
237
+
238
+ CSS = """
239
+ #col-container { max-width: 1150px; margin: 0 auto; }
240
+ .dark .gradio-container { color: var(--body-text-color); }
241
+ """
242
+
243
+ with gr.Blocks(title="K2-Type-0.9B") as demo:
244
+ with gr.Column(elem_id="col-container"):
245
+ gr.Markdown(
246
+ "# ⚖️ K2-Type-0.9B — typed decision model\n"
247
+ "Give a **state** (text or JSON) and up to three typed questions — **yes/no**, **choice**, or "
248
+ "**ordered score**. The model returns a calibrated probability for every option of every question from "
249
+ "**one forward pass**; questions can't see each other, so adding one never changes another's answer. "
250
+ "It never generates text.\n\n"
251
+ "[Model card](https://huggingface.co/IFM/K2-Type-0.9B) · "
252
+ "base: [IFM/K2-Horizon-0.9B](https://huggingface.co/IFM/K2-Horizon-0.9B)"
253
+ )
254
+ with gr.Tab("Question builder"):
255
+ with gr.Row():
256
+ with gr.Column(scale=5):
257
+ state = gr.Textbox(label="State (text or JSON)", lines=7, value=TICKET)
258
+ qboxes = []
259
+ defaults = EXAMPLES[0][1:]
260
+ for i in range(3):
261
+ with gr.Group():
262
+ with gr.Row():
263
+ qt = gr.Dropdown(list(TYPE_LABELS), value=defaults[3 * i], label=f"Question {i + 1} type",
264
+ scale=1)
265
+ qi = gr.Textbox(label=f"Question {i + 1} (leave empty to skip)",
266
+ value=defaults[3 * i + 1], scale=3)
267
+ qo = gr.Textbox(
268
+ label="Options — Choice: one per line, `name: description` · Score: levels low→high · "
269
+ "Yes/No: optional `true: …` / `false: …`",
270
+ value=defaults[3 * i + 2], lines=3)
271
+ qboxes += [qt, qi, qo]
272
+ run = gr.Button("Decide", variant="primary")
273
+ with gr.Column(scale=4):
274
+ summary = gr.Markdown()
275
+ labels = [gr.Label(label=f"Question {i + 1}", num_top_classes=10) for i in range(3)]
276
+ with gr.Accordion("Raw request / response", open=False):
277
+ raw = gr.JSON()
278
+ run.click(decide, inputs=[state, *qboxes], outputs=[summary, *labels, raw], api_name="decide")
279
+ gr.Examples(
280
+ examples=EXAMPLES,
281
+ inputs=[state, *qboxes],
282
+ outputs=[summary, *labels, raw],
283
+ fn=decide,
284
+ cache_examples=False,
285
+ run_on_click=True,
286
+ )
287
+ with gr.Tab("Raw /v1/systemone"):
288
+ gr.Markdown("Send any number of questions in TypeSafe's `/v1/systemone` wire format.")
289
+ with gr.Row():
290
+ req_box = gr.Code(value=RAW_EXAMPLE, language="json", label="Request", lines=22)
291
+ res_box = gr.JSON(label="Response")
292
+ raw_btn = gr.Button("Send", variant="primary")
293
+ raw_btn.click(systemone, inputs=req_box, outputs=res_box, api_name="systemone")
294
+
295
+ demo.launch(theme=gr.themes.Citrus(), css=CSS, mcp_server=True)
requirements.txt ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ torch
2
+ transformers>=5.17,<6
3
+ safetensors>=0.5
4
+ fastapi