Spaces:
Running on Zero
Running on Zero
Upload app.py with huggingface_hub
Browse files
app.py
CHANGED
|
@@ -286,12 +286,49 @@ def _decide_impl(screenshot, goal, ax_tree, app, task_family, options_text, moda
|
|
| 286 |
)
|
| 287 |
progress(0.9, desc="Softmax over option letters...")
|
| 288 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 289 |
best = max(range(len(options)), key=lambda i: probs[i])
|
| 290 |
lines = [
|
| 291 |
-
|
| 292 |
-
|
| 293 |
-
|
| 294 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 295 |
"",
|
| 296 |
"Ranked option probabilities (softmax over the option letters at the final position):",
|
| 297 |
_format_options_text(options, probs),
|
|
|
|
| 286 |
)
|
| 287 |
progress(0.9, desc="Softmax over option letters...")
|
| 288 |
|
| 289 |
+
# The benchmark's decision rule (eval.scoring / train_4b_v2.group_log_probs)
|
| 290 |
+
# is PER ELEMENT: among each element's own candidate actions, take the
|
| 291 |
+
# argmax after re-normalizing over just that element's options. The global
|
| 292 |
+
# argmax over all letters is a different, stricter question, so both are
|
| 293 |
+
# reported.
|
| 294 |
+
by_element: dict[str, list[int]] = {}
|
| 295 |
+
for i, o in enumerate(options):
|
| 296 |
+
by_element.setdefault(o["label"], []).append(i)
|
| 297 |
+
|
| 298 |
+
element_lines = []
|
| 299 |
+
actionable = []
|
| 300 |
+
for label, idxs in by_element.items():
|
| 301 |
+
if len(idxs) > 1:
|
| 302 |
+
total = sum(probs[i] for i in idxs)
|
| 303 |
+
renorm = [(probs[i] / total if total > 0 else 0.0, i) for i in idxs]
|
| 304 |
+
renorm.sort(key=lambda t: -t[0])
|
| 305 |
+
p_top, i_top = renorm[0]
|
| 306 |
+
o = options[i_top]
|
| 307 |
+
element_lines.append(
|
| 308 |
+
f'• {o["role"]} "{label}" → {o["action"]}'
|
| 309 |
+
+ (f" (with entity '{o['entity_id']}')" if o.get("entity_id") else "")
|
| 310 |
+
+ f" [{p_top:.1%} of this element's mass]"
|
| 311 |
+
)
|
| 312 |
+
if o["action"] != "skip":
|
| 313 |
+
actionable.append(element_lines[-1])
|
| 314 |
+
else:
|
| 315 |
+
i = idxs[0]
|
| 316 |
+
o = options[i]
|
| 317 |
+
element_lines.append(
|
| 318 |
+
f'• {o["role"]} "{label}" → {o["action"]} (only candidate)'
|
| 319 |
+
)
|
| 320 |
+
|
| 321 |
best = max(range(len(options)), key=lambda i: probs[i])
|
| 322 |
lines = [
|
| 323 |
+
"Per-element decisions (renormalized within each element's options — the "
|
| 324 |
+
"benchmark's decision rule):",
|
| 325 |
+
*element_lines,
|
| 326 |
+
"",
|
| 327 |
+
"Single best option overall (global argmax over the letter softmax):",
|
| 328 |
+
f'{LETTERS[best]}. {options[best]["role"]} "{options[best]["label"]}" '
|
| 329 |
+
f'-> {options[best]["action"]}'
|
| 330 |
+
+ (f" (with entity '{options[best]['entity_id']}')" if options[best].get("entity_id") else "")
|
| 331 |
+
+ f" — p={probs[best]:.3f}",
|
| 332 |
"",
|
| 333 |
"Ranked option probabilities (softmax over the option letters at the final position):",
|
| 334 |
_format_options_text(options, probs),
|