multimodalart HF Staff commited on
Commit
a4fa0be
·
verified ·
1 Parent(s): 92951cb

Upload app.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. app.py +41 -4
app.py CHANGED
@@ -286,12 +286,49 @@ def _decide_impl(screenshot, goal, ax_tree, app, task_family, options_text, moda
286
  )
287
  progress(0.9, desc="Softmax over option letters...")
288
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
289
  best = max(range(len(options)), key=lambda i: probs[i])
290
  lines = [
291
- f"Decision: {LETTERS[best]} — {options[best]['role']} \"{options[best]['label']}\" "
292
- f"-> {options[best]['action']}"
293
- + (f" (with entity '{options[best]['entity_id']}')" if options[best].get("entity_id") else ""),
294
- f"Confidence: {probs[best]:.1%}",
 
 
 
 
 
295
  "",
296
  "Ranked option probabilities (softmax over the option letters at the final position):",
297
  _format_options_text(options, probs),
 
286
  )
287
  progress(0.9, desc="Softmax over option letters...")
288
 
289
+ # The benchmark's decision rule (eval.scoring / train_4b_v2.group_log_probs)
290
+ # is PER ELEMENT: among each element's own candidate actions, take the
291
+ # argmax after re-normalizing over just that element's options. The global
292
+ # argmax over all letters is a different, stricter question, so both are
293
+ # reported.
294
+ by_element: dict[str, list[int]] = {}
295
+ for i, o in enumerate(options):
296
+ by_element.setdefault(o["label"], []).append(i)
297
+
298
+ element_lines = []
299
+ actionable = []
300
+ for label, idxs in by_element.items():
301
+ if len(idxs) > 1:
302
+ total = sum(probs[i] for i in idxs)
303
+ renorm = [(probs[i] / total if total > 0 else 0.0, i) for i in idxs]
304
+ renorm.sort(key=lambda t: -t[0])
305
+ p_top, i_top = renorm[0]
306
+ o = options[i_top]
307
+ element_lines.append(
308
+ f'• {o["role"]} "{label}" → {o["action"]}'
309
+ + (f" (with entity '{o['entity_id']}')" if o.get("entity_id") else "")
310
+ + f" [{p_top:.1%} of this element's mass]"
311
+ )
312
+ if o["action"] != "skip":
313
+ actionable.append(element_lines[-1])
314
+ else:
315
+ i = idxs[0]
316
+ o = options[i]
317
+ element_lines.append(
318
+ f'• {o["role"]} "{label}" → {o["action"]} (only candidate)'
319
+ )
320
+
321
  best = max(range(len(options)), key=lambda i: probs[i])
322
  lines = [
323
+ "Per-element decisions (renormalized within each element's options — the "
324
+ "benchmark's decision rule):",
325
+ *element_lines,
326
+ "",
327
+ "Single best option overall (global argmax over the letter softmax):",
328
+ f'{LETTERS[best]}. {options[best]["role"]} "{options[best]["label"]}" '
329
+ f'-> {options[best]["action"]}'
330
+ + (f" (with entity '{options[best]['entity_id']}')" if options[best].get("entity_id") else "")
331
+ + f" — p={probs[best]:.3f}",
332
  "",
333
  "Ranked option probabilities (softmax over the option letters at the final position):",
334
  _format_options_text(options, probs),