"""Held-out eval: operation accuracy and target top-1 (given the gold operation) on unseen pages. python finetune/eval.py out/pages.jsonl out/eval_cases.jsonl [subfolder] """ import json, os, sys, time sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) sys.path.insert(0, "/home/ckl/projects/S/laya-upstream") # laya with Agent.accelerate() from common_ft import build_request, goal_done_question import laya def main(): pages = [json.loads(l) for l in open(sys.argv[1])]; cases = [json.loads(l) for l in open(sys.argv[2])] agent = laya.load(sys.argv[3], subfolder=sys.argv[4] if len(sys.argv) > 4 else None) agent.cfg["max_len"], agent.cfg["head_max_len"] = int(os.environ.get("LAYA_MAXLEN", "1024")), int(os.environ.get("LAYA_HEAD", agent.cfg.get("head_max_len_train", 512))) agent.accelerate() cases = [c for c in cases if not c.get("noul_only")] + [c for c in cases if c.get("noul_only")] op_ok = tgt_ok = tgt_n = 0; ranks = []; t = time.time(); by_kind = {}; by_width = {} NOUL = os.environ.get("NOUL") == "1"; nl = {"tp": 0, "fn": 0, "tn": 0, "fp": 0} for c in cases: state, questions, targets, controls = build_request(c.get("page_obj") or pages[c["page"]], c["goal"], c.get("history", [])) if c.get("noul_only"): # completion-only case (a failed run's stop state): counts only for the noul metric if NOUL: r = agent.predict(state, {"goal_done": goal_done_question(c["goal"])})["answers"] nl["fp" if r["goal_done"]["noul"] >= 0.5 else "tn"] += 1; nl["hard_n"] = nl.get("hard_n", 0) + 1 nl["hard_fp"] = nl.get("hard_fp", 0) + (r["goal_done"]["noul"] >= 0.5) continue if NOUL: questions["goal_done"] = goal_done_question(c["goal"]) r = agent.predict(state, questions)["answers"] if NOUL: yes, said = c["gold_op"] == "DONE", r["goal_done"]["noul"] >= 0.5 nl[("tp" if said else "fn") if yes else ("fp" if said else "tn")] += 1 op_hit = r["operation"]["choice"] == c["gold_op"]; op_ok += op_hit said_done = r["operation"]["choice"] == "DONE" fd = by_kind.setdefault("__done", [0, 0, 0, 0]) # [done n, done said, not-done n, not-done said DONE] if c["gold_op"] == "DONE": fd[0] += 1; fd[1] += said_done else: fd[2] += 1; fd[3] += said_done k = by_kind.setdefault(f"{c.get('source', 'live'):9s} {c['gold_op']}", [0, 0, 0]); k[0] += 1; k[1] += op_hit if c.get("gold_index") is not None: a = r[c["gold_op"].lower() + "_target"]; probs = a["probabilities"] order = sorted(probs, key=probs.get, reverse=True); rank = order.index(c["gold_index"]) + 1 tgt_n += 1; tgt_ok += rank == 1; ranks.append(rank / len(probs)); k[2] += rank == 1 n = len(probs); w = by_width.setdefault("<=20" if n <= 20 else "21-35" if n <= 35 else "36-60" if n <= 60 else ">60", [0, 0]) w[0] += 1; w[1] += rank == 1 n_op = sum(not c.get("noul_only") for c in cases) dt = (time.time() - t) / len(cases) * 1000 print(f"{sys.argv[3]}/{sys.argv[4] if len(sys.argv) > 4 else ''}: cases {n_op} operation acc {op_ok/max(1, n_op):.3f} " f"target top-1 {tgt_ok/max(1,tgt_n):.3f} (n={tgt_n}, mean normalized rank {sum(ranks)/max(1,len(ranks)):.3f}) {dt:.0f} ms/case") fd = by_kind.pop("__done", [0, 0, 0, 0]) print(f" op DONE: recall {fd[1]/max(1,fd[0]):.3f} (n={fd[0]}) premature DONE on not-done states {fd[3]/max(1,fd[2]):.4f} (n={fd[2]})") for kind, (n, o, tg) in sorted(by_kind.items()): print(f" {kind:19s} n={n:4d} op acc {o/n:.2f}" + (f" target top-1 {tg/n:.2f}" if not kind.endswith("DONE") else "")) if NOUL: pos, neg = nl["tp"] + nl["fn"], nl["tn"] + nl["fp"] if nl.get("hard_n"): print(f" goal_done (noul): failed runs' stop states judged done {nl['hard_fp']/nl['hard_n']:.3f} (n={nl['hard_n']})") print(f" goal_done (noul): done recall {nl['tp']/max(1,pos):.3f} (n={pos}) not-done judged done {nl['fp']/max(1,neg):.3f} (n={neg}) " f"acc {(nl['tp']+nl['tn'])/max(1,pos+neg):.3f}") for w in ("<=20", "21-35", "36-60", ">60"): if w in by_width: n, ok = by_width[w]; print(f" width {w:6s} n={n:4d} target top-1 {ok/n:.3f}") if __name__ == "__main__": main()