"""Goal-contrast twins against "reached a page about the thing = done". python finetune/make_contrast.py Failure seen on real sites (suite C traces): after searching, the policy says DONE on the results page even when the goal asks to OPEN a result ("find the recipe page of Margarita") or a sub-page ("open the discussion page of OpenRC"). The training data has ~930 DONE labels on results pages (mostly right: "search for X"), no counter-examples on the same pages, and ~40 wrong ones ("Open the page about X" labelled DONE on search results). 1. relabel: an "open ..." goal marked DONE on a results page -> CLICK the result whose label matches the goal's entity (dropped when no result matches); written as source "contrast_fix" 2. twins on results pages: a correct DONE("search for Q") page gets a second goal that names one of its results ("Open the page for ", ...) -> CLICK that result 3. twins on entity pages: a DONE page that links a sub-page (Discussion/Talk/Reviews/Episodes/...) gets a goal asking for that sub-page -> CLICK it Every twin reuses the exact page and history of a real case, so only the goal differs between DONE and CLICK. """ import json, os, random, re, sys O = os.path.join(os.path.dirname(os.path.abspath(__file__)), "out") RESULTS_URL = re.compile(r"[?&](q|s|query|search|keyword|term|st|k)=|/search", re.I) OPEN_GOAL = re.compile(r"^\s*(open|go to the (entry|page)|find the (page|recipe|profile|details)|show me the (page|details|profile|recipe))", re.I) NAV = re.compile(r"^(home|search|sign in|log in|login|register|menu|help|about|contact|next|previous|prev|more|skip|cookie|privacy|terms|" r"\d+|page \d+|filter|sort|advanced search|clear|reset|close|back|top)\b", re.I) SUBPAGES = ["Discussion", "Talk", "Reviews", "Episodes", "Cast", "Specifications", "Comments", "History", "Versions", "Files", "Dependencies", "Photos", "Ingredients", "Issues", "Releases", "Changelog", "Documentation", "Details", "Seasons"] OPEN_T = ["Open the page for {x}.", "Find {x} and open its page.", "Go to the details page of {x}.", "Open {x}.", "Find the page of {x}.", "Show me the {x} page.", "Search for {q} and open {x}."] SUB_T = ["Open the {s} page of {x}.", "Show the {s} of {x}.", "Go to the {s} section for {x}.", "Open {x}'s {s}."] def clicks(pg): return [a for a in pg["actions"] if a["kind"] == "click" and a.get("role") in ("link", "button", "gridcell", "row", "heading", "generic")] def query_of(url, goal): m = re.search(r"[?&](?:q|s|query|search|keyword|term|st|k)=([^&#]+)", url or "") if m: from urllib.parse import unquote_plus return unquote_plus(m.group(1)).strip() m = re.search(r"['\"]([^'\"]{2,40})['\"]", goal) return m.group(1) if m else "" def entity_of(goal): m = re.search(r"['\"]([^'\"]{2,60})['\"]", goal) or re.search(r"(?:about|for|of|entry for)\s+(.+?)[.?!]*$", goal) return m.group(1).strip() if m else "" def main(): rng = random.Random(0) pages = [json.loads(l) for l in open(os.path.join(O, "pages.jsonl"))] out = open(sys.argv[1], "w"); n = {"fix": 0, "fix_drop": 0, "twin_result": 0, "twin_sub": 0} for f in ("done_cases.jsonl", "step2_cases.jsonl", "rollout_cases.jsonl", "rollout2_cases.jsonl", "cases.jsonl", "gym_cases.jsonl"): for line in open(os.path.join(O, f)): c = json.loads(line) if c.get("gold_op") != "DONE": continue pg = c.get("page_obj") or (pages[c["page"]] if c.get("page", -1) >= 0 else None) if not pg: continue url, goal = pg.get("url") or c.get("url") or "", c["goal"] base = {k: v for k, v in c.items() if k not in ("gold_op", "gold_id", "kind", "label", "goal")} base.update(page=-1, page_obj=pg) cands = [a for a in clicks(pg) if 3 <= len(a["label"].strip()) <= 90 and not NAV.match(a["label"].strip())] if RESULTS_URL.search(url) and OPEN_GOAL.search(goal) and f != "gym_cases.jsonl": # 1. a wrong DONE: the goal asks to open something that is only listed here ent = entity_of(goal).lower() hit = [a for a in cands if ent and ent in a["label"].lower()] if hit: a = hit[0] out.write(json.dumps({**base, "goal": goal, "gold_op": "CLICK", "gold_id": a["id"], "kind": "click", "label": a["label"], "source": "contrast_fix", "fix_of": f}, ensure_ascii=False) + "\n"); n["fix"] += 1 else: n["fix_drop"] += 1 continue if RESULTS_URL.search(url): # 2. same results page, a goal that names one of the results q = query_of(url, goal) # only real result links: they contain a word of the query (no fallback to arbitrary links) res = [a for a in cands if a.get("role") in ("link", "heading", "gridcell", "row") and len(a["label"].split()) >= 1 and q and any(w in a["label"].lower() for w in q.lower().split() if len(w) > 2) and a["label"].strip().lower() != q.lower()] for a in rng.sample(res, min(2 if f != "gym_cases.jsonl" else 1, len(res))): x = a["label"].strip().split("\n")[0][:60] g = rng.choice(OPEN_T).format(x=f"'{x}'", q=f"'{q}'" if q else f"'{x}'") out.write(json.dumps({**base, "goal": g, "gold_op": "CLICK", "gold_id": a["id"], "kind": "click", "label": a["label"], "source": "contrast"}, ensure_ascii=False) + "\n"); n["twin_result"] += 1 elif f != "gym_cases.jsonl": # 3. an entity page (real sites only: the webgym shop's nav links are not sub-pages of anything) subs = [(s, a) for a in clicks(pg) for s in SUBPAGES if a["label"].strip().lower() in (s.lower(), s.lower() + "s")] if subs: s, a = rng.choice(subs) x = re.split(r" [-|–—] ", pg.get("title") or "")[0].strip()[:50] if x: out.write(json.dumps({**base, "goal": rng.choice(SUB_T).format(s=s.lower(), x=f"'{x}'"), "gold_op": "CLICK", "gold_id": a["id"], "kind": "click", "label": a["label"], "source": "contrast"}, ensure_ascii=False) + "\n") n["twin_sub"] += 1 print("contrast", n, "->", sys.argv[1]) if __name__ == "__main__": main()