laya-browser / code /finetune /make_contrast.py
cklxx's picture
v19s: WebChain real-site trajectories, format v5, webgym x7 + DAgger, harness fixes; replaces v17s
454b3e6 verified
Raw History Blame Contribute Delete
6.66 kB
"""Goal-contrast twins against "reached a page about the thing = done".
python finetune/make_contrast.py <out.jsonl>
Failure seen on real sites (suite C traces): after searching, the policy says DONE on the results page even when the goal
asks to OPEN a result ("find the recipe page of Margarita") or a sub-page ("open the discussion page of OpenRC"). The
training data has ~930 DONE labels on results pages (mostly right: "search for X"), no counter-examples on the same pages,
and ~40 wrong ones ("Open the page about X" labelled DONE on search results).
1. relabel: an "open ..." goal marked DONE on a results page -> CLICK the result whose label matches the goal's entity
(dropped when no result matches); written as source "contrast_fix"
2. twins on results pages: a correct DONE("search for Q") page gets a second goal that names one of its results
("Open the page for <result>", ...) -> CLICK that result
3. twins on entity pages: a DONE page that links a sub-page (Discussion/Talk/Reviews/Episodes/...) gets a goal asking
for that sub-page -> CLICK it
Every twin reuses the exact page and history of a real case, so only the goal differs between DONE and CLICK.
"""
import json, os, random, re, sys
O = os.path.join(os.path.dirname(os.path.abspath(__file__)), "out")
RESULTS_URL = re.compile(r"[?&](q|s|query|search|keyword|term|st|k)=|/search", re.I)
OPEN_GOAL = re.compile(r"^\s*(open|go to the (entry|page)|find the (page|recipe|profile|details)|show me the (page|details|profile|recipe))", re.I)
NAV = re.compile(r"^(home|search|sign in|log in|login|register|menu|help|about|contact|next|previous|prev|more|skip|cookie|privacy|terms|"
r"\d+|page \d+|filter|sort|advanced search|clear|reset|close|back|top)\b", re.I)
SUBPAGES = ["Discussion", "Talk", "Reviews", "Episodes", "Cast", "Specifications", "Comments", "History", "Versions", "Files",
"Dependencies", "Photos", "Ingredients", "Issues", "Releases", "Changelog", "Documentation", "Details", "Seasons"]
OPEN_T = ["Open the page for {x}.", "Find {x} and open its page.", "Go to the details page of {x}.", "Open {x}.",
"Find the page of {x}.", "Show me the {x} page.", "Search for {q} and open {x}."]
SUB_T = ["Open the {s} page of {x}.", "Show the {s} of {x}.", "Go to the {s} section for {x}.", "Open {x}'s {s}."]
def clicks(pg):
return [a for a in pg["actions"] if a["kind"] == "click" and a.get("role") in ("link", "button", "gridcell", "row", "heading", "generic")]
def query_of(url, goal):
m = re.search(r"[?&](?:q|s|query|search|keyword|term|st|k)=([^&#]+)", url or "")
if m:
from urllib.parse import unquote_plus
return unquote_plus(m.group(1)).strip()
m = re.search(r"['\"]([^'\"]{2,40})['\"]", goal)
return m.group(1) if m else ""
def entity_of(goal):
m = re.search(r"['\"]([^'\"]{2,60})['\"]", goal) or re.search(r"(?:about|for|of|entry for)\s+(.+?)[.?!]*$", goal)
return m.group(1).strip() if m else ""
def main():
rng = random.Random(0)
pages = [json.loads(l) for l in open(os.path.join(O, "pages.jsonl"))]
out = open(sys.argv[1], "w"); n = {"fix": 0, "fix_drop": 0, "twin_result": 0, "twin_sub": 0}
for f in ("done_cases.jsonl", "step2_cases.jsonl", "rollout_cases.jsonl", "rollout2_cases.jsonl", "cases.jsonl", "gym_cases.jsonl"):
for line in open(os.path.join(O, f)):
c = json.loads(line)
if c.get("gold_op") != "DONE":
continue
pg = c.get("page_obj") or (pages[c["page"]] if c.get("page", -1) >= 0 else None)
if not pg:
continue
url, goal = pg.get("url") or c.get("url") or "", c["goal"]
base = {k: v for k, v in c.items() if k not in ("gold_op", "gold_id", "kind", "label", "goal")}
base.update(page=-1, page_obj=pg)
cands = [a for a in clicks(pg) if 3 <= len(a["label"].strip()) <= 90 and not NAV.match(a["label"].strip())]
if RESULTS_URL.search(url) and OPEN_GOAL.search(goal) and f != "gym_cases.jsonl":
# 1. a wrong DONE: the goal asks to open something that is only listed here
ent = entity_of(goal).lower()
hit = [a for a in cands if ent and ent in a["label"].lower()]
if hit:
a = hit[0]
out.write(json.dumps({**base, "goal": goal, "gold_op": "CLICK", "gold_id": a["id"], "kind": "click", "label": a["label"],
"source": "contrast_fix", "fix_of": f}, ensure_ascii=False) + "\n"); n["fix"] += 1
else:
n["fix_drop"] += 1
continue
if RESULTS_URL.search(url):
# 2. same results page, a goal that names one of the results
q = query_of(url, goal)
# only real result links: they contain a word of the query (no fallback to arbitrary links)
res = [a for a in cands if a.get("role") in ("link", "heading", "gridcell", "row") and len(a["label"].split()) >= 1
and q and any(w in a["label"].lower() for w in q.lower().split() if len(w) > 2)
and a["label"].strip().lower() != q.lower()]
for a in rng.sample(res, min(2 if f != "gym_cases.jsonl" else 1, len(res))):
x = a["label"].strip().split("\n")[0][:60]
g = rng.choice(OPEN_T).format(x=f"'{x}'", q=f"'{q}'" if q else f"'{x}'")
out.write(json.dumps({**base, "goal": g, "gold_op": "CLICK", "gold_id": a["id"], "kind": "click", "label": a["label"],
"source": "contrast"}, ensure_ascii=False) + "\n"); n["twin_result"] += 1
elif f != "gym_cases.jsonl":
# 3. an entity page (real sites only: the webgym shop's nav links are not sub-pages of anything)
subs = [(s, a) for a in clicks(pg) for s in SUBPAGES if a["label"].strip().lower() in (s.lower(), s.lower() + "s")]
if subs:
s, a = rng.choice(subs)
x = re.split(r" [-|–—] ", pg.get("title") or "")[0].strip()[:50]
if x:
out.write(json.dumps({**base, "goal": rng.choice(SUB_T).format(s=s.lower(), x=f"'{x}'"), "gold_op": "CLICK",
"gold_id": a["id"], "kind": "click", "label": a["label"], "source": "contrast"}, ensure_ascii=False) + "\n")
n["twin_sub"] += 1
print("contrast", n, "->", sys.argv[1])
if __name__ == "__main__":
main()