File size: 6,657 Bytes
454b3e6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
"""Goal-contrast twins against "reached a page about the thing = done".

    python finetune/make_contrast.py <out.jsonl>

Failure seen on real sites (suite C traces): after searching, the policy says DONE on the results page even when the goal
asks to OPEN a result ("find the recipe page of Margarita") or a sub-page ("open the discussion page of OpenRC").  The
training data has ~930 DONE labels on results pages (mostly right: "search for X"), no counter-examples on the same pages,
and ~40 wrong ones ("Open the page about X" labelled DONE on search results).

  1. relabel: an "open ..." goal marked DONE on a results page -> CLICK the result whose label matches the goal's entity
     (dropped when no result matches); written as source "contrast_fix"
  2. twins on results pages: a correct DONE("search for Q") page gets a second goal that names one of its results
     ("Open the page for <result>", ...) -> CLICK that result
  3. twins on entity pages: a DONE page that links a sub-page (Discussion/Talk/Reviews/Episodes/...) gets a goal asking
     for that sub-page -> CLICK it
Every twin reuses the exact page and history of a real case, so only the goal differs between DONE and CLICK.
"""
import json, os, random, re, sys

O = os.path.join(os.path.dirname(os.path.abspath(__file__)), "out")
RESULTS_URL = re.compile(r"[?&](q|s|query|search|keyword|term|st|k)=|/search", re.I)
OPEN_GOAL = re.compile(r"^\s*(open|go to the (entry|page)|find the (page|recipe|profile|details)|show me the (page|details|profile|recipe))", re.I)
NAV = re.compile(r"^(home|search|sign in|log in|login|register|menu|help|about|contact|next|previous|prev|more|skip|cookie|privacy|terms|"
                 r"\d+|page \d+|filter|sort|advanced search|clear|reset|close|back|top)\b", re.I)
SUBPAGES = ["Discussion", "Talk", "Reviews", "Episodes", "Cast", "Specifications", "Comments", "History", "Versions", "Files",
            "Dependencies", "Photos", "Ingredients", "Issues", "Releases", "Changelog", "Documentation", "Details", "Seasons"]
OPEN_T = ["Open the page for {x}.", "Find {x} and open its page.", "Go to the details page of {x}.", "Open {x}.",
          "Find the page of {x}.", "Show me the {x} page.", "Search for {q} and open {x}."]
SUB_T = ["Open the {s} page of {x}.", "Show the {s} of {x}.", "Go to the {s} section for {x}.", "Open {x}'s {s}."]


def clicks(pg):
    return [a for a in pg["actions"] if a["kind"] == "click" and a.get("role") in ("link", "button", "gridcell", "row", "heading", "generic")]


def query_of(url, goal):
    m = re.search(r"[?&](?:q|s|query|search|keyword|term|st|k)=([^&#]+)", url or "")
    if m:
        from urllib.parse import unquote_plus
        return unquote_plus(m.group(1)).strip()
    m = re.search(r"['\"]([^'\"]{2,40})['\"]", goal)
    return m.group(1) if m else ""


def entity_of(goal):
    m = re.search(r"['\"]([^'\"]{2,60})['\"]", goal) or re.search(r"(?:about|for|of|entry for)\s+(.+?)[.?!]*$", goal)
    return m.group(1).strip() if m else ""


def main():
    rng = random.Random(0)
    pages = [json.loads(l) for l in open(os.path.join(O, "pages.jsonl"))]
    out = open(sys.argv[1], "w"); n = {"fix": 0, "fix_drop": 0, "twin_result": 0, "twin_sub": 0}
    for f in ("done_cases.jsonl", "step2_cases.jsonl", "rollout_cases.jsonl", "rollout2_cases.jsonl", "cases.jsonl", "gym_cases.jsonl"):
        for line in open(os.path.join(O, f)):
            c = json.loads(line)
            if c.get("gold_op") != "DONE":
                continue
            pg = c.get("page_obj") or (pages[c["page"]] if c.get("page", -1) >= 0 else None)
            if not pg:
                continue
            url, goal = pg.get("url") or c.get("url") or "", c["goal"]
            base = {k: v for k, v in c.items() if k not in ("gold_op", "gold_id", "kind", "label", "goal")}
            base.update(page=-1, page_obj=pg)
            cands = [a for a in clicks(pg) if 3 <= len(a["label"].strip()) <= 90 and not NAV.match(a["label"].strip())]
            if RESULTS_URL.search(url) and OPEN_GOAL.search(goal) and f != "gym_cases.jsonl":
                # 1. a wrong DONE: the goal asks to open something that is only listed here
                ent = entity_of(goal).lower()
                hit = [a for a in cands if ent and ent in a["label"].lower()]
                if hit:
                    a = hit[0]
                    out.write(json.dumps({**base, "goal": goal, "gold_op": "CLICK", "gold_id": a["id"], "kind": "click", "label": a["label"],
                                          "source": "contrast_fix", "fix_of": f}, ensure_ascii=False) + "\n"); n["fix"] += 1
                else:
                    n["fix_drop"] += 1
                continue
            if RESULTS_URL.search(url):
                # 2. same results page, a goal that names one of the results
                q = query_of(url, goal)
                # only real result links: they contain a word of the query (no fallback to arbitrary links)
                res = [a for a in cands if a.get("role") in ("link", "heading", "gridcell", "row") and len(a["label"].split()) >= 1
                       and q and any(w in a["label"].lower() for w in q.lower().split() if len(w) > 2)
                       and a["label"].strip().lower() != q.lower()]
                for a in rng.sample(res, min(2 if f != "gym_cases.jsonl" else 1, len(res))):
                    x = a["label"].strip().split("\n")[0][:60]
                    g = rng.choice(OPEN_T).format(x=f"'{x}'", q=f"'{q}'" if q else f"'{x}'")
                    out.write(json.dumps({**base, "goal": g, "gold_op": "CLICK", "gold_id": a["id"], "kind": "click", "label": a["label"],
                                          "source": "contrast"}, ensure_ascii=False) + "\n"); n["twin_result"] += 1
            elif f != "gym_cases.jsonl":
                # 3. an entity page (real sites only: the webgym shop's nav links are not sub-pages of anything)
                subs = [(s, a) for a in clicks(pg) for s in SUBPAGES if a["label"].strip().lower() in (s.lower(), s.lower() + "s")]
                if subs:
                    s, a = rng.choice(subs)
                    x = re.split(r" [-|–—] ", pg.get("title") or "")[0].strip()[:50]
                    if x:
                        out.write(json.dumps({**base, "goal": rng.choice(SUB_T).format(s=s.lower(), x=f"'{x}'"), "gold_op": "CLICK",
                                              "gold_id": a["id"], "kind": "click", "label": a["label"], "source": "contrast"}, ensure_ascii=False) + "\n")
                        n["twin_sub"] += 1
    print("contrast", n, "->", sys.argv[1])


if __name__ == "__main__":
    main()