File size: 5,982 Bytes
adf912b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
"""Mind2Web (osunlp/Mind2Web from ModelScope) -> cases in this repo's format (page_obj with jev-style actions).

    python finetune/convert_mind2web.py finetune/data/mind2web/data/train/*.json finetune/out/m2w_cases.jsonl

Each Mind2Web action becomes one case: goal = confirmed_task, history = previous action_reprs, gold = the positive
candidate; the element table = positive + up to MAX_NEG sampled negative candidates rendered like jev's snapshot.js
(label from text / aria-label / placeholder / alt / title / value, role from tag).
"""
import json, random, re, sys
from bs4 import BeautifulSoup

MAX_NEG = 44
ROLE = {"a": "link", "button": "button", "input": "textbox", "textarea": "textbox", "select": "combobox", "option": "option",
        "img": "img", "li": "listitem", "label": "label", "span": "generic", "div": "generic", "svg": "img", "h1": "heading",
        "h2": "heading", "h3": "heading", "p": "text", "td": "cell", "th": "columnheader", "tr": "row", "ul": "list"}


def label_of(el):
    attrs = el.attrs
    for k in ("aria-label", "placeholder", "alt", "title"):
        v = attrs.get(k)
        if isinstance(v, list): v = " ".join(v)
        if v and v.strip(): return v.strip()
    txt = " ".join(el.get_text(" ", strip=True).split())
    if txt: return txt[:80]
    v = attrs.get("value")
    if v: return str(v).strip()[:80]
    return (attrs.get("name") or attrs.get("id") or el.name or "")[:60]


def kind_of(el, op):
    t = el.name
    typ = (el.attrs.get("type") or "").lower()
    if t == "select": return "select"
    if t == "textarea" or (t == "input" and typ in ("", "text", "search", "email", "password", "number", "tel", "url")) or el.attrs.get("contenteditable"):
        return "fill"
    return "click"


def convert_action(task, ai, action, rng):
    soup = BeautifulSoup(action["cleaned_html"], "lxml")
    by_id = {}
    for el in soup.find_all(attrs={"backend_node_id": True}):
        by_id[el.attrs["backend_node_id"]] = el
    pos = action["pos_candidates"]
    if not pos: return None
    gold_el = by_id.get(pos[0]["backend_node_id"])
    if gold_el is None: return None
    negs = [c for c in action["neg_candidates"] if c["backend_node_id"] in by_id]
    rng.shuffle(negs)
    cands = [pos[0]] + negs[:MAX_NEG]
    rng.shuffle(cands)
    op = action["operation"]["op"]
    # values typed/selected in earlier steps: a field that already holds its value must show it (jev's rules key on that)
    filled = {}
    for r in task["action_reprs"][:ai]:
        m = re.match(r"\[(\w+)\]\s+(.*?)\s+->\s+(TYPE|SELECT):\s*(.*)$", r)
        if m: filled[" ".join(m.group(2).split()).lower()] = m.group(4).strip()
    actions, gold_id, node = [], None, 0
    for c in cands:
        el = by_id[c["backend_node_id"]]
        lab = label_of(el)
        if not lab: continue
        node += 1
        is_gold = c["backend_node_id"] == pos[0]["backend_node_id"]
        kind = {"CLICK": "click", "TYPE": "fill", "SELECT": "select"}[op] if is_gold else kind_of(el, op)
        role = ROLE.get(el.name, "generic")
        base = {"node": node, "label": lab, "role": role}
        if kind == "select":
            opts = [o.get_text(" ", strip=True) for o in el.find_all("option")][:8] or [action["operation"]["value"] or "option"]
            if is_gold and action["operation"]["value"] and action["operation"]["value"] not in opts:
                opts = [action["operation"]["value"]] + opts[:7]
            for oi, o in enumerate(opts):
                a = {**base, "id": f"select:{node}:{oi}", "kind": "select", "value": o, "label": f"{lab} → {o}", "current_value": ""}
                actions.append(a)
                if is_gold and (o == action["operation"]["value"] or (oi == 0 and not action["operation"]["value"])): gold_id = a["id"]
            continue
        a = {**base, "id": f"{kind}:{node}", "kind": kind}
        if kind == "fill":
            a["value"] = filled.get(lab.lower(), "")
            a["current_value"] = a["value"]
        actions.append(a)
        if is_gold: gold_id = a["id"]
    if gold_id is None: return None
    for k, lab in (("wait", "Wait for the page to update"), ("scroll_down", "Scroll down"), ("scroll_up", "Scroll up")):
        actions.append({"id": k, "kind": "wait" if k == "wait" else "scroll", "label": lab, "node": None, "delta": 600 if k == "scroll_down" else -600})
    text = " ".join(soup.get_text(" ", strip=True).split())[:6000]
    hist = []
    for r in task["action_reprs"][:ai]:
        lab_, _, opv = r.rpartition(" -> ")
        kind_, _, val = opv.partition(": ")
        hist.append({"action": lab_.strip(), "kind": {"CLICK": "click", "TYPE": "fill", "SELECT": "select"}.get(kind_, kind_.lower()),
                     "text": val or None, "page_changed": kind_ == "CLICK"})
    gold_op = {"CLICK": "CLICK", "TYPE": "TYPE_TEXT", "SELECT": "SELECT"}[op]
    return {"page": -1, "url": f"https://{task['website']}.com/", "title": task["website"], "goal": task["confirmed_task"], "gold_op": gold_op,
            "gold_id": gold_id, "kind": {"CLICK": "click", "TYPE": "fill", "SELECT": "select"}[op], "label": label_of(gold_el), "history": hist,
            "source": "mind2web", "task_id": task["annotation_id"], "website": task["website"],
            "page_obj": {"url": f"https://{task['website']}.com/", "title": task["website"], "text": text, "actions": actions}}


def main():
    files, out = sys.argv[1:-1], sys.argv[-1]
    rng = random.Random(0); n = 0; skipped = 0
    with open(out, "w") as f:
        for fn in files:
            tasks = json.load(open(fn))
            for t in tasks:
                for ai, a in enumerate(t["actions"]):
                    c = convert_action(t, ai, a, rng)
                    if c is None: skipped += 1; continue
                    f.write(json.dumps(c, ensure_ascii=False) + "\n"); n += 1
            print(f"{fn}: total {n} cases, skipped {skipped}", flush=True)
    print("wrote", n)

if __name__ == "__main__":
    main()