"""NNetNav-live (stanfordnlp/nnetnav-live, WebArena-style accessibility-tree trajectories on live sites) -> jev-format cases. python finetune/convert_nnetnav.py finetune/data/nnetnav/train.jsonl finetune/out/nnetnav_cases.jsonl [max_cases=20000] [seed=3] Each record is one step: the user message holds OBSERVATION (accessibility tree), URL, OBJECTIVE, PREVIOUS ACTIONS; the assistant message ends with ```action```. Mapping to jev's action space: click [id] -> CLICK on the element (clickable link/button/checkbox/... nodes) type [id] [text] [enter] -> TYPE_TEXT on the element (textbox/searchbox/combobox); the implicit Enter is dropped scroll [down|up] -> SCROLL_DOWN / SCROLL_UP (page-level controls) stop [...] -> DONE hover / press / goto / go_back / new_tab / tab_focus / close_tab -> skipped The element table is the tree's interactive nodes in document order (labels = accessible name, role kept), the page text is the tree's static text. Cases carry `website` so build_items.py can hold out whole sites, and `source: "nnetnav"`. Sampling: the action mix is capped so CLICK does not swamp TYPE_TEXT / SCROLL / DONE (see CAP). """ import json, random, re, sys from collections import Counter from urllib.parse import urlparse NODE = re.compile(r"^\s*\[(\d+)\] (\w+) '((?:[^'\\]|\\.)*)'(.*)$") STATIC = re.compile(r"^\s*StaticText '((?:[^'\\]|\\.)*)'") ACTION = re.compile(r"```\s*(\w+)(?:\s+\[([^\]]*)\])?(?:\s+\[((?:[^\]\\]|\\.)*)\])?(?:\s+\[([^\]]*)\])?\s*```") PREV = re.compile(r"^(\d+): (.*)$") CLICK_ROLES = {"link", "button", "checkbox", "radio", "menuitem", "tab", "option", "switch", "menuitemcheckbox", "menuitemradio", "treeitem", "combobox", "listbox", "img", "image", "gridcell", "cell", "row", "listitem", "heading", "generic", "LabelText", "DisclosureTriangle", "Details", "Summary", "slider", "spinbutton"} FILL_ROLES = {"textbox", "searchbox", "textarea"} ROLE_OUT = {"img": "image", "image": "image", "LabelText": "label", "DisclosureTriangle": "button", "Summary": "button", "Details": "group", "combobox": "combobox"} CAP = {"CLICK": 0.55, "TYPE_TEXT": 0.2, "SCROLL_DOWN": 0.1, "SCROLL_UP": 0.03, "DONE": 0.12} TEXT_CHARS = 6000 MAX_ELEMENTS = 45 def parse_obs(user): """-> (tree_lines, url, objective, previous action strings)""" m = re.search(r"OBSERVATION:\n(.*?)\nURL: (\S+)\nOBJECTIVE: (.*?)\nPREVIOUS ACTIONS:\n(.*)$", user, re.S) if not m: return None tree, url, objective, prev = m.groups() prevs = [PREV.match(l).group(2).strip() for l in prev.strip().splitlines() if PREV.match(l)] prevs = [p for p in prevs if p and p != "None"] return tree.splitlines(), url, objective.strip(), prevs def elements_of(tree_lines): """Interactive nodes -> jev actions; static text -> page text. Returns (actions, text, title, id->action).""" actions, texts, by_id, title = [], [], {}, "" for line in tree_lines: if line.startswith("RootWebArea"): m = re.match(r"RootWebArea '((?:[^'\\]|\\.)*)'", line) title = m.group(1) if m else "" continue sm = STATIC.match(line) if sm: texts.append(sm.group(1)); continue nm = NODE.match(line) if not nm: continue nid, role, name, rest = nm.groups() name = name.replace("\\'", "'").strip() clickable = "clickable" in rest if role in FILL_ROLES or (role == "combobox" and "hasPopup" not in rest and "autocomplete" in rest): val = re.search(r"value='((?:[^'\\]|\\.)*)'", rest) a = {"id": f"e{len(actions)+1}", "node": int(nid), "kind": "fill", "label": name or role, "role": "textbox" if role == "textarea" else role, "value": val.group(1) if val else ""} actions.append(a); by_id[nid] = a texts.append(name) continue if role in CLICK_ROLES and (clickable or role in ("link", "button", "checkbox", "radio", "tab", "menuitem", "option")): if not name and role in ("generic", "listitem", "row", "cell", "gridcell", "heading"): continue a = {"id": f"e{len(actions)+1}", "node": int(nid), "kind": "click", "label": name or role, "role": ROLE_OUT.get(role, role)} for attr in ("checked", "expanded", "selected"): mm = re.search(attr + r"=(\w+)", rest) if mm: a[attr] = mm.group(1) == "True" actions.append(a); by_id[nid] = a texts.append(name) continue if name and role in ("heading", "paragraph", "cell", "gridcell", "LabelText", "caption", "time", "code", "emphasis", "strong"): texts.append(name) return actions, "\n".join(t for t in texts if t)[:TEXT_CHARS], title, by_id def parse_action(assistant): m = ACTION.findall(assistant) if not m: return None verb, a1, a2, a3 = m[-1] return verb.lower(), a1, a2.replace("\\]", "]") if a2 else a2, a3 def history_of(prevs, tree_by_id_stack): """jev history from the textual previous actions. Labels come from 'where [id] is ...' suffixes when present.""" hist = [] for p in prevs[-10:]: pa = ACTION.findall("```" + p + "```") if not pa: continue verb, a1, a2, _ = pa[0] verb = verb.lower() where = re.search(r"where \[\d+\] is (\w+) '((?:[^'\\]|\\.)*)'", p) label = where.group(2) if where else (f"[{a1}]" if a1 else "") if verb == "click": hist.append({"action": label, "kind": "click", "text": None, "page_changed": None}) elif verb == "type": hist.append({"action": label, "kind": "fill", "text": a2, "page_changed": None}) elif verb == "scroll": hist.append({"action": "Scroll down" if "down" in a1 else "Scroll up", "kind": "scroll", "text": None, "page_changed": False}) elif verb == "hover": hist.append({"action": label, "kind": "click", "text": None, "page_changed": None}) else: hist.append({"action": p[:60], "kind": verb, "text": None, "page_changed": None}) return hist def main(): src, dst = sys.argv[1], sys.argv[2] max_cases = int(sys.argv[3]) if len(sys.argv) > 3 else 20000 rng = random.Random(int(sys.argv[4]) if len(sys.argv) > 4 else 3) counts, kept, skipped = Counter(), [], Counter() with open(src) as f: for line in f: try: d = json.loads(line) except Exception: skipped["json"] += 1; continue msgs = d.get("messages") or [] user = next((m["content"] for m in msgs if m["role"] == "user"), None) asst = next((m["content"] for m in reversed(msgs) if m["role"] == "assistant"), None) or d.get("output", "") if not user or not asst: skipped["msgs"] += 1; continue obs = parse_obs(user) act = parse_action(asst) if not obs or not act: skipped["parse"] += 1; continue tree, url, objective, prevs = obs verb, a1, a2, a3 = act actions, text, title, by_id = elements_of(tree) if len(actions) < 2: skipped["few_elements"] += 1; continue # keep the element table within laya's option budget (head_max_len 768 ~ 45 options, like the Mind2Web # conversion): the gold element plus a random sample of the others, document order preserved if len(actions) > MAX_ELEMENTS: gold_a = by_id.get(a1) if verb in ("click", "type") else None pool = [a for a in actions if a is not gold_a] keep = set(id(a) for a in rng.sample(pool, MAX_ELEMENTS - (1 if gold_a else 0))) actions = [a for a in actions if id(a) in keep or a is gold_a] for j, a in enumerate(actions): a["id"] = f"e{j+1}" # page-level controls, as jev would offer them (scroll offered unless the tree is tiny) if len(actions) > 8: actions.append({"id": "scroll_down", "kind": "scroll", "label": "Scroll down", "delta": 560}) if any(p.startswith("scroll [down") for p in prevs): actions.append({"id": "scroll_up", "kind": "scroll", "label": "Scroll up", "delta": -560}) actions.append({"id": "wait", "kind": "wait", "label": "Wait for the page to update"}) if verb == "click" and a1 in by_id and by_id[a1]["kind"] == "click": gop, gid, kind, label = "CLICK", by_id[a1]["id"], "click", by_id[a1]["label"] elif verb == "type" and a1 in by_id and by_id[a1]["kind"] == "fill": gop, gid, kind, label = "TYPE_TEXT", by_id[a1]["id"], "fill", by_id[a1]["label"] elif verb == "scroll" and "down" in (a1 or "") and any(a["id"] == "scroll_down" for a in actions): gop, gid, kind, label = "SCROLL_DOWN", "scroll_down", "scroll", "Scroll down" elif verb == "scroll" and "up" in (a1 or "") and any(a["id"] == "scroll_up" for a in actions): gop, gid, kind, label = "SCROLL_UP", "scroll_up", "scroll", "Scroll up" elif verb == "stop": gop, gid, kind, label = "DONE", "DONE", "done", "" else: skipped["verb:" + verb] += 1; continue site = urlparse(url).netloc kept.append({"page": -1, "url": url, "title": title, "goal": objective, "gold_op": gop, "gold_id": gid, "kind": kind, "label": label, "history": history_of(prevs, None), "source": "nnetnav", "website": site, "skill": "nnetnav", "page_obj": {"url": url, "title": title, "text": text, "actions": actions}}) counts[gop] += 1 print("parsed", len(kept), dict(counts), "skipped", dict(skipped)) # cap the mix: sample per operation up to CAP share of max_cases rng.shuffle(kept) out, taken = [], Counter() for c in kept: if taken[c["gold_op"]] < CAP[c["gold_op"]] * max_cases: out.append(c); taken[c["gold_op"]] += 1 with open(dst, "w") as f: for c in out: f.write(json.dumps(c, ensure_ascii=False) + "\n") print("wrote", len(out), dict(taken), "->", dst) if __name__ == "__main__": main()