diff --git a/.gitattributes b/.gitattributes index a6344aac8c09253b3b630fb776ae94478aa0275b..9a22c005d074b9b4e29d17dd6a56058b0d460873 100644 --- a/.gitattributes +++ b/.gitattributes @@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text *.zip filter=lfs diff=lfs merge=lfs -text *.zst filter=lfs diff=lfs merge=lfs -text *tfevents* filter=lfs diff=lfs merge=lfs -text +v10s/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000000000000000000000000000000000000..79e349e948dd747efa4ebb210b96a8d5e3815fdf --- /dev/null +++ b/README.md @@ -0,0 +1,111 @@ +--- +license: apache-2.0 +base_model: convaiinnovations/laya +language: +- en +- multilingual +tags: +- laya +- system-1 +- browser-agent +- web-navigation +- decision-model +- modernbert +- mind2web +- tilelang +datasets: +- osunlp/Mind2Web +--- + +# laya-browser — laya fine-tuned as a browser-agent decision head (drop-in replacement for TypeSafe Jev) + +**laya** ([convaiinnovations/laya](https://huggingface.co/convaiinnovations/laya)) is a non-autoregressive "System 1" decision model: +one bidirectional encoder pass answers several typed questions (`choice` / `score` / `noul`) with calibrated probabilities, no text generation. +Out of the box it is near chance at browser decisions ("which element should I click for this goal?" — top-1 0.10 among ~45 candidates). + +This repo is what it took to turn it into a usable decision head for [browser-use/jev-ultrafast](https://github.com/browser-use/jev-ultrafast), +whose `/v1/systemone` request format is identical to laya's `predict(state, questions)`. Everything was done locally on one RTX 4070 Ti SUPER (16 GB), +no paid API: the text helper and the DAgger teacher are a local Qwen3-8B-AWQ served by sglang. + +## What changed relative to the original laya + +| | original laya (typed-decisions) | this repo | +|---|---|---| +| browser decision quality (16 real tasks × 3 runs) | 0 % | **v10: 58 %**, v10s: 50 % | +| element top-1 on held-out pages (2,734 decisions, ~45 candidates) | 0.10 | **0.66** (v10), 0.63 (v10s) | +| operation accuracy (CLICK / TYPE_TEXT / SELECT / DONE) | 0.54 | 0.88 | +| latency per browser step (3 questions, 30–65 candidates) | 50–200 ms | 41–50 ms (v10), **17–23 ms** (v10s) | +| backbone | ModernBERT-large 421M | v10: same; v10s: mmBERT-base 322M | +| input format | jev's state verbatim (element table as JSON inside the state, truncated by the 1024-token window) | **format v2/v3**: elements live only in the option list (full label + role + current value), state keeps title / URL / history / 1.2–1.5k chars of text, `head_max_len` 512 → 768 | +| training data | LocalLLaMA/typed-decisions | 5,244 reverse-generated goals on 421 crawled pages (Qwen writes "the goal a user would state to need this element"), 700 real DONE states (clicks actually executed), 659 step-2 negatives, [Mind2Web](https://huggingface.co/datasets/osunlp/Mind2Web) train (7,296 steps, candidates re-rendered as an element table), 177 on-policy DAgger corrections | +| training | — | laya's RLCD recipe (noisy-logit policy gradient + soft CE), single GPU, no gradient checkpointing, 4 epochs (~2 h for v10, ~1 h for v10s), post-hoc temperature | +| inference | HF eager + autocast | optional TileLang fast path ([PR #25 to laya](https://github.com/NandhaKishorM/laya/pull/25)): fused GEMM/GEGLU/LayerNorm/RoPE, sliding-window flash attention, bf16-resident weights, CUDA graphs — 4–5× lower per-call latency, identical answers | + +### Things that did **not** work (so you don't repeat them) +- Templated DONE goals ("Open the page titled X, stop once it is open") leak phrasing: the model learns *stop when ⇒ DONE*. DONE samples must be real landing pages after an executed action. +- If every DONE sample has exactly one prior action and every click sample has none, the model learns *any history ⇒ DONE*. Add mid-task negatives (step-2 goals on landing pages). +- Mind2Web alone kills DONE / TYPE_TEXT (no DONE there, CLICK dominates): re-weight rare operations (DONE ×4, TYPE_TEXT/SELECT ×3). +- Cutting page text to 3,000 chars saved nothing (the sequence is dominated by the head) and cost 0.04 top-1. +- `torch.compile` on variable-length batches recompiles per shape: 6× slower. +- Confidence-gated escalation to Qwen3-8B (System 2) made things *worse* (58 % → 42 %): on these pages the fine-tuned 322M/421M model is a better decider than an 8B general LLM. Use a stronger System 2 or none. +- jev's DOM reader hides password fields by design (login tasks are impossible) and never sees collapsed menus (Wikipedia's "Random article"). + +### What still fails +Multi-step "type then submit / pick a suggestion" flows, anything that needs scrolling first (the training data had no scroll samples), ` +

状态 (JSON 或纯文本)

+

问题 (JSON)

+ +
+""" + +class H(BaseHTTPRequestHandler): + def log_message(self, *a): pass + def do_GET(self): + self.send_response(200); self.send_header("Content-Type", "text/html; charset=utf-8"); self.end_headers() + self.wfile.write(HTML.encode()) + def do_POST(self): + n = int(self.headers.get("Content-Length", 0)); req = json.loads(self.rfile.read(n)) + try: + st = req["state"].strip() + try: st = json.loads(st) + except Exception: st = {"text": st} + qs = json.loads(req["questions"]) + agent = get_agent(req.get("model", "multilingual")) + t = time.time(); r = agent.predict(st, qs); ms = (time.time() - t) * 1000 + body = {"answers": r["answers"], "raw": r, "ms": ms} + except Exception as e: + body = {"error": f"{type(e).__name__}: {e}"} + data = json.dumps(body, ensure_ascii=False, default=str).encode() + self.send_response(200); self.send_header("Content-Type", "application/json"); self.end_headers(); self.wfile.write(data) + +if __name__ == "__main__": + port = int(sys.argv[1]) if len(sys.argv) > 1 else 7860 + get_agent("multilingual") + print(f"open http://127.0.0.1:{port}") + ThreadingHTTPServer(("127.0.0.1", port), H).serve_forever() diff --git a/code/env.sh b/code/env.sh new file mode 100644 index 0000000000000000000000000000000000000000..daf78f8449da3397a0ed3be7edf2e9a93f359101 --- /dev/null +++ b/code/env.sh @@ -0,0 +1,6 @@ +export HF_ENDPOINT=https://hf-mirror.com +export USE_TF=0 +export UV_INDEX_URL=https://pypi.tuna.tsinghua.edu.cn/simple +export PIP_INDEX_URL=https://pypi.tuna.tsinghua.edu.cn/simple +unset http_proxy https_proxy all_proxy HTTP_PROXY HTTPS_PROXY ALL_PROXY +export PATH="$HOME/.local/bin:$PATH" diff --git a/code/finetune/README.md b/code/finetune/README.md new file mode 100644 index 0000000000000000000000000000000000000000..d576fa1c79f05ccb4a83214e7753db8803099d56 --- /dev/null +++ b/code/finetune/README.md @@ -0,0 +1,83 @@ +# 微调 laya 做浏览器 agent 决策头(替代 TypeSafe Jev) + +流水线(全部本地、无付费 API): + +| 步骤 | 脚本 | 说明 | +|---|---|---| +| 1 抓页面 | `collect_pages.py` | headless Chromium + browser-harness,90 个真实页面的元素表 + 正文 | +| 2 反向生成目标 | `gen_goals.py` | 随机选一个元素当 gold,本地 Qwen3-8B-AWQ(sglang)写出"要用它的用户目标",1121 条 | +| 3 真实 DONE 样本 | `make_done_cases.py` | 在浏览器里真的执行点击,落地页 + 历史 = DONE,250 条 | +| 4 构建样本 | `build_items.py` | 复刻 jev-ultrafast 的 systemone 请求格式(`common_ft.py`),head_max_len 512;页面按 index%5 留出 | +| 5 训练 | `train.py` | laya 官方 RLCD 配方(噪声 logit 策略梯度 + soft CE)单卡版,4 轮约 10 分钟 | +| 6 评测 | `eval.py` | 留出页面上的操作准确率 / 目标 top-1 | + +```fish +bash finetune/run_v2.sh # 4-6 步 +bash finetune/run_all.sh # 含零样本基线 +``` + +## 加入魔搭 Mind2Web 后(v3/v4) + +`osunlp/Mind2Web`(魔搭,6.7 GB,12-25 MB/s 不走代理)→ `convert_mind2web.py`:按 backend_node_id 从 cleaned_html 还原元素文本/角色, +正例 + 44 个负例打乱成 jev 格式元素表,历史用 action_reprs,7296 条;按网站哈希留出 20%。v4 = 自采 + 真实 DONE + 全部 Mind2Web, +DONE / TYPE_TEXT / SELECT 操作样本 3 倍过采样,16020 条训练样本,3 轮 64 分钟。`bash finetune/run_v4.sh`。 + +评测 1642 条(自采留出页 232 + Mind2Web 未见网站 1410): + +| 模型 | 操作准确率 | 目标 top-1 | live DONE | m2w CLICK 目标 | m2w TYPE_TEXT 操作 | m2w SELECT 操作 | +|---|---|---|---|---|---|---| +| typed-decisions 零样本 | 0.26 | 0.13 | 0.39 | 0.03 | 0.00 | 0.00 | +| v3(无过采样) | 0.790 | 0.496 | 0.58 | 0.40 | 0.04 | 0.20 | +| **v4(过采样)** | **0.795** | **0.524** | **0.82** | **0.43** | **0.80** | **0.55** | + +真实浏览器 6 任务:零样本 0/6 → v2 2/6 → **v4 3/6**(python.org Downloads 2.0 s、Travel 分类 0.3 s、Wikipedia 随机文章), +Wikipedia 搜索任务首次正确选 TYPE_TEXT 并由本地 Qwen 填入 "Python programming language",但之后重复填同一字段而没有提交—— +训练数据里缺"字段已填好 → 下一步提交"的样本。GitHub Issues 点成 Releases、HN new 提前 DONE。 + +## 最终对比(2026-09-21,16 个真实任务 × 3 次,apps/browser_suite.py) + +| 模型 | 底座 | 一步延迟 | 留出目标 top-1 | 真实任务通过率 | +门控 τ=0.7(Qwen3-8B 兜底) | +|---|---|---|---|---|---| +| v10 | ModernBERT-large 421M | 41–50 ms | 0.656 | **58%**(28/48) | 42%,升级率 78% | +| v10s | mmBERT-base 322M,格式 v3 | **17–23 ms** | 0.631 | 50%(24/48) | 52%,升级率 85% | + +结果高度双峰:9 个任务 3/3 稳定通过(导航、分类、勾选、HN 各页、DuckDuckGo 搜索),7 个任务 0/3 稳定失败 +(Wikipedia 两个搜索任务、GitHub Issues、下拉选择、翻页需滚动、arXiv 搜索、Google Flights)。 +门控把低置信步骤交给 Qwen3-8B 反而更差:在这些页面上微调后的 laya 比 8B 通用模型判断得更准,System 2 需要更强的模型或专门提示。 +失败任务的共性是"输入后提交 / 选建议"和"需要先滚动",训练数据里几乎没有滚动样本。 +v10 数据:5244 条自采目标(421 页)+ 700 真实 DONE + step-2 + Mind2Web 全量 + DAgger,格式 v2,head 768,4 轮 2.2 小时。 + +## 输入格式 v2(v9 起) + +`LAYA_FMT=v2 LAYA_HEAD=768`:元素表不再塞进 state(1024 token 里会被截掉),只在选项里保留完整标签 + 角色 + 已填值, +state 只留标题 / URL / 历史 / 1500 字正文,head_max_len 512→768。v9 只用 Mind2Web + 该格式,真实任务从 6/16 到 **10/16**, +Mind2Web 点击目标 0.44→0.51。checkpoint 的 `rl_agent_config.json` 记录 `laya_fmt` / `head_max_len_train`,服务端自动跟随。 + +注意:jev 的 `snapshot.js` 有意隐藏密码框(`!['password','file','hidden'].includes(e.type)`),密码登录任务在该框架里不可能完成,任务集已移除。 + +## 真实任务集(apps/browser_suite.py,16 个任务,自动验收) + +| 版本 | 通过 | 主要失败模式 | +|---|---|---| +| v4 | 3/14 | 过早 DONE(6 个),填完不提交(2 个),点错(3 个) | +| v5(DONE 过采样 2x + 字段带已填值 + 温度 1.40) | 4/14 | 登录/搜索仍在填一个字段后 DONE | + +v5 的根因:真实 DONE 样本历史恰好都是 1 步、自采点击样本历史都是 0 步,模型学到"有历史 ⇒ DONE"。 +v6 加 `gen_step2.py`:在 250 个落地页上保留历史再生成 659 条新目标作为负样本。 + +## 结果(v1/v2,仅自采数据) + +留出 18 个未见页面、232 条: + +| 模型 | 操作准确率 | 目标 top-1 | DONE 判断 | 每条延迟 | +|---|---|---|---|---| +| typed-decisions 零样本 | 0.543 | 0.103(随机) | 0.39 | 254 ms | +| multilingual 零样本 | 0.127 | 0.124 | 0.00 | 249 ms | +| **微调 v2(1886 样本)** | **0.875** | **0.665** | 0.66 | 55 ms | + +真实浏览器 6 个任务(jev-ultrafast 循环 + 本地 laya 服务):python.org Downloads 2.4 s 完成、books.toscrape Travel +分类 0.3 s 完成;HN "new"、GitHub Issues、Wikipedia 两个任务失败(点错元素或提前 DONE)。零样本时 6 个全失败。 + +已知问题:置信度全是 1.00(未做温度校准);数据只有 90 个页面,泛化有限。下一步是把页面扩到 500+、 +每轮把线上失败样本加回训练集、拟合温度。v1 的教训:模板化的 DONE 样本会让模型学到"看到 stop when 就答 DONE", +DONE 样本必须来自真实执行后的落地页。 diff --git a/code/finetune/build_items.py b/code/finetune/build_items.py new file mode 100644 index 0000000000000000000000000000000000000000..2d8af27716b305243e45336c77fd64a35ab995ca --- /dev/null +++ b/code/finetune/build_items.py @@ -0,0 +1,67 @@ +"""cases.jsonl + pages.jsonl -> tokenized training items (train split) and eval cases (held-out pages). + + python finetune/build_items.py out/pages.jsonl out/cases.jsonl out/ +""" +import json, os, sys +import torch +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from common_ft import build_request, gold_for +from transformers import AutoTokenizer +from laya.common import QTYPES, build_sequence, render_options + +MAX_LEN, HEAD_MAX_LEN = 1024, int(os.environ.get('LAYA_HEAD', '512')) +EVAL_EVERY = 5 # pages with index % 5 == 0 are held out + + +def main(): + pages_f, cases_f, out = sys.argv[1:4] + pages = [json.loads(l) for l in open(pages_f)] + cases = [json.loads(l) for l in open(cases_f)] + for extra in sys.argv[4:]: + cases += [json.loads(l) for l in open(extra)] + snap = os.environ.get("LAYA_BASE") + tok = AutoTokenizer.from_pretrained(os.path.join(snap, "tokenizer")) + items, n_skip, ev = [], 0, [] + import hashlib + STOPS = [" Stop when it is open.", " Stop once that page is visible.", " Then stop.", " Finish when it has loaded.", ""] + for c in cases: + if c["kind"] == "done" and "page_obj" not in c: + continue # template DONE cases leak phrasing; real ones come from make_done_cases.py + page = c.get("page_obj") or pages[c["page"]] + h = int(hashlib.md5(c["goal"].encode()).hexdigest(), 16) + goal = c["goal"] + STOPS[h % len(STOPS)] + c = {**c, "goal": goal} + state, questions, targets, controls = build_request(page, goal, c.get("history", [])) + gop, gidx = gold_for(c, targets, controls) + if gop is None: + n_skip += 1; continue + held = (hashlib.md5(c["website"].encode()).digest()[0] % EVAL_EVERY == 0) if c.get("source") == "mind2web" else (c["page"] % EVAL_EVERY == 0) + if held: + ev.append({**c, "gold_index": gidx}); continue + golds = {"operation": gop} + if gidx is not None: + golds[gop.lower() + "_target"] = gidx + for qid, gold in golds.items(): + q = questions[qid]; keys = list(q["criteria"]) + target = [1.0 if k == gold else 0.0 for k in keys] + qq = {"t": "choice", "ins": json.dumps(q["instructions"]), "crit": q["criteria"]} + seq, markers = build_sequence(tok, state, qq, MAX_LEN, HEAD_MAX_LEN) + if len(markers) != len(render_options(qq)): + n_skip += 1; continue + item = {"ids": seq, "markers": markers, "qtype": QTYPES["choice"], "target": target, "label": keys.index(gold), "qid": qid, "gold_op": gop} + # class balance: CLICK dominates the operation question, so repeat the rare operations + reps = {"DONE": 4, "TYPE_TEXT": 3, "SELECT": 3}.get(gop, 1) if qid == "operation" else 1 + if c.get("source") == "dagger": + reps *= 5 # on-policy teacher corrections from real tasks: few but exactly where the policy fails + items.extend([item] * reps) + torch.save(items, os.path.join(out, "train_items.pt")) + with open(os.path.join(out, "eval_cases.jsonl"), "w") as f: + for c in ev: f.write(json.dumps(c, ensure_ascii=False) + "\n") + lens = [len(i["ids"]) for i in items] + import collections + print("operation label counts:", dict(collections.Counter(i["gold_op"] for i in items if i["qid"] == "operation"))) + print(f"train items {len(items)} (op {sum(i['qid']=='operation' for i in items)}, target {sum(i['qid']!='operation' for i in items)}), " + f"eval cases {len(ev)}, skipped {n_skip}, seq len mean {sum(lens)/len(lens):.0f} max {max(lens)}") + +if __name__ == "__main__": + main() diff --git a/code/finetune/calibrate.py b/code/finetune/calibrate.py new file mode 100644 index 0000000000000000000000000000000000000000..2beb510badcc13be00508bbcf5644a0dea8d7b67 --- /dev/null +++ b/code/finetune/calibrate.py @@ -0,0 +1,43 @@ +"""Fit per-question-type temperatures on held-out cases (NLL), write them into rl_agent_config.json. + + python finetune/calibrate.py out/pages.jsonl out/eval_cases.jsonl +""" +import json, os, sys +import numpy as np, torch +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))); sys.path.insert(0, "/home/ckl/projects/S/laya-upstream") +from common_ft import build_request, gold_for +import laya +from laya.common import QTYPES, build_sequence, collate_items + +def main(): + pages = [json.loads(l) for l in open(sys.argv[1])]; cases = [json.loads(l) for l in open(sys.argv[2])]; ck = sys.argv[3] + agent = laya.load(ck); agent.cfg["max_len"], agent.cfg["head_max_len"] = 1024, int(os.environ.get("LAYA_HEAD", agent.cfg.get("head_max_len_train", 512))); agent.accelerate() + Z, T, K = [], [], [] + for c in cases[::2]: + state, questions, targets, controls = build_request(c.get("page_obj") or pages[c["page"]], c["goal"], c.get("history", [])) + gop, gidx = gold_for(c, targets, controls) + golds = {"operation": gop} + if gidx is not None: golds[gop.lower() + "_target"] = gidx + items, keys = [], [] + for qid, gold in golds.items(): + q = questions[qid]; qq = {"t": "choice", "ins": json.dumps(q["instructions"]), "crit": q["criteria"]} + seq, markers = build_sequence(agent.tok, state, qq, 1024, agent.cfg['head_max_len']) + items.append({"ids": seq, "markers": markers, "qtype": QTYPES["choice"]}); keys.append((list(q["criteria"]), gold)) + b = collate_items([items], agent.tok.pad_token_id) + with torch.no_grad(), torch.autocast("cuda", dtype=agent.dtype): + logits, _ = agent.model(b["input_ids"].cuda(), b["attention_mask"].cuda(), b["marker_pos"].cuda(), b["marker_mask"].cuda(), b["qtype"].cuda()) + for r, (ks, gold) in enumerate(keys): + z = logits[r, :len(ks)].float().cpu(); Z.append(z); T.append(ks.index(gold)) + kmax = max(len(z) for z in Z); M = torch.full((len(Z), kmax), -1e4) + for i, z in enumerate(Z): M[i, :len(z)] = z + y = torch.tensor(T) + def nll(t): return torch.nn.functional.cross_entropy(M / t, y).item() + ts = np.exp(np.linspace(np.log(0.2), np.log(10), 200)); best = min(ts, key=nll) + acc = (M.argmax(-1) == y).float().mean().item() + conf0 = torch.softmax(M, -1).max(-1).values.mean().item(); conf1 = torch.softmax(M / best, -1).max(-1).values.mean().item() + print(f"n={len(Z)} acc={acc:.3f} T=1: nll {nll(1.0):.3f} mean conf {conf0:.3f} | T={best:.2f}: nll {nll(best):.3f} mean conf {conf1:.3f}") + cfgp = os.path.join(ck, "rl_agent_config.json"); cfg = json.load(open(cfgp)); cfg["temperature"] = [float(best), 1.0, 1.0] + json.dump(cfg, open(cfgp, "w"), indent=2); print("wrote temperature", best, "->", cfgp) + +if __name__ == "__main__": + main() diff --git a/code/finetune/collect_pages.py b/code/finetune/collect_pages.py new file mode 100644 index 0000000000000000000000000000000000000000..c09261280c28fd70593d8cc14bab14fa98ddecfe --- /dev/null +++ b/code/finetune/collect_pages.py @@ -0,0 +1,107 @@ +"""Crawl real pages with browser-harness and dump their jev-ultrafast observations (element tables + page text). + + python finetune/collect_pages.py out/pages.jsonl [max_pages=80] + +Starts from SEEDS, follows a few random same-site links from each page to diversify. One JSON line per page. +""" +import json, os, random, sys, time +sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast") +os.environ.setdefault("BU_CDP_URL", "http://127.0.0.1:9222") +from jev_ultrafast.browser import Browser, StalePage +from jev_ultrafast.model import action_space + +SEEDS = [ + "https://en.wikipedia.org/wiki/Main_Page", "https://en.wikipedia.org/wiki/Special:Random", "https://en.wikipedia.org/wiki/Python_(programming_language)", + "https://news.ycombinator.com/", "https://news.ycombinator.com/newest", "https://news.ycombinator.com/login", + "https://github.com/", "https://github.com/tile-ai/tilelang", "https://github.com/browser-use/browser-use/issues", "https://github.com/login", + "https://www.python.org/", "https://docs.python.org/3/", "https://pypi.org/", "https://pypi.org/project/laya/", + "https://archlinux.org/", "https://wiki.archlinux.org/", "https://developer.mozilla.org/en-US/", "https://duckduckgo.com/", + "https://www.bing.com/", "https://stackoverflow.com/questions", "https://www.reddit.com/", "https://arxiv.org/", + "https://arxiv.org/list/cs.LG/recent", "https://huggingface.co/models", "https://huggingface.co/convaiinnovations/laya", + "https://www.saucedemo.com/", "https://the-internet.herokuapp.com/", "https://the-internet.herokuapp.com/login", + "https://demo.opencart.com/", "https://www.demoblaze.com/", "https://books.toscrape.com/", "https://quotes.toscrape.com/login", + "https://www.google.com/travel/flights?hl=en", "https://www.booking.com/", "https://www.airbnb.com/", "https://www.amazon.com/", + "https://www.ebay.com/", "https://www.imdb.com/", "https://www.nytimes.com/", "https://www.bbc.com/", + "https://www.openstreetmap.org/", "https://weather.com/", "https://www.wolframalpha.com/", "https://translate.google.com/", + "https://www.gnu.org/", "https://kernel.org/", "https://www.rust-lang.org/", "https://go.dev/", "https://nodejs.org/en", + "https://www.npmjs.com/", "https://crates.io/", "https://docs.rs/", "https://www.kaggle.com/", "https://paperswithcode.com/", + # round 2: more sites, more form-heavy pages + "https://en.wikipedia.org/wiki/Special:Search", "https://en.wikipedia.org/wiki/Portal:Current_events", "https://de.wikipedia.org/", "https://zh.wikipedia.org/", + "https://news.ycombinator.com/ask", "https://news.ycombinator.com/show", "https://news.ycombinator.com/jobs", "https://news.ycombinator.com/submit", + "https://github.com/explore", "https://github.com/trending", "https://github.com/pytorch/pytorch", "https://github.com/pytorch/pytorch/pulls", + "https://github.com/pytorch/pytorch/issues", "https://gitlab.com/explore", "https://gitee.com/explore", "https://about.gitlab.com/", + "https://www.python.org/downloads/", "https://docs.python.org/3/tutorial/", "https://docs.python.org/3/library/", "https://peps.python.org/", + "https://pypi.org/search/?q=torch", "https://pypi.org/project/torch/", "https://pypi.org/account/login/", "https://pypi.org/help/", + "https://the-internet.herokuapp.com/dropdown", "https://the-internet.herokuapp.com/checkboxes", "https://the-internet.herokuapp.com/forgot_password", + "https://the-internet.herokuapp.com/inputs", "https://the-internet.herokuapp.com/tables", "https://the-internet.herokuapp.com/javascript_alerts", + "https://demo.opencart.com/index.php?route=product/category&path=20", "https://demo.opencart.com/index.php?route=account/login", + "https://demo.opencart.com/index.php?route=account/register", "https://demo.opencart.com/index.php?route=product/search&search=mac", + "https://www.demoblaze.com/cart.html", "https://www.demoblaze.com/prod.html?idp_=1", "https://books.toscrape.com/catalogue/page-2.html", + "https://books.toscrape.com/catalogue/category/books/mystery_3/index.html", "https://quotes.toscrape.com/tag/love/", "https://quotes.toscrape.com/page/2/", + "https://www.saucedemo.com/inventory.html", "https://automationexercise.com/", "https://automationexercise.com/login", "https://automationexercise.com/products", + "https://practicetestautomation.com/practice-test-login/", "https://demoqa.com/", "https://demoqa.com/text-box", "https://demoqa.com/select-menu", + "https://demoqa.com/webtables", "https://www.selenium.dev/selenium/web/web-form.html", "https://formy-project.herokuapp.com/", "https://formy-project.herokuapp.com/form", + "https://parabank.parasoft.com/parabank/index.htm", "https://parabank.parasoft.com/parabank/register.htm", "https://www.globalsqa.com/angularJs-protractor/BankingProject/", + "https://opensource-demo.orangehrmlive.com/", "https://magento.softwaretestingboard.com/", "https://magento.softwaretestingboard.com/women.html", + "https://www.airbnb.com/s/London/homes", "https://www.booking.com/searchresults.html?ss=Paris", "https://www.google.com/travel/hotels?hl=en", + "https://www.google.com/maps?hl=en", "https://www.google.com/search?q=tilelang&hl=en", "https://duckduckgo.com/?q=modernbert", "https://www.bing.com/search?q=laya", + "https://arxiv.org/list/cs.CL/new", "https://arxiv.org/abs/2412.13663", "https://arxiv.org/search/?query=flash+attention&searchtype=all", + "https://huggingface.co/datasets", "https://huggingface.co/spaces", "https://huggingface.co/docs", "https://huggingface.co/login", "https://huggingface.co/answerdotai/ModernBERT-base", + "https://www.modelscope.cn/models", "https://www.modelscope.cn/datasets", "https://www.kaggle.com/datasets", "https://www.kaggle.com/competitions", + "https://stackoverflow.com/", "https://stackoverflow.com/questions/tagged/python", "https://superuser.com/", "https://askubuntu.com/", + "https://www.reddit.com/r/MachineLearning/", "https://old.reddit.com/", "https://old.reddit.com/r/python/", "https://lobste.rs/", + "https://www.bbc.com/news", "https://www.bbc.com/sport", "https://www.theguardian.com/international", "https://www.reuters.com/", "https://apnews.com/", + "https://www.imdb.com/chart/top/", "https://www.imdb.com/find/?q=inception", "https://www.rottentomatoes.com/", "https://www.goodreads.com/", + "https://www.openstreetmap.org/search?query=Berlin", "https://www.wikidata.org/", "https://commons.wikimedia.org/", "https://www.wiktionary.org/", + "https://developer.mozilla.org/en-US/docs/Web/JavaScript", "https://developer.mozilla.org/en-US/docs/Web/HTML/Element/select", "https://web.dev/", + "https://www.rust-lang.org/learn", "https://doc.rust-lang.org/book/", "https://go.dev/doc/", "https://pkg.go.dev/", "https://nodejs.org/en/download", + "https://www.npmjs.com/package/react", "https://react.dev/", "https://vuejs.org/", "https://tailwindcss.com/docs", "https://getbootstrap.com/docs/", + "https://www.wolframalpha.com/input?i=2%2B2", "https://translate.google.com/?sl=en&tl=zh-CN&text=hello", "https://www.deepl.com/translator", + "https://weather.com/weather/today/l/USNY0996", "https://www.timeanddate.com/", "https://www.xe.com/currencyconverter/", "https://www.calculator.net/", + "https://archlinux.org/packages/", "https://aur.archlinux.org/", "https://wiki.archlinux.org/title/Installation_guide", "https://www.kernel.org/doc/", + "https://www.debian.org/", "https://ubuntu.com/download", "https://www.gnu.org/software/", "https://www.fsf.org/", +] + +def observe(url, timeout=25): + b = Browser(url) + try: + page = b.observe(screenshot=False) + links = b.evaluate("""(() => { const out=[]; for (const a of document.querySelectorAll('a[href]')) { + const h=a.href; if (h.startsWith(location.origin) && !h.includes('#') && h!==location.href) out.push(h); } return out.slice(0,400); })()""") or [] + return page, links + finally: + b.close() + +def main(): + out, max_pages = sys.argv[1], int(sys.argv[2]) if len(sys.argv) > 2 else 80 + os.makedirs(os.path.dirname(out) or ".", exist_ok=True) + seen = set() + if os.path.exists(out): + for line in open(out): + seen.add(json.loads(line)["url"]) + rng = random.Random(0) + queue = list(SEEDS); rng.shuffle(queue) + n = len(seen) + with open(out, "a") as f: + while queue and n < max_pages: + url = queue.pop(0) + if url in seen: + continue + t = time.time() + try: + page, links = observe(url) + except Exception as e: + print(f"skip {url}: {type(e).__name__}: {str(e)[:80]}", flush=True); continue + elements, targets, controls = action_space(page["actions"]) + if len(elements) < 5 or len(elements) > 160: + print(f"skip {url}: {len(elements)} elements", flush=True); continue + seen.add(page["url"]); n += 1 + f.write(json.dumps({"url": page["url"], "title": page["title"], "text": page["text"], "actions": page["actions"], + "scroll": page.get("scroll")}, ensure_ascii=False) + "\n"); f.flush() + print(f"[{n}] {len(elements):3d} elements {time.time()-t:4.1f}s {page['title'][:60]}", flush=True) + rng.shuffle(links) + queue.extend(l for l in links[:3] if l not in seen) + print("done", n, "pages") + +if __name__ == "__main__": + main() diff --git a/code/finetune/common_ft.py b/code/finetune/common_ft.py new file mode 100644 index 0000000000000000000000000000000000000000..e622adae91c045dc904c23905e70de0c92231039 --- /dev/null +++ b/code/finetune/common_ft.py @@ -0,0 +1,77 @@ +"""Shared: build the exact jev-ultrafast systemone request for a page + goal, with the server-side compaction.""" +import importlib.util, json, sys, types + +JEV = "/home/ckl/projects/S/jev-ultrafast/jev_ultrafast" +# load model.py / questions.py without running the package __init__ (which pulls in browser-harness) +if "jev_ultrafast" not in sys.modules: + _pkg = types.ModuleType("jev_ultrafast"); _pkg.__path__ = [JEV]; sys.modules["jev_ultrafast"] = _pkg + for _name in ("questions", "model"): + _spec = importlib.util.spec_from_file_location(f"jev_ultrafast.{_name}", f"{JEV}/{_name}.py") + _m = importlib.util.module_from_spec(_spec); sys.modules[_spec.name] = _m; _spec.loader.exec_module(_m) +from jev_ultrafast.model import action_space # noqa: E402 +from jev_ultrafast.questions import NEXT_ACTION, TARGET # noqa: E402 + +import os +FMT = os.environ.get("LAYA_FMT", "v1") +# v1: jev's state verbatim (page text up to 6000 chars + the whole element table as JSON) -- the 1024-token budget truncates +# most of it, so the model often never sees the candidates' context. 3000 chars was tried (v7): -0.04 top-1. +# v2: elements live only in the option list (full label + role + value); state keeps title/url/history and 1500 chars of text. +# v3: v2 + option labels capped at 50 chars and 1200 chars of text (~30% fewer tokens; for the 322M base to hit ~20 ms/step) +PAGE_TEXT_CHARS = {"v2": 1500, "v3": 1200}.get(FMT, 6000) +LABEL_CHARS = 50 if FMT == "v3" else 10000 + +LABELS = { + "CLICK": "Click an element, button, menu option, autocomplete suggestion, or calendar day.", + "TYPE_TEXT": "Enter or replace text in an editable field. A small LLM will supply the value from the goal.", + "SELECT": "Select an observed dropdown value.", +} + + +def compact(v): + if isinstance(v, dict) and "element" in v: + s = str(v["element"])[:LABEL_CHARS] + if v.get("role"): + s += f" ({v['role']})" + if v.get("current_value"): + s += f" = {str(v['current_value'])[:30]!r}" + for k in ("checked", "selected", "expanded"): + if k in v: + s += f" {k}={v[k]}" + return s + return v + + +def build_request(page, goal, history=()): + """Mirror of jev_ultrafast.model.choose() up to the HTTP call. Returns (state, questions, targets, controls).""" + elements, targets, controls = action_space(page["actions"]) + operations = {key: LABELS[key] for key in targets} + operations.update({key: value["label"] for key, value in controls.items()}) + operations.update(DONE="Every requirement is visibly satisfied.", BLOCKED="No supported operation can progress.") + questions = {"operation": {"type": "choice", "criteria": operations, "instructions": {"goal": goal, "rules": NEXT_ACTION}}} + for operation, candidates in targets.items(): + questions[operation.lower() + "_target"] = { + "type": "choice", + "criteria": {index: {"element": f"[{index}] {a['label']}", "current_value": a.get("current_value", a.get("value", "")), + **{k: a[k] for k in ("role", "checked", "selected", "expanded") if k in a}} for index, a in candidates.items()}, + "instructions": {"goal": goal, "operation": operation, "rules": [NEXT_ACTION, TARGET]}, + } + state = {"page": {"url": page["url"], "title": page["title"], "text": page["text"][:PAGE_TEXT_CHARS]}, + "recent_actions": [{k: h.get(k) for k in ("action", "kind", "text", "page_changed")} for h in list(history)[-10:]]} + if FMT not in ("v2", "v3"): + state["elements"] = elements + for q in questions.values(): + q["criteria"] = {k: compact(v) for k, v in q["criteria"].items()} + return state, questions, targets, controls + + +def gold_for(case, targets, controls): + """(gold operation key, gold target index or None) for a case in the operation/target question vocab.""" + op = case["gold_op"] + if op == "DONE": + return "DONE", None + if op in ("SCROLL_DOWN", "SCROLL_UP", "WAIT"): # page-level controls: operation question only + return (op, None) if op in {k.upper() for k in controls} else (None, None) + for index, a in targets.get(op, {}).items(): + if a["id"] == case["gold_id"]: + return op, index + return None, None diff --git a/code/finetune/convert_mind2web.py b/code/finetune/convert_mind2web.py new file mode 100644 index 0000000000000000000000000000000000000000..34789dbf9bcec6b1c23b616c65c689a28ee29860 --- /dev/null +++ b/code/finetune/convert_mind2web.py @@ -0,0 +1,116 @@ +"""Mind2Web (osunlp/Mind2Web from ModelScope) -> cases in this repo's format (page_obj with jev-style actions). + + python finetune/convert_mind2web.py finetune/data/mind2web/data/train/*.json finetune/out/m2w_cases.jsonl + +Each Mind2Web action becomes one case: goal = confirmed_task, history = previous action_reprs, gold = the positive +candidate; the element table = positive + up to MAX_NEG sampled negative candidates rendered like jev's snapshot.js +(label from text / aria-label / placeholder / alt / title / value, role from tag). +""" +import json, random, re, sys +from bs4 import BeautifulSoup + +MAX_NEG = 44 +ROLE = {"a": "link", "button": "button", "input": "textbox", "textarea": "textbox", "select": "combobox", "option": "option", + "img": "img", "li": "listitem", "label": "label", "span": "generic", "div": "generic", "svg": "img", "h1": "heading", + "h2": "heading", "h3": "heading", "p": "text", "td": "cell", "th": "columnheader", "tr": "row", "ul": "list"} + + +def label_of(el): + attrs = el.attrs + for k in ("aria-label", "placeholder", "alt", "title"): + v = attrs.get(k) + if isinstance(v, list): v = " ".join(v) + if v and v.strip(): return v.strip() + txt = " ".join(el.get_text(" ", strip=True).split()) + if txt: return txt[:80] + v = attrs.get("value") + if v: return str(v).strip()[:80] + return (attrs.get("name") or attrs.get("id") or el.name or "")[:60] + + +def kind_of(el, op): + t = el.name + typ = (el.attrs.get("type") or "").lower() + if t == "select": return "select" + if t == "textarea" or (t == "input" and typ in ("", "text", "search", "email", "password", "number", "tel", "url")) or el.attrs.get("contenteditable"): + return "fill" + return "click" + + +def convert_action(task, ai, action, rng): + soup = BeautifulSoup(action["cleaned_html"], "lxml") + by_id = {} + for el in soup.find_all(attrs={"backend_node_id": True}): + by_id[el.attrs["backend_node_id"]] = el + pos = action["pos_candidates"] + if not pos: return None + gold_el = by_id.get(pos[0]["backend_node_id"]) + if gold_el is None: return None + negs = [c for c in action["neg_candidates"] if c["backend_node_id"] in by_id] + rng.shuffle(negs) + cands = [pos[0]] + negs[:MAX_NEG] + rng.shuffle(cands) + op = action["operation"]["op"] + # values typed/selected in earlier steps: a field that already holds its value must show it (jev's rules key on that) + filled = {} + for r in task["action_reprs"][:ai]: + m = re.match(r"\[(\w+)\]\s+(.*?)\s+->\s+(TYPE|SELECT):\s*(.*)$", r) + if m: filled[" ".join(m.group(2).split()).lower()] = m.group(4).strip() + actions, gold_id, node = [], None, 0 + for c in cands: + el = by_id[c["backend_node_id"]] + lab = label_of(el) + if not lab: continue + node += 1 + is_gold = c["backend_node_id"] == pos[0]["backend_node_id"] + kind = {"CLICK": "click", "TYPE": "fill", "SELECT": "select"}[op] if is_gold else kind_of(el, op) + role = ROLE.get(el.name, "generic") + base = {"node": node, "label": lab, "role": role} + if kind == "select": + opts = [o.get_text(" ", strip=True) for o in el.find_all("option")][:8] or [action["operation"]["value"] or "option"] + if is_gold and action["operation"]["value"] and action["operation"]["value"] not in opts: + opts = [action["operation"]["value"]] + opts[:7] + for oi, o in enumerate(opts): + a = {**base, "id": f"select:{node}:{oi}", "kind": "select", "value": o, "label": f"{lab} → {o}", "current_value": ""} + actions.append(a) + if is_gold and (o == action["operation"]["value"] or (oi == 0 and not action["operation"]["value"])): gold_id = a["id"] + continue + a = {**base, "id": f"{kind}:{node}", "kind": kind} + if kind == "fill": + a["value"] = filled.get(lab.lower(), "") + a["current_value"] = a["value"] + actions.append(a) + if is_gold: gold_id = a["id"] + if gold_id is None: return None + for k, lab in (("wait", "Wait for the page to update"), ("scroll_down", "Scroll down"), ("scroll_up", "Scroll up")): + actions.append({"id": k, "kind": "wait" if k == "wait" else "scroll", "label": lab, "node": None, "delta": 600 if k == "scroll_down" else -600}) + text = " ".join(soup.get_text(" ", strip=True).split())[:6000] + hist = [] + for r in task["action_reprs"][:ai]: + lab_, _, opv = r.rpartition(" -> ") + kind_, _, val = opv.partition(": ") + hist.append({"action": lab_.strip(), "kind": {"CLICK": "click", "TYPE": "fill", "SELECT": "select"}.get(kind_, kind_.lower()), + "text": val or None, "page_changed": kind_ == "CLICK"}) + gold_op = {"CLICK": "CLICK", "TYPE": "TYPE_TEXT", "SELECT": "SELECT"}[op] + return {"page": -1, "url": f"https://{task['website']}.com/", "title": task["website"], "goal": task["confirmed_task"], "gold_op": gold_op, + "gold_id": gold_id, "kind": {"CLICK": "click", "TYPE": "fill", "SELECT": "select"}[op], "label": label_of(gold_el), "history": hist, + "source": "mind2web", "task_id": task["annotation_id"], "website": task["website"], + "page_obj": {"url": f"https://{task['website']}.com/", "title": task["website"], "text": text, "actions": actions}} + + +def main(): + files, out = sys.argv[1:-1], sys.argv[-1] + rng = random.Random(0); n = 0; skipped = 0 + with open(out, "w") as f: + for fn in files: + tasks = json.load(open(fn)) + for t in tasks: + for ai, a in enumerate(t["actions"]): + c = convert_action(t, ai, a, rng) + if c is None: skipped += 1; continue + f.write(json.dumps(c, ensure_ascii=False) + "\n"); n += 1 + print(f"{fn}: total {n} cases, skipped {skipped}", flush=True) + print("wrote", n) + +if __name__ == "__main__": + main() diff --git a/code/finetune/dagger.py b/code/finetune/dagger.py new file mode 100644 index 0000000000000000000000000000000000000000..434b9c99e6e3736b0a3c0ee9bb2be11b289fdeea --- /dev/null +++ b/code/finetune/dagger.py @@ -0,0 +1,121 @@ +"""DAgger-style harvesting: run real tasks with the current laya agent, and at every step ask the local Qwen teacher which +action is right given the page's element table. Disagreements (and agreements) become training cases with the *agent's* +on-policy states, so the next model learns exactly where this one goes wrong. + + python finetune/dagger.py out/dagger_cases.jsonl [tasks.jsonl] (services: chromium 9222, laya 8791, sglang 30000) + +tasks.jsonl lines: {"url": ..., "goal": ...}; default = apps/browser_suite.TASKS plus extra goals below. +""" +import json, os, sys, time +import httpx +sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast"); sys.path.insert(0, "/home/ckl/projects/S/laya/apps") +os.environ.update(BU_CDP_URL="http://127.0.0.1:9222", TYPESAFE_BASE_URL="http://127.0.0.1:8791", TYPESAFE_API_KEY="local", + TEXT_MODEL_API_KEY="local", TEXT_MODEL_BASE_URL="http://127.0.0.1:30000/v1", TEXT_MODEL="Qwen/Qwen3-8B-AWQ", + TEXT_MODEL_EXTRA_JSON='{"chat_template_kwargs": {"enable_thinking": false}}') +from jev_ultrafast import Agent +from jev_ultrafast.model import action_space +from browser_suite import TASKS + +EXTRA = [ + ("https://en.wikipedia.org/wiki/Main_Page", "Open the article about Albert Einstein."), + ("https://news.ycombinator.com/", "Open the 'past' page."), + ("https://news.ycombinator.com/", "Open the comments of the first story on the front page."), + ("https://github.com/tile-ai/tilelang", "Open the Pull requests tab."), + ("https://github.com/tile-ai/tilelang", "Open the README's 'examples' folder."), + ("https://www.python.org/", "Open the documentation page."), + ("https://docs.python.org/3/", "Open the tutorial."), + ("https://books.toscrape.com/", "Open the 'Mystery' category and then open the first book in it."), + ("https://books.toscrape.com/", "Go to page 2 of the catalogue."), + ("https://the-internet.herokuapp.com/", "Open the 'Dropdown' example and select 'Option 2'."), + ("https://the-internet.herokuapp.com/", "Open the 'Checkboxes' example and tick the first checkbox."), + ("https://quotes.toscrape.com/", "Open the quotes tagged 'love'."), + ("https://quotes.toscrape.com/", "Go to the next page of quotes."), + ("https://arxiv.org/", "Open the listing of new submissions in cs.CL."), + ("https://pypi.org/", "Search PyPI for 'tilelang' and open the project page."), + ("https://duckduckgo.com/", "Search for 'ModernBERT paper'."), + ("https://www.saucedemo.com/", "Log in with username 'standard_user' and password 'secret_sauce', then add the 'Sauce Labs Backpack' to the cart."), + ("https://demo.opencart.com/", "Open the 'Desktops' category from the top menu."), + ("https://www.demoblaze.com/", "Open the 'Laptops' category."), + ("https://huggingface.co/models", "Search models for 'laya'."), +] +TEACHER = os.environ["TEXT_MODEL_BASE_URL"] + "/chat/completions" +client = httpx.Client(timeout=180) +SYS = """You are the teacher for a browser agent. You see the user's goal, the actions taken so far, the current page (title, url, text excerpt) and a +numbered table of the controls on it. Decide the single best NEXT step: +- {"operation": "CLICK", "index": n} click control n +- {"operation": "TYPE_TEXT", "index": n} type into text control n (a separate helper supplies the value) +- {"operation": "SELECT", "index": n, "value": "..."} choose that option of select control n +- {"operation": "DONE"} every requirement of the goal is already visibly satisfied on this page +- {"operation": "WAIT"} / {"operation": "SCROLL_DOWN"} / {"operation": "SCROLL_UP"} +Do not re-do satisfied steps; a field that already shows the requested value is done. Return JSON only.""" + +def teach(goal, history, page, elements): + table = [{"index": e["index"], "label": e["label"][:70], "role": e.get("role"), "ops": e["operations"], **({"value": e["value"]} if e.get("value") else {})} for e in elements[:120]] + user = {"goal": goal, "actions_so_far": [{k: h.get(k) for k in ("action", "kind", "text")} for h in history[-8:]], + "page": {"title": page["title"], "url": page["url"], "text": page["text"][:2500]}, "controls": table} + body = {"model": os.environ["TEXT_MODEL"], "max_tokens": 120, "temperature": 0.0, "response_format": {"type": "json_object"}, + "chat_template_kwargs": {"enable_thinking": False}, + "messages": [{"role": "system", "content": SYS}, {"role": "user", "content": json.dumps(user, ensure_ascii=False)}]} + r = client.post(TEACHER, json=body).json() + return json.loads(r["choices"][0]["message"]["content"]) + +def to_case(goal, history, page, verdict): + elements, targets, controls = action_space(page["actions"]) + op = str(verdict.get("operation", "")).upper() + if op in ("DONE",): + return {"page": -1, "url": page["url"], "title": page["title"], "goal": goal, "gold_op": "DONE", "gold_id": "DONE", "kind": "done", "label": "", + "history": history, "source": "dagger", "page_obj": {k: page[k] for k in ("url", "title", "text", "actions")}} + if op in ("CLICK", "TYPE_TEXT", "SELECT"): + idx = str(verdict.get("index")) + cands = targets.get(op, {}) + if op == "SELECT": + hit = [k for k, a in cands.items() if k.split(":")[0] == idx and (a["value"] == verdict.get("value") or a["label"].endswith(str(verdict.get("value"))))] + key = hit[0] if hit else None + else: + key = idx if idx in cands else None + if key is None: return None + a = cands[key] + return {"page": -1, "url": page["url"], "title": page["title"], "goal": goal, "gold_op": op, "gold_id": a["id"], "kind": a["kind"], "label": a["label"], + "history": history, "source": "dagger", "page_obj": {k: page[k] for k in ("url", "title", "text", "actions")}} + return None + +def main(): + out = sys.argv[1] + tasks = [(u, g) for _, u, g, _ in TASKS] + EXTRA + if len(sys.argv) > 2: + tasks += [(json.loads(l)["url"], json.loads(l)["goal"]) for l in open(sys.argv[2])] + n_cases = n_dis = 0 + with open(out, "a") as f: + for url, goal in tasks: + t0 = time.time() + try: + with Agent(url, goal) as agent: + steps = 0 + while agent.state["status"] not in ("done", "blocked") and steps < 12: + page = agent.state["page"]; history = list(agent.state["history"]) + elements = action_space(page["actions"])[0] + try: + verdict = teach(goal, history, page, elements) + except Exception as e: + print(" teacher fail", str(e)[:60]); break + case = to_case(goal, [{k: h.get(k) for k in ("action", "kind", "text", "page_changed")} for h in history], page, verdict) + if case: + f.write(json.dumps(case, ensure_ascii=False) + "\n"); f.flush(); n_cases += 1 + # step the agent with its own policy + try: + st = agent.command("tick") + except Exception as e: + print(" tick fail", type(e).__name__, str(e)[:50]); break + d = st["decisions"][-1] if st["decisions"] else None + agent_choice = (d["operation"], d.get("target")) if d else None + teacher_choice = (str(verdict.get("operation", "")).upper(), str(verdict.get("index")) if verdict.get("index") is not None else None) + if agent_choice and agent_choice[0] != teacher_choice[0] or (agent_choice and agent_choice[1] != teacher_choice[1] and teacher_choice[0] in ("CLICK", "TYPE_TEXT")): + n_dis += 1 + steps += 1 + except Exception as e: + print(f" task fail {type(e).__name__}: {str(e)[:60]}") + print(f"{goal[:60]:60s} cases={n_cases} disagreements={n_dis} {time.time()-t0:.0f}s", flush=True) + print("wrote", n_cases, "cases ->", out) + +if __name__ == "__main__": + main() diff --git a/code/finetune/eval.py b/code/finetune/eval.py new file mode 100644 index 0000000000000000000000000000000000000000..0cb6f5f53172f26e5e5aec337e73909f4fb6e77f --- /dev/null +++ b/code/finetune/eval.py @@ -0,0 +1,33 @@ +"""Held-out eval: operation accuracy and target top-1 (given the gold operation) on unseen pages. + + python finetune/eval.py out/pages.jsonl out/eval_cases.jsonl [subfolder] +""" +import json, os, sys, time +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, "/home/ckl/projects/S/laya-upstream") # laya with Agent.accelerate() +from common_ft import build_request +import laya + +def main(): + pages = [json.loads(l) for l in open(sys.argv[1])]; cases = [json.loads(l) for l in open(sys.argv[2])] + agent = laya.load(sys.argv[3], subfolder=sys.argv[4] if len(sys.argv) > 4 else None) + agent.cfg["max_len"], agent.cfg["head_max_len"] = 1024, int(os.environ.get("LAYA_HEAD", agent.cfg.get("head_max_len_train", 512))) + agent.accelerate() + op_ok = tgt_ok = tgt_n = 0; ranks = []; t = time.time(); by_kind = {} + for c in cases: + state, questions, targets, controls = build_request(c.get("page_obj") or pages[c["page"]], c["goal"], c.get("history", [])) + r = agent.predict(state, questions)["answers"] + op_hit = r["operation"]["choice"] == c["gold_op"]; op_ok += op_hit + k = by_kind.setdefault(f"{c.get('source', 'live'):9s} {c['gold_op']}", [0, 0, 0]); k[0] += 1; k[1] += op_hit + if c.get("gold_index") is not None: + a = r[c["gold_op"].lower() + "_target"]; probs = a["probabilities"] + order = sorted(probs, key=probs.get, reverse=True); rank = order.index(c["gold_index"]) + 1 + tgt_n += 1; tgt_ok += rank == 1; ranks.append(rank / len(probs)); k[2] += rank == 1 + dt = (time.time() - t) / len(cases) * 1000 + print(f"{sys.argv[3]}/{sys.argv[4] if len(sys.argv) > 4 else ''}: cases {len(cases)} operation acc {op_ok/len(cases):.3f} " + f"target top-1 {tgt_ok/max(1,tgt_n):.3f} (n={tgt_n}, mean normalized rank {sum(ranks)/max(1,len(ranks)):.3f}) {dt:.0f} ms/case") + for kind, (n, o, tg) in sorted(by_kind.items()): + print(f" {kind:19s} n={n:4d} op acc {o/n:.2f}" + (f" target top-1 {tg/n:.2f}" if not kind.endswith("DONE") else "")) + +if __name__ == "__main__": + main() diff --git a/code/finetune/gen_goals.py b/code/finetune/gen_goals.py new file mode 100644 index 0000000000000000000000000000000000000000..8addc79b6bff1845788083f149bd6a6976750bb3 --- /dev/null +++ b/code/finetune/gen_goals.py @@ -0,0 +1,85 @@ +"""Reverse-generate browser goals with a local LLM: pick an element as gold, ask Qwen to write the user goal for it. + + python finetune/gen_goals.py out/pages.jsonl out/cases.jsonl [per_page=12] + +Each case: {url, title, goal, gold_op, gold_id, gold_node, kind, label}. Also one DONE case per page. +""" +import json, os, random, sys, threading +from concurrent.futures import ThreadPoolExecutor +import httpx +sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast") +from jev_ultrafast.model import action_space + +LLM = os.environ.get("TEXT_MODEL_BASE_URL", "http://127.0.0.1:30000/v1") + "/chat/completions" +MODEL = os.environ.get("TEXT_MODEL", "Qwen/Qwen3-8B-AWQ") +client = httpx.Client(timeout=120) + +SYS = """You write realistic browser-automation goals. Given a web page and ONE target control on it, write the goal a user +would give to an assistant such that the assistant's NEXT step is to use exactly that control. Rules: +- One or two sentences, natural language, from the user's perspective, mention what they want (not the UI mechanics). +- The goal must single out the target among the other listed controls; do not mention element numbers. +- For a text field, the goal must imply typing a concrete value into it (include the value). +- For a dropdown option, the goal must imply choosing that option. +- Vary phrasing: sometimes terse ("open the login page"), sometimes contextual ("I want to read about X, take me there"). +Return JSON: {"goal": "..."}""" + +def ask(page, target, others): + user = {"page": {"title": page["title"], "url": page["url"], "text_excerpt": page["text"][:700]}, + "target": {"kind": target["kind"], "label": target["label"], "role": target.get("role"), + "value": target.get("value", target.get("current_value", ""))}, + "other_controls_on_page": [o["label"][:60] for o in others]} + body = {"model": MODEL, "max_tokens": 200, "temperature": 0.9, "response_format": {"type": "json_object"}, + "chat_template_kwargs": {"enable_thinking": False}, + "messages": [{"role": "system", "content": SYS}, {"role": "user", "content": json.dumps(user, ensure_ascii=False)}]} + r = client.post(LLM, json=body).json() + goal = json.loads(r["choices"][0]["message"]["content"])["goal"] + return goal.strip() + +def main(): + src, out, per_page = sys.argv[1], sys.argv[2], int(sys.argv[3]) if len(sys.argv) > 3 else 12 + pages = [json.loads(l) for l in open(src)] + rng = random.Random(1) + jobs = [] + for pi, page in enumerate(pages): + elements, targets, controls = action_space(page["actions"]) + cands = [a for a in page["actions"] if a["kind"] in ("click", "fill", "select")] + # de-duplicate by label, prefer informative labels + seen, uniq = set(), [] + for a in cands: + lab = a["label"].split(" → ")[0].strip() + if len(lab) < 2 or lab.lower() in seen: continue + seen.add(lab.lower()); uniq.append(a) + rng.shuffle(uniq) + fills = [a for a in uniq if a["kind"] == "fill"][:3] + picks = fills + [a for a in uniq if a["kind"] != "fill"][: max(0, per_page - len(fills))] + for a in picks: + others = rng.sample([o for o in uniq if o is not a], min(10, len(uniq) - 1)) + jobs.append((pi, page, a, others)) + print(f"{len(pages)} pages -> {len(jobs)} goal jobs", flush=True) + lock = threading.Lock(); done = [0] + def work(job): + pi, page, a, others = job + try: + goal = ask(page, a, others) + except Exception as e: + print("fail", type(e).__name__, str(e)[:60], flush=True); return None + with lock: + done[0] += 1 + if done[0] % 50 == 0: print(f" {done[0]}/{len(jobs)}", flush=True) + op = {"click": "CLICK", "fill": "TYPE_TEXT", "select": "SELECT"}[a["kind"]] + return {"page": pi, "url": page["url"], "title": page["title"], "goal": goal, "gold_op": op, "gold_id": a["id"], + "gold_node": a.get("node"), "kind": a["kind"], "label": a["label"]} + with ThreadPoolExecutor(16) as ex: + cases = [c for c in ex.map(work, jobs) if c] + for pi, page in enumerate(pages): # DONE cases: the goal is already satisfied by the current page + cases.append({"page": pi, "url": page["url"], "title": page["title"], "gold_op": "DONE", "gold_id": "DONE", "kind": "done", + "label": "", "goal": rng.choice([f"Open the page titled '{page['title'][:70]}'. Stop once it is open.", + f"Go to {page['url']} and stop when it has loaded.", + f"Navigate to the '{page['title'][:50]}' page."])}) + with open(out, "w") as f: + for c in cases: f.write(json.dumps(c, ensure_ascii=False) + "\n") + print("wrote", len(cases), "cases ->", out) + for c in rng.sample(cases, 8): print(f" [{c['gold_op']:9s}] {c['label'][:35]:35s} <- {c['goal'][:90]}") + +if __name__ == "__main__": + main() diff --git a/code/finetune/gen_step2.py b/code/finetune/gen_step2.py new file mode 100644 index 0000000000000000000000000000000000000000..12cd0d8f2f37cefccb11ef092c1f0db0a1a81fbf --- /dev/null +++ b/code/finetune/gen_step2.py @@ -0,0 +1,50 @@ +"""Step-2 negatives: on landing pages from done_cases (history = one executed click), reverse-generate NEW goals whose +next step is another element on that page. Breaks the 'any history => DONE' shortcut. + + python finetune/gen_step2.py out/done_cases.jsonl out/step2_cases.jsonl [per_page=3] +""" +import json, random, sys, threading +from concurrent.futures import ThreadPoolExecutor +sys.path.insert(0, "/home/ckl/projects/S/laya/finetune") +from gen_goals import ask + +def main(): + src, out, per = sys.argv[1], sys.argv[2], int(sys.argv[3]) if len(sys.argv) > 3 else 3 + dones = [json.loads(l) for l in open(src)] + rng = random.Random(7); jobs = [] + for d in dones: + page = d["page_obj"] + cands = [a for a in page["actions"] if a["kind"] in ("click", "fill", "select") and len(a["label"].split(" → ")[0].strip()) > 1 + and a["label"] != d["label"]] + seen, uniq = set(), [] + for a in cands: + k = a["label"].lower() + if k in seen: continue + seen.add(k); uniq.append(a) + rng.shuffle(uniq) + fills = [a for a in uniq if a["kind"] == "fill"][:1] + for a in fills + [a for a in uniq if a["kind"] != "fill"][: per - len(fills)]: + jobs.append((d, a, rng.sample([o for o in uniq if o is not a], min(10, len(uniq) - 1)))) + print(f"{len(dones)} landing pages -> {len(jobs)} jobs", flush=True) + lock = threading.Lock(); n = [0] + def work(job): + d, a, others = job + try: + goal = ask(d["page_obj"], a, others) + except Exception as e: + print("fail", str(e)[:60], flush=True); return None + with lock: + n[0] += 1 + if n[0] % 100 == 0: print(f" {n[0]}/{len(jobs)}", flush=True) + op = {"click": "CLICK", "fill": "TYPE_TEXT", "select": "SELECT"}[a["kind"]] + # keep the history: the previous click is unrelated to the new goal, which is what happens mid-task all the time + return {**{k: d[k] for k in ("page", "url", "title", "history", "page_obj")}, "goal": goal, "gold_op": op, "gold_id": a["id"], + "gold_node": a.get("node"), "kind": a["kind"], "label": a["label"], "source": "live"} + with ThreadPoolExecutor(16) as ex: + cases = [c for c in ex.map(work, jobs) if c] + with open(out, "w") as f: + for c in cases: f.write(json.dumps(c, ensure_ascii=False) + "\n") + print("wrote", len(cases)) + +if __name__ == "__main__": + main() diff --git a/code/finetune/make_done_cases.py b/code/finetune/make_done_cases.py new file mode 100644 index 0000000000000000000000000000000000000000..f8ee54716cae97d405dbb541732ea6ec9bab8634 --- /dev/null +++ b/code/finetune/make_done_cases.py @@ -0,0 +1,47 @@ +"""Turn click cases into realistic DONE cases by actually executing the click in the browser. + + python finetune/make_done_cases.py out/pages.jsonl out/cases.jsonl out/done_cases.jsonl [max=250] + +For a click case (page A, goal, gold element): open A, click the gold element, observe the landing page B. +If the page changed, emit {goal, gold_op: DONE, page_obj: B, history: [that click]} with the same goal phrasing. +""" +import json, os, random, sys, time +sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast") +os.environ.setdefault("BU_CDP_URL", "http://127.0.0.1:9222") +from jev_ultrafast.browser import Browser + +def main(): + pages_f, cases_f, out = sys.argv[1:4]; mx = int(sys.argv[4]) if len(sys.argv) > 4 else 250 + pages = [json.loads(l) for l in open(pages_f)] + cases = [c for c in (json.loads(l) for l in open(cases_f)) if c["kind"] == "click"] + rng = random.Random(3); rng.shuffle(cases) + n_ok = n_try = 0 + with open(out, "w") as f: + for c in cases: + if n_ok >= mx: break + src = pages[c["page"]]; n_try += 1 + try: + b = Browser(src["url"]) + try: + page = b.observe(screenshot=False) + act = next((a for a in page["actions"] if a["kind"] == "click" and a["label"] == c["label"]), None) + if act is None: + continue + b.act(act, page); time.sleep(0.3) + dest = b.observe(screenshot=False) + finally: + b.close() + except Exception as e: + print("fail", type(e).__name__, str(e)[:60], flush=True); continue + if dest["url"] == page["url"] and dest["title"] == page["title"]: + continue + hist = [{"action": act["label"], "kind": "click", "text": None, "page_changed": True}] + f.write(json.dumps({"page": c["page"], "url": dest["url"], "title": dest["title"], "goal": c["goal"], "gold_op": "DONE", + "gold_id": "DONE", "kind": "done", "label": act["label"], "history": hist, + "page_obj": {k: dest[k] for k in ("url", "title", "text", "actions")}}, ensure_ascii=False) + "\n"); f.flush() + n_ok += 1 + if n_ok % 25 == 0: print(f" {n_ok} done cases from {n_try} tries", flush=True) + print("wrote", n_ok, "DONE cases") + +if __name__ == "__main__": + main() diff --git a/code/finetune/rollouts.py b/code/finetune/rollouts.py new file mode 100644 index 0000000000000000000000000000000000000000..1aa08f83f7f50da22979e22620c60d59cb2d9865 --- /dev/null +++ b/code/finetune/rollouts.py @@ -0,0 +1,109 @@ +"""Scripted multi-step trajectories executed in the real browser, for the three skills the model lacks: + scroll : goal targets an element that is only visible after scrolling -> [SCROLL_DOWN, CLICK, DONE] + search : goal asks to search for a phrase in a text field -> [TYPE_TEXT, CLICK submit/suggestion, DONE] + select : goal asks to choose an option of a