Feature Extraction
Transformers
Safetensors
English
multilingual
laya_browser
laya
custom_code
system-1
browser-agent
web-navigation
decision-model
mmbert
mind2web
tilelang
Instructions to use cklxx/laya-browser with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use cklxx/laya-browser with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("feature-extraction", model="cklxx/laya-browser", trust_remote_code=True)# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("cklxx/laya-browser", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
code: rollouts, teacher eval, corrected suite checks
Browse files- code/apps/browser_suite.py +6 -2
- code/finetune/README.md +11 -0
- code/finetune/run_suite_fixed.sh +14 -0
- code/finetune/run_teacher_eval.sh +29 -0
- code/finetune/run_v11.sh +21 -0
- code/finetune/teacher_eval.py +51 -0
code/apps/browser_suite.py
CHANGED
|
@@ -57,13 +57,17 @@ def run(name, url, goal, check, max_steps=20):
|
|
| 57 |
for state in agent.run():
|
| 58 |
steps = len(state["history"]); status = state["status"]; page = state["page"]
|
| 59 |
if steps >= max_steps: break
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 60 |
except Exception as e:
|
| 61 |
status = f"error:{type(e).__name__}"
|
| 62 |
wall = time.time() - t0
|
| 63 |
ok = bool(page and check(page["url"], page["title"], page.get("text", "")))
|
| 64 |
if page and name == "internet-dropdown":
|
| 65 |
-
ok = any(a.get("kind") == "select" and a.get("current_value")
|
| 66 |
-
any(a.get("kind") == "select" and a.get("value") == "2" and a.get("selected") for a in page["actions"])
|
| 67 |
if page and name == "internet-checkbox":
|
| 68 |
cbs = [a for a in page["actions"] if a.get("role") == "checkbox"]
|
| 69 |
ok = bool(cbs) and bool(cbs[0].get("checked"))
|
|
|
|
| 57 |
for state in agent.run():
|
| 58 |
steps = len(state["history"]); status = state["status"]; page = state["page"]
|
| 59 |
if steps >= max_steps: break
|
| 60 |
+
time.sleep(1.5) # let a slow navigation (GitHub's Turbo, etc.) land before judging
|
| 61 |
+
try:
|
| 62 |
+
page = agent.browser.observe(screenshot=False)
|
| 63 |
+
except Exception:
|
| 64 |
+
pass
|
| 65 |
except Exception as e:
|
| 66 |
status = f"error:{type(e).__name__}"
|
| 67 |
wall = time.time() - t0
|
| 68 |
ok = bool(page and check(page["url"], page["title"], page.get("text", "")))
|
| 69 |
if page and name == "internet-dropdown":
|
| 70 |
+
ok = any(a.get("kind") == "select" and str(a.get("current_value", "")).strip() in ("2", "Option 2") for a in page["actions"])
|
|
|
|
| 71 |
if page and name == "internet-checkbox":
|
| 72 |
cbs = [a for a in page["actions"] if a.get("role") == "checkbox"]
|
| 73 |
ok = bool(cbs) and bool(cbs[0].get("checked"))
|
code/finetune/README.md
CHANGED
|
@@ -34,6 +34,17 @@ DONE / TYPE_TEXT / SELECT 操作样本 3 倍过采样,16020 条训练样本,
|
|
| 34 |
Wikipedia 搜索任务首次正确选 TYPE_TEXT 并由本地 Qwen 填入 "Python programming language",但之后重复填同一字段而没有提交——
|
| 35 |
训练数据里缺"字段已填好 → 下一步提交"的样本。GitHub Issues 点成 Releases、HN new 提前 DONE。
|
| 36 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 37 |
## 最终对比(2026-09-21,16 个真实任务 × 3 次,apps/browser_suite.py)
|
| 38 |
|
| 39 |
| 模型 | 底座 | 一步延迟 | 留出目标 top-1 | 真实任务通过率 | +门控 τ=0.7(Qwen3-8B 兜底) |
|
|
|
|
| 34 |
Wikipedia 搜索任务首次正确选 TYPE_TEXT 并由本地 Qwen 填入 "Python programming language",但之后重复填同一字段而没有提交——
|
| 35 |
训练数据里缺"字段已填好 → 下一步提交"的样本。GitHub Issues 点成 Releases、HN new 提前 DONE。
|
| 36 |
|
| 37 |
+
## 最终对比 v2(2026-09-21,修正检查脚本后:下拉按 current_value 文本判定、导航后等 1.5 s 再判)
|
| 38 |
+
|
| 39 |
+
| 模型 | 真实任务 16×3 | 留出 top-1 | 每步 |
|
| 40 |
+
|---|---|---|---|
|
| 41 |
+
| v10(421M) | 50% | 0.656 | 41–50 ms |
|
| 42 |
+
| **v10s(322M)** | **62%** | 0.631 | 17–23 ms |
|
| 43 |
+
| v11s(322M + 682 条滚动/搜索/下拉脚本轨迹) | 56% | 0.623 | 19 ms |
|
| 44 |
+
|
| 45 |
+
10 个任务 3/3 稳过、6 个 3/3 稳挂;v10s/v11s 差异小于站点随机波动。老师对比:27B 思考预算 300 最好(0.861/0.603,4.7 s/步),
|
| 46 |
+
仍不超过 laya v11s(0.890/0.623,0.021 s);8B/27B 都不适合做 DAgger 老师或 System-2 兜底。
|
| 47 |
+
|
| 48 |
## 最终对比(2026-09-21,16 个真实任务 × 3 次,apps/browser_suite.py)
|
| 49 |
|
| 50 |
| 模型 | 底座 | 一步延迟 | 留出目标 top-1 | 真实任务通过率 | +门控 τ=0.7(Qwen3-8B 兜底) |
|
code/finetune/run_suite_fixed.sh
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# after the teacher eval: re-run the 16-task suite x3 for v11s, v10s and v10 with the corrected checks
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh; O=finetune/out
|
| 5 |
+
until grep -q "TEACHER_EVAL_DONE" $O/teacher_eval.log; do sleep 60; done
|
| 6 |
+
bash $S/stop_all.sh >/dev/null 2>&1; bash $S/bonsai.sh stop >/dev/null 2>&1
|
| 7 |
+
curl -s -m 3 http://127.0.0.1:9222/json/version >/dev/null || (nohup chromium --headless=new --remote-debugging-port=9222 --user-data-dir=$S/chrome-profile --window-size=1120,780 --no-first-run --lang=en-US about:blank >/dev/null 2>&1 &); sleep 3
|
| 8 |
+
for ck in laya-browser-v11s laya-browser-v10s laya-browser-v10; do
|
| 9 |
+
echo "== suite $ck x3 (fixed checks)"
|
| 10 |
+
ESCALATE_TAU=0 bash $S/restart_s1.sh $PWD/$O/$ck 999 >/dev/null
|
| 11 |
+
(cd ../jev-ultrafast && REPEATS=3 SUITE_OUT=$PWD/../laya/$O/suite_fixed_$ck.json timeout 3000 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -E "^==|per task")
|
| 12 |
+
done
|
| 13 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 14 |
+
echo SUITE_FIXED_DONE
|
code/finetune/run_teacher_eval.sh
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# after v11: compare teachers on 80 held-out cases (40 Mind2Web human-labelled + 40 live), then laya v10s / v11s on the same cases.
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 5 |
+
O=finetune/out; P=.venv/bin/python
|
| 6 |
+
until grep -q "V11_DONE" $O/v11.log; do sleep 60; done
|
| 7 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 8 |
+
echo "== 27B (bonsai), thinking off"
|
| 9 |
+
bash $S/bonsai.sh start
|
| 10 |
+
$P finetune/teacher_eval.py $O/pages.jsonl $O/eval_cases.jsonl http://127.0.0.1:30001/v1 "bonsai-27b think=off" 80
|
| 11 |
+
echo "== 27B, thinking full"
|
| 12 |
+
$P finetune/teacher_eval.py $O/pages.jsonl $O/eval_cases.jsonl http://127.0.0.1:30001/v1 "bonsai-27b think=full" 40 --think
|
| 13 |
+
bash $S/bonsai.sh stop
|
| 14 |
+
echo "== 27B, thinking budget 300 tokens (--reasoning-budget 300)"
|
| 15 |
+
bash $S/bonsai.sh start --reasoning-budget 300
|
| 16 |
+
$P finetune/teacher_eval.py $O/pages.jsonl $O/eval_cases.jsonl http://127.0.0.1:30001/v1 "bonsai-27b think=budget300" 80 --think
|
| 17 |
+
bash $S/bonsai.sh stop
|
| 18 |
+
echo "== 27B, thinking with --reasoning-effort low (server flag)"
|
| 19 |
+
bash $S/bonsai.sh start --reasoning-effort low
|
| 20 |
+
$P finetune/teacher_eval.py $O/pages.jsonl $O/eval_cases.jsonl http://127.0.0.1:30001/v1 "bonsai-27b think=low" 40 --think
|
| 21 |
+
bash $S/bonsai.sh stop
|
| 22 |
+
echo "== Qwen3-8B-AWQ, thinking off"
|
| 23 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 24 |
+
$P finetune/teacher_eval.py $O/pages.jsonl $O/eval_cases.jsonl http://127.0.0.1:30000/v1 "qwen3-8b think=off" 80
|
| 25 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 26 |
+
echo "== laya v10s / v11s on the full eval set (for reference)"
|
| 27 |
+
LAYA_FMT=v3 $P finetune/eval.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v10s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch" | head -1
|
| 28 |
+
LAYA_FMT=v3 $P finetune/eval.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v11s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch" | head -1
|
| 29 |
+
echo TEACHER_EVAL_DONE
|
code/finetune/run_v11.sh
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# v11s: v10s recipe + scripted scroll / search-submit / select trajectories (rollout_cases, x3), then 16 tasks x3.
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 5 |
+
O=finetune/out; P=.venv/bin/python
|
| 6 |
+
until ! pgrep -f "finetune/rollouts.py" >/dev/null; do sleep 60; done
|
| 7 |
+
echo "== rollout cases: $(wc -l < $O/rollout_cases.jsonl)"
|
| 8 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/multilingual | head -1)
|
| 9 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True CKPT=0 COMPILE=0 LAYA_FMT=v3 LAYA_HEAD=768
|
| 10 |
+
echo "== build v11s"
|
| 11 |
+
$P finetune/build_items.py $O/pages.jsonl $O/cases.jsonl $O/ $O/done_cases.jsonl $O/step2_cases.jsonl $O/m2w_cases.jsonl $O/dagger_cases.jsonl $O/rollout_cases.jsonl
|
| 12 |
+
echo "== train v11s (4 epochs)"; $P finetune/train.py $O/train_items.pt $O/laya-browser-v11s 4 2>&1 | grep --line-buffered -E "=== epoch|saved|Error|Traceback"
|
| 13 |
+
echo "== calibrate"; $P finetune/calibrate.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v11s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 14 |
+
echo "== eval v11s"; $P finetune/eval.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v11s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 15 |
+
echo "== suite v11s x3"
|
| 16 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 17 |
+
export ESCALATE_TAU=0
|
| 18 |
+
bash $S/restart_s1.sh $PWD/$O/laya-browser-v11s 999 >/dev/null
|
| 19 |
+
(cd ../jev-ultrafast && REPEATS=3 SUITE_OUT=$PWD/../laya/$O/suite_final_v11s.json timeout 3000 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -E "^==|per task")
|
| 20 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 21 |
+
echo V11_DONE
|
code/finetune/teacher_eval.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Teacher quality on held-out cases: operation accuracy and target top-1 for an LLM teacher (OpenAI-compatible endpoint).
|
| 2 |
+
|
| 3 |
+
python finetune/teacher_eval.py out/pages.jsonl out/eval_cases.jsonl <base_url> <label> [n=80] [--think] [--budget N]
|
| 4 |
+
|
| 5 |
+
Same element table / goal / history the laya model sees (format v3), same gold. Prints accuracy and seconds per decision.
|
| 6 |
+
"""
|
| 7 |
+
import json, os, random, sys, time
|
| 8 |
+
import httpx
|
| 9 |
+
os.environ.setdefault("LAYA_FMT", "v3")
|
| 10 |
+
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
| 11 |
+
from common_ft import build_request, gold_for
|
| 12 |
+
|
| 13 |
+
SYS = """You are the decision head of a browser agent (like TypeSafe Jev). Given the goal, the actions so far, the page and a numbered table of
|
| 14 |
+
controls with the operations each supports, output the single best NEXT step as JSON only:
|
| 15 |
+
{"operation": "CLICK"|"TYPE_TEXT"|"SELECT"|"DONE"|"BLOCKED"|"WAIT"|"SCROLL_DOWN"|"SCROLL_UP", "target": "<control key or null>"}
|
| 16 |
+
Rules: use TYPE_TEXT (not CLICK) for a text field the goal needs filled; SELECT for a dropdown option; CLICK for links/buttons/checkboxes.
|
| 17 |
+
DONE only if every requirement of the goal is already visibly satisfied. A field already showing the requested value is done.
|
| 18 |
+
If the needed control is not in the table and the page is taller than the viewport, SCROLL_DOWN."""
|
| 19 |
+
|
| 20 |
+
def main():
|
| 21 |
+
pages = [json.loads(l) for l in open(sys.argv[1])]; cases = [json.loads(l) for l in open(sys.argv[2])]
|
| 22 |
+
base, label = sys.argv[3].rstrip("/"), sys.argv[4]
|
| 23 |
+
n = int(next((a for a in sys.argv[5:] if a.isdigit()), "80")); think = "--think" in sys.argv
|
| 24 |
+
budget = int(sys.argv[sys.argv.index("--budget") + 1]) if "--budget" in sys.argv else None
|
| 25 |
+
rng = random.Random(0); rng.shuffle(cases)
|
| 26 |
+
sel = [c for c in cases if c.get("source") == "mind2web"][: n // 2] + [c for c in cases if c.get("source") != "mind2web"][: n // 2]
|
| 27 |
+
op_ok = tgt_ok = tgt_n = 0; secs = []; fails = 0
|
| 28 |
+
for c in sel:
|
| 29 |
+
state, questions, targets, controls = build_request(c.get("page_obj") or pages[c["page"]], c["goal"], c.get("history", []))
|
| 30 |
+
gop, gidx = gold_for(c, targets, controls)
|
| 31 |
+
if gop is None: continue
|
| 32 |
+
ctr = [{"op": qid[:-7].upper(), "target": k, "control": v} for qid, q in questions.items() if qid.endswith("_target") for k, v in q["criteria"].items()]
|
| 33 |
+
user = {"goal": c["goal"], "actions_so_far": state.get("recent_actions", []), "page": state["page"], "operations": list(questions["operation"]["criteria"]), "controls": ctr[:150]}
|
| 34 |
+
body = {"model": "x", "temperature": 0.0, "max_tokens": 1500 if think else 60, "response_format": {"type": "json_object"},
|
| 35 |
+
"chat_template_kwargs": {"enable_thinking": think},
|
| 36 |
+
"messages": [{"role": "system", "content": SYS}, {"role": "user", "content": json.dumps(user, ensure_ascii=False)}]}
|
| 37 |
+
if budget is not None: body["reasoning_budget"] = budget
|
| 38 |
+
t = time.time()
|
| 39 |
+
try:
|
| 40 |
+
r = httpx.post(base + "/chat/completions", json=body, timeout=600).json(); v = json.loads(r["choices"][0]["message"]["content"] or "{}")
|
| 41 |
+
except Exception as e:
|
| 42 |
+
fails += 1; continue
|
| 43 |
+
secs.append(time.time() - t)
|
| 44 |
+
op = str(v.get("operation", "")).upper(); op_ok += op == gop
|
| 45 |
+
if gidx is not None:
|
| 46 |
+
tgt_n += 1; tgt_ok += (op == gop and str(v.get("target")) == str(gidx))
|
| 47 |
+
m = len(secs)
|
| 48 |
+
print(f"{label:28s} n={m:3d} op acc {op_ok/max(1,m):.3f} target top-1 {tgt_ok/max(1,tgt_n):.3f} {sum(secs)/max(1,m):.2f} s/decision fails={fails}")
|
| 49 |
+
|
| 50 |
+
if __name__ == "__main__":
|
| 51 |
+
main()
|