Duplicate from cklxx/laya-browser
Browse filesCo-authored-by: chenkailun <cklxx@users.noreply.huggingface.co>
This view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +39 -0
- README.md +132 -0
- assets/laya_browser_demo.gif +3 -0
- assets/laya_browser_demo.mp4 +3 -0
- code/.python-version +1 -0
- code/README.md +35 -0
- code/apps/BROWSER_AGENT.md +34 -0
- code/apps/browser_diag.py +56 -0
- code/apps/browser_suite.py +92 -0
- code/apps/browser_task.py +18 -0
- code/apps/common.py +61 -0
- code/apps/fast_batch.py +89 -0
- code/apps/hn_radar.py +44 -0
- code/apps/inbox_triage.py +43 -0
- code/apps/make_demo.py +112 -0
- code/apps/moderator.py +47 -0
- code/apps/profile_step.py +17 -0
- code/apps/systemone_server.py +169 -0
- code/apps/web_console.py +71 -0
- code/env.sh +6 -0
- code/finetune/README.md +94 -0
- code/finetune/build_items.py +67 -0
- code/finetune/calibrate.py +43 -0
- code/finetune/collect_pages.py +107 -0
- code/finetune/common_ft.py +77 -0
- code/finetune/convert_mind2web.py +116 -0
- code/finetune/dagger.py +121 -0
- code/finetune/eval.py +33 -0
- code/finetune/gen_goals.py +85 -0
- code/finetune/gen_step2.py +50 -0
- code/finetune/make_done_cases.py +47 -0
- code/finetune/rollouts.py +109 -0
- code/finetune/run_after_v6.sh +30 -0
- code/finetune/run_all.sh +16 -0
- code/finetune/run_final.sh +20 -0
- code/finetune/run_gated.sh +17 -0
- code/finetune/run_suite_fixed.sh +14 -0
- code/finetune/run_teacher_eval.sh +29 -0
- code/finetune/run_train.sh +9 -0
- code/finetune/run_v10.sh +32 -0
- code/finetune/run_v10s.sh +24 -0
- code/finetune/run_v11.sh +21 -0
- code/finetune/run_v2.sh +11 -0
- code/finetune/run_v3.sh +11 -0
- code/finetune/run_v4.sh +13 -0
- code/finetune/run_v5.sh +14 -0
- code/finetune/run_v6.sh +14 -0
- code/finetune/run_v7.sh +20 -0
- code/finetune/run_v8.sh +18 -0
- code/finetune/run_v9.sh +29 -0
.gitattributes
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
+
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
+
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
+
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
+
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
+
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
+
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
+
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
+
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
+
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
+
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
+
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
+
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
+
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
+
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
+
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
+
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
+
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
+
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
+
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
+
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
+
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
+
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
+
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
+
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
+
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
+
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
+
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
+
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
+
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
+
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
+
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
+
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
+
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
v10s/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
assets/laya_browser_demo.gif filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
assets/laya_browser_demo.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
v11s/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
README.md
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
base_model: convaiinnovations/laya
|
| 4 |
+
language:
|
| 5 |
+
- en
|
| 6 |
+
- multilingual
|
| 7 |
+
tags:
|
| 8 |
+
- laya
|
| 9 |
+
- system-1
|
| 10 |
+
- browser-agent
|
| 11 |
+
- web-navigation
|
| 12 |
+
- decision-model
|
| 13 |
+
- modernbert
|
| 14 |
+
- mind2web
|
| 15 |
+
- tilelang
|
| 16 |
+
datasets:
|
| 17 |
+
- osunlp/Mind2Web
|
| 18 |
+
---
|
| 19 |
+
|
| 20 |
+
# laya-browser — laya fine-tuned as a browser-agent decision head (drop-in replacement for TypeSafe Jev)
|
| 21 |
+
|
| 22 |
+

|
| 23 |
+
|
| 24 |
+
*46-second demo ([mp4](assets/laya_browser_demo.mp4)): v10s drives headless Chromium through jev-ultrafast, one encoder pass per step; recorded with `code/apps/make_demo.py`.*
|
| 25 |
+
|
| 26 |
+
**laya** ([convaiinnovations/laya](https://huggingface.co/convaiinnovations/laya)) is a non-autoregressive "System 1" decision model:
|
| 27 |
+
one bidirectional encoder pass answers several typed questions (`choice` / `score` / `noul`) with calibrated probabilities, no text generation.
|
| 28 |
+
Out of the box it is near chance at browser decisions ("which element should I click for this goal?" — top-1 0.10 among ~45 candidates).
|
| 29 |
+
|
| 30 |
+
This repo is what it took to turn it into a usable decision head for [browser-use/jev-ultrafast](https://github.com/browser-use/jev-ultrafast),
|
| 31 |
+
whose `/v1/systemone` request format is identical to laya's `predict(state, questions)`. Everything was done locally on one RTX 4070 Ti SUPER (16 GB),
|
| 32 |
+
no paid API: the text helper and the DAgger teacher are a local Qwen3-8B-AWQ served by sglang.
|
| 33 |
+
|
| 34 |
+
## What changed relative to the original laya
|
| 35 |
+
|
| 36 |
+
| | original laya (typed-decisions) | this repo |
|
| 37 |
+
|---|---|---|
|
| 38 |
+
| browser decision quality (16 real tasks × 3 runs, corrected checks) | 0 % | **v10s: 62 %**, v11s: 56 %, v10: 50 % (10 tasks pass 3/3, 6 fail 3/3; see below) |
|
| 39 |
+
| element top-1 on held-out pages (2,734 decisions, ~45 candidates) | 0.10 | **0.66** (v10), 0.63 (v10s), 0.62 (v11s) |
|
| 40 |
+
| operation accuracy (CLICK / TYPE_TEXT / SELECT / DONE) | 0.54 | 0.88–0.89 |
|
| 41 |
+
| latency per browser step (3 questions, 30–65 candidates) | 50–200 ms | 41–50 ms (v10), **17–23 ms** (v10s / v11s) |
|
| 42 |
+
| backbone | ModernBERT-large 421M | v10: same; v10s / v11s: mmBERT-base 322M |
|
| 43 |
+
| input format | jev's state verbatim (element table as JSON inside the state, truncated by the 1024-token window) | **format v2/v3**: elements live only in the option list (full label + role + current value), state keeps title / URL / history / 1.2–1.5k chars of text, `head_max_len` 512 → 768 |
|
| 44 |
+
| training data | LocalLLaMA/typed-decisions | 5,244 reverse-generated goals on 421 crawled pages (Qwen writes "the goal a user would state to need this element"), 700 real DONE states (clicks actually executed), 659 step-2 negatives, [Mind2Web](https://huggingface.co/datasets/osunlp/Mind2Web) train (7,296 steps, candidates re-rendered as an element table), 177 on-policy DAgger corrections |
|
| 45 |
+
| training | — | laya's RLCD recipe (noisy-logit policy gradient + soft CE), single GPU, no gradient checkpointing, 4 epochs (~2 h for v10, ~1 h for v10s), post-hoc temperature |
|
| 46 |
+
| inference | HF eager + autocast | optional TileLang fast path ([PR #25 to laya](https://github.com/NandhaKishorM/laya/pull/25)): fused GEMM/GEGLU/LayerNorm/RoPE, sliding-window flash attention, bf16-resident weights, CUDA graphs — 4–5× lower per-call latency, identical answers |
|
| 47 |
+
|
| 48 |
+
### Things that did **not** work (so you don't repeat them)
|
| 49 |
+
- Templated DONE goals ("Open the page titled X, stop once it is open") leak phrasing: the model learns *stop when ⇒ DONE*. DONE samples must be real landing pages after an executed action.
|
| 50 |
+
- If every DONE sample has exactly one prior action and every click sample has none, the model learns *any history ⇒ DONE*. Add mid-task negatives (step-2 goals on landing pages).
|
| 51 |
+
- Mind2Web alone kills DONE / TYPE_TEXT (no DONE there, CLICK dominates): re-weight rare operations (DONE ×4, TYPE_TEXT/SELECT ×3).
|
| 52 |
+
- Cutting page text to 3,000 chars saved nothing (the sequence is dominated by the head) and cost 0.04 top-1.
|
| 53 |
+
- `torch.compile` on variable-length batches recompiles per shape: 6× slower.
|
| 54 |
+
- Confidence-gated escalation to Qwen3-8B (System 2) made things *worse* (58 % → 42 %): on these pages the fine-tuned 322M/421M model is a better decider than an 8B general LLM. Use a stronger System 2 or none.
|
| 55 |
+
- jev's DOM reader hides password fields by design (login tasks are impossible) and never sees collapsed menus (Wikipedia's "Random article").
|
| 56 |
+
|
| 57 |
+
### What still fails
|
| 58 |
+
The suite is bimodal: 10 tasks pass 3/3 (category / tab / page navigation, checkbox, `<select>`, HN pages, DuckDuckGo search in some runs) and 6 fail 3/3:
|
| 59 |
+
"type then submit / pick a suggestion" flows (Wikipedia search ×2, arXiv), pagination that needs a scroll first (the model clicks the first
|
| 60 |
+
visible item instead), and Google Flights. v11s added 682 scripted scroll / search-submit / select trajectories (`code/finetune/rollouts.py`):
|
| 61 |
+
SELECT and DuckDuckGo improved, the scroll case did not — run-to-run variance on live sites (HN front page changes, DDG 50x pages) is larger
|
| 62 |
+
than the v10s ↔ v11s difference, so treat the two as equivalent.
|
| 63 |
+
|
| 64 |
+
### Teachers vs the fine-tuned student (80 held-out decisions, same element tables)
|
| 65 |
+
| decider | op acc | target top-1 | s / decision |
|
| 66 |
+
|---|---|---|---|
|
| 67 |
+
| Ternary-Bonsai-2-27B (local, thinking off) | 0.825 | 0.554 | 1.56 |
|
| 68 |
+
| Bonsai-27B, thinking budget 300 tokens | 0.861 | 0.603 | 4.7 |
|
| 69 |
+
| Bonsai-27B, thinking unrestricted | 0.625 | 0.474 | 11 |
|
| 70 |
+
| Qwen3-8B-AWQ (35 % of requests failed, survivors only) | 0.904 | 0.617 | 0.38 |
|
| 71 |
+
| **laya v11s (322M, this repo; full 2,734-case set)** | **0.890** | **0.623** | **0.021** |
|
| 72 |
+
|
| 73 |
+
A 27B general model with a short thinking budget matches the 322M fine-tuned student at 200× the latency; neither 8B nor 27B is a useful
|
| 74 |
+
DAgger teacher or System-2 fallback here. Further gains need a stronger teacher or more targeted trajectories.
|
| 75 |
+
|
| 76 |
+
## Files
|
| 77 |
+
|
| 78 |
+
```
|
| 79 |
+
v10/ ModernBERT-large 421M, format v2, head_max_len 768 (best held-out top-1)
|
| 80 |
+
v10s/ mmBERT-base 322M, format v3, head_max_len 768 (17–23 ms per step; best on the live suite)
|
| 81 |
+
v11s/ v10s data + 682 scripted scroll / search-submit / select trajectories (equivalent to v10s within noise)
|
| 82 |
+
code/ finetune pipeline, laya systemone server, task suite, TileLang kernels, jev-ultrafast patch
|
| 83 |
+
results/ per-run suite JSONs and logs behind every number above
|
| 84 |
+
```
|
| 85 |
+
Each checkpoint is a laya checkpoint directory (`model.safetensors`, `encoder/`, `tokenizer/`, `rl_agent_config.json`); the config records
|
| 86 |
+
`laya_fmt` and `head_max_len_train` so the server applies the matching input format automatically.
|
| 87 |
+
|
| 88 |
+
## Use
|
| 89 |
+
|
| 90 |
+
```bash
|
| 91 |
+
huggingface-cli download cklxx/laya-browser --local-dir laya-browser
|
| 92 |
+
cd laya-browser/code && uv sync --extra fast # pinned uv.lock (Python 3.12, torch 2.11, tilelang 0.1.14)
|
| 93 |
+
uv run python verify.py v10s # downloads v10s if needed, answers one recorded browser step
|
| 94 |
+
uv run python verify.py v10s --fast # same through the TileLang fast path
|
| 95 |
+
```
|
| 96 |
+
Verified from a clean environment on 2026-09-21 (RTX 4070 Ti SUPER): `TYPE_TEXT → [2] Search Wikipedia (searchbox)`, 35 ms per step
|
| 97 |
+
stock / 28 ms with the fast path on a 65-option, 2.5k-token step. Extras: `--extra data` (Mind2Web conversion, dataset eval),
|
| 98 |
+
`--extra browser` (live suite / crawling; also needs jev-ultrafast with `code/jev-ultrafast.patch` applied and a Chromium with
|
| 99 |
+
`--remote-debugging-port=9222`).
|
| 100 |
+
|
| 101 |
+
```python
|
| 102 |
+
import laya
|
| 103 |
+
agent = laya.load("laya-browser/v10s") # a local laya checkpoint dir
|
| 104 |
+
agent.cfg["head_max_len"] = agent.cfg["head_max_len_train"]
|
| 105 |
+
# state / questions exactly as jev-ultrafast's model.choose() builds them, after the format-v3 transform in code/apps/systemone_server.py
|
| 106 |
+
result = agent.predict(state, questions)
|
| 107 |
+
```
|
| 108 |
+
|
| 109 |
+
As a TypeSafe replacement for jev-ultrafast:
|
| 110 |
+
|
| 111 |
+
```bash
|
| 112 |
+
# in code/: laya systemone-compatible server (format transform + optional gating + DAgger logging)
|
| 113 |
+
python apps/systemone_server.py 8791 /path/to/laya-browser/v10s 999
|
| 114 |
+
# in jev-ultrafast (apply code/jev-ultrafast.patch): TYPESAFE_BASE_URL=http://127.0.0.1:8791
|
| 115 |
+
```
|
| 116 |
+
|
| 117 |
+
`code/apps/browser_suite.py` runs the 16-task real-browser suite with automatic outcome checks (`REPEATS=3`).
|
| 118 |
+
|
| 119 |
+
## Reproduce
|
| 120 |
+
|
| 121 |
+
`code/finetune/README.md` documents every step (crawl → reverse-generate goals → execute clicks for DONE → step-2 → Mind2Web conversion →
|
| 122 |
+
DAgger → build → train → calibrate → eval → suite) with the exact scripts (`run_v10.sh`, `run_v10s.sh`, `run_final.sh`) and all intermediate
|
| 123 |
+
numbers from v1 to v10s.
|
| 124 |
+
|
| 125 |
+
## GPU cost
|
| 126 |
+
|
| 127 |
+
v10s: ~0.65 GB weights, ~1.5 GB VRAM resident with CUDA graphs, 17–23 ms per 3-question browser step, 3 ms for a single-question call.
|
| 128 |
+
v10: ~0.85 GB weights, ~1.8 GB VRAM, 41–50 ms per step. The Qwen text helper (only needed for TYPE_TEXT values) is separate.
|
| 129 |
+
|
| 130 |
+
## License
|
| 131 |
+
|
| 132 |
+
Apache-2.0, same as laya. Mind2Web is used under its own license for training only.
|
assets/laya_browser_demo.gif
ADDED
|
Git LFS Details
|
assets/laya_browser_demo.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7eaddd1fd8ac8e49ef219a2224b51800f5d7153e0493e42d5e51ab96ab9a9c0c
|
| 3 |
+
size 888825
|
code/.python-version
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
3.12
|
code/README.md
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Laya 本机加速 + 应用
|
| 2 |
+
|
| 3 |
+
[convaiinnovations/laya](https://huggingface.co/convaiinnovations/laya):非自回归“System 1”决策模型(ModernBERT/mmBERT 编码器 + 决策头),
|
| 4 |
+
一次前向同时回答多个 `choice / score / noul` 问题,输出校准概率,不生成文本。
|
| 5 |
+
|
| 6 |
+
## 环境
|
| 7 |
+
|
| 8 |
+
```fish
|
| 9 |
+
source env.sh # HF 镜像 + 清华 PyPI + 关闭代理
|
| 10 |
+
.venv/bin/python ... # Python 3.12, torch 2.11+cu130, tilelang 0.1.14
|
| 11 |
+
```
|
| 12 |
+
|
| 13 |
+
## 应用(apps/)
|
| 14 |
+
|
| 15 |
+
| 脚本 | 说明 |
|
| 16 |
+
|---|---|
|
| 17 |
+
| `apps/inbox_triage.py [正文]` | 多语言工单分诊:部门 / 紧急度 / 流失风险 / 情绪 |
|
| 18 |
+
| `apps/moderator.py [file]` | 评论审核台:有害 / 垃圾 / 主题 / 严重度,按风险排序 |
|
| 19 |
+
| `apps/hn_radar.py [N]` | 拉 Hacker News 热帖实时打标签 |
|
| 20 |
+
| `apps/web_console.py [port]` | 零依赖 Web 决策台,自定义问题 JSON,实时概率条 |
|
| 21 |
+
|
| 22 |
+
在任何应用里加两行即可启用加速:
|
| 23 |
+
```python
|
| 24 |
+
import sys; sys.path.insert(0, "kernels")
|
| 25 |
+
from fast_laya import accelerate; accelerate(agent)
|
| 26 |
+
```
|
| 27 |
+
|
| 28 |
+
## TileLang 加速(kernels/)
|
| 29 |
+
|
| 30 |
+
* `tl_kernels.py` — GEMM(+bias/act 融合)、GEMM+GEGLU 融合、残差+LayerNorm 融合、原地 RoPE、
|
| 31 |
+
带 padding 掩码与滑动窗口的 flash attention。行数 M 为运行时符号,每个 kernel 只编译一次。
|
| 32 |
+
* `fast_laya.py` — 用上述 kernel 重写整个编码器 + 决策头前向,bf16 权重常驻,按 (batch, L) 桶捕获 CUDA Graph。
|
| 33 |
+
`accelerate(agent)` 原地替换 `agent.model.forward`。
|
| 34 |
+
* `test_kernels.py` / `verify_and_bench.py` — 单 kernel 对拍、端到端数值一致性与延迟对比。
|
| 35 |
+
* `tune.py` — tile 参数扫描。
|
code/apps/BROWSER_AGENT.md
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# laya 作为浏览器 agent 的决策模型(替代 TypeSafe Jev)
|
| 2 |
+
|
| 3 |
+
[jev-ultrafast](https://github.com/browser-use/jev-ultrafast) 的 `/v1/systemone` 请求/响应格式与 `laya.predict(state, questions)` 完全一致,
|
| 4 |
+
所以本地 laya(+TileLang 加速)可以直接替代 TypeSafe API。克隆在 `../jev-ultrafast`,`model.py` 已打补丁支持
|
| 5 |
+
`TYPESAFE_BASE_URL` 和 `TEXT_MODEL_EXTRA_JSON`。
|
| 6 |
+
|
| 7 |
+
```fish
|
| 8 |
+
# 1. 无头 Chromium(远程调试端口 9222)
|
| 9 |
+
chromium --headless=new --remote-debugging-port=9222 --user-data-dir=/tmp/chrome-profile about:blank &
|
| 10 |
+
# 2. 本地 laya systemone 服务(typed | multilingual | english;第 3 个参数是选项分块宽度)
|
| 11 |
+
source env.sh; .venv/bin/python apps/systemone_server.py 8791 typed 12 &
|
| 12 |
+
# 3. TYPE_TEXT 用本地 sglang + Qwen(可选)
|
| 13 |
+
HF_HUB_OFFLINE=1 ~/sglang-venv/bin/python -m sglang.launch_server --model-path Qwen/Qwen3-8B-AWQ --port 30000 \
|
| 14 |
+
--mem-fraction-static 0.5 --context-length 8192 --reasoning-parser qwen3 &
|
| 15 |
+
# 4. 跑任务
|
| 16 |
+
cd ../jev-ultrafast; .venv/bin/python ../laya/apps/browser_task.py "https://en.wikipedia.org/wiki/Main_Page" "Click the 'Random article' link."
|
| 17 |
+
# 诊断:真实元素表上直接问 laya 该点哪个(格式 jev|compact,后面是打乱平均次数)
|
| 18 |
+
.venv/bin/python ../laya/apps/browser_diag.py compact 4
|
| 19 |
+
```
|
| 20 |
+
|
| 21 |
+
`systemone_server.py` 做了两件适配:把 jev 的 dict 型元素描述压成一行字符串;超过 N 个选项的 choice 问题
|
| 22 |
+
分块粗筛 + 决赛(两次前向),53 个元素约 150 ms。
|
| 23 |
+
|
| 24 |
+
## 结论(2026-09-20,零样本,Wikipedia 首页 46 个可点元素)
|
| 25 |
+
|
| 26 |
+
| checkpoint | jev 原格式 top-1 | 精简格式 top-1 | 精简+4 次打乱平均 top-1 | 平均排名 |
|
| 27 |
+
|---|---|---|---|---|
|
| 28 |
+
| laya (english) | 0/3 | 0/3 | 1/3 | 2.7 / 46 |
|
| 29 |
+
| laya-multilingual | 0/3 | 0/3 | 0/3 | 5.3 / 46 |
|
| 30 |
+
| laya-typed-decisions | 0/3 | 1/3 | 1/3 | 3.7 / 46 |
|
| 31 |
+
|
| 32 |
+
管线(Chromium → browser-harness → jev 循环 → 本地 laya)跑通且每步 60-150 ms,但三个 checkpoint 零样本都
|
| 33 |
+
无法可靠选中目标元素(强烈偏向第 1 个选项),端到端任务全部失败。这与 laya 自己的说明一致:typed-decisions
|
| 34 |
+
零样本 0.36,需要针对任务微调。要真正替代 Jev,下一步是用 Qwen 当 teacher 在录制的页面状态上生成决策数据,微调 laya。
|
code/apps/browser_diag.py
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Direct diagnostic: can laya pick the right element on a real Wikipedia element table, under different prompt formats?"""
|
| 2 |
+
import json, os, random, sys, time
|
| 3 |
+
import httpx
|
| 4 |
+
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast")
|
| 5 |
+
os.environ["BU_CDP_URL"] = "http://127.0.0.1:9222"
|
| 6 |
+
from jev_ultrafast.browser import Browser
|
| 7 |
+
from jev_ultrafast.model import action_space
|
| 8 |
+
from jev_ultrafast.questions import NEXT_ACTION, TARGET
|
| 9 |
+
|
| 10 |
+
S1 = "http://127.0.0.1:8791/v1/systemone"
|
| 11 |
+
page = None
|
| 12 |
+
def get_page():
|
| 13 |
+
global page
|
| 14 |
+
if page is None:
|
| 15 |
+
b = Browser("https://en.wikipedia.org/wiki/Main_Page"); page = b.observe(screenshot=False); b.close()
|
| 16 |
+
return page
|
| 17 |
+
|
| 18 |
+
GOALS = [("Click the 'Random article' link in the navigation.", "Random article"),
|
| 19 |
+
("Open the site's search box so a query can be typed.", "Search"),
|
| 20 |
+
("Log in to Wikipedia.", "Log in"),
|
| 21 |
+
("Open the Talk page for the main page.", "Talk"),
|
| 22 |
+
("Read the community portal.", "Community portal")]
|
| 23 |
+
|
| 24 |
+
def ask(state, questions):
|
| 25 |
+
r = httpx.post(S1, json={"model": "x", "state": state, "questions": questions}, timeout=120).json()
|
| 26 |
+
return r["answers"]
|
| 27 |
+
|
| 28 |
+
def run(fmt, K=1):
|
| 29 |
+
p = get_page(); elements, targets, controls = action_space(p["actions"])
|
| 30 |
+
cands = targets["CLICK"]
|
| 31 |
+
hits, ranks = 0, []
|
| 32 |
+
for goal, want in GOALS:
|
| 33 |
+
gold = [i for i, a in cands.items() if a["label"].split(" → ")[0].strip().lower() == want.lower()]
|
| 34 |
+
if not gold: print(" no gold for", want); continue
|
| 35 |
+
if fmt == "jev":
|
| 36 |
+
crit = {i: {"element": f"[{i}] {a['label']}", "current_value": a.get("current_value", a.get("value", "")), **{k: a[k] for k in ("role", "checked", "selected", "expanded") if k in a}} for i, a in cands.items()}
|
| 37 |
+
state = {"page": {k: p[k] for k in ("url", "title", "text")}, "elements": elements, "recent_actions": []}
|
| 38 |
+
ins = {"goal": goal, "operation": "CLICK", "rules": [NEXT_ACTION, TARGET]}
|
| 39 |
+
else:
|
| 40 |
+
crit = {i: a["label"].split(" → ")[0][:60] for i, a in cands.items()}
|
| 41 |
+
state = {"goal": goal, "page_title": p["title"], "url": p["url"]}
|
| 42 |
+
ins = f"Goal: {goal} Which element should be clicked next?"
|
| 43 |
+
probs = {i: 0.0 for i in crit}
|
| 44 |
+
for k in range(K):
|
| 45 |
+
keys = list(crit); random.Random(k).shuffle(keys) if K > 1 else None
|
| 46 |
+
a = ask(state, {"t": {"type": "choice", "instructions": ins, "criteria": {i: crit[i] for i in keys}}})["t"]
|
| 47 |
+
for i, v in a["probabilities"].items(): probs[i] += v / K
|
| 48 |
+
order = sorted(probs, key=probs.get, reverse=True)
|
| 49 |
+
rank = min(order.index(g) for g in gold) + 1
|
| 50 |
+
ranks.append(rank); hits += rank == 1
|
| 51 |
+
print(f" {want:18s} gold={gold} top={order[:3]} p_gold={max(probs[g] for g in gold):.3f} rank={rank}")
|
| 52 |
+
print(f" => top1 {hits}/{len(ranks)} mean rank {sum(ranks)/len(ranks):.1f} of {len(cands)}")
|
| 53 |
+
|
| 54 |
+
fmt = sys.argv[1]; K = int(sys.argv[2]) if len(sys.argv) > 2 else 1
|
| 55 |
+
print(f"== format={fmt} permutations={K} variant={httpx.get('http://127.0.0.1:8791/').json()['variant']}")
|
| 56 |
+
run(fmt, K)
|
code/apps/browser_suite.py
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Real-task suite for the browser agent with automatic outcome checks (URL / title / page text).
|
| 2 |
+
|
| 3 |
+
python apps/browser_suite.py [name-filter] (services: chromium 9222, laya systemone 8791, sglang 30000)
|
| 4 |
+
|
| 5 |
+
Prints one line per task: PASS/FAIL, steps, wall time, and a summary table.
|
| 6 |
+
"""
|
| 7 |
+
import json, os, re, sys, time
|
| 8 |
+
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast")
|
| 9 |
+
os.environ.update(BU_CDP_URL="http://127.0.0.1:9222", TYPESAFE_BASE_URL="http://127.0.0.1:8791", TYPESAFE_API_KEY="local",
|
| 10 |
+
TEXT_MODEL_API_KEY="local", TEXT_MODEL_BASE_URL="http://127.0.0.1:30000/v1", TEXT_MODEL="Qwen/Qwen3-8B-AWQ",
|
| 11 |
+
TEXT_MODEL_EXTRA_JSON='{"chat_template_kwargs": {"enable_thinking": false}}')
|
| 12 |
+
from jev_ultrafast import Agent
|
| 13 |
+
|
| 14 |
+
# (name, url, goal, check(url, title, text) -> bool)
|
| 15 |
+
TASKS = [
|
| 16 |
+
# NOTE: Wikipedia's 'Random article' link sits in a collapsed menu that the DOM reader never observes -> search task instead
|
| 17 |
+
("wiki-einstein", "https://en.wikipedia.org/wiki/Main_Page", "Open the Wikipedia article about Albert Einstein.",
|
| 18 |
+
lambda u, t, x: "Albert_Einstein" in u),
|
| 19 |
+
("wiki-search", "https://en.wikipedia.org/wiki/Main_Page", "Search Wikipedia for 'Python programming language' and open the article about the Python language.",
|
| 20 |
+
lambda u, t, x: "Python" in t),
|
| 21 |
+
("hn-new", "https://news.ycombinator.com/", "Open the 'new' page that lists the newest submissions.",
|
| 22 |
+
lambda u, t, x: u.rstrip("/").endswith("/newest")),
|
| 23 |
+
("hn-login-page", "https://news.ycombinator.com/", "Go to the login page.",
|
| 24 |
+
lambda u, t, x: "login" in u),
|
| 25 |
+
("gh-issues", "https://github.com/tile-ai/tilelang", "Open the Issues tab of this repository.",
|
| 26 |
+
lambda u, t, x: "/issues" in u),
|
| 27 |
+
("py-downloads", "https://www.python.org/", "Go to the Downloads page.",
|
| 28 |
+
lambda u, t, x: "/downloads" in u),
|
| 29 |
+
("books-travel", "https://books.toscrape.com/", "Open the 'Travel' category.",
|
| 30 |
+
lambda u, t, x: "travel" in u),
|
| 31 |
+
("books-open-book", "https://books.toscrape.com/", "Open the product page of the book 'A Light in the Attic'.",
|
| 32 |
+
lambda u, t, x: "a-light-in-the-attic" in u),
|
| 33 |
+
# NOTE: jev's snapshot.js hides password fields by design, so password logins are impossible in this framework.
|
| 34 |
+
("internet-dropdown", "https://the-internet.herokuapp.com/dropdown", "Select 'Option 2' in the dropdown.",
|
| 35 |
+
lambda u, t, x: False), # checked via page state below
|
| 36 |
+
("internet-checkbox", "https://the-internet.herokuapp.com/checkboxes", "Tick the first checkbox.",
|
| 37 |
+
lambda u, t, x: False),
|
| 38 |
+
("books-page2", "https://books.toscrape.com/", "Go to page 2 of the catalogue.",
|
| 39 |
+
lambda u, t, x: "page-2" in u),
|
| 40 |
+
("quotes-tag-love", "https://quotes.toscrape.com/", "Show the quotes tagged 'love'.",
|
| 41 |
+
lambda u, t, x: "/tag/love" in u),
|
| 42 |
+
("hn-past", "https://news.ycombinator.com/", "Open the 'past' page (front pages from previous days).",
|
| 43 |
+
lambda u, t, x: "/front" in u),
|
| 44 |
+
("ddg-search", "https://duckduckgo.com/", "Search for 'tilelang github' and show the results.",
|
| 45 |
+
lambda u, t, x: "q=" in u and "tilelang" in u.lower()),
|
| 46 |
+
("arxiv-search", "https://arxiv.org/", "Search arXiv for 'flash attention' papers and show the results list.",
|
| 47 |
+
lambda u, t, x: "search" in u and "flash" in u.lower()),
|
| 48 |
+
("flights", "https://www.google.com/travel/flights?hl=en", "Find one-way flights from Zurich to London on September 28, 2026, for one adult in economy. Stop when matching flight options are visible.",
|
| 49 |
+
lambda u, t, x: ("ZRH" in x or "Zurich" in x or "Zürich" in x) and "London" in x and re.search(r"\b\d{1,2}:\d{2}\b", x) is not None and "one way" in x.lower()),
|
| 50 |
+
]
|
| 51 |
+
|
| 52 |
+
def run(name, url, goal, check, max_steps=20):
|
| 53 |
+
t0 = time.time(); steps = 0; status = "error"; page = None
|
| 54 |
+
try:
|
| 55 |
+
with Agent(url, goal) as agent:
|
| 56 |
+
page = agent.state["page"]
|
| 57 |
+
for state in agent.run():
|
| 58 |
+
steps = len(state["history"]); status = state["status"]; page = state["page"]
|
| 59 |
+
if steps >= max_steps: break
|
| 60 |
+
time.sleep(1.5) # let a slow navigation (GitHub's Turbo, etc.) land before judging
|
| 61 |
+
try:
|
| 62 |
+
page = agent.browser.observe(screenshot=False)
|
| 63 |
+
except Exception:
|
| 64 |
+
pass
|
| 65 |
+
except Exception as e:
|
| 66 |
+
status = f"error:{type(e).__name__}"
|
| 67 |
+
wall = time.time() - t0
|
| 68 |
+
ok = bool(page and check(page["url"], page["title"], page.get("text", "")))
|
| 69 |
+
if page and name == "internet-dropdown":
|
| 70 |
+
ok = any(a.get("kind") == "select" and str(a.get("current_value", "")).strip() in ("2", "Option 2") for a in page["actions"])
|
| 71 |
+
if page and name == "internet-checkbox":
|
| 72 |
+
cbs = [a for a in page["actions"] if a.get("role") == "checkbox"]
|
| 73 |
+
ok = bool(cbs) and bool(cbs[0].get("checked"))
|
| 74 |
+
return ok, steps, status, wall, (page or {}).get("url", "")
|
| 75 |
+
|
| 76 |
+
if __name__ == "__main__":
|
| 77 |
+
flt = sys.argv[1] if len(sys.argv) > 1 else ""
|
| 78 |
+
repeats = int(os.environ.get("REPEATS", "1"))
|
| 79 |
+
rows = []
|
| 80 |
+
for name, url, goal, check in TASKS:
|
| 81 |
+
if flt and flt not in name: continue
|
| 82 |
+
for rep in range(repeats):
|
| 83 |
+
ok, steps, status, wall, final = run(name, url, goal, check)
|
| 84 |
+
rows.append((name, ok, steps, status, wall))
|
| 85 |
+
print(f"{'PASS' if ok else 'FAIL'} {name:16s} steps={steps:2d} status={status:9s} {wall:5.1f}s {final[:70]}", flush=True)
|
| 86 |
+
n = sum(r[1] for r in rows)
|
| 87 |
+
print(f"\n== {n}/{len(rows)} passed ({100*n/len(rows):.0f}%, {repeats} run(s) per task) | median wall {sorted(r[4] for r in rows)[len(rows)//2]:.1f}s")
|
| 88 |
+
if repeats > 1:
|
| 89 |
+
per = {}
|
| 90 |
+
for r in rows: per.setdefault(r[0], []).append(r[1])
|
| 91 |
+
print(" per task: " + " ".join(f"{k}={sum(v)}/{len(v)}" for k, v in per.items()))
|
| 92 |
+
json.dump([dict(zip(("name", "pass", "steps", "status", "wall"), r)) for r in rows], open(os.environ.get("SUITE_OUT", "/tmp/suite.json"), "w"), indent=1)
|
code/apps/browser_task.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os, sys, json, time
|
| 2 |
+
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast")
|
| 3 |
+
os.environ.update(BU_CDP_URL="http://127.0.0.1:9222", TYPESAFE_BASE_URL="http://127.0.0.1:8791", TYPESAFE_API_KEY="local",
|
| 4 |
+
TEXT_MODEL_API_KEY="local", TEXT_MODEL_BASE_URL="http://127.0.0.1:30000/v1", TEXT_MODEL="Qwen/Qwen3-4B-AWQ",
|
| 5 |
+
TEXT_MODEL_EXTRA_JSON='{"chat_template_kwargs": {"enable_thinking": false}}')
|
| 6 |
+
from jev_ultrafast import Agent
|
| 7 |
+
url = sys.argv[1]; goal = sys.argv[2]
|
| 8 |
+
t = time.time()
|
| 9 |
+
with Agent(url, goal) as agent:
|
| 10 |
+
print("elements on first page:", len(agent.snapshot()["elements"]), "| title:", agent.state["page"]["title"])
|
| 11 |
+
for state in agent.run():
|
| 12 |
+
h = state["history"][-1] if state["history"] else None
|
| 13 |
+
d = state["decisions"][-1] if state["decisions"] else None
|
| 14 |
+
print(f"{state['elapsed_ms']:6d} ms status={state['status']:9s} op={d['operation'] if d else None:9s} "
|
| 15 |
+
f"conf={d['confidence'] if d else 0:.2f} model={d['latency_ms'] if d else 0:4d}ms "
|
| 16 |
+
f"action={h['action'][:60] if h else None} text={h['text'] if h else None}")
|
| 17 |
+
if len(state["history"]) > 25: break
|
| 18 |
+
print("FINAL:", state["status"], "| url:", state["page"]["url"], "| title:", state["page"]["title"], "| wall", round(time.time() - t, 1), "s")
|
code/apps/common.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Shared helpers: model loading + pretty printing."""
|
| 2 |
+
import os, sys, time
|
| 3 |
+
os.environ.setdefault("USE_TF", "0")
|
| 4 |
+
os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
|
| 5 |
+
|
| 6 |
+
_AGENTS = {}
|
| 7 |
+
|
| 8 |
+
def get_agent(variant="english"):
|
| 9 |
+
"""variant: english | multilingual | typed"""
|
| 10 |
+
if variant in _AGENTS:
|
| 11 |
+
return _AGENTS[variant]
|
| 12 |
+
import laya
|
| 13 |
+
t = time.time()
|
| 14 |
+
if os.path.isdir(variant):
|
| 15 |
+
a = laya.load(variant)
|
| 16 |
+
elif variant == "english":
|
| 17 |
+
# download only the root checkpoint (the other subfolders are several GB each)
|
| 18 |
+
from huggingface_hub import snapshot_download
|
| 19 |
+
path = snapshot_download("convaiinnovations/laya", ignore_patterns=["multilingual/*", "typed-decisions/*"])
|
| 20 |
+
a = laya.load(path)
|
| 21 |
+
elif variant == "multilingual":
|
| 22 |
+
a = laya.load("convaiinnovations/laya", subfolder="multilingual")
|
| 23 |
+
else:
|
| 24 |
+
from huggingface_hub import snapshot_download
|
| 25 |
+
path = snapshot_download("convaiinnovations/laya", allow_patterns=["typed-decisions/*"])
|
| 26 |
+
a = laya.load(path, subfolder="typed-decisions")
|
| 27 |
+
print(f"[laya] loaded {variant} in {time.time()-t:.1f}s", file=sys.stderr)
|
| 28 |
+
if os.environ.get("LAYA_FAST", "1") == "1":
|
| 29 |
+
try:
|
| 30 |
+
sys.path.insert(0, os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "kernels"))
|
| 31 |
+
try:
|
| 32 |
+
from fast_laya import accelerate # laya/kernels layout
|
| 33 |
+
accelerate(a)
|
| 34 |
+
except ImportError:
|
| 35 |
+
from fast import FastLaya # code/kernels layout (this repo)
|
| 36 |
+
fl = FastLaya(a.model, max_len=a.cfg.get("max_len", 1024)); a._fast = fl; a.model.forward = fl.forward
|
| 37 |
+
print("[laya] TileLang fast path enabled (LAYA_FAST=0 to disable)", file=sys.stderr)
|
| 38 |
+
except Exception as e:
|
| 39 |
+
print(f"[laya] fast path unavailable: {e}", file=sys.stderr)
|
| 40 |
+
_AGENTS[variant] = a
|
| 41 |
+
return a
|
| 42 |
+
|
| 43 |
+
def bar(p, width=20):
|
| 44 |
+
n = int(round(p * width))
|
| 45 |
+
return "█" * n + "░" * (width - n)
|
| 46 |
+
|
| 47 |
+
def show(result, indent=" "):
|
| 48 |
+
"""Pretty-print a laya predict() result."""
|
| 49 |
+
for name, a in result["answers"].items():
|
| 50 |
+
t = a["type"]
|
| 51 |
+
if t == "choice":
|
| 52 |
+
print(f"{indent}{name}: {a['choice']} (conf {a['confidence']:.2f})")
|
| 53 |
+
for k, v in sorted(a["probabilities"].items(), key=lambda kv: -kv[1]):
|
| 54 |
+
print(f"{indent} {bar(v)} {v:5.2f} {k}")
|
| 55 |
+
elif t == "score":
|
| 56 |
+
print(f"{indent}{name}: score={a['score']:.2f} (conf {a['confidence']:.2f})")
|
| 57 |
+
for k, v in a["probabilities"].items():
|
| 58 |
+
print(f"{indent} {bar(v)} {v:5.2f} {a['legend'][k]}")
|
| 59 |
+
else:
|
| 60 |
+
p = a["noul"]
|
| 61 |
+
print(f"{indent}{name}: P(true)={p:.2f} {bar(p)}")
|
code/apps/fast_batch.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""One-shot batched predict for laya: tokenize the shared state once, build every question's sequence from cached ids,
|
| 2 |
+
single forward, vectorised post-processing. Same outputs as agent.predict() (up to float rounding).
|
| 3 |
+
|
| 4 |
+
from fast_batch import predict_fast, profile_step
|
| 5 |
+
"""
|
| 6 |
+
import json, time
|
| 7 |
+
import numpy as np, torch
|
| 8 |
+
from laya.common import QTYPES, collate_items, confidence_from_probs, render_options, serialize_state, temp_bucket
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
def build_items(agent, state, questions):
|
| 12 |
+
tok = agent.tok
|
| 13 |
+
max_len, head_max_len = agent.cfg.get("max_len", 512), agent.cfg.get("head_max_len", 192)
|
| 14 |
+
mask_tok, mask_id = tok.mask_token, tok.mask_token_id
|
| 15 |
+
st_ids = None # shared state tokens, computed lazily once
|
| 16 |
+
items, meta = [], []
|
| 17 |
+
for qid, qdef in questions.items():
|
| 18 |
+
q = agent._to_internal(qdef)
|
| 19 |
+
opts = render_options(q)
|
| 20 |
+
ins = str(q["ins"]).replace(mask_tok, " ")
|
| 21 |
+
head_ids = tok("%s question: %s" % (q["t"], ins), add_special_tokens=False)["input_ids"]
|
| 22 |
+
opt_txt = [" " + o.replace(mask_tok, " ") for o in opts]
|
| 23 |
+
opt_enc = tok(opt_txt, add_special_tokens=False)["input_ids"] # one batched tokenizer call for all options
|
| 24 |
+
opt_ids = [[mask_id] + o[:48] for o in opt_enc]
|
| 25 |
+
opt_budget = head_max_len - sum(len(o) for o in opt_ids)
|
| 26 |
+
if opt_budget < 16:
|
| 27 |
+
per = max(4, (head_max_len - 16) // max(1, len(opt_ids)))
|
| 28 |
+
opt_ids = [o[:per] for o in opt_ids]
|
| 29 |
+
opt_budget = head_max_len - sum(len(o) for o in opt_ids)
|
| 30 |
+
head_ids = head_ids[: max(8, opt_budget)]
|
| 31 |
+
ids = [tok.cls_token_id] + head_ids + [tok.sep_token_id]
|
| 32 |
+
markers = []
|
| 33 |
+
for o in opt_ids:
|
| 34 |
+
markers.append(len(ids)); ids.extend(o)
|
| 35 |
+
ids.append(tok.sep_token_id)
|
| 36 |
+
room = max(0, max_len - len(ids) - 1)
|
| 37 |
+
if st_ids is None:
|
| 38 |
+
st_ids = tok(serialize_state(state).replace(mask_tok, " "), add_special_tokens=False)["input_ids"]
|
| 39 |
+
ids = (ids + st_ids[:room] + [tok.sep_token_id])[:max_len]
|
| 40 |
+
markers = [m for m in markers if m < max_len]
|
| 41 |
+
if len(markers) != len(opts):
|
| 42 |
+
raise ValueError("question %r options exceed head_max_len=%d" % (qid, head_max_len))
|
| 43 |
+
items.append({"ids": ids, "markers": markers, "qtype": QTYPES[q["t"]]})
|
| 44 |
+
meta.append((qid, q, len(markers)))
|
| 45 |
+
return items, meta
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
@torch.no_grad()
|
| 49 |
+
def predict_fast(agent, state, questions, timing=None):
|
| 50 |
+
t0 = time.perf_counter()
|
| 51 |
+
items, meta = build_items(agent, state, questions)
|
| 52 |
+
b = collate_items([items], agent.tok.pad_token_id)
|
| 53 |
+
t1 = time.perf_counter()
|
| 54 |
+
dev = agent.device
|
| 55 |
+
with torch.autocast(device_type=dev.type, dtype=agent.dtype, enabled=dev.type == "cuda"):
|
| 56 |
+
logits, act = agent.model(b["input_ids"].to(dev, non_blocking=True), b["attention_mask"].to(dev, non_blocking=True),
|
| 57 |
+
b["marker_pos"].to(dev, non_blocking=True), b["marker_mask"].to(dev, non_blocking=True), b["qtype"].to(dev, non_blocking=True))
|
| 58 |
+
logits = logits.float().cpu().numpy(); act = torch.softmax(act.float(), -1).cpu().numpy()
|
| 59 |
+
t2 = time.perf_counter()
|
| 60 |
+
answers = {}
|
| 61 |
+
for r, (qid, q, k) in enumerate(meta):
|
| 62 |
+
qt = QTYPES[q["t"]]
|
| 63 |
+
t_scale = agent.temperature_by_options.get(temp_bucket(qt, k), agent.temperature[qt])
|
| 64 |
+
z = logits[r, :k] / max(1e-3, float(t_scale)); p = np.exp(z - z.max()); p /= p.sum()
|
| 65 |
+
conf = round(confidence_from_probs(p, k), 4); ext = {"act_probability": round(float(act[r, 0]), 4)}
|
| 66 |
+
if q["t"] == "choice":
|
| 67 |
+
keys = list(q["crit"].keys())
|
| 68 |
+
answers[qid] = {"type": "choice", "choice": keys[int(p.argmax())], "probabilities": {kk: round(float(v), 4) for kk, v in zip(keys, p)}, "confidence": conf, "action": ext}
|
| 69 |
+
elif q["t"] == "score":
|
| 70 |
+
answers[qid] = {"type": "score", "score": round(float((np.arange(k) * p).sum()), 4), "legend": {str(i): c for i, c in enumerate(q["crit"])},
|
| 71 |
+
"probabilities": {str(i): round(float(v), 4) for i, v in enumerate(p)}, "confidence": conf, "action": ext}
|
| 72 |
+
else:
|
| 73 |
+
answers[qid] = {"type": "noul", "noul": round(float(p[1]), 4), "confidence": round(max(float(p[1]), 1 - float(p[1])), 4), "action": ext}
|
| 74 |
+
t3 = time.perf_counter()
|
| 75 |
+
if timing is not None:
|
| 76 |
+
timing.update(tokenize_ms=(t1 - t0) * 1e3, forward_ms=(t2 - t1) * 1e3, post_ms=(t3 - t2) * 1e3, tokens=int(b["attention_mask"].sum()))
|
| 77 |
+
return {"model": "laya-rl-agent", "answers": answers, "usage": {"input_tokens": int(b["attention_mask"].sum()), "output_tokens": 0}}
|
| 78 |
+
|
| 79 |
+
|
| 80 |
+
def profile_step(agent, state, questions, n=20):
|
| 81 |
+
"""Compare agent.predict vs predict_fast on one recorded browser step."""
|
| 82 |
+
for _ in range(3): agent.predict(state, questions); predict_fast(agent, state, questions)
|
| 83 |
+
torch.cuda.synchronize(); t = time.perf_counter()
|
| 84 |
+
for _ in range(n): agent.predict(state, questions)
|
| 85 |
+
torch.cuda.synchronize(); slow = (time.perf_counter() - t) / n * 1e3
|
| 86 |
+
tm = {}; torch.cuda.synchronize(); t = time.perf_counter()
|
| 87 |
+
for _ in range(n): predict_fast(agent, state, questions, tm)
|
| 88 |
+
torch.cuda.synchronize(); fast = (time.perf_counter() - t) / n * 1e3
|
| 89 |
+
return {"predict_ms": slow, "predict_fast_ms": fast, **tm}
|
code/apps/hn_radar.py
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Hacker News 雷达:拉取实时热帖标题,用 laya 给每条打 主题/是否硬核技术/是否值得读 标签。
|
| 2 |
+
用法: python apps/hn_radar.py [N=30]
|
| 3 |
+
"""
|
| 4 |
+
import os, sys, json, time, urllib.request
|
| 5 |
+
from common import get_agent, bar
|
| 6 |
+
|
| 7 |
+
QUESTIONS = {
|
| 8 |
+
"topic": {"type": "choice", "instructions": "What is this Hacker News story about?",
|
| 9 |
+
"criteria": {"ai": "machine learning, LLMs, models", "systems": "OS, compilers, databases, hardware, GPUs",
|
| 10 |
+
"web": "frontend, browsers, web frameworks", "security": "vulnerabilities, hacking, privacy",
|
| 11 |
+
"business": "startups, funding, layoffs, policy", "science": "physics, biology, space, math",
|
| 12 |
+
"other": "anything else"}},
|
| 13 |
+
"technical_depth": {"type": "score", "instructions": "How technically deep is this likely to be?",
|
| 14 |
+
"criteria": ["fluff", "medium", "deep dive"]},
|
| 15 |
+
"showhn": {"type": "noul", "instructions": "Is this a project someone built and is showing off?"},
|
| 16 |
+
}
|
| 17 |
+
|
| 18 |
+
def fetch(n):
|
| 19 |
+
"""One request to the HN Algolia API (front page)."""
|
| 20 |
+
r = json.load(urllib.request.urlopen(f"https://hn.algolia.com/api/v1/search?tags=front_page&hitsPerPage={n}", timeout=20))
|
| 21 |
+
return [{"title": h["title"], "url": h.get("url") or "", "score": h.get("points", 0)} for h in r["hits"] if h.get("title")]
|
| 22 |
+
|
| 23 |
+
def main():
|
| 24 |
+
n = int(sys.argv[1]) if len(sys.argv) > 1 else 30
|
| 25 |
+
print(f"fetching {n} HN top stories...")
|
| 26 |
+
items = fetch(n)
|
| 27 |
+
agent = get_agent(os.environ.get("LAYA_VARIANT", "multilingual"))
|
| 28 |
+
t = time.time()
|
| 29 |
+
res = [agent.predict({"title": it["title"], "url": it.get("url", "")}, QUESTIONS) for it in items]
|
| 30 |
+
dt = time.time() - t
|
| 31 |
+
print(f"classified {len(items)} in {dt*1000:.0f} ms\n")
|
| 32 |
+
by_topic = {}
|
| 33 |
+
for it, r in zip(items, res):
|
| 34 |
+
a = r["answers"]
|
| 35 |
+
by_topic.setdefault(a["topic"]["choice"], []).append((round(a["technical_depth"]["score"],1), a["showhn"]["noul"], it))
|
| 36 |
+
for topic, lst in sorted(by_topic.items(), key=lambda kv: -len(kv[1])):
|
| 37 |
+
print(f"## {topic} ({len(lst)})")
|
| 38 |
+
for depth, show_p, it in sorted(lst, key=lambda x: -(x[0] or 0)):
|
| 39 |
+
tag = " [show]" if show_p > 0.5 else ""
|
| 40 |
+
print(f" depth={depth} ↑{it.get('score',0):<4} {it['title'][:70]}{tag}")
|
| 41 |
+
print()
|
| 42 |
+
|
| 43 |
+
if __name__ == "__main__":
|
| 44 |
+
main()
|
code/apps/inbox_triage.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""多语言工单/邮件分诊:一次前向传播同时回答 部门 / 紧急度 / 流失风险 / 情绪。
|
| 2 |
+
用法: python apps/inbox_triage.py # 跑内置样例(中/英/日/西/印地语)
|
| 3 |
+
python apps/inbox_triage.py "你的邮件正文"
|
| 4 |
+
"""
|
| 5 |
+
import sys, json, time
|
| 6 |
+
from common import get_agent, show
|
| 7 |
+
|
| 8 |
+
QUESTIONS = {
|
| 9 |
+
"department": {"type": "choice", "instructions": "Which team should handle this message?",
|
| 10 |
+
"criteria": {"billing": "invoices, payments, refunds, double charge",
|
| 11 |
+
"technical": "bugs, crashes, outages, login failures",
|
| 12 |
+
"sales": "pricing, quotes, upgrades, enterprise plans",
|
| 13 |
+
"shipping": "delivery, tracking, lost or damaged package"}},
|
| 14 |
+
"urgency": {"type": "score", "instructions": "How urgent is this?",
|
| 15 |
+
"criteria": ["not urgent", "soon", "blocking"]},
|
| 16 |
+
"churn_risk": {"type": "noul", "instructions": "Does the user threaten to cancel or leave?"},
|
| 17 |
+
"sentiment": {"type": "choice", "instructions": "What is the customer's tone?",
|
| 18 |
+
"criteria": {"angry": "furious, threatening, insulting",
|
| 19 |
+
"frustrated": "annoyed but civil", "neutral": "matter of fact",
|
| 20 |
+
"happy": "grateful, positive"}},
|
| 21 |
+
}
|
| 22 |
+
|
| 23 |
+
SAMPLES = [
|
| 24 |
+
{"subject": "Duplicate charge on invoice 4411",
|
| 25 |
+
"body": "We were billed twice for March. Please refund the duplicate or we're moving to a competitor."},
|
| 26 |
+
{"subject": "登录一直失败", "body": "从昨天开始整个团队都登录不了后台,报 502,我们的上线被卡住了,急!"},
|
| 27 |
+
{"subject": "見積もりのお願い", "body": "エンタープライズプランの料金と年間契約の割引について教えてください。"},
|
| 28 |
+
{"subject": "Paquete perdido", "body": "El rastreo dice entregado pero no recibí nada. Llevo una semana esperando, estoy muy molesto."},
|
| 29 |
+
{"subject": "धन्यवाद", "body": "आपकी टीम ने मेरी समस्या बहुत जल्दी हल कर दी। बहुत बहुत धन्यवाद!"},
|
| 30 |
+
]
|
| 31 |
+
|
| 32 |
+
def main():
|
| 33 |
+
agent = get_agent("multilingual")
|
| 34 |
+
states = [{"subject": "(cli)", "body": " ".join(sys.argv[1:])}] if len(sys.argv) > 1 else SAMPLES
|
| 35 |
+
for s in states:
|
| 36 |
+
t = time.time()
|
| 37 |
+
r = agent.predict(s, QUESTIONS)
|
| 38 |
+
dt = (time.time() - t) * 1000
|
| 39 |
+
print(f"\n=== {s['subject']} [{dt:.0f} ms]\n {s['body'][:80]}")
|
| 40 |
+
show(r)
|
| 41 |
+
|
| 42 |
+
if __name__ == "__main__":
|
| 43 |
+
main()
|
code/apps/make_demo.py
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Record a short demo video of laya driving a real browser through jev-ultrafast.
|
| 2 |
+
|
| 3 |
+
python apps/make_demo.py out_dir (services: chromium :9222, laya systemone :8791; no text model needed)
|
| 4 |
+
|
| 5 |
+
Runs a few click-only tasks with frame recording, overlays goal / laya's decision / per-step latency, renders MP4 + GIF via ffmpeg.
|
| 6 |
+
"""
|
| 7 |
+
import base64, json, os, shutil, subprocess, sys, time
|
| 8 |
+
from pathlib import Path
|
| 9 |
+
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast")
|
| 10 |
+
os.environ.update(BU_CDP_URL="http://127.0.0.1:9222", TYPESAFE_BASE_URL="http://127.0.0.1:8791", TYPESAFE_API_KEY="local", TEXT_MODEL_API_KEY="local")
|
| 11 |
+
from PIL import Image, ImageDraw, ImageFont
|
| 12 |
+
from jev_ultrafast import Agent
|
| 13 |
+
|
| 14 |
+
TASKS = [
|
| 15 |
+
("https://books.toscrape.com/", "Open the 'Travel' category."),
|
| 16 |
+
("https://books.toscrape.com/", "Open the product page of the book 'A Light in the Attic'."),
|
| 17 |
+
("https://www.python.org/", "Go to the Downloads page."),
|
| 18 |
+
("https://quotes.toscrape.com/", "Show the quotes tagged 'love'."),
|
| 19 |
+
("https://news.ycombinator.com/", "Open the 'past' page (front pages from previous days)."),
|
| 20 |
+
("https://the-internet.herokuapp.com/checkboxes", "Tick the first checkbox."),
|
| 21 |
+
]
|
| 22 |
+
W, H = 1120, 780; BAR_T, BAR_B = 64, 96; FPS = 24
|
| 23 |
+
F_REG = "/usr/share/fonts/TTF/DejaVuSans.ttf"; F_BOLD = "/usr/share/fonts/TTF/DejaVuSans-Bold.ttf"; F_MONO = "/usr/share/fonts/TTF/DejaVuSansMono.ttf"
|
| 24 |
+
def font(n, bold=False): return ImageFont.truetype(F_BOLD if bold else F_REG, n)
|
| 25 |
+
def mono(n): return ImageFont.truetype(F_MONO, n)
|
| 26 |
+
BG, INK, MUTED, GREEN, BLUE = "#0f1115", "#e8e8e8", "#8b93a1", "#3ddc84", "#4f8cff"
|
| 27 |
+
|
| 28 |
+
def canvas(page_img=None):
|
| 29 |
+
im = Image.new("RGB", (W, BAR_T + H + BAR_B), BG)
|
| 30 |
+
if page_img is not None: im.paste(page_img.resize((W, H)), (0, BAR_T))
|
| 31 |
+
return im
|
| 32 |
+
|
| 33 |
+
def frame(page_img, goal, step_txt, decision_txt, lat_txt, elapsed_txt, status=None):
|
| 34 |
+
im = canvas(page_img); d = ImageDraw.Draw(im)
|
| 35 |
+
d.text((16, 12), "laya-browser", font=font(22, True), fill=GREEN); d.text((190, 16), "System-1 decision head · 322M · local RTX 4070", font=font(15), fill=MUTED)
|
| 36 |
+
d.text((16, 38), "goal: " + goal, font=font(16), fill=INK)
|
| 37 |
+
y = BAR_T + H + 12
|
| 38 |
+
d.text((16, y), step_txt, font=font(15, True), fill=MUTED)
|
| 39 |
+
d.text((16, y + 24), decision_txt, font=mono(19), fill=INK)
|
| 40 |
+
d.text((16, y + 54), lat_txt, font=mono(15), fill=BLUE); d.text((W - 200, y + 54), elapsed_txt, font=mono(15), fill=MUTED)
|
| 41 |
+
if status: d.text((W - 200, y), status, font=font(17, True), fill=GREEN if status == "DONE" else "#ff6b6b")
|
| 42 |
+
return im
|
| 43 |
+
|
| 44 |
+
def title_card(lines, sub):
|
| 45 |
+
im = canvas(); d = ImageDraw.Draw(im); y = 260
|
| 46 |
+
for i, l in enumerate(lines):
|
| 47 |
+
d.text((60, y), l, font=font(40 if i == 0 else 26, i == 0), fill=GREEN if i == 0 else INK); y += 64 if i == 0 else 40
|
| 48 |
+
y += 20
|
| 49 |
+
for l in sub: d.text((60, y), l, font=font(20), fill=MUTED); y += 32
|
| 50 |
+
return im
|
| 51 |
+
|
| 52 |
+
def run_task(url, goal, rec):
|
| 53 |
+
frames, decisions = [], []
|
| 54 |
+
with Agent(url, goal, record_dir=str(rec), screenshots=True) as agent:
|
| 55 |
+
t0 = time.time(); page0 = agent.state["page"]
|
| 56 |
+
frames.append((0, Image.open(rec / "000000.jpg").convert("RGB"), None))
|
| 57 |
+
while agent.state["status"] not in ("done", "blocked") and len(agent.state["history"]) < 8:
|
| 58 |
+
st = agent.command("tick")
|
| 59 |
+
d = st["decisions"][-1] if st["decisions"] else None
|
| 60 |
+
h = st["history"][-1] if st["history"] else None
|
| 61 |
+
lat = d["latency_ms"] if d else 0
|
| 62 |
+
label = (h["action"] if h and d and h.get("latency_ms") == lat else None)
|
| 63 |
+
dec = {"op": d["operation"] if d else "?", "target": label or (d.get("target") if d else ""), "conf": d["confidence"] if d else 0, "lat": lat,
|
| 64 |
+
"elapsed": st["elapsed_ms"], "status": st["status"]}
|
| 65 |
+
decisions.append(dec)
|
| 66 |
+
shots = sorted(p for p in rec.glob("*.jpg") if p.stem != "000000")
|
| 67 |
+
img = Image.open(shots[-1]).convert("RGB") if shots else frames[-1][1]
|
| 68 |
+
frames.append((st["elapsed_ms"], img, dec))
|
| 69 |
+
final = agent.state
|
| 70 |
+
return frames, decisions, final
|
| 71 |
+
|
| 72 |
+
def main():
|
| 73 |
+
out = Path(sys.argv[1]); shutil.rmtree(out, ignore_errors=True); (out / "frames").mkdir(parents=True)
|
| 74 |
+
n = 0
|
| 75 |
+
def emit(im, secs):
|
| 76 |
+
nonlocal n
|
| 77 |
+
for _ in range(int(secs * FPS)):
|
| 78 |
+
im.save(out / "frames" / f"{n:06d}.png"); n += 1
|
| 79 |
+
emit(title_card(["laya-browser", "one bidirectional encoder pass per step -> operation + target + calibrated confidence",
|
| 80 |
+
"no text generation, ~20 ms per decision on an RTX 4070, 1.5 GB VRAM"],
|
| 81 |
+
["fine-tuned from convaiinnovations/laya on crawled pages + Mind2Web + on-policy corrections",
|
| 82 |
+
"driving browser-use/jev-ultrafast through its TypeSafe-compatible /v1/systemone API"]), 3.5)
|
| 83 |
+
totals = {"steps": 0, "lat": [], "wall": 0.0, "tasks_ok": 0}
|
| 84 |
+
for i, (url, goal) in enumerate(TASKS):
|
| 85 |
+
rec = out / f"rec{i}"; rec.mkdir()
|
| 86 |
+
try:
|
| 87 |
+
t = time.time(); frames, decisions, final = run_task(url, goal, rec); wall = time.time() - t
|
| 88 |
+
except Exception as e:
|
| 89 |
+
print("task failed:", goal, type(e).__name__, str(e)[:60]); continue
|
| 90 |
+
ok = final["status"] == "done"
|
| 91 |
+
totals["steps"] += len(decisions); totals["lat"] += [d["lat"] for d in decisions]; totals["wall"] += wall; totals["tasks_ok"] += ok
|
| 92 |
+
print(f"{goal[:50]:50s} steps={len(decisions)} status={final['status']} wall={wall:.1f}s decisions={[ (d['op'], d['lat']) for d in decisions]}", flush=True)
|
| 93 |
+
# observed page, then each decision: show the page it decided on with the decision, then the resulting page
|
| 94 |
+
emit(frame(frames[0][1], goal, "step 0 · observe page", "laya reads the element table ...", "", "t = 0 ms"), 1.2)
|
| 95 |
+
for k, (ms, img, dec) in enumerate(frames[1:], 1):
|
| 96 |
+
prev = frames[k - 1][1]
|
| 97 |
+
dtxt = f"{dec['op']} -> {str(dec['target'])[:48]}" if dec["op"] not in ("DONE", "BLOCKED") else dec["op"]
|
| 98 |
+
emit(frame(prev, goal, f"step {k} · decide", dtxt, f"laya: {dec['lat']} ms (confidence {dec['conf']:.2f})", f"t = {dec['elapsed']} ms"), 1.3)
|
| 99 |
+
if dec["op"] not in ("DONE", "BLOCKED"):
|
| 100 |
+
emit(frame(img, goal, f"step {k} · executed", dtxt, f"laya: {dec['lat']} ms", f"t = {dec['elapsed']} ms"), 1.0)
|
| 101 |
+
else:
|
| 102 |
+
emit(frame(img, goal, f"step {k}", dtxt, f"laya: {dec['lat']} ms", f"t = {dec['elapsed']} ms", status=dec["op"]), 1.6)
|
| 103 |
+
lat = sorted(totals["lat"]); med = lat[len(lat) // 2] if lat else 0
|
| 104 |
+
emit(title_card(["that's laya as a browser agent's System 1", f"{totals['tasks_ok']}/{len(TASKS)} tasks · {totals['steps']} decisions · median {med} ms per decision",
|
| 105 |
+
f"total wall time {totals['wall']:.1f} s including page loads"],
|
| 106 |
+
["model + code + numbers: huggingface.co/cklxx/laya-browser", "TileLang fused kernels + CUDA graphs: github.com/NandhaKishorM/laya/pull/25"]), 4.0)
|
| 107 |
+
subprocess.run(["ffmpeg", "-y", "-loglevel", "error", "-framerate", str(FPS), "-i", str(out / "frames" / "%06d.png"), "-c:v", "libx264", "-pix_fmt", "yuv420p", "-crf", "23", str(out / "laya_browser_demo.mp4")], check=True)
|
| 108 |
+
subprocess.run(["ffmpeg", "-y", "-loglevel", "error", "-i", str(out / "laya_browser_demo.mp4"), "-vf", "fps=8,scale=700:-1:flags=lanczos,split[s0][s1];[s0]palettegen=max_colors=128[p];[s1][p]paletteuse=dither=bayer", str(out / "laya_browser_demo.gif")], check=True)
|
| 109 |
+
print("video:", out / "laya_browser_demo.mp4", f"{(out / 'laya_browser_demo.mp4').stat().st_size/1e6:.1f} MB", "| gif:", f"{(out / 'laya_browser_demo.gif').stat().st_size/1e6:.1f} MB", "| frames", n, f"({n/FPS:.0f}s)")
|
| 110 |
+
|
| 111 |
+
if __name__ == "__main__":
|
| 112 |
+
main()
|
code/apps/moderator.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""内容审核台:给一批评论打 有害/垃圾/需人工复核 的校准概率,并按风险排序。
|
| 2 |
+
用法: python apps/moderator.py [file.txt] # 每行一条评论;无参数用内置样例
|
| 3 |
+
"""
|
| 4 |
+
import sys, time
|
| 5 |
+
from common import get_agent, bar
|
| 6 |
+
|
| 7 |
+
QUESTIONS = {
|
| 8 |
+
"toxic": {"type": "noul", "instructions": "Is this comment abusive, hateful or harassing toward someone?"},
|
| 9 |
+
"spam": {"type": "noul", "instructions": "Is this comment spam or unsolicited advertising?"},
|
| 10 |
+
"topic": {"type": "choice", "instructions": "What is the comment mainly about?",
|
| 11 |
+
"criteria": {"product": "the product or service itself", "politics": "political opinion",
|
| 12 |
+
"personal": "attacks or remarks about a person", "offtopic": "unrelated chatter"}},
|
| 13 |
+
"severity": {"type": "score", "instructions": "How severe is the policy violation, if any?",
|
| 14 |
+
"criteria": ["none", "mild", "serious", "ban-worthy"]},
|
| 15 |
+
}
|
| 16 |
+
|
| 17 |
+
SAMPLES = [
|
| 18 |
+
"This update is great, the new editor is so much faster!",
|
| 19 |
+
"Buy cheap followers now!!! visit my profile link, 50% off today only",
|
| 20 |
+
"You are a worthless idiot and everyone here knows it.",
|
| 21 |
+
"这个功能真的太难用了,建议回滚到上个版本。",
|
| 22 |
+
"滚出去,你这种垃圾不配在这发言。",
|
| 23 |
+
"Honestly both parties are the same, nothing will change.",
|
| 24 |
+
"Does anyone know if the API supports webhooks?",
|
| 25 |
+
]
|
| 26 |
+
|
| 27 |
+
def main():
|
| 28 |
+
agent = get_agent("multilingual")
|
| 29 |
+
texts = [l.strip() for l in open(sys.argv[1], encoding="utf-8") if l.strip()] if len(sys.argv) > 1 else SAMPLES
|
| 30 |
+
t = time.time()
|
| 31 |
+
results = [agent.predict({"comment": c}, QUESTIONS) for c in texts]
|
| 32 |
+
dt = time.time() - t
|
| 33 |
+
rows = []
|
| 34 |
+
for c, r in zip(texts, results):
|
| 35 |
+
a = r["answers"]
|
| 36 |
+
risk = max(a["toxic"]["noul"], a["spam"]["noul"])
|
| 37 |
+
rows.append((risk, a["toxic"]["noul"], a["spam"]["noul"],
|
| 38 |
+
a["topic"]["choice"], round(a["severity"]["score"],1), c))
|
| 39 |
+
rows.sort(reverse=True)
|
| 40 |
+
print(f"{len(texts)} comments in {dt*1000:.0f} ms ({dt*1000/len(texts):.0f} ms each)\n")
|
| 41 |
+
print(f"{'risk':>5} {'toxic':>5} {'spam':>5} {'topic':9} {'sev':>4} comment")
|
| 42 |
+
for risk, tox, spam, topic, sev, c in rows:
|
| 43 |
+
flag = "🚨" if risk > 0.7 else ("⚠️ " if risk > 0.4 else " ")
|
| 44 |
+
print(f"{flag}{risk:5.2f} {tox:5.2f} {spam:5.2f} {topic:9} {str(sev):>4} {c[:60]}")
|
| 45 |
+
|
| 46 |
+
if __name__ == "__main__":
|
| 47 |
+
main()
|
code/apps/profile_step.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Time one real browser step (recorded in finetune/out/dagger_cases.jsonl) end to end inside the server process.
|
| 2 |
+
python apps/profile_step.py <checkpoint dir>"""
|
| 3 |
+
import json, os, sys
|
| 4 |
+
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))); sys.path.insert(0, "/home/ckl/projects/S/laya/finetune")
|
| 5 |
+
os.environ.setdefault("LAYA_FMT", "v2")
|
| 6 |
+
from common import get_agent
|
| 7 |
+
from common_ft import build_request
|
| 8 |
+
from fast_batch import profile_step
|
| 9 |
+
agent = get_agent(sys.argv[1])
|
| 10 |
+
if agent.cfg.get("head_max_len_train"): agent.cfg["head_max_len"] = agent.cfg["head_max_len_train"]
|
| 11 |
+
cases = [json.loads(l) for l in open("/home/ckl/projects/S/laya/finetune/out/dagger_cases.jsonl")]
|
| 12 |
+
for c in cases[:4]:
|
| 13 |
+
page = c["page_obj"]; state, questions, _, _ = build_request(page, c["goal"], c.get("history", []))
|
| 14 |
+
n_opts = sum(len(q["criteria"]) for q in questions.values())
|
| 15 |
+
r = profile_step(agent, state, questions)
|
| 16 |
+
print(f"{len(questions)} questions / {n_opts:3d} options / {r['tokens']:4d} tokens: agent.predict {r['predict_ms']:6.1f} ms | fast {r['predict_fast_ms']:6.1f} ms "
|
| 17 |
+
f"(tokenize {r['tokenize_ms']:.1f} + forward {r['forward_ms']:.1f} + post {r['post_ms']:.1f})")
|
code/apps/systemone_server.py
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""TypeSafe-compatible /v1/systemone endpoint backed by a local laya checkpoint (+ TileLang fast path).
|
| 2 |
+
|
| 3 |
+
python apps/systemone_server.py [port=8791] [variant=typed|multilingual|english]
|
| 4 |
+
|
| 5 |
+
Request body: {"model": ..., "state": {...}, "questions": {...}} -> {"answers": ..., "model": ..., "usage": ...}
|
| 6 |
+
"""
|
| 7 |
+
import json, os, sys, time, traceback
|
| 8 |
+
from http.server import ThreadingHTTPServer, BaseHTTPRequestHandler
|
| 9 |
+
from common import get_agent
|
| 10 |
+
from fast_batch import predict_fast
|
| 11 |
+
|
| 12 |
+
PORT = int(sys.argv[1]) if len(sys.argv) > 1 else 8791
|
| 13 |
+
VARIANT = sys.argv[2] if len(sys.argv) > 2 else "typed"
|
| 14 |
+
MAXOPT = int(sys.argv[3]) if len(sys.argv) > 3 else 12 # laya's option budget: keep choice questions at most this wide
|
| 15 |
+
agent = get_agent(VARIANT)
|
| 16 |
+
FMT = os.environ.get("LAYA_FMT", agent.cfg.get("laya_fmt", "v1")) # fine-tuned checkpoints record their format in rl_agent_config.json
|
| 17 |
+
if agent.cfg.get("head_max_len_train"):
|
| 18 |
+
agent.cfg["head_max_len"] = agent.cfg["head_max_len_train"]
|
| 19 |
+
print(f"[systemone] format={FMT} head_max_len={agent.cfg.get('head_max_len')}", flush=True)
|
| 20 |
+
LOG = []
|
| 21 |
+
|
| 22 |
+
# ---- System 1 / System 2 gating: below ESCALATE_TAU confidence, ask the LLM teacher (same element table) and return its
|
| 23 |
+
# decision in laya's answer format. Every escalation is also appended to ESCALATE_LOG as a DAgger case.
|
| 24 |
+
TAU = float(os.environ.get("ESCALATE_TAU", "0")) # 0 = off
|
| 25 |
+
ESC_URL = os.environ.get("TEXT_MODEL_BASE_URL", "http://127.0.0.1:30000/v1") + "/chat/completions"
|
| 26 |
+
ESC_MODEL = os.environ.get("TEXT_MODEL", "Qwen/Qwen3-8B-AWQ")
|
| 27 |
+
ESC_LOG = os.environ.get("ESCALATE_LOG", "")
|
| 28 |
+
STATS = {"calls": 0, "escalated": 0}
|
| 29 |
+
ESC_SYS = """You are the System-2 fallback for a browser agent. Given the goal, the actions so far, the page and a numbered list of
|
| 30 |
+
controls (each with the operations it supports), pick the single best NEXT step. Answer JSON:
|
| 31 |
+
{"operation": "CLICK"|"TYPE_TEXT"|"SELECT"|"DONE"|"BLOCKED"|"WAIT"|"SCROLL_DOWN"|"SCROLL_UP", "target": "<option key of the chosen control, or null>"}
|
| 32 |
+
DONE only if the goal is already visibly satisfied. A field that already shows the requested value is done; do not re-type it."""
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
def escalate(state, questions, answers):
|
| 36 |
+
import httpx
|
| 37 |
+
ops = questions["operation"]["criteria"]
|
| 38 |
+
controls = []
|
| 39 |
+
for qid, q in questions.items():
|
| 40 |
+
if qid.endswith("_target"):
|
| 41 |
+
op = qid[:-7].upper()
|
| 42 |
+
for key, desc in q["criteria"].items():
|
| 43 |
+
controls.append({"op": op, "target": key, "control": desc})
|
| 44 |
+
user = {"goal": (questions["operation"]["instructions"] or {}).get("goal") if isinstance(questions["operation"]["instructions"], dict) else "",
|
| 45 |
+
"actions_so_far": state.get("recent_actions", [])[-8:], "page": state.get("page", {}), "operations": list(ops),
|
| 46 |
+
"controls": controls[:150], "system1_guess": {k: v.get("choice") for k, v in answers.items()}}
|
| 47 |
+
body = {"model": ESC_MODEL, "max_tokens": 80, "temperature": 0.0, "response_format": {"type": "json_object"},
|
| 48 |
+
"chat_template_kwargs": {"enable_thinking": False},
|
| 49 |
+
"messages": [{"role": "system", "content": ESC_SYS}, {"role": "user", "content": json.dumps(user, ensure_ascii=False)}]}
|
| 50 |
+
r = httpx.post(ESC_URL, json=body, timeout=120).json()
|
| 51 |
+
v = json.loads(r["choices"][0]["message"]["content"])
|
| 52 |
+
op = str(v.get("operation", "")).upper(); tgt = v.get("target")
|
| 53 |
+
if op not in ops:
|
| 54 |
+
return answers, False
|
| 55 |
+
def one_hot(keys, k, p=0.97):
|
| 56 |
+
rest = (1 - p) / max(1, len(keys) - 1)
|
| 57 |
+
return {kk: (p if kk == k else rest) for kk in keys}
|
| 58 |
+
answers["operation"] = {**answers["operation"], "choice": op, "probabilities": one_hot(list(ops), op), "confidence": 0.9, "system2": True}
|
| 59 |
+
tq = op.lower() + "_target"
|
| 60 |
+
if tq in questions:
|
| 61 |
+
keys = list(questions[tq]["criteria"]); tgt = str(tgt)
|
| 62 |
+
if tgt not in keys:
|
| 63 |
+
return answers, False
|
| 64 |
+
answers[tq] = {**answers[tq], "choice": tgt, "probabilities": one_hot(keys, tgt), "confidence": 0.9, "system2": True}
|
| 65 |
+
if ESC_LOG:
|
| 66 |
+
with open(ESC_LOG, "a") as f:
|
| 67 |
+
f.write(json.dumps({"state": state, "operation": op, "target": tgt if tq in questions else None}, ensure_ascii=False) + "\n")
|
| 68 |
+
return answers, True
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def compact(v):
|
| 72 |
+
"""Shrink jev-ultrafast element criteria ({'element': '[3] Search', 'role': 'button', ...}) into one short string
|
| 73 |
+
so more options fit laya's head token budget."""
|
| 74 |
+
if isinstance(v, dict) and "element" in v:
|
| 75 |
+
s = str(v["element"])[:50 if FMT == "v3" else 10000]
|
| 76 |
+
if v.get("role"):
|
| 77 |
+
s += f" ({v['role']})"
|
| 78 |
+
if v.get("current_value"):
|
| 79 |
+
s += f" = {str(v['current_value'])[:30]!r}"
|
| 80 |
+
for k in ("checked", "selected", "expanded"):
|
| 81 |
+
if k in v:
|
| 82 |
+
s += f" {k}={v[k]}"
|
| 83 |
+
return s
|
| 84 |
+
return v
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
def predict(state, questions):
|
| 88 |
+
"""agent.predict with coarse-to-fine handling of wide choice questions.
|
| 89 |
+
|
| 90 |
+
A choice with more than MAXOPT options is split into interleaved chunks; every chunk is a question in the same
|
| 91 |
+
forward pass as the normal questions, then the chunk winners compete in a second pass.
|
| 92 |
+
p(option) = p_final(winner of its chunk) * p_chunk(option)."""
|
| 93 |
+
qs, plan = {}, {}
|
| 94 |
+
if isinstance(state, dict) and isinstance(state.get("page"), dict) and isinstance(state["page"].get("text"), str):
|
| 95 |
+
if FMT in ("v2", "v3"): # mirror finetune/common_ft.py
|
| 96 |
+
state = {"page": {**state["page"], "text": state["page"]["text"][:1500 if FMT == "v2" else 1200]}, "recent_actions": state.get("recent_actions", [])}
|
| 97 |
+
else:
|
| 98 |
+
state = {**state, "page": {**state["page"], "text": state["page"]["text"][:6000]}}
|
| 99 |
+
for qid, q in questions.items():
|
| 100 |
+
q = dict(q)
|
| 101 |
+
if isinstance(q.get("criteria"), dict):
|
| 102 |
+
q["criteria"] = {k: compact(v) for k, v in q["criteria"].items()}
|
| 103 |
+
keys = list(q["criteria"]) if q["type"] == "choice" and isinstance(q.get("criteria"), dict) else []
|
| 104 |
+
if len(keys) <= MAXOPT:
|
| 105 |
+
qs[qid] = q
|
| 106 |
+
continue
|
| 107 |
+
n = -(-len(keys) // MAXOPT)
|
| 108 |
+
chunks = [keys[i::n] for i in range(n)]
|
| 109 |
+
plan[qid] = (q, chunks)
|
| 110 |
+
for ci, ch in enumerate(chunks):
|
| 111 |
+
qs[f"{qid}__chunk{ci}"] = {**q, "criteria": {k: q["criteria"][k] for k in ch}}
|
| 112 |
+
r = predict_fast(agent, state, qs)
|
| 113 |
+
r["passes"] = 1
|
| 114 |
+
if plan:
|
| 115 |
+
chunk_ans = {qid: [r["answers"].pop(f"{qid}__chunk{ci}") for ci in range(len(chunks))] for qid, (q, chunks) in plan.items()}
|
| 116 |
+
finals = {qid: {**q, "criteria": {a["choice"]: q["criteria"][a["choice"]] for a in chunk_ans[qid]}} for qid, (q, _) in plan.items()}
|
| 117 |
+
r2 = predict_fast(agent, state, finals)
|
| 118 |
+
r["passes"] = 2
|
| 119 |
+
r["usage"]["input_tokens"] += r2["usage"]["input_tokens"]
|
| 120 |
+
for qid, (q, chunks) in plan.items():
|
| 121 |
+
fa = r2["answers"][qid]
|
| 122 |
+
probs = {}
|
| 123 |
+
for ca, ch in zip(chunk_ans[qid], chunks):
|
| 124 |
+
pf = fa["probabilities"][ca["choice"]]
|
| 125 |
+
for k in ch:
|
| 126 |
+
probs[k] = pf * ca["probabilities"][k]
|
| 127 |
+
tot = sum(probs.values()) or 1.0
|
| 128 |
+
probs = {k: round(v / tot, 6) for k, v in probs.items()}
|
| 129 |
+
choice = max(probs, key=probs.get)
|
| 130 |
+
r["answers"][qid] = {"type": "choice", "choice": choice, "probabilities": probs, "confidence": fa["confidence"],
|
| 131 |
+
"action": fa.get("action", {}), "coarse_to_fine": {"chunks": len(chunks), "winners": [a["choice"] for a in chunk_ans[qid]]}}
|
| 132 |
+
return r
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
class H(BaseHTTPRequestHandler):
|
| 136 |
+
def log_message(self, *a): pass
|
| 137 |
+
def _send(self, code, body):
|
| 138 |
+
data = json.dumps(body, ensure_ascii=False).encode()
|
| 139 |
+
self.send_response(code); self.send_header("Content-Type", "application/json"); self.send_header("Content-Length", str(len(data))); self.end_headers(); self.wfile.write(data)
|
| 140 |
+
def do_GET(self):
|
| 141 |
+
self._send(200, {"ok": True, "variant": VARIANT, "calls": len(LOG), "tau": TAU, "escalated": STATS["escalated"], "recent": LOG[-5:]})
|
| 142 |
+
def do_POST(self):
|
| 143 |
+
n = int(self.headers.get("Content-Length", 0)); req = json.loads(self.rfile.read(n) or b"{}")
|
| 144 |
+
try:
|
| 145 |
+
t = time.perf_counter()
|
| 146 |
+
r = predict(req["state"], req["questions"])
|
| 147 |
+
STATS["calls"] += 1
|
| 148 |
+
if TAU > 0:
|
| 149 |
+
a = r["answers"]; op = a["operation"]["choice"]; tq = op.lower() + "_target"
|
| 150 |
+
conf = min(a["operation"]["confidence"], a[tq]["confidence"] if tq in a else 1.0)
|
| 151 |
+
if conf < TAU:
|
| 152 |
+
try:
|
| 153 |
+
r["answers"], esc = escalate(req["state"], req["questions"], a)
|
| 154 |
+
STATS["escalated"] += esc
|
| 155 |
+
except Exception as e:
|
| 156 |
+
print("[escalate] failed:", str(e)[:80], flush=True)
|
| 157 |
+
ms = (time.perf_counter() - t) * 1000
|
| 158 |
+
r["model"] = f"laya-{VARIANT}"
|
| 159 |
+
nq = len(req["questions"]); nopt = sum(len(q.get("criteria") or []) for q in req["questions"].values())
|
| 160 |
+
LOG.append({"ms": round(ms, 1), "questions": nq, "options": nopt, "tokens": r["usage"]["input_tokens"], "passes": r["passes"]})
|
| 161 |
+
print(f"[systemone] {nq} q / {nopt} opts / {r['usage']['input_tokens']} tok / {r['passes']} pass -> {ms:.1f} ms " +
|
| 162 |
+
", ".join(f"{k}={v.get('choice', v.get('score', v.get('noul')))}({v['confidence']:.2f})" for k, v in r["answers"].items()), flush=True)
|
| 163 |
+
self._send(200, r)
|
| 164 |
+
except Exception as e:
|
| 165 |
+
traceback.print_exc(); self._send(400, {"error": f"{type(e).__name__}: {e}"})
|
| 166 |
+
|
| 167 |
+
if __name__ == "__main__":
|
| 168 |
+
print(f"laya systemone server on http://127.0.0.1:{PORT}/v1/systemone ({VARIANT})", flush=True)
|
| 169 |
+
ThreadingHTTPServer(("127.0.0.1", PORT), H).serve_forever()
|
code/apps/web_console.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""决策台 Web UI(零依赖,stdlib http.server):粘贴任意文本,自定义问题,实时看概率条。
|
| 2 |
+
用法: python apps/web_console.py [port=7860] 然后打开 http://127.0.0.1:7860
|
| 3 |
+
"""
|
| 4 |
+
import sys, json, time
|
| 5 |
+
from http.server import ThreadingHTTPServer, BaseHTTPRequestHandler
|
| 6 |
+
from common import get_agent
|
| 7 |
+
|
| 8 |
+
HTML = r"""<!doctype html><meta charset=utf-8><title>Laya 决策台</title>
|
| 9 |
+
<style>
|
| 10 |
+
body{font:14px system-ui;margin:0;background:#0f1115;color:#e6e6e6;display:grid;grid-template-columns:1fr 1fr;gap:16px;padding:16px;height:100vh;box-sizing:border-box}
|
| 11 |
+
textarea{width:100%;box-sizing:border-box;background:#1a1d24;color:#eee;border:1px solid #333;border-radius:6px;padding:8px;font:13px ui-monospace,monospace}
|
| 12 |
+
button{background:#4f8cff;color:#fff;border:0;padding:8px 16px;border-radius:6px;font-size:14px;cursor:pointer}
|
| 13 |
+
select{background:#1a1d24;color:#eee;border:1px solid #333;padding:6px;border-radius:6px}
|
| 14 |
+
.q{margin:10px 0;padding:10px;background:#171a21;border-radius:8px}.q h3{margin:0 0 6px;font-size:14px}
|
| 15 |
+
.row{display:flex;align-items:center;gap:8px;margin:2px 0}.bar{height:10px;background:#4f8cff;border-radius:3px}
|
| 16 |
+
.lab{width:120px;text-align:right;color:#aaa}.p{width:40px;color:#aaa}
|
| 17 |
+
#out{overflow:auto}.meta{color:#888;font-size:12px}
|
| 18 |
+
</style>
|
| 19 |
+
<div><h2>Laya 决策台 <span class=meta>单次前向传播 · 不生成文本 · 校准概率</span></h2>
|
| 20 |
+
<label>模型 <select id=model><option value=multilingual>multilingual (100+ 语言)</option><option value=english>english</option><option value=typed>typed-decisions</option></select></label>
|
| 21 |
+
<p><b>状态 (JSON 或纯文本)</b><br><textarea id=state rows=8>{"subject": "登录一直失败", "body": "从昨天开始整个团队都登录不了后台,报 502,我们的上线被卡住了。再不解决就退订。"}</textarea></p>
|
| 22 |
+
<p><b>问题 (JSON)</b><br><textarea id=questions rows=16>{
|
| 23 |
+
"department": {"type": "choice", "instructions": "Which team should handle this?",
|
| 24 |
+
"criteria": {"billing": "invoices, refunds", "technical": "bugs, outages, login", "sales": "pricing, plans"}},
|
| 25 |
+
"urgency": {"type": "score", "instructions": "How urgent is this?", "criteria": ["not urgent", "soon", "blocking"]},
|
| 26 |
+
"churn_risk": {"type": "noul", "instructions": "Does the user threaten to cancel?"}
|
| 27 |
+
}</textarea></p>
|
| 28 |
+
<button onclick=run()>预测 (Ctrl+Enter)</button> <span id=t class=meta></span></div>
|
| 29 |
+
<div id=out></div>
|
| 30 |
+
<script>
|
| 31 |
+
async function run(){
|
| 32 |
+
const body={model:model.value,state:state.value,questions:questions.value};
|
| 33 |
+
t.textContent='...';
|
| 34 |
+
const r=await fetch('/predict',{method:'POST',body:JSON.stringify(body)});const j=await r.json();
|
| 35 |
+
if(j.error){out.innerHTML='<pre style="color:#f66">'+j.error+'</pre>';t.textContent='';return}
|
| 36 |
+
t.textContent=j.ms.toFixed(0)+' ms';
|
| 37 |
+
let h='';for(const [k,a] of Object.entries(j.answers)){
|
| 38 |
+
h+='<div class=q><h3>'+k+' → <span style=color:#8fd>'+(a.choice??(a.score!==undefined?'score '+a.score.toFixed(2):'')??'')+(a.noul!==undefined?'P(true)='+a.noul.toFixed(2):'')+'</span>'+(a.confidence!==undefined?' <span class=meta>conf '+a.confidence.toFixed(2)+'</span>':'')+'</h3>';
|
| 39 |
+
let d=a.probabilities||(a.noul!==undefined?{true:a.noul,false:1-a.noul}:{});if(a.legend)d=Object.fromEntries(Object.entries(d).map(([k,v])=>[k+' '+a.legend[k],v]));
|
| 40 |
+
if(Array.isArray(d))d=Object.fromEntries(d.map((v,i)=>[i,v]));
|
| 41 |
+
for(const [l,p] of Object.entries(d).sort((x,y)=>y[1]-x[1]))h+='<div class=row><span class=lab>'+l+'</span><div class=bar style=width:'+(p*300)+'px></div><span class=p>'+p.toFixed(2)+'</span></div>';
|
| 42 |
+
h+='</div>'}
|
| 43 |
+
h+='<pre class=meta>'+JSON.stringify(j.raw,null,1)+'</pre>';out.innerHTML=h}
|
| 44 |
+
document.addEventListener('keydown',e=>{if(e.ctrlKey&&e.key=='Enter')run()});
|
| 45 |
+
</script>"""
|
| 46 |
+
|
| 47 |
+
class H(BaseHTTPRequestHandler):
|
| 48 |
+
def log_message(self, *a): pass
|
| 49 |
+
def do_GET(self):
|
| 50 |
+
self.send_response(200); self.send_header("Content-Type", "text/html; charset=utf-8"); self.end_headers()
|
| 51 |
+
self.wfile.write(HTML.encode())
|
| 52 |
+
def do_POST(self):
|
| 53 |
+
n = int(self.headers.get("Content-Length", 0)); req = json.loads(self.rfile.read(n))
|
| 54 |
+
try:
|
| 55 |
+
st = req["state"].strip()
|
| 56 |
+
try: st = json.loads(st)
|
| 57 |
+
except Exception: st = {"text": st}
|
| 58 |
+
qs = json.loads(req["questions"])
|
| 59 |
+
agent = get_agent(req.get("model", "multilingual"))
|
| 60 |
+
t = time.time(); r = agent.predict(st, qs); ms = (time.time() - t) * 1000
|
| 61 |
+
body = {"answers": r["answers"], "raw": r, "ms": ms}
|
| 62 |
+
except Exception as e:
|
| 63 |
+
body = {"error": f"{type(e).__name__}: {e}"}
|
| 64 |
+
data = json.dumps(body, ensure_ascii=False, default=str).encode()
|
| 65 |
+
self.send_response(200); self.send_header("Content-Type", "application/json"); self.end_headers(); self.wfile.write(data)
|
| 66 |
+
|
| 67 |
+
if __name__ == "__main__":
|
| 68 |
+
port = int(sys.argv[1]) if len(sys.argv) > 1 else 7860
|
| 69 |
+
get_agent("multilingual")
|
| 70 |
+
print(f"open http://127.0.0.1:{port}")
|
| 71 |
+
ThreadingHTTPServer(("127.0.0.1", port), H).serve_forever()
|
code/env.sh
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
export HF_ENDPOINT=https://hf-mirror.com
|
| 2 |
+
export USE_TF=0
|
| 3 |
+
export UV_INDEX_URL=https://pypi.tuna.tsinghua.edu.cn/simple
|
| 4 |
+
export PIP_INDEX_URL=https://pypi.tuna.tsinghua.edu.cn/simple
|
| 5 |
+
unset http_proxy https_proxy all_proxy HTTP_PROXY HTTPS_PROXY ALL_PROXY
|
| 6 |
+
export PATH="$HOME/.local/bin:$PATH"
|
code/finetune/README.md
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# 微调 laya 做浏览器 agent 决策头(替代 TypeSafe Jev)
|
| 2 |
+
|
| 3 |
+
流水线(全部本地、无付费 API):
|
| 4 |
+
|
| 5 |
+
| 步骤 | 脚本 | 说明 |
|
| 6 |
+
|---|---|---|
|
| 7 |
+
| 1 抓页面 | `collect_pages.py` | headless Chromium + browser-harness,90 个真实页面的元素表 + 正文 |
|
| 8 |
+
| 2 反向生成目标 | `gen_goals.py` | 随机选一个元素当 gold,本地 Qwen3-8B-AWQ(sglang)写出"要用它的用户目标",1121 条 |
|
| 9 |
+
| 3 真实 DONE 样本 | `make_done_cases.py` | 在浏览器里真的执行点击,落地页 + 历史 = DONE,250 条 |
|
| 10 |
+
| 4 构建样本 | `build_items.py` | 复刻 jev-ultrafast 的 systemone 请求格式(`common_ft.py`),head_max_len 512;页面按 index%5 留出 |
|
| 11 |
+
| 5 训练 | `train.py` | laya 官方 RLCD 配方(噪声 logit 策略梯度 + soft CE)单卡版,4 轮约 10 分钟 |
|
| 12 |
+
| 6 评测 | `eval.py` | 留出页面上的操作准确率 / 目标 top-1 |
|
| 13 |
+
|
| 14 |
+
```fish
|
| 15 |
+
bash finetune/run_v2.sh # 4-6 步
|
| 16 |
+
bash finetune/run_all.sh # 含零样本基线
|
| 17 |
+
```
|
| 18 |
+
|
| 19 |
+
## 加入魔搭 Mind2Web 后(v3/v4)
|
| 20 |
+
|
| 21 |
+
`osunlp/Mind2Web`(魔搭,6.7 GB,12-25 MB/s 不走代理)→ `convert_mind2web.py`:按 backend_node_id 从 cleaned_html 还原元素文本/角色,
|
| 22 |
+
正例 + 44 个负例打乱成 jev 格式元素表,历史用 action_reprs,7296 条;按网站哈希留出 20%。v4 = 自采 + 真实 DONE + 全部 Mind2Web,
|
| 23 |
+
DONE / TYPE_TEXT / SELECT 操作样本 3 倍过采样,16020 条训练样本,3 轮 64 分钟。`bash finetune/run_v4.sh`。
|
| 24 |
+
|
| 25 |
+
评测 1642 条(自采留出页 232 + Mind2Web 未见网站 1410):
|
| 26 |
+
|
| 27 |
+
| 模型 | 操作准确率 | 目标 top-1 | live DONE | m2w CLICK 目标 | m2w TYPE_TEXT 操作 | m2w SELECT 操作 |
|
| 28 |
+
|---|---|---|---|---|---|---|
|
| 29 |
+
| typed-decisions 零样本 | 0.26 | 0.13 | 0.39 | 0.03 | 0.00 | 0.00 |
|
| 30 |
+
| v3(无过采样) | 0.790 | 0.496 | 0.58 | 0.40 | 0.04 | 0.20 |
|
| 31 |
+
| **v4(过采样)** | **0.795** | **0.524** | **0.82** | **0.43** | **0.80** | **0.55** |
|
| 32 |
+
|
| 33 |
+
真实浏览器 6 任务:零样本 0/6 → v2 2/6 → **v4 3/6**(python.org Downloads 2.0 s、Travel 分类 0.3 s、Wikipedia 随机文章),
|
| 34 |
+
Wikipedia 搜索任务首次正确选 TYPE_TEXT 并由本地 Qwen 填入 "Python programming language",但之后重复填同一字段而没有提交——
|
| 35 |
+
训练数据里缺"字段已填好 → 下一步提交"的样本。GitHub Issues 点成 Releases、HN new 提前 DONE。
|
| 36 |
+
|
| 37 |
+
## 最终对比 v2(2026-09-21,修正检查脚本后:下拉按 current_value 文本判定、导航后等 1.5 s 再判)
|
| 38 |
+
|
| 39 |
+
| 模型 | 真实任务 16×3 | 留出 top-1 | 每步 |
|
| 40 |
+
|---|---|---|---|
|
| 41 |
+
| v10(421M) | 50% | 0.656 | 41–50 ms |
|
| 42 |
+
| **v10s(322M)** | **62%** | 0.631 | 17–23 ms |
|
| 43 |
+
| v11s(322M + 682 条滚动/搜索/下拉脚本轨迹) | 56% | 0.623 | 19 ms |
|
| 44 |
+
|
| 45 |
+
10 个任务 3/3 稳过、6 个 3/3 稳挂;v10s/v11s 差异小于站点随机波动。老师对比:27B 思考预算 300 最好(0.861/0.603,4.7 s/步),
|
| 46 |
+
仍不超过 laya v11s(0.890/0.623,0.021 s);8B/27B 都不适合做 DAgger 老师或 System-2 兜底。
|
| 47 |
+
|
| 48 |
+
## 最终对比(2026-09-21,16 个真实任务 × 3 次,apps/browser_suite.py)
|
| 49 |
+
|
| 50 |
+
| 模型 | 底座 | 一步延迟 | 留出目标 top-1 | 真实任务通过率 | +门控 τ=0.7(Qwen3-8B 兜底) |
|
| 51 |
+
|---|---|---|---|---|---|
|
| 52 |
+
| v10 | ModernBERT-large 421M | 41–50 ms | 0.656 | **58%**(28/48) | 42%,升级率 78% |
|
| 53 |
+
| v10s | mmBERT-base 322M,格式 v3 | **17–23 ms** | 0.631 | 50%(24/48) | 52%,升级率 85% |
|
| 54 |
+
|
| 55 |
+
结果高度双峰:9 个任务 3/3 稳定通过(导航、分类、勾选、HN 各页、DuckDuckGo 搜索),7 个任务 0/3 稳定失败
|
| 56 |
+
(Wikipedia 两个搜索任务、GitHub Issues、下拉选择、翻页需滚动、arXiv 搜索、Google Flights)。
|
| 57 |
+
门控把低置信步骤交给 Qwen3-8B 反而更差:在这些页面上微调后的 laya 比 8B 通用模型判断得更准,System 2 需要更强的模型或专门提示。
|
| 58 |
+
失败任务的共性是"输入后提交 / 选建议"和"需要先滚动",训练数据里几乎没有滚动样本。
|
| 59 |
+
v10 数据:5244 条自采目标(421 页)+ 700 真实 DONE + step-2 + Mind2Web 全量 + DAgger,格式 v2,head 768,4 轮 2.2 小时。
|
| 60 |
+
|
| 61 |
+
## 输入格式 v2(v9 起)
|
| 62 |
+
|
| 63 |
+
`LAYA_FMT=v2 LAYA_HEAD=768`:元素表不再塞进 state(1024 token 里会被截掉),只在选项里保留完整标签 + 角色 + 已填值,
|
| 64 |
+
state 只留标题 / URL / 历史 / 1500 字正文,head_max_len 512→768。v9 只用 Mind2Web + 该格式,真实任务从 6/16 到 **10/16**,
|
| 65 |
+
Mind2Web 点击目标 0.44→0.51。checkpoint 的 `rl_agent_config.json` 记录 `laya_fmt` / `head_max_len_train`,服务端自动跟随。
|
| 66 |
+
|
| 67 |
+
注意:jev 的 `snapshot.js` 有意隐藏密码框(`!['password','file','hidden'].includes(e.type)`),密码登录任务在该框架里不可能完成,任务集已移除。
|
| 68 |
+
|
| 69 |
+
## 真实任务集(apps/browser_suite.py,16 个任务,自动验收)
|
| 70 |
+
|
| 71 |
+
| 版本 | 通过 | 主要失败模式 |
|
| 72 |
+
|---|---|---|
|
| 73 |
+
| v4 | 3/14 | 过早 DONE(6 个),填完不提交���2 个),点错(3 个) |
|
| 74 |
+
| v5(DONE 过采样 2x + 字段带已填值 + 温度 1.40) | 4/14 | 登录/搜索仍在填一个字段后 DONE |
|
| 75 |
+
|
| 76 |
+
v5 的根因:真实 DONE 样本历史恰好都是 1 步、自采点击样本历史都是 0 步,模型学到"有历史 ⇒ DONE"。
|
| 77 |
+
v6 加 `gen_step2.py`:在 250 个落地页上保留历史再生成 659 条新目标作为负样本。
|
| 78 |
+
|
| 79 |
+
## 结果(v1/v2,仅自采数据)
|
| 80 |
+
|
| 81 |
+
留出 18 个未见页面、232 条:
|
| 82 |
+
|
| 83 |
+
| 模型 | 操作准确率 | 目标 top-1 | DONE 判断 | 每条延迟 |
|
| 84 |
+
|---|---|---|---|---|
|
| 85 |
+
| typed-decisions 零样本 | 0.543 | 0.103(随机) | 0.39 | 254 ms |
|
| 86 |
+
| multilingual 零样本 | 0.127 | 0.124 | 0.00 | 249 ms |
|
| 87 |
+
| **微调 v2(1886 样本)** | **0.875** | **0.665** | 0.66 | 55 ms |
|
| 88 |
+
|
| 89 |
+
真实浏览器 6 个任务(jev-ultrafast 循环 + 本地 laya 服务):python.org Downloads 2.4 s 完成、books.toscrape Travel
|
| 90 |
+
分类 0.3 s 完成;HN "new"、GitHub Issues、Wikipedia 两个任务失败(点错元素或提前 DONE)。零样本时 6 个全失败。
|
| 91 |
+
|
| 92 |
+
已知问题:置信度全是 1.00(未做温度校准);数据只有 90 个页面,泛化有限。下一步是把页面扩到 500+、
|
| 93 |
+
每轮把线上失败样本加回训练集、拟合温度。v1 的教训:模板化的 DONE 样本会让模型学到"看到 stop when 就答 DONE",
|
| 94 |
+
DONE 样本必须来自真实执行后的落地页。
|
code/finetune/build_items.py
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""cases.jsonl + pages.jsonl -> tokenized training items (train split) and eval cases (held-out pages).
|
| 2 |
+
|
| 3 |
+
python finetune/build_items.py out/pages.jsonl out/cases.jsonl out/
|
| 4 |
+
"""
|
| 5 |
+
import json, os, sys
|
| 6 |
+
import torch
|
| 7 |
+
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
| 8 |
+
from common_ft import build_request, gold_for
|
| 9 |
+
from transformers import AutoTokenizer
|
| 10 |
+
from laya.common import QTYPES, build_sequence, render_options
|
| 11 |
+
|
| 12 |
+
MAX_LEN, HEAD_MAX_LEN = 1024, int(os.environ.get('LAYA_HEAD', '512'))
|
| 13 |
+
EVAL_EVERY = 5 # pages with index % 5 == 0 are held out
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
def main():
|
| 17 |
+
pages_f, cases_f, out = sys.argv[1:4]
|
| 18 |
+
pages = [json.loads(l) for l in open(pages_f)]
|
| 19 |
+
cases = [json.loads(l) for l in open(cases_f)]
|
| 20 |
+
for extra in sys.argv[4:]:
|
| 21 |
+
cases += [json.loads(l) for l in open(extra)]
|
| 22 |
+
snap = os.environ.get("LAYA_BASE")
|
| 23 |
+
tok = AutoTokenizer.from_pretrained(os.path.join(snap, "tokenizer"))
|
| 24 |
+
items, n_skip, ev = [], 0, []
|
| 25 |
+
import hashlib
|
| 26 |
+
STOPS = [" Stop when it is open.", " Stop once that page is visible.", " Then stop.", " Finish when it has loaded.", ""]
|
| 27 |
+
for c in cases:
|
| 28 |
+
if c["kind"] == "done" and "page_obj" not in c:
|
| 29 |
+
continue # template DONE cases leak phrasing; real ones come from make_done_cases.py
|
| 30 |
+
page = c.get("page_obj") or pages[c["page"]]
|
| 31 |
+
h = int(hashlib.md5(c["goal"].encode()).hexdigest(), 16)
|
| 32 |
+
goal = c["goal"] + STOPS[h % len(STOPS)]
|
| 33 |
+
c = {**c, "goal": goal}
|
| 34 |
+
state, questions, targets, controls = build_request(page, goal, c.get("history", []))
|
| 35 |
+
gop, gidx = gold_for(c, targets, controls)
|
| 36 |
+
if gop is None:
|
| 37 |
+
n_skip += 1; continue
|
| 38 |
+
held = (hashlib.md5(c["website"].encode()).digest()[0] % EVAL_EVERY == 0) if c.get("source") == "mind2web" else (c["page"] % EVAL_EVERY == 0)
|
| 39 |
+
if held:
|
| 40 |
+
ev.append({**c, "gold_index": gidx}); continue
|
| 41 |
+
golds = {"operation": gop}
|
| 42 |
+
if gidx is not None:
|
| 43 |
+
golds[gop.lower() + "_target"] = gidx
|
| 44 |
+
for qid, gold in golds.items():
|
| 45 |
+
q = questions[qid]; keys = list(q["criteria"])
|
| 46 |
+
target = [1.0 if k == gold else 0.0 for k in keys]
|
| 47 |
+
qq = {"t": "choice", "ins": json.dumps(q["instructions"]), "crit": q["criteria"]}
|
| 48 |
+
seq, markers = build_sequence(tok, state, qq, MAX_LEN, HEAD_MAX_LEN)
|
| 49 |
+
if len(markers) != len(render_options(qq)):
|
| 50 |
+
n_skip += 1; continue
|
| 51 |
+
item = {"ids": seq, "markers": markers, "qtype": QTYPES["choice"], "target": target, "label": keys.index(gold), "qid": qid, "gold_op": gop}
|
| 52 |
+
# class balance: CLICK dominates the operation question, so repeat the rare operations
|
| 53 |
+
reps = {"DONE": 4, "TYPE_TEXT": 3, "SELECT": 3}.get(gop, 1) if qid == "operation" else 1
|
| 54 |
+
if c.get("source") == "dagger":
|
| 55 |
+
reps *= 5 # on-policy teacher corrections from real tasks: few but exactly where the policy fails
|
| 56 |
+
items.extend([item] * reps)
|
| 57 |
+
torch.save(items, os.path.join(out, "train_items.pt"))
|
| 58 |
+
with open(os.path.join(out, "eval_cases.jsonl"), "w") as f:
|
| 59 |
+
for c in ev: f.write(json.dumps(c, ensure_ascii=False) + "\n")
|
| 60 |
+
lens = [len(i["ids"]) for i in items]
|
| 61 |
+
import collections
|
| 62 |
+
print("operation label counts:", dict(collections.Counter(i["gold_op"] for i in items if i["qid"] == "operation")))
|
| 63 |
+
print(f"train items {len(items)} (op {sum(i['qid']=='operation' for i in items)}, target {sum(i['qid']!='operation' for i in items)}), "
|
| 64 |
+
f"eval cases {len(ev)}, skipped {n_skip}, seq len mean {sum(lens)/len(lens):.0f} max {max(lens)}")
|
| 65 |
+
|
| 66 |
+
if __name__ == "__main__":
|
| 67 |
+
main()
|
code/finetune/calibrate.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Fit per-question-type temperatures on held-out cases (NLL), write them into rl_agent_config.json.
|
| 2 |
+
|
| 3 |
+
python finetune/calibrate.py out/pages.jsonl out/eval_cases.jsonl <checkpoint dir>
|
| 4 |
+
"""
|
| 5 |
+
import json, os, sys
|
| 6 |
+
import numpy as np, torch
|
| 7 |
+
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))); sys.path.insert(0, "/home/ckl/projects/S/laya-upstream")
|
| 8 |
+
from common_ft import build_request, gold_for
|
| 9 |
+
import laya
|
| 10 |
+
from laya.common import QTYPES, build_sequence, collate_items
|
| 11 |
+
|
| 12 |
+
def main():
|
| 13 |
+
pages = [json.loads(l) for l in open(sys.argv[1])]; cases = [json.loads(l) for l in open(sys.argv[2])]; ck = sys.argv[3]
|
| 14 |
+
agent = laya.load(ck); agent.cfg["max_len"], agent.cfg["head_max_len"] = 1024, int(os.environ.get("LAYA_HEAD", agent.cfg.get("head_max_len_train", 512))); agent.accelerate()
|
| 15 |
+
Z, T, K = [], [], []
|
| 16 |
+
for c in cases[::2]:
|
| 17 |
+
state, questions, targets, controls = build_request(c.get("page_obj") or pages[c["page"]], c["goal"], c.get("history", []))
|
| 18 |
+
gop, gidx = gold_for(c, targets, controls)
|
| 19 |
+
golds = {"operation": gop}
|
| 20 |
+
if gidx is not None: golds[gop.lower() + "_target"] = gidx
|
| 21 |
+
items, keys = [], []
|
| 22 |
+
for qid, gold in golds.items():
|
| 23 |
+
q = questions[qid]; qq = {"t": "choice", "ins": json.dumps(q["instructions"]), "crit": q["criteria"]}
|
| 24 |
+
seq, markers = build_sequence(agent.tok, state, qq, 1024, agent.cfg['head_max_len'])
|
| 25 |
+
items.append({"ids": seq, "markers": markers, "qtype": QTYPES["choice"]}); keys.append((list(q["criteria"]), gold))
|
| 26 |
+
b = collate_items([items], agent.tok.pad_token_id)
|
| 27 |
+
with torch.no_grad(), torch.autocast("cuda", dtype=agent.dtype):
|
| 28 |
+
logits, _ = agent.model(b["input_ids"].cuda(), b["attention_mask"].cuda(), b["marker_pos"].cuda(), b["marker_mask"].cuda(), b["qtype"].cuda())
|
| 29 |
+
for r, (ks, gold) in enumerate(keys):
|
| 30 |
+
z = logits[r, :len(ks)].float().cpu(); Z.append(z); T.append(ks.index(gold))
|
| 31 |
+
kmax = max(len(z) for z in Z); M = torch.full((len(Z), kmax), -1e4)
|
| 32 |
+
for i, z in enumerate(Z): M[i, :len(z)] = z
|
| 33 |
+
y = torch.tensor(T)
|
| 34 |
+
def nll(t): return torch.nn.functional.cross_entropy(M / t, y).item()
|
| 35 |
+
ts = np.exp(np.linspace(np.log(0.2), np.log(10), 200)); best = min(ts, key=nll)
|
| 36 |
+
acc = (M.argmax(-1) == y).float().mean().item()
|
| 37 |
+
conf0 = torch.softmax(M, -1).max(-1).values.mean().item(); conf1 = torch.softmax(M / best, -1).max(-1).values.mean().item()
|
| 38 |
+
print(f"n={len(Z)} acc={acc:.3f} T=1: nll {nll(1.0):.3f} mean conf {conf0:.3f} | T={best:.2f}: nll {nll(best):.3f} mean conf {conf1:.3f}")
|
| 39 |
+
cfgp = os.path.join(ck, "rl_agent_config.json"); cfg = json.load(open(cfgp)); cfg["temperature"] = [float(best), 1.0, 1.0]
|
| 40 |
+
json.dump(cfg, open(cfgp, "w"), indent=2); print("wrote temperature", best, "->", cfgp)
|
| 41 |
+
|
| 42 |
+
if __name__ == "__main__":
|
| 43 |
+
main()
|
code/finetune/collect_pages.py
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Crawl real pages with browser-harness and dump their jev-ultrafast observations (element tables + page text).
|
| 2 |
+
|
| 3 |
+
python finetune/collect_pages.py out/pages.jsonl [max_pages=80]
|
| 4 |
+
|
| 5 |
+
Starts from SEEDS, follows a few random same-site links from each page to diversify. One JSON line per page.
|
| 6 |
+
"""
|
| 7 |
+
import json, os, random, sys, time
|
| 8 |
+
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast")
|
| 9 |
+
os.environ.setdefault("BU_CDP_URL", "http://127.0.0.1:9222")
|
| 10 |
+
from jev_ultrafast.browser import Browser, StalePage
|
| 11 |
+
from jev_ultrafast.model import action_space
|
| 12 |
+
|
| 13 |
+
SEEDS = [
|
| 14 |
+
"https://en.wikipedia.org/wiki/Main_Page", "https://en.wikipedia.org/wiki/Special:Random", "https://en.wikipedia.org/wiki/Python_(programming_language)",
|
| 15 |
+
"https://news.ycombinator.com/", "https://news.ycombinator.com/newest", "https://news.ycombinator.com/login",
|
| 16 |
+
"https://github.com/", "https://github.com/tile-ai/tilelang", "https://github.com/browser-use/browser-use/issues", "https://github.com/login",
|
| 17 |
+
"https://www.python.org/", "https://docs.python.org/3/", "https://pypi.org/", "https://pypi.org/project/laya/",
|
| 18 |
+
"https://archlinux.org/", "https://wiki.archlinux.org/", "https://developer.mozilla.org/en-US/", "https://duckduckgo.com/",
|
| 19 |
+
"https://www.bing.com/", "https://stackoverflow.com/questions", "https://www.reddit.com/", "https://arxiv.org/",
|
| 20 |
+
"https://arxiv.org/list/cs.LG/recent", "https://huggingface.co/models", "https://huggingface.co/convaiinnovations/laya",
|
| 21 |
+
"https://www.saucedemo.com/", "https://the-internet.herokuapp.com/", "https://the-internet.herokuapp.com/login",
|
| 22 |
+
"https://demo.opencart.com/", "https://www.demoblaze.com/", "https://books.toscrape.com/", "https://quotes.toscrape.com/login",
|
| 23 |
+
"https://www.google.com/travel/flights?hl=en", "https://www.booking.com/", "https://www.airbnb.com/", "https://www.amazon.com/",
|
| 24 |
+
"https://www.ebay.com/", "https://www.imdb.com/", "https://www.nytimes.com/", "https://www.bbc.com/",
|
| 25 |
+
"https://www.openstreetmap.org/", "https://weather.com/", "https://www.wolframalpha.com/", "https://translate.google.com/",
|
| 26 |
+
"https://www.gnu.org/", "https://kernel.org/", "https://www.rust-lang.org/", "https://go.dev/", "https://nodejs.org/en",
|
| 27 |
+
"https://www.npmjs.com/", "https://crates.io/", "https://docs.rs/", "https://www.kaggle.com/", "https://paperswithcode.com/",
|
| 28 |
+
# round 2: more sites, more form-heavy pages
|
| 29 |
+
"https://en.wikipedia.org/wiki/Special:Search", "https://en.wikipedia.org/wiki/Portal:Current_events", "https://de.wikipedia.org/", "https://zh.wikipedia.org/",
|
| 30 |
+
"https://news.ycombinator.com/ask", "https://news.ycombinator.com/show", "https://news.ycombinator.com/jobs", "https://news.ycombinator.com/submit",
|
| 31 |
+
"https://github.com/explore", "https://github.com/trending", "https://github.com/pytorch/pytorch", "https://github.com/pytorch/pytorch/pulls",
|
| 32 |
+
"https://github.com/pytorch/pytorch/issues", "https://gitlab.com/explore", "https://gitee.com/explore", "https://about.gitlab.com/",
|
| 33 |
+
"https://www.python.org/downloads/", "https://docs.python.org/3/tutorial/", "https://docs.python.org/3/library/", "https://peps.python.org/",
|
| 34 |
+
"https://pypi.org/search/?q=torch", "https://pypi.org/project/torch/", "https://pypi.org/account/login/", "https://pypi.org/help/",
|
| 35 |
+
"https://the-internet.herokuapp.com/dropdown", "https://the-internet.herokuapp.com/checkboxes", "https://the-internet.herokuapp.com/forgot_password",
|
| 36 |
+
"https://the-internet.herokuapp.com/inputs", "https://the-internet.herokuapp.com/tables", "https://the-internet.herokuapp.com/javascript_alerts",
|
| 37 |
+
"https://demo.opencart.com/index.php?route=product/category&path=20", "https://demo.opencart.com/index.php?route=account/login",
|
| 38 |
+
"https://demo.opencart.com/index.php?route=account/register", "https://demo.opencart.com/index.php?route=product/search&search=mac",
|
| 39 |
+
"https://www.demoblaze.com/cart.html", "https://www.demoblaze.com/prod.html?idp_=1", "https://books.toscrape.com/catalogue/page-2.html",
|
| 40 |
+
"https://books.toscrape.com/catalogue/category/books/mystery_3/index.html", "https://quotes.toscrape.com/tag/love/", "https://quotes.toscrape.com/page/2/",
|
| 41 |
+
"https://www.saucedemo.com/inventory.html", "https://automationexercise.com/", "https://automationexercise.com/login", "https://automationexercise.com/products",
|
| 42 |
+
"https://practicetestautomation.com/practice-test-login/", "https://demoqa.com/", "https://demoqa.com/text-box", "https://demoqa.com/select-menu",
|
| 43 |
+
"https://demoqa.com/webtables", "https://www.selenium.dev/selenium/web/web-form.html", "https://formy-project.herokuapp.com/", "https://formy-project.herokuapp.com/form",
|
| 44 |
+
"https://parabank.parasoft.com/parabank/index.htm", "https://parabank.parasoft.com/parabank/register.htm", "https://www.globalsqa.com/angularJs-protractor/BankingProject/",
|
| 45 |
+
"https://opensource-demo.orangehrmlive.com/", "https://magento.softwaretestingboard.com/", "https://magento.softwaretestingboard.com/women.html",
|
| 46 |
+
"https://www.airbnb.com/s/London/homes", "https://www.booking.com/searchresults.html?ss=Paris", "https://www.google.com/travel/hotels?hl=en",
|
| 47 |
+
"https://www.google.com/maps?hl=en", "https://www.google.com/search?q=tilelang&hl=en", "https://duckduckgo.com/?q=modernbert", "https://www.bing.com/search?q=laya",
|
| 48 |
+
"https://arxiv.org/list/cs.CL/new", "https://arxiv.org/abs/2412.13663", "https://arxiv.org/search/?query=flash+attention&searchtype=all",
|
| 49 |
+
"https://huggingface.co/datasets", "https://huggingface.co/spaces", "https://huggingface.co/docs", "https://huggingface.co/login", "https://huggingface.co/answerdotai/ModernBERT-base",
|
| 50 |
+
"https://www.modelscope.cn/models", "https://www.modelscope.cn/datasets", "https://www.kaggle.com/datasets", "https://www.kaggle.com/competitions",
|
| 51 |
+
"https://stackoverflow.com/", "https://stackoverflow.com/questions/tagged/python", "https://superuser.com/", "https://askubuntu.com/",
|
| 52 |
+
"https://www.reddit.com/r/MachineLearning/", "https://old.reddit.com/", "https://old.reddit.com/r/python/", "https://lobste.rs/",
|
| 53 |
+
"https://www.bbc.com/news", "https://www.bbc.com/sport", "https://www.theguardian.com/international", "https://www.reuters.com/", "https://apnews.com/",
|
| 54 |
+
"https://www.imdb.com/chart/top/", "https://www.imdb.com/find/?q=inception", "https://www.rottentomatoes.com/", "https://www.goodreads.com/",
|
| 55 |
+
"https://www.openstreetmap.org/search?query=Berlin", "https://www.wikidata.org/", "https://commons.wikimedia.org/", "https://www.wiktionary.org/",
|
| 56 |
+
"https://developer.mozilla.org/en-US/docs/Web/JavaScript", "https://developer.mozilla.org/en-US/docs/Web/HTML/Element/select", "https://web.dev/",
|
| 57 |
+
"https://www.rust-lang.org/learn", "https://doc.rust-lang.org/book/", "https://go.dev/doc/", "https://pkg.go.dev/", "https://nodejs.org/en/download",
|
| 58 |
+
"https://www.npmjs.com/package/react", "https://react.dev/", "https://vuejs.org/", "https://tailwindcss.com/docs", "https://getbootstrap.com/docs/",
|
| 59 |
+
"https://www.wolframalpha.com/input?i=2%2B2", "https://translate.google.com/?sl=en&tl=zh-CN&text=hello", "https://www.deepl.com/translator",
|
| 60 |
+
"https://weather.com/weather/today/l/USNY0996", "https://www.timeanddate.com/", "https://www.xe.com/currencyconverter/", "https://www.calculator.net/",
|
| 61 |
+
"https://archlinux.org/packages/", "https://aur.archlinux.org/", "https://wiki.archlinux.org/title/Installation_guide", "https://www.kernel.org/doc/",
|
| 62 |
+
"https://www.debian.org/", "https://ubuntu.com/download", "https://www.gnu.org/software/", "https://www.fsf.org/",
|
| 63 |
+
]
|
| 64 |
+
|
| 65 |
+
def observe(url, timeout=25):
|
| 66 |
+
b = Browser(url)
|
| 67 |
+
try:
|
| 68 |
+
page = b.observe(screenshot=False)
|
| 69 |
+
links = b.evaluate("""(() => { const out=[]; for (const a of document.querySelectorAll('a[href]')) {
|
| 70 |
+
const h=a.href; if (h.startsWith(location.origin) && !h.includes('#') && h!==location.href) out.push(h); } return out.slice(0,400); })()""") or []
|
| 71 |
+
return page, links
|
| 72 |
+
finally:
|
| 73 |
+
b.close()
|
| 74 |
+
|
| 75 |
+
def main():
|
| 76 |
+
out, max_pages = sys.argv[1], int(sys.argv[2]) if len(sys.argv) > 2 else 80
|
| 77 |
+
os.makedirs(os.path.dirname(out) or ".", exist_ok=True)
|
| 78 |
+
seen = set()
|
| 79 |
+
if os.path.exists(out):
|
| 80 |
+
for line in open(out):
|
| 81 |
+
seen.add(json.loads(line)["url"])
|
| 82 |
+
rng = random.Random(0)
|
| 83 |
+
queue = list(SEEDS); rng.shuffle(queue)
|
| 84 |
+
n = len(seen)
|
| 85 |
+
with open(out, "a") as f:
|
| 86 |
+
while queue and n < max_pages:
|
| 87 |
+
url = queue.pop(0)
|
| 88 |
+
if url in seen:
|
| 89 |
+
continue
|
| 90 |
+
t = time.time()
|
| 91 |
+
try:
|
| 92 |
+
page, links = observe(url)
|
| 93 |
+
except Exception as e:
|
| 94 |
+
print(f"skip {url}: {type(e).__name__}: {str(e)[:80]}", flush=True); continue
|
| 95 |
+
elements, targets, controls = action_space(page["actions"])
|
| 96 |
+
if len(elements) < 5 or len(elements) > 160:
|
| 97 |
+
print(f"skip {url}: {len(elements)} elements", flush=True); continue
|
| 98 |
+
seen.add(page["url"]); n += 1
|
| 99 |
+
f.write(json.dumps({"url": page["url"], "title": page["title"], "text": page["text"], "actions": page["actions"],
|
| 100 |
+
"scroll": page.get("scroll")}, ensure_ascii=False) + "\n"); f.flush()
|
| 101 |
+
print(f"[{n}] {len(elements):3d} elements {time.time()-t:4.1f}s {page['title'][:60]}", flush=True)
|
| 102 |
+
rng.shuffle(links)
|
| 103 |
+
queue.extend(l for l in links[:3] if l not in seen)
|
| 104 |
+
print("done", n, "pages")
|
| 105 |
+
|
| 106 |
+
if __name__ == "__main__":
|
| 107 |
+
main()
|
code/finetune/common_ft.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Shared: build the exact jev-ultrafast systemone request for a page + goal, with the server-side compaction."""
|
| 2 |
+
import importlib.util, json, sys, types
|
| 3 |
+
|
| 4 |
+
JEV = "/home/ckl/projects/S/jev-ultrafast/jev_ultrafast"
|
| 5 |
+
# load model.py / questions.py without running the package __init__ (which pulls in browser-harness)
|
| 6 |
+
if "jev_ultrafast" not in sys.modules:
|
| 7 |
+
_pkg = types.ModuleType("jev_ultrafast"); _pkg.__path__ = [JEV]; sys.modules["jev_ultrafast"] = _pkg
|
| 8 |
+
for _name in ("questions", "model"):
|
| 9 |
+
_spec = importlib.util.spec_from_file_location(f"jev_ultrafast.{_name}", f"{JEV}/{_name}.py")
|
| 10 |
+
_m = importlib.util.module_from_spec(_spec); sys.modules[_spec.name] = _m; _spec.loader.exec_module(_m)
|
| 11 |
+
from jev_ultrafast.model import action_space # noqa: E402
|
| 12 |
+
from jev_ultrafast.questions import NEXT_ACTION, TARGET # noqa: E402
|
| 13 |
+
|
| 14 |
+
import os
|
| 15 |
+
FMT = os.environ.get("LAYA_FMT", "v1")
|
| 16 |
+
# v1: jev's state verbatim (page text up to 6000 chars + the whole element table as JSON) -- the 1024-token budget truncates
|
| 17 |
+
# most of it, so the model often never sees the candidates' context. 3000 chars was tried (v7): -0.04 top-1.
|
| 18 |
+
# v2: elements live only in the option list (full label + role + value); state keeps title/url/history and 1500 chars of text.
|
| 19 |
+
# v3: v2 + option labels capped at 50 chars and 1200 chars of text (~30% fewer tokens; for the 322M base to hit ~20 ms/step)
|
| 20 |
+
PAGE_TEXT_CHARS = {"v2": 1500, "v3": 1200}.get(FMT, 6000)
|
| 21 |
+
LABEL_CHARS = 50 if FMT == "v3" else 10000
|
| 22 |
+
|
| 23 |
+
LABELS = {
|
| 24 |
+
"CLICK": "Click an element, button, menu option, autocomplete suggestion, or calendar day.",
|
| 25 |
+
"TYPE_TEXT": "Enter or replace text in an editable field. A small LLM will supply the value from the goal.",
|
| 26 |
+
"SELECT": "Select an observed dropdown value.",
|
| 27 |
+
}
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def compact(v):
|
| 31 |
+
if isinstance(v, dict) and "element" in v:
|
| 32 |
+
s = str(v["element"])[:LABEL_CHARS]
|
| 33 |
+
if v.get("role"):
|
| 34 |
+
s += f" ({v['role']})"
|
| 35 |
+
if v.get("current_value"):
|
| 36 |
+
s += f" = {str(v['current_value'])[:30]!r}"
|
| 37 |
+
for k in ("checked", "selected", "expanded"):
|
| 38 |
+
if k in v:
|
| 39 |
+
s += f" {k}={v[k]}"
|
| 40 |
+
return s
|
| 41 |
+
return v
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
def build_request(page, goal, history=()):
|
| 45 |
+
"""Mirror of jev_ultrafast.model.choose() up to the HTTP call. Returns (state, questions, targets, controls)."""
|
| 46 |
+
elements, targets, controls = action_space(page["actions"])
|
| 47 |
+
operations = {key: LABELS[key] for key in targets}
|
| 48 |
+
operations.update({key: value["label"] for key, value in controls.items()})
|
| 49 |
+
operations.update(DONE="Every requirement is visibly satisfied.", BLOCKED="No supported operation can progress.")
|
| 50 |
+
questions = {"operation": {"type": "choice", "criteria": operations, "instructions": {"goal": goal, "rules": NEXT_ACTION}}}
|
| 51 |
+
for operation, candidates in targets.items():
|
| 52 |
+
questions[operation.lower() + "_target"] = {
|
| 53 |
+
"type": "choice",
|
| 54 |
+
"criteria": {index: {"element": f"[{index}] {a['label']}", "current_value": a.get("current_value", a.get("value", "")),
|
| 55 |
+
**{k: a[k] for k in ("role", "checked", "selected", "expanded") if k in a}} for index, a in candidates.items()},
|
| 56 |
+
"instructions": {"goal": goal, "operation": operation, "rules": [NEXT_ACTION, TARGET]},
|
| 57 |
+
}
|
| 58 |
+
state = {"page": {"url": page["url"], "title": page["title"], "text": page["text"][:PAGE_TEXT_CHARS]},
|
| 59 |
+
"recent_actions": [{k: h.get(k) for k in ("action", "kind", "text", "page_changed")} for h in list(history)[-10:]]}
|
| 60 |
+
if FMT not in ("v2", "v3"):
|
| 61 |
+
state["elements"] = elements
|
| 62 |
+
for q in questions.values():
|
| 63 |
+
q["criteria"] = {k: compact(v) for k, v in q["criteria"].items()}
|
| 64 |
+
return state, questions, targets, controls
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
def gold_for(case, targets, controls):
|
| 68 |
+
"""(gold operation key, gold target index or None) for a case in the operation/target question vocab."""
|
| 69 |
+
op = case["gold_op"]
|
| 70 |
+
if op == "DONE":
|
| 71 |
+
return "DONE", None
|
| 72 |
+
if op in ("SCROLL_DOWN", "SCROLL_UP", "WAIT"): # page-level controls: operation question only
|
| 73 |
+
return (op, None) if op in {k.upper() for k in controls} else (None, None)
|
| 74 |
+
for index, a in targets.get(op, {}).items():
|
| 75 |
+
if a["id"] == case["gold_id"]:
|
| 76 |
+
return op, index
|
| 77 |
+
return None, None
|
code/finetune/convert_mind2web.py
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Mind2Web (osunlp/Mind2Web from ModelScope) -> cases in this repo's format (page_obj with jev-style actions).
|
| 2 |
+
|
| 3 |
+
python finetune/convert_mind2web.py finetune/data/mind2web/data/train/*.json finetune/out/m2w_cases.jsonl
|
| 4 |
+
|
| 5 |
+
Each Mind2Web action becomes one case: goal = confirmed_task, history = previous action_reprs, gold = the positive
|
| 6 |
+
candidate; the element table = positive + up to MAX_NEG sampled negative candidates rendered like jev's snapshot.js
|
| 7 |
+
(label from text / aria-label / placeholder / alt / title / value, role from tag).
|
| 8 |
+
"""
|
| 9 |
+
import json, random, re, sys
|
| 10 |
+
from bs4 import BeautifulSoup
|
| 11 |
+
|
| 12 |
+
MAX_NEG = 44
|
| 13 |
+
ROLE = {"a": "link", "button": "button", "input": "textbox", "textarea": "textbox", "select": "combobox", "option": "option",
|
| 14 |
+
"img": "img", "li": "listitem", "label": "label", "span": "generic", "div": "generic", "svg": "img", "h1": "heading",
|
| 15 |
+
"h2": "heading", "h3": "heading", "p": "text", "td": "cell", "th": "columnheader", "tr": "row", "ul": "list"}
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
def label_of(el):
|
| 19 |
+
attrs = el.attrs
|
| 20 |
+
for k in ("aria-label", "placeholder", "alt", "title"):
|
| 21 |
+
v = attrs.get(k)
|
| 22 |
+
if isinstance(v, list): v = " ".join(v)
|
| 23 |
+
if v and v.strip(): return v.strip()
|
| 24 |
+
txt = " ".join(el.get_text(" ", strip=True).split())
|
| 25 |
+
if txt: return txt[:80]
|
| 26 |
+
v = attrs.get("value")
|
| 27 |
+
if v: return str(v).strip()[:80]
|
| 28 |
+
return (attrs.get("name") or attrs.get("id") or el.name or "")[:60]
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
def kind_of(el, op):
|
| 32 |
+
t = el.name
|
| 33 |
+
typ = (el.attrs.get("type") or "").lower()
|
| 34 |
+
if t == "select": return "select"
|
| 35 |
+
if t == "textarea" or (t == "input" and typ in ("", "text", "search", "email", "password", "number", "tel", "url")) or el.attrs.get("contenteditable"):
|
| 36 |
+
return "fill"
|
| 37 |
+
return "click"
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
def convert_action(task, ai, action, rng):
|
| 41 |
+
soup = BeautifulSoup(action["cleaned_html"], "lxml")
|
| 42 |
+
by_id = {}
|
| 43 |
+
for el in soup.find_all(attrs={"backend_node_id": True}):
|
| 44 |
+
by_id[el.attrs["backend_node_id"]] = el
|
| 45 |
+
pos = action["pos_candidates"]
|
| 46 |
+
if not pos: return None
|
| 47 |
+
gold_el = by_id.get(pos[0]["backend_node_id"])
|
| 48 |
+
if gold_el is None: return None
|
| 49 |
+
negs = [c for c in action["neg_candidates"] if c["backend_node_id"] in by_id]
|
| 50 |
+
rng.shuffle(negs)
|
| 51 |
+
cands = [pos[0]] + negs[:MAX_NEG]
|
| 52 |
+
rng.shuffle(cands)
|
| 53 |
+
op = action["operation"]["op"]
|
| 54 |
+
# values typed/selected in earlier steps: a field that already holds its value must show it (jev's rules key on that)
|
| 55 |
+
filled = {}
|
| 56 |
+
for r in task["action_reprs"][:ai]:
|
| 57 |
+
m = re.match(r"\[(\w+)\]\s+(.*?)\s+->\s+(TYPE|SELECT):\s*(.*)$", r)
|
| 58 |
+
if m: filled[" ".join(m.group(2).split()).lower()] = m.group(4).strip()
|
| 59 |
+
actions, gold_id, node = [], None, 0
|
| 60 |
+
for c in cands:
|
| 61 |
+
el = by_id[c["backend_node_id"]]
|
| 62 |
+
lab = label_of(el)
|
| 63 |
+
if not lab: continue
|
| 64 |
+
node += 1
|
| 65 |
+
is_gold = c["backend_node_id"] == pos[0]["backend_node_id"]
|
| 66 |
+
kind = {"CLICK": "click", "TYPE": "fill", "SELECT": "select"}[op] if is_gold else kind_of(el, op)
|
| 67 |
+
role = ROLE.get(el.name, "generic")
|
| 68 |
+
base = {"node": node, "label": lab, "role": role}
|
| 69 |
+
if kind == "select":
|
| 70 |
+
opts = [o.get_text(" ", strip=True) for o in el.find_all("option")][:8] or [action["operation"]["value"] or "option"]
|
| 71 |
+
if is_gold and action["operation"]["value"] and action["operation"]["value"] not in opts:
|
| 72 |
+
opts = [action["operation"]["value"]] + opts[:7]
|
| 73 |
+
for oi, o in enumerate(opts):
|
| 74 |
+
a = {**base, "id": f"select:{node}:{oi}", "kind": "select", "value": o, "label": f"{lab} → {o}", "current_value": ""}
|
| 75 |
+
actions.append(a)
|
| 76 |
+
if is_gold and (o == action["operation"]["value"] or (oi == 0 and not action["operation"]["value"])): gold_id = a["id"]
|
| 77 |
+
continue
|
| 78 |
+
a = {**base, "id": f"{kind}:{node}", "kind": kind}
|
| 79 |
+
if kind == "fill":
|
| 80 |
+
a["value"] = filled.get(lab.lower(), "")
|
| 81 |
+
a["current_value"] = a["value"]
|
| 82 |
+
actions.append(a)
|
| 83 |
+
if is_gold: gold_id = a["id"]
|
| 84 |
+
if gold_id is None: return None
|
| 85 |
+
for k, lab in (("wait", "Wait for the page to update"), ("scroll_down", "Scroll down"), ("scroll_up", "Scroll up")):
|
| 86 |
+
actions.append({"id": k, "kind": "wait" if k == "wait" else "scroll", "label": lab, "node": None, "delta": 600 if k == "scroll_down" else -600})
|
| 87 |
+
text = " ".join(soup.get_text(" ", strip=True).split())[:6000]
|
| 88 |
+
hist = []
|
| 89 |
+
for r in task["action_reprs"][:ai]:
|
| 90 |
+
lab_, _, opv = r.rpartition(" -> ")
|
| 91 |
+
kind_, _, val = opv.partition(": ")
|
| 92 |
+
hist.append({"action": lab_.strip(), "kind": {"CLICK": "click", "TYPE": "fill", "SELECT": "select"}.get(kind_, kind_.lower()),
|
| 93 |
+
"text": val or None, "page_changed": kind_ == "CLICK"})
|
| 94 |
+
gold_op = {"CLICK": "CLICK", "TYPE": "TYPE_TEXT", "SELECT": "SELECT"}[op]
|
| 95 |
+
return {"page": -1, "url": f"https://{task['website']}.com/", "title": task["website"], "goal": task["confirmed_task"], "gold_op": gold_op,
|
| 96 |
+
"gold_id": gold_id, "kind": {"CLICK": "click", "TYPE": "fill", "SELECT": "select"}[op], "label": label_of(gold_el), "history": hist,
|
| 97 |
+
"source": "mind2web", "task_id": task["annotation_id"], "website": task["website"],
|
| 98 |
+
"page_obj": {"url": f"https://{task['website']}.com/", "title": task["website"], "text": text, "actions": actions}}
|
| 99 |
+
|
| 100 |
+
|
| 101 |
+
def main():
|
| 102 |
+
files, out = sys.argv[1:-1], sys.argv[-1]
|
| 103 |
+
rng = random.Random(0); n = 0; skipped = 0
|
| 104 |
+
with open(out, "w") as f:
|
| 105 |
+
for fn in files:
|
| 106 |
+
tasks = json.load(open(fn))
|
| 107 |
+
for t in tasks:
|
| 108 |
+
for ai, a in enumerate(t["actions"]):
|
| 109 |
+
c = convert_action(t, ai, a, rng)
|
| 110 |
+
if c is None: skipped += 1; continue
|
| 111 |
+
f.write(json.dumps(c, ensure_ascii=False) + "\n"); n += 1
|
| 112 |
+
print(f"{fn}: total {n} cases, skipped {skipped}", flush=True)
|
| 113 |
+
print("wrote", n)
|
| 114 |
+
|
| 115 |
+
if __name__ == "__main__":
|
| 116 |
+
main()
|
code/finetune/dagger.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""DAgger-style harvesting: run real tasks with the current laya agent, and at every step ask the local Qwen teacher which
|
| 2 |
+
action is right given the page's element table. Disagreements (and agreements) become training cases with the *agent's*
|
| 3 |
+
on-policy states, so the next model learns exactly where this one goes wrong.
|
| 4 |
+
|
| 5 |
+
python finetune/dagger.py out/dagger_cases.jsonl [tasks.jsonl] (services: chromium 9222, laya 8791, sglang 30000)
|
| 6 |
+
|
| 7 |
+
tasks.jsonl lines: {"url": ..., "goal": ...}; default = apps/browser_suite.TASKS plus extra goals below.
|
| 8 |
+
"""
|
| 9 |
+
import json, os, sys, time
|
| 10 |
+
import httpx
|
| 11 |
+
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast"); sys.path.insert(0, "/home/ckl/projects/S/laya/apps")
|
| 12 |
+
os.environ.update(BU_CDP_URL="http://127.0.0.1:9222", TYPESAFE_BASE_URL="http://127.0.0.1:8791", TYPESAFE_API_KEY="local",
|
| 13 |
+
TEXT_MODEL_API_KEY="local", TEXT_MODEL_BASE_URL="http://127.0.0.1:30000/v1", TEXT_MODEL="Qwen/Qwen3-8B-AWQ",
|
| 14 |
+
TEXT_MODEL_EXTRA_JSON='{"chat_template_kwargs": {"enable_thinking": false}}')
|
| 15 |
+
from jev_ultrafast import Agent
|
| 16 |
+
from jev_ultrafast.model import action_space
|
| 17 |
+
from browser_suite import TASKS
|
| 18 |
+
|
| 19 |
+
EXTRA = [
|
| 20 |
+
("https://en.wikipedia.org/wiki/Main_Page", "Open the article about Albert Einstein."),
|
| 21 |
+
("https://news.ycombinator.com/", "Open the 'past' page."),
|
| 22 |
+
("https://news.ycombinator.com/", "Open the comments of the first story on the front page."),
|
| 23 |
+
("https://github.com/tile-ai/tilelang", "Open the Pull requests tab."),
|
| 24 |
+
("https://github.com/tile-ai/tilelang", "Open the README's 'examples' folder."),
|
| 25 |
+
("https://www.python.org/", "Open the documentation page."),
|
| 26 |
+
("https://docs.python.org/3/", "Open the tutorial."),
|
| 27 |
+
("https://books.toscrape.com/", "Open the 'Mystery' category and then open the first book in it."),
|
| 28 |
+
("https://books.toscrape.com/", "Go to page 2 of the catalogue."),
|
| 29 |
+
("https://the-internet.herokuapp.com/", "Open the 'Dropdown' example and select 'Option 2'."),
|
| 30 |
+
("https://the-internet.herokuapp.com/", "Open the 'Checkboxes' example and tick the first checkbox."),
|
| 31 |
+
("https://quotes.toscrape.com/", "Open the quotes tagged 'love'."),
|
| 32 |
+
("https://quotes.toscrape.com/", "Go to the next page of quotes."),
|
| 33 |
+
("https://arxiv.org/", "Open the listing of new submissions in cs.CL."),
|
| 34 |
+
("https://pypi.org/", "Search PyPI for 'tilelang' and open the project page."),
|
| 35 |
+
("https://duckduckgo.com/", "Search for 'ModernBERT paper'."),
|
| 36 |
+
("https://www.saucedemo.com/", "Log in with username 'standard_user' and password 'secret_sauce', then add the 'Sauce Labs Backpack' to the cart."),
|
| 37 |
+
("https://demo.opencart.com/", "Open the 'Desktops' category from the top menu."),
|
| 38 |
+
("https://www.demoblaze.com/", "Open the 'Laptops' category."),
|
| 39 |
+
("https://huggingface.co/models", "Search models for 'laya'."),
|
| 40 |
+
]
|
| 41 |
+
TEACHER = os.environ["TEXT_MODEL_BASE_URL"] + "/chat/completions"
|
| 42 |
+
client = httpx.Client(timeout=180)
|
| 43 |
+
SYS = """You are the teacher for a browser agent. You see the user's goal, the actions taken so far, the current page (title, url, text excerpt) and a
|
| 44 |
+
numbered table of the controls on it. Decide the single best NEXT step:
|
| 45 |
+
- {"operation": "CLICK", "index": n} click control n
|
| 46 |
+
- {"operation": "TYPE_TEXT", "index": n} type into text control n (a separate helper supplies the value)
|
| 47 |
+
- {"operation": "SELECT", "index": n, "value": "..."} choose that option of select control n
|
| 48 |
+
- {"operation": "DONE"} every requirement of the goal is already visibly satisfied on this page
|
| 49 |
+
- {"operation": "WAIT"} / {"operation": "SCROLL_DOWN"} / {"operation": "SCROLL_UP"}
|
| 50 |
+
Do not re-do satisfied steps; a field that already shows the requested value is done. Return JSON only."""
|
| 51 |
+
|
| 52 |
+
def teach(goal, history, page, elements):
|
| 53 |
+
table = [{"index": e["index"], "label": e["label"][:70], "role": e.get("role"), "ops": e["operations"], **({"value": e["value"]} if e.get("value") else {})} for e in elements[:120]]
|
| 54 |
+
user = {"goal": goal, "actions_so_far": [{k: h.get(k) for k in ("action", "kind", "text")} for h in history[-8:]],
|
| 55 |
+
"page": {"title": page["title"], "url": page["url"], "text": page["text"][:2500]}, "controls": table}
|
| 56 |
+
body = {"model": os.environ["TEXT_MODEL"], "max_tokens": 120, "temperature": 0.0, "response_format": {"type": "json_object"},
|
| 57 |
+
"chat_template_kwargs": {"enable_thinking": False},
|
| 58 |
+
"messages": [{"role": "system", "content": SYS}, {"role": "user", "content": json.dumps(user, ensure_ascii=False)}]}
|
| 59 |
+
r = client.post(TEACHER, json=body).json()
|
| 60 |
+
return json.loads(r["choices"][0]["message"]["content"])
|
| 61 |
+
|
| 62 |
+
def to_case(goal, history, page, verdict):
|
| 63 |
+
elements, targets, controls = action_space(page["actions"])
|
| 64 |
+
op = str(verdict.get("operation", "")).upper()
|
| 65 |
+
if op in ("DONE",):
|
| 66 |
+
return {"page": -1, "url": page["url"], "title": page["title"], "goal": goal, "gold_op": "DONE", "gold_id": "DONE", "kind": "done", "label": "",
|
| 67 |
+
"history": history, "source": "dagger", "page_obj": {k: page[k] for k in ("url", "title", "text", "actions")}}
|
| 68 |
+
if op in ("CLICK", "TYPE_TEXT", "SELECT"):
|
| 69 |
+
idx = str(verdict.get("index"))
|
| 70 |
+
cands = targets.get(op, {})
|
| 71 |
+
if op == "SELECT":
|
| 72 |
+
hit = [k for k, a in cands.items() if k.split(":")[0] == idx and (a["value"] == verdict.get("value") or a["label"].endswith(str(verdict.get("value"))))]
|
| 73 |
+
key = hit[0] if hit else None
|
| 74 |
+
else:
|
| 75 |
+
key = idx if idx in cands else None
|
| 76 |
+
if key is None: return None
|
| 77 |
+
a = cands[key]
|
| 78 |
+
return {"page": -1, "url": page["url"], "title": page["title"], "goal": goal, "gold_op": op, "gold_id": a["id"], "kind": a["kind"], "label": a["label"],
|
| 79 |
+
"history": history, "source": "dagger", "page_obj": {k: page[k] for k in ("url", "title", "text", "actions")}}
|
| 80 |
+
return None
|
| 81 |
+
|
| 82 |
+
def main():
|
| 83 |
+
out = sys.argv[1]
|
| 84 |
+
tasks = [(u, g) for _, u, g, _ in TASKS] + EXTRA
|
| 85 |
+
if len(sys.argv) > 2:
|
| 86 |
+
tasks += [(json.loads(l)["url"], json.loads(l)["goal"]) for l in open(sys.argv[2])]
|
| 87 |
+
n_cases = n_dis = 0
|
| 88 |
+
with open(out, "a") as f:
|
| 89 |
+
for url, goal in tasks:
|
| 90 |
+
t0 = time.time()
|
| 91 |
+
try:
|
| 92 |
+
with Agent(url, goal) as agent:
|
| 93 |
+
steps = 0
|
| 94 |
+
while agent.state["status"] not in ("done", "blocked") and steps < 12:
|
| 95 |
+
page = agent.state["page"]; history = list(agent.state["history"])
|
| 96 |
+
elements = action_space(page["actions"])[0]
|
| 97 |
+
try:
|
| 98 |
+
verdict = teach(goal, history, page, elements)
|
| 99 |
+
except Exception as e:
|
| 100 |
+
print(" teacher fail", str(e)[:60]); break
|
| 101 |
+
case = to_case(goal, [{k: h.get(k) for k in ("action", "kind", "text", "page_changed")} for h in history], page, verdict)
|
| 102 |
+
if case:
|
| 103 |
+
f.write(json.dumps(case, ensure_ascii=False) + "\n"); f.flush(); n_cases += 1
|
| 104 |
+
# step the agent with its own policy
|
| 105 |
+
try:
|
| 106 |
+
st = agent.command("tick")
|
| 107 |
+
except Exception as e:
|
| 108 |
+
print(" tick fail", type(e).__name__, str(e)[:50]); break
|
| 109 |
+
d = st["decisions"][-1] if st["decisions"] else None
|
| 110 |
+
agent_choice = (d["operation"], d.get("target")) if d else None
|
| 111 |
+
teacher_choice = (str(verdict.get("operation", "")).upper(), str(verdict.get("index")) if verdict.get("index") is not None else None)
|
| 112 |
+
if agent_choice and agent_choice[0] != teacher_choice[0] or (agent_choice and agent_choice[1] != teacher_choice[1] and teacher_choice[0] in ("CLICK", "TYPE_TEXT")):
|
| 113 |
+
n_dis += 1
|
| 114 |
+
steps += 1
|
| 115 |
+
except Exception as e:
|
| 116 |
+
print(f" task fail {type(e).__name__}: {str(e)[:60]}")
|
| 117 |
+
print(f"{goal[:60]:60s} cases={n_cases} disagreements={n_dis} {time.time()-t0:.0f}s", flush=True)
|
| 118 |
+
print("wrote", n_cases, "cases ->", out)
|
| 119 |
+
|
| 120 |
+
if __name__ == "__main__":
|
| 121 |
+
main()
|
code/finetune/eval.py
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Held-out eval: operation accuracy and target top-1 (given the gold operation) on unseen pages.
|
| 2 |
+
|
| 3 |
+
python finetune/eval.py out/pages.jsonl out/eval_cases.jsonl <checkpoint dir> [subfolder]
|
| 4 |
+
"""
|
| 5 |
+
import json, os, sys, time
|
| 6 |
+
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
| 7 |
+
sys.path.insert(0, "/home/ckl/projects/S/laya-upstream") # laya with Agent.accelerate()
|
| 8 |
+
from common_ft import build_request
|
| 9 |
+
import laya
|
| 10 |
+
|
| 11 |
+
def main():
|
| 12 |
+
pages = [json.loads(l) for l in open(sys.argv[1])]; cases = [json.loads(l) for l in open(sys.argv[2])]
|
| 13 |
+
agent = laya.load(sys.argv[3], subfolder=sys.argv[4] if len(sys.argv) > 4 else None)
|
| 14 |
+
agent.cfg["max_len"], agent.cfg["head_max_len"] = 1024, int(os.environ.get("LAYA_HEAD", agent.cfg.get("head_max_len_train", 512)))
|
| 15 |
+
agent.accelerate()
|
| 16 |
+
op_ok = tgt_ok = tgt_n = 0; ranks = []; t = time.time(); by_kind = {}
|
| 17 |
+
for c in cases:
|
| 18 |
+
state, questions, targets, controls = build_request(c.get("page_obj") or pages[c["page"]], c["goal"], c.get("history", []))
|
| 19 |
+
r = agent.predict(state, questions)["answers"]
|
| 20 |
+
op_hit = r["operation"]["choice"] == c["gold_op"]; op_ok += op_hit
|
| 21 |
+
k = by_kind.setdefault(f"{c.get('source', 'live'):9s} {c['gold_op']}", [0, 0, 0]); k[0] += 1; k[1] += op_hit
|
| 22 |
+
if c.get("gold_index") is not None:
|
| 23 |
+
a = r[c["gold_op"].lower() + "_target"]; probs = a["probabilities"]
|
| 24 |
+
order = sorted(probs, key=probs.get, reverse=True); rank = order.index(c["gold_index"]) + 1
|
| 25 |
+
tgt_n += 1; tgt_ok += rank == 1; ranks.append(rank / len(probs)); k[2] += rank == 1
|
| 26 |
+
dt = (time.time() - t) / len(cases) * 1000
|
| 27 |
+
print(f"{sys.argv[3]}/{sys.argv[4] if len(sys.argv) > 4 else ''}: cases {len(cases)} operation acc {op_ok/len(cases):.3f} "
|
| 28 |
+
f"target top-1 {tgt_ok/max(1,tgt_n):.3f} (n={tgt_n}, mean normalized rank {sum(ranks)/max(1,len(ranks)):.3f}) {dt:.0f} ms/case")
|
| 29 |
+
for kind, (n, o, tg) in sorted(by_kind.items()):
|
| 30 |
+
print(f" {kind:19s} n={n:4d} op acc {o/n:.2f}" + (f" target top-1 {tg/n:.2f}" if not kind.endswith("DONE") else ""))
|
| 31 |
+
|
| 32 |
+
if __name__ == "__main__":
|
| 33 |
+
main()
|
code/finetune/gen_goals.py
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Reverse-generate browser goals with a local LLM: pick an element as gold, ask Qwen to write the user goal for it.
|
| 2 |
+
|
| 3 |
+
python finetune/gen_goals.py out/pages.jsonl out/cases.jsonl [per_page=12]
|
| 4 |
+
|
| 5 |
+
Each case: {url, title, goal, gold_op, gold_id, gold_node, kind, label}. Also one DONE case per page.
|
| 6 |
+
"""
|
| 7 |
+
import json, os, random, sys, threading
|
| 8 |
+
from concurrent.futures import ThreadPoolExecutor
|
| 9 |
+
import httpx
|
| 10 |
+
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast")
|
| 11 |
+
from jev_ultrafast.model import action_space
|
| 12 |
+
|
| 13 |
+
LLM = os.environ.get("TEXT_MODEL_BASE_URL", "http://127.0.0.1:30000/v1") + "/chat/completions"
|
| 14 |
+
MODEL = os.environ.get("TEXT_MODEL", "Qwen/Qwen3-8B-AWQ")
|
| 15 |
+
client = httpx.Client(timeout=120)
|
| 16 |
+
|
| 17 |
+
SYS = """You write realistic browser-automation goals. Given a web page and ONE target control on it, write the goal a user
|
| 18 |
+
would give to an assistant such that the assistant's NEXT step is to use exactly that control. Rules:
|
| 19 |
+
- One or two sentences, natural language, from the user's perspective, mention what they want (not the UI mechanics).
|
| 20 |
+
- The goal must single out the target among the other listed controls; do not mention element numbers.
|
| 21 |
+
- For a text field, the goal must imply typing a concrete value into it (include the value).
|
| 22 |
+
- For a dropdown option, the goal must imply choosing that option.
|
| 23 |
+
- Vary phrasing: sometimes terse ("open the login page"), sometimes contextual ("I want to read about X, take me there").
|
| 24 |
+
Return JSON: {"goal": "..."}"""
|
| 25 |
+
|
| 26 |
+
def ask(page, target, others):
|
| 27 |
+
user = {"page": {"title": page["title"], "url": page["url"], "text_excerpt": page["text"][:700]},
|
| 28 |
+
"target": {"kind": target["kind"], "label": target["label"], "role": target.get("role"),
|
| 29 |
+
"value": target.get("value", target.get("current_value", ""))},
|
| 30 |
+
"other_controls_on_page": [o["label"][:60] for o in others]}
|
| 31 |
+
body = {"model": MODEL, "max_tokens": 200, "temperature": 0.9, "response_format": {"type": "json_object"},
|
| 32 |
+
"chat_template_kwargs": {"enable_thinking": False},
|
| 33 |
+
"messages": [{"role": "system", "content": SYS}, {"role": "user", "content": json.dumps(user, ensure_ascii=False)}]}
|
| 34 |
+
r = client.post(LLM, json=body).json()
|
| 35 |
+
goal = json.loads(r["choices"][0]["message"]["content"])["goal"]
|
| 36 |
+
return goal.strip()
|
| 37 |
+
|
| 38 |
+
def main():
|
| 39 |
+
src, out, per_page = sys.argv[1], sys.argv[2], int(sys.argv[3]) if len(sys.argv) > 3 else 12
|
| 40 |
+
pages = [json.loads(l) for l in open(src)]
|
| 41 |
+
rng = random.Random(1)
|
| 42 |
+
jobs = []
|
| 43 |
+
for pi, page in enumerate(pages):
|
| 44 |
+
elements, targets, controls = action_space(page["actions"])
|
| 45 |
+
cands = [a for a in page["actions"] if a["kind"] in ("click", "fill", "select")]
|
| 46 |
+
# de-duplicate by label, prefer informative labels
|
| 47 |
+
seen, uniq = set(), []
|
| 48 |
+
for a in cands:
|
| 49 |
+
lab = a["label"].split(" → ")[0].strip()
|
| 50 |
+
if len(lab) < 2 or lab.lower() in seen: continue
|
| 51 |
+
seen.add(lab.lower()); uniq.append(a)
|
| 52 |
+
rng.shuffle(uniq)
|
| 53 |
+
fills = [a for a in uniq if a["kind"] == "fill"][:3]
|
| 54 |
+
picks = fills + [a for a in uniq if a["kind"] != "fill"][: max(0, per_page - len(fills))]
|
| 55 |
+
for a in picks:
|
| 56 |
+
others = rng.sample([o for o in uniq if o is not a], min(10, len(uniq) - 1))
|
| 57 |
+
jobs.append((pi, page, a, others))
|
| 58 |
+
print(f"{len(pages)} pages -> {len(jobs)} goal jobs", flush=True)
|
| 59 |
+
lock = threading.Lock(); done = [0]
|
| 60 |
+
def work(job):
|
| 61 |
+
pi, page, a, others = job
|
| 62 |
+
try:
|
| 63 |
+
goal = ask(page, a, others)
|
| 64 |
+
except Exception as e:
|
| 65 |
+
print("fail", type(e).__name__, str(e)[:60], flush=True); return None
|
| 66 |
+
with lock:
|
| 67 |
+
done[0] += 1
|
| 68 |
+
if done[0] % 50 == 0: print(f" {done[0]}/{len(jobs)}", flush=True)
|
| 69 |
+
op = {"click": "CLICK", "fill": "TYPE_TEXT", "select": "SELECT"}[a["kind"]]
|
| 70 |
+
return {"page": pi, "url": page["url"], "title": page["title"], "goal": goal, "gold_op": op, "gold_id": a["id"],
|
| 71 |
+
"gold_node": a.get("node"), "kind": a["kind"], "label": a["label"]}
|
| 72 |
+
with ThreadPoolExecutor(16) as ex:
|
| 73 |
+
cases = [c for c in ex.map(work, jobs) if c]
|
| 74 |
+
for pi, page in enumerate(pages): # DONE cases: the goal is already satisfied by the current page
|
| 75 |
+
cases.append({"page": pi, "url": page["url"], "title": page["title"], "gold_op": "DONE", "gold_id": "DONE", "kind": "done",
|
| 76 |
+
"label": "", "goal": rng.choice([f"Open the page titled '{page['title'][:70]}'. Stop once it is open.",
|
| 77 |
+
f"Go to {page['url']} and stop when it has loaded.",
|
| 78 |
+
f"Navigate to the '{page['title'][:50]}' page."])})
|
| 79 |
+
with open(out, "w") as f:
|
| 80 |
+
for c in cases: f.write(json.dumps(c, ensure_ascii=False) + "\n")
|
| 81 |
+
print("wrote", len(cases), "cases ->", out)
|
| 82 |
+
for c in rng.sample(cases, 8): print(f" [{c['gold_op']:9s}] {c['label'][:35]:35s} <- {c['goal'][:90]}")
|
| 83 |
+
|
| 84 |
+
if __name__ == "__main__":
|
| 85 |
+
main()
|
code/finetune/gen_step2.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Step-2 negatives: on landing pages from done_cases (history = one executed click), reverse-generate NEW goals whose
|
| 2 |
+
next step is another element on that page. Breaks the 'any history => DONE' shortcut.
|
| 3 |
+
|
| 4 |
+
python finetune/gen_step2.py out/done_cases.jsonl out/step2_cases.jsonl [per_page=3]
|
| 5 |
+
"""
|
| 6 |
+
import json, random, sys, threading
|
| 7 |
+
from concurrent.futures import ThreadPoolExecutor
|
| 8 |
+
sys.path.insert(0, "/home/ckl/projects/S/laya/finetune")
|
| 9 |
+
from gen_goals import ask
|
| 10 |
+
|
| 11 |
+
def main():
|
| 12 |
+
src, out, per = sys.argv[1], sys.argv[2], int(sys.argv[3]) if len(sys.argv) > 3 else 3
|
| 13 |
+
dones = [json.loads(l) for l in open(src)]
|
| 14 |
+
rng = random.Random(7); jobs = []
|
| 15 |
+
for d in dones:
|
| 16 |
+
page = d["page_obj"]
|
| 17 |
+
cands = [a for a in page["actions"] if a["kind"] in ("click", "fill", "select") and len(a["label"].split(" → ")[0].strip()) > 1
|
| 18 |
+
and a["label"] != d["label"]]
|
| 19 |
+
seen, uniq = set(), []
|
| 20 |
+
for a in cands:
|
| 21 |
+
k = a["label"].lower()
|
| 22 |
+
if k in seen: continue
|
| 23 |
+
seen.add(k); uniq.append(a)
|
| 24 |
+
rng.shuffle(uniq)
|
| 25 |
+
fills = [a for a in uniq if a["kind"] == "fill"][:1]
|
| 26 |
+
for a in fills + [a for a in uniq if a["kind"] != "fill"][: per - len(fills)]:
|
| 27 |
+
jobs.append((d, a, rng.sample([o for o in uniq if o is not a], min(10, len(uniq) - 1))))
|
| 28 |
+
print(f"{len(dones)} landing pages -> {len(jobs)} jobs", flush=True)
|
| 29 |
+
lock = threading.Lock(); n = [0]
|
| 30 |
+
def work(job):
|
| 31 |
+
d, a, others = job
|
| 32 |
+
try:
|
| 33 |
+
goal = ask(d["page_obj"], a, others)
|
| 34 |
+
except Exception as e:
|
| 35 |
+
print("fail", str(e)[:60], flush=True); return None
|
| 36 |
+
with lock:
|
| 37 |
+
n[0] += 1
|
| 38 |
+
if n[0] % 100 == 0: print(f" {n[0]}/{len(jobs)}", flush=True)
|
| 39 |
+
op = {"click": "CLICK", "fill": "TYPE_TEXT", "select": "SELECT"}[a["kind"]]
|
| 40 |
+
# keep the history: the previous click is unrelated to the new goal, which is what happens mid-task all the time
|
| 41 |
+
return {**{k: d[k] for k in ("page", "url", "title", "history", "page_obj")}, "goal": goal, "gold_op": op, "gold_id": a["id"],
|
| 42 |
+
"gold_node": a.get("node"), "kind": a["kind"], "label": a["label"], "source": "live"}
|
| 43 |
+
with ThreadPoolExecutor(16) as ex:
|
| 44 |
+
cases = [c for c in ex.map(work, jobs) if c]
|
| 45 |
+
with open(out, "w") as f:
|
| 46 |
+
for c in cases: f.write(json.dumps(c, ensure_ascii=False) + "\n")
|
| 47 |
+
print("wrote", len(cases))
|
| 48 |
+
|
| 49 |
+
if __name__ == "__main__":
|
| 50 |
+
main()
|
code/finetune/make_done_cases.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Turn click cases into realistic DONE cases by actually executing the click in the browser.
|
| 2 |
+
|
| 3 |
+
python finetune/make_done_cases.py out/pages.jsonl out/cases.jsonl out/done_cases.jsonl [max=250]
|
| 4 |
+
|
| 5 |
+
For a click case (page A, goal, gold element): open A, click the gold element, observe the landing page B.
|
| 6 |
+
If the page changed, emit {goal, gold_op: DONE, page_obj: B, history: [that click]} with the same goal phrasing.
|
| 7 |
+
"""
|
| 8 |
+
import json, os, random, sys, time
|
| 9 |
+
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast")
|
| 10 |
+
os.environ.setdefault("BU_CDP_URL", "http://127.0.0.1:9222")
|
| 11 |
+
from jev_ultrafast.browser import Browser
|
| 12 |
+
|
| 13 |
+
def main():
|
| 14 |
+
pages_f, cases_f, out = sys.argv[1:4]; mx = int(sys.argv[4]) if len(sys.argv) > 4 else 250
|
| 15 |
+
pages = [json.loads(l) for l in open(pages_f)]
|
| 16 |
+
cases = [c for c in (json.loads(l) for l in open(cases_f)) if c["kind"] == "click"]
|
| 17 |
+
rng = random.Random(3); rng.shuffle(cases)
|
| 18 |
+
n_ok = n_try = 0
|
| 19 |
+
with open(out, "w") as f:
|
| 20 |
+
for c in cases:
|
| 21 |
+
if n_ok >= mx: break
|
| 22 |
+
src = pages[c["page"]]; n_try += 1
|
| 23 |
+
try:
|
| 24 |
+
b = Browser(src["url"])
|
| 25 |
+
try:
|
| 26 |
+
page = b.observe(screenshot=False)
|
| 27 |
+
act = next((a for a in page["actions"] if a["kind"] == "click" and a["label"] == c["label"]), None)
|
| 28 |
+
if act is None:
|
| 29 |
+
continue
|
| 30 |
+
b.act(act, page); time.sleep(0.3)
|
| 31 |
+
dest = b.observe(screenshot=False)
|
| 32 |
+
finally:
|
| 33 |
+
b.close()
|
| 34 |
+
except Exception as e:
|
| 35 |
+
print("fail", type(e).__name__, str(e)[:60], flush=True); continue
|
| 36 |
+
if dest["url"] == page["url"] and dest["title"] == page["title"]:
|
| 37 |
+
continue
|
| 38 |
+
hist = [{"action": act["label"], "kind": "click", "text": None, "page_changed": True}]
|
| 39 |
+
f.write(json.dumps({"page": c["page"], "url": dest["url"], "title": dest["title"], "goal": c["goal"], "gold_op": "DONE",
|
| 40 |
+
"gold_id": "DONE", "kind": "done", "label": act["label"], "history": hist,
|
| 41 |
+
"page_obj": {k: dest[k] for k in ("url", "title", "text", "actions")}}, ensure_ascii=False) + "\n"); f.flush()
|
| 42 |
+
n_ok += 1
|
| 43 |
+
if n_ok % 25 == 0: print(f" {n_ok} done cases from {n_try} tries", flush=True)
|
| 44 |
+
print("wrote", n_ok, "DONE cases")
|
| 45 |
+
|
| 46 |
+
if __name__ == "__main__":
|
| 47 |
+
main()
|
code/finetune/rollouts.py
ADDED
|
@@ -0,0 +1,109 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Scripted multi-step trajectories executed in the real browser, for the three skills the model lacks:
|
| 2 |
+
scroll : goal targets an element that is only visible after scrolling -> [SCROLL_DOWN, CLICK, DONE]
|
| 3 |
+
search : goal asks to search for a phrase in a text field -> [TYPE_TEXT, CLICK submit/suggestion, DONE]
|
| 4 |
+
select : goal asks to choose an option of a <select> -> [SELECT, DONE]
|
| 5 |
+
|
| 6 |
+
python finetune/rollouts.py out/pages.jsonl out/rollout_cases.jsonl [max_pages=300]
|
| 7 |
+
|
| 8 |
+
Goals are templated (varied phrasing); no LLM needed. Pages are the crawled ones; the suite's exact URLs are skipped.
|
| 9 |
+
"""
|
| 10 |
+
import json, os, random, re, sys, time
|
| 11 |
+
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast")
|
| 12 |
+
os.environ.setdefault("BU_CDP_URL", "http://127.0.0.1:9222")
|
| 13 |
+
from jev_ultrafast.browser import Browser, StalePage
|
| 14 |
+
|
| 15 |
+
SUITE_URLS = {"https://en.wikipedia.org/wiki/Main_Page", "https://news.ycombinator.com/", "https://github.com/tile-ai/tilelang", "https://www.python.org/",
|
| 16 |
+
"https://books.toscrape.com/", "https://the-internet.herokuapp.com/dropdown", "https://the-internet.herokuapp.com/checkboxes",
|
| 17 |
+
"https://quotes.toscrape.com/", "https://duckduckgo.com/", "https://arxiv.org/", "https://www.google.com/travel/flights?hl=en"}
|
| 18 |
+
SCROLL_T = ["Open '{x}'.", "I want to see '{x}', take me there.", "Go to {x}.", "Find and click '{x}' on this page.", "Open the '{x}' link further down the page.", "Navigate to '{x}'."]
|
| 19 |
+
SEARCH_T = ["Search for '{q}'.", "Look up '{q}' using the search box and show the results.", "Find results for '{q}'.", "Use the site search to find '{q}'.", "Search this site for {q}."]
|
| 20 |
+
SELECT_T = ["Select '{o}' in the '{f}' dropdown.", "Choose {o} for {f}.", "Set '{f}' to '{o}'.", "Pick the option '{o}'."]
|
| 21 |
+
QUERIES = ["python tutorial", "flash attention", "climate change", "rust async", "linear algebra", "tilelang", "modernbert", "black holes", "sourdough bread", "gpu kernels"]
|
| 22 |
+
|
| 23 |
+
def lab(a): return a["label"].split(" → ")[0].strip()
|
| 24 |
+
def case(page, goal, gold_op, gold_id, kind, label, hist, tag):
|
| 25 |
+
return {"page": -1, "url": page["url"], "title": page["title"], "goal": goal, "gold_op": gold_op, "gold_id": gold_id, "kind": kind, "label": label,
|
| 26 |
+
"history": hist, "source": "rollout", "skill": tag, "page_obj": {k: page[k] for k in ("url", "title", "text", "actions")}}
|
| 27 |
+
def hist_of(a, text=None, changed=True): return {"action": a["label"], "kind": a["kind"], "text": text, "page_changed": changed}
|
| 28 |
+
def find(page, pred):
|
| 29 |
+
return next((a for a in page["actions"] if pred(a)), None)
|
| 30 |
+
|
| 31 |
+
def do_scroll(b, page, rng, out):
|
| 32 |
+
if not page.get("scroll") or page["scroll"]["height"] < 1300: return 0
|
| 33 |
+
before = {a["label"] for a in page["actions"] if a["kind"] == "click"}
|
| 34 |
+
sd = find(page, lambda a: a["id"] == "scroll_down")
|
| 35 |
+
if not sd: return 0
|
| 36 |
+
b.act(sd, page); time.sleep(0.3); p2 = b.observe(screenshot=False)
|
| 37 |
+
new = [a for a in p2["actions"] if a["kind"] == "click" and a["label"] not in before and 3 <= len(lab(a)) <= 60 and a.get("role") in ("link", "button")]
|
| 38 |
+
if not new: return 0
|
| 39 |
+
n = 0
|
| 40 |
+
for tgt in rng.sample(new, min(2, len(new))):
|
| 41 |
+
goal = rng.choice(SCROLL_T).format(x=lab(tgt))
|
| 42 |
+
out.write(json.dumps(case(page, goal, "SCROLL_DOWN", "scroll_down", "scroll", "Scroll down", [], "scroll"), ensure_ascii=False) + "\n")
|
| 43 |
+
out.write(json.dumps(case(p2, goal, "CLICK", tgt["id"], "click", tgt["label"], [hist_of(sd, changed=False)], "scroll"), ensure_ascii=False) + "\n")
|
| 44 |
+
n += 2
|
| 45 |
+
# execute one click for a DONE state
|
| 46 |
+
tgt = new[0]
|
| 47 |
+
try:
|
| 48 |
+
b.act(tgt, p2); time.sleep(0.4); p3 = b.observe(screenshot=False)
|
| 49 |
+
if p3["url"] != p2["url"]:
|
| 50 |
+
goal = rng.choice(SCROLL_T).format(x=lab(tgt))
|
| 51 |
+
out.write(json.dumps(case(p3, goal, "DONE", "DONE", "done", "", [hist_of(sd, changed=False), hist_of(tgt)], "scroll"), ensure_ascii=False) + "\n"); n += 1
|
| 52 |
+
except Exception: pass
|
| 53 |
+
return n
|
| 54 |
+
|
| 55 |
+
def do_search(b, page, rng, out):
|
| 56 |
+
field = find(page, lambda a: a["kind"] == "fill" and re.search(r"search|query|find|keyword", a["label"], re.I))
|
| 57 |
+
if not field: return 0
|
| 58 |
+
q = rng.choice(QUERIES); goal = rng.choice(SEARCH_T).format(q=q)
|
| 59 |
+
out.write(json.dumps(case(page, goal, "TYPE_TEXT", field["id"], "fill", field["label"], [], "search"), ensure_ascii=False) + "\n")
|
| 60 |
+
b.act(field, page, text=q); time.sleep(0.4); p2 = b.observe(screenshot=False)
|
| 61 |
+
# submit: a search/go button, or a suggestion containing the query
|
| 62 |
+
sub = find(p2, lambda a: a["kind"] == "click" and (re.search(r"^(search|go|submit|find)\b", lab(a), re.I) or (a.get("role") in ("option", "listitem", "link") and q.split()[0].lower() in a["label"].lower())))
|
| 63 |
+
if not sub: return 1
|
| 64 |
+
out.write(json.dumps(case(p2, goal, "CLICK", sub["id"], "click", sub["label"], [hist_of(field, q, False)], "search"), ensure_ascii=False) + "\n")
|
| 65 |
+
try:
|
| 66 |
+
b.act(sub, p2); time.sleep(0.6); p3 = b.observe(screenshot=False)
|
| 67 |
+
if p3["url"] != p2["url"] or p3["title"] != p2["title"]:
|
| 68 |
+
out.write(json.dumps(case(p3, goal, "DONE", "DONE", "done", "", [hist_of(field, q, False), hist_of(sub)], "search"), ensure_ascii=False) + "\n"); return 3
|
| 69 |
+
except Exception: pass
|
| 70 |
+
return 2
|
| 71 |
+
|
| 72 |
+
def do_select(b, page, rng, out):
|
| 73 |
+
sels = [a for a in page["actions"] if a["kind"] == "select" and a.get("value") and a["value"] != a.get("current_value")]
|
| 74 |
+
if not sels: return 0
|
| 75 |
+
by_field = {}
|
| 76 |
+
for a in sels: by_field.setdefault(a["node"], []).append(a)
|
| 77 |
+
node, opts = rng.choice(list(by_field.items()))
|
| 78 |
+
opt = rng.choice(opts); f = lab(opt); o = opt["label"].split(" → ")[-1].strip()
|
| 79 |
+
goal = rng.choice(SELECT_T).format(o=o, f=f)
|
| 80 |
+
out.write(json.dumps(case(page, goal, "SELECT", opt["id"], "select", opt["label"], [], "select"), ensure_ascii=False) + "\n")
|
| 81 |
+
try:
|
| 82 |
+
b.act(opt, page); time.sleep(0.3); p2 = b.observe(screenshot=False)
|
| 83 |
+
out.write(json.dumps(case(p2, goal, "DONE", "DONE", "done", "", [hist_of(opt, changed=False)], "select"), ensure_ascii=False) + "\n"); return 2
|
| 84 |
+
except Exception: return 1
|
| 85 |
+
|
| 86 |
+
def main():
|
| 87 |
+
pages = [json.loads(l) for l in open(sys.argv[1])]; out_f = sys.argv[2]; mx = int(sys.argv[3]) if len(sys.argv) > 3 else 300
|
| 88 |
+
rng = random.Random(11); rng.shuffle(pages)
|
| 89 |
+
counts = {"scroll": 0, "search": 0, "select": 0}; done = 0
|
| 90 |
+
with open(out_f, "w") as out:
|
| 91 |
+
for pg in pages:
|
| 92 |
+
if done >= mx: break
|
| 93 |
+
if pg["url"] in SUITE_URLS: continue
|
| 94 |
+
for skill, fn in (("scroll", do_scroll), ("search", do_search), ("select", do_select)):
|
| 95 |
+
try:
|
| 96 |
+
b = Browser(pg["url"])
|
| 97 |
+
try:
|
| 98 |
+
page = b.observe(screenshot=False); n = fn(b, page, rng, out)
|
| 99 |
+
finally:
|
| 100 |
+
b.close()
|
| 101 |
+
counts[skill] += n
|
| 102 |
+
except Exception as e:
|
| 103 |
+
pass
|
| 104 |
+
done += 1
|
| 105 |
+
if done % 25 == 0: print(f" {done} pages {counts}", flush=True)
|
| 106 |
+
print("wrote", counts, "->", out_f)
|
| 107 |
+
|
| 108 |
+
if __name__ == "__main__":
|
| 109 |
+
main()
|
code/finetune/run_after_v6.sh
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# after v6: (1) real-task suite with v6, (2) training-speed benchmark of the three cheap wins, (3) v7 = v6 data with
|
| 3 |
+
# truncated page text + no checkpointing + torch.compile, 3 epochs, calibrate, eval, suite.
|
| 4 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 5 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 6 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 7 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
|
| 8 |
+
P=.venv/bin/python
|
| 9 |
+
until grep -qE "V6_DONE|Traceback" finetune/out/v6.log; do sleep 30; done
|
| 10 |
+
echo "== suite v6"
|
| 11 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 12 |
+
bash $S/restart_s1.sh $PWD/finetune/out/laya-browser-v6 999 >/dev/null
|
| 13 |
+
(cd ../jev-ultrafast && SUITE_OUT=$PWD/../laya/finetune/out/suite_v6.json timeout 1500 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -vE "Warn|TileLang")
|
| 14 |
+
bash $S/stop_all.sh >/dev/null 2>&1; sleep 3
|
| 15 |
+
echo "== speed benchmark (256 items, 1 epoch)"
|
| 16 |
+
echo "-- old items (1016 tok), CKPT=1 COMPILE=0"; CKPT=1 COMPILE=0 LIMIT=256 $P finetune/train.py finetune/out/train_items.pt /tmp/bench_ck 1 2>&1 | grep "=== epoch"
|
| 17 |
+
echo "-- old items, CKPT=0 COMPILE=0"; CKPT=0 COMPILE=0 LIMIT=256 $P finetune/train.py finetune/out/train_items.pt /tmp/bench_ck 1 2>&1 | grep -E "=== epoch|OutOfMemory"
|
| 18 |
+
echo "-- old items, CKPT=0 COMPILE=1"; CKPT=0 COMPILE=1 LIMIT=256 $P finetune/train.py finetune/out/train_items.pt /tmp/bench_ck 1 2>&1 | grep -E "=== epoch|OutOfMemory|Error"
|
| 19 |
+
echo "== rebuild items with 3000-char page text"
|
| 20 |
+
$P finetune/build_items.py finetune/out/pages.jsonl finetune/out/cases.jsonl finetune/out/ finetune/out/done_cases.jsonl finetune/out/step2_cases.jsonl finetune/out/m2w_cases.jsonl
|
| 21 |
+
echo "-- new items, CKPT=0 COMPILE=1"; CKPT=0 COMPILE=1 LIMIT=256 $P finetune/train.py finetune/out/train_items.pt /tmp/bench_ck 1 2>&1 | grep -E "=== epoch|OutOfMemory|Error"
|
| 22 |
+
echo "== train v7 (3 epochs, fast settings)"; CKPT=0 COMPILE=1 $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser-v7 3 2>&1 | grep -E "=== epoch|saved|Error|Traceback"
|
| 23 |
+
echo "== calibrate v7"; $P finetune/calibrate.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v7 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 24 |
+
echo "== eval v7"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v7 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 25 |
+
echo "== suite v7"
|
| 26 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 27 |
+
bash $S/restart_s1.sh $PWD/finetune/out/laya-browser-v7 999 >/dev/null
|
| 28 |
+
(cd ../jev-ultrafast && SUITE_OUT=$PWD/../laya/finetune/out/suite_v7.json timeout 1500 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -vE "Warn|TileLang")
|
| 29 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 30 |
+
echo AFTER_V6_DONE
|
code/finetune/run_all.sh
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# build items -> zero-shot baselines -> fine-tune -> eval. Logs to finetune/out/run.log
|
| 3 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 4 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 5 |
+
P=.venv/bin/python
|
| 6 |
+
for p in $(pgrep -f "sglang.launch_server"); do kill $p; done; sleep 3
|
| 7 |
+
uv pip install --python $P "httpx[http2]" 2>&1 | tail -1
|
| 8 |
+
set -e
|
| 9 |
+
echo "== build items"; $P finetune/build_items.py finetune/out/pages.jsonl finetune/out/cases.jsonl finetune/out/
|
| 10 |
+
echo "== zero-shot baselines"
|
| 11 |
+
$P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl "$LAYA_BASE" 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 12 |
+
$P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl convaiinnovations/laya multilingual 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 13 |
+
echo "== train"; $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser 4 2>&1 | grep -vE "Warn|warn"
|
| 14 |
+
echo "== eval fine-tuned"
|
| 15 |
+
$P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl finetune/out/laya-browser 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 16 |
+
echo RUN_ALL_DONE
|
code/finetune/run_final.sh
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# Final comparison after v10s: v10 vs v10s, gating off vs tau=0.7, 3 runs per task; plus per-step latency profiles.
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 5 |
+
O=finetune/out; P=.venv/bin/python
|
| 6 |
+
until grep -q "V10S_DONE" $O/v10s.log; do sleep 60; done
|
| 7 |
+
echo "== per-step latency"
|
| 8 |
+
for ck in laya-browser-v10 laya-browser-v10s; do echo "-- $ck"; $P apps/profile_step.py $PWD/$O/$ck 2>&1 | grep -vE "TileLang|Warn|warn|Fetch|loaded"; done
|
| 9 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 10 |
+
for ck in laya-browser-v10 laya-browser-v10s; do
|
| 11 |
+
for tau in 0 0.7; do
|
| 12 |
+
echo "== suite $ck tau=$tau x3"
|
| 13 |
+
export ESCALATE_TAU=$tau ESCALATE_LOG=$PWD/$O/escalations_final_${ck}_$tau.jsonl; rm -f $ESCALATE_LOG
|
| 14 |
+
bash $S/restart_s1.sh $PWD/$O/$ck 999 >/dev/null
|
| 15 |
+
(cd ../jev-ultrafast && REPEATS=3 SUITE_OUT=$PWD/../laya/$O/suite_final_${ck}_$tau.json timeout 3000 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -E "^==|per task")
|
| 16 |
+
curl -s http://127.0.0.1:8791/ | python3 -c "import json,sys; d=json.load(sys.stdin); print(f\" server calls={d['calls']} escalated={d['escalated']} ({100*d['escalated']/max(1,d['calls']):.0f}%)\")"
|
| 17 |
+
done
|
| 18 |
+
done
|
| 19 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 20 |
+
echo FINAL_DONE
|
code/finetune/run_gated.sh
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# after v10: real-task suite with System-1/System-2 gating at several confidence thresholds; escalations logged as DAgger cases
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 5 |
+
O=finetune/out
|
| 6 |
+
until grep -q "V10_DONE" $O/v10.log; do sleep 60; done
|
| 7 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 8 |
+
for tau in 0.5 0.7 0.9; do
|
| 9 |
+
echo "== suite v10 + gating tau=$tau"
|
| 10 |
+
export ESCALATE_TAU=$tau ESCALATE_LOG=$PWD/$O/escalations_tau$tau.jsonl
|
| 11 |
+
rm -f $ESCALATE_LOG
|
| 12 |
+
bash $S/restart_s1.sh $PWD/$O/laya-browser-v10 999 >/dev/null
|
| 13 |
+
(cd ../jev-ultrafast && SUITE_OUT=$PWD/../laya/$O/suite_v10_tau$tau.json timeout 1500 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -vE "Warn|TileLang")
|
| 14 |
+
curl -s http://127.0.0.1:8791/ | python3 -c "import json,sys; d=json.load(sys.stdin); print(f\" server calls={d['calls']} escalated={d['escalated']} ({100*d['escalated']/max(1,d['calls']):.0f}%)\")"
|
| 15 |
+
done
|
| 16 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 17 |
+
echo GATED_DONE
|
code/finetune/run_suite_fixed.sh
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# after the teacher eval: re-run the 16-task suite x3 for v11s, v10s and v10 with the corrected checks
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh; O=finetune/out
|
| 5 |
+
until grep -q "TEACHER_EVAL_DONE" $O/teacher_eval.log; do sleep 60; done
|
| 6 |
+
bash $S/stop_all.sh >/dev/null 2>&1; bash $S/bonsai.sh stop >/dev/null 2>&1
|
| 7 |
+
curl -s -m 3 http://127.0.0.1:9222/json/version >/dev/null || (nohup chromium --headless=new --remote-debugging-port=9222 --user-data-dir=$S/chrome-profile --window-size=1120,780 --no-first-run --lang=en-US about:blank >/dev/null 2>&1 &); sleep 3
|
| 8 |
+
for ck in laya-browser-v11s laya-browser-v10s laya-browser-v10; do
|
| 9 |
+
echo "== suite $ck x3 (fixed checks)"
|
| 10 |
+
ESCALATE_TAU=0 bash $S/restart_s1.sh $PWD/$O/$ck 999 >/dev/null
|
| 11 |
+
(cd ../jev-ultrafast && REPEATS=3 SUITE_OUT=$PWD/../laya/$O/suite_fixed_$ck.json timeout 3000 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -E "^==|per task")
|
| 12 |
+
done
|
| 13 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 14 |
+
echo SUITE_FIXED_DONE
|
code/finetune/run_teacher_eval.sh
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# after v11: compare teachers on 80 held-out cases (40 Mind2Web human-labelled + 40 live), then laya v10s / v11s on the same cases.
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 5 |
+
O=finetune/out; P=.venv/bin/python
|
| 6 |
+
until grep -q "V11_DONE" $O/v11.log; do sleep 60; done
|
| 7 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 8 |
+
echo "== 27B (bonsai), thinking off"
|
| 9 |
+
bash $S/bonsai.sh start
|
| 10 |
+
$P finetune/teacher_eval.py $O/pages.jsonl $O/eval_cases.jsonl http://127.0.0.1:30001/v1 "bonsai-27b think=off" 80
|
| 11 |
+
echo "== 27B, thinking full"
|
| 12 |
+
$P finetune/teacher_eval.py $O/pages.jsonl $O/eval_cases.jsonl http://127.0.0.1:30001/v1 "bonsai-27b think=full" 40 --think
|
| 13 |
+
bash $S/bonsai.sh stop
|
| 14 |
+
echo "== 27B, thinking budget 300 tokens (--reasoning-budget 300)"
|
| 15 |
+
bash $S/bonsai.sh start --reasoning-budget 300
|
| 16 |
+
$P finetune/teacher_eval.py $O/pages.jsonl $O/eval_cases.jsonl http://127.0.0.1:30001/v1 "bonsai-27b think=budget300" 80 --think
|
| 17 |
+
bash $S/bonsai.sh stop
|
| 18 |
+
echo "== 27B, thinking with --reasoning-effort low (server flag)"
|
| 19 |
+
bash $S/bonsai.sh start --reasoning-effort low
|
| 20 |
+
$P finetune/teacher_eval.py $O/pages.jsonl $O/eval_cases.jsonl http://127.0.0.1:30001/v1 "bonsai-27b think=low" 40 --think
|
| 21 |
+
bash $S/bonsai.sh stop
|
| 22 |
+
echo "== Qwen3-8B-AWQ, thinking off"
|
| 23 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 24 |
+
$P finetune/teacher_eval.py $O/pages.jsonl $O/eval_cases.jsonl http://127.0.0.1:30000/v1 "qwen3-8b think=off" 80
|
| 25 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 26 |
+
echo "== laya v10s / v11s on the full eval set (for reference)"
|
| 27 |
+
LAYA_FMT=v3 $P finetune/eval.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v10s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch" | head -1
|
| 28 |
+
LAYA_FMT=v3 $P finetune/eval.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v11s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch" | head -1
|
| 29 |
+
echo TEACHER_EVAL_DONE
|
code/finetune/run_train.sh
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 3 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 4 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
|
| 5 |
+
P=.venv/bin/python
|
| 6 |
+
echo "== train"; $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser 4 2>&1 | grep -vE "Warn|warn"
|
| 7 |
+
echo "== eval fine-tuned"
|
| 8 |
+
$P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 9 |
+
echo TRAIN_DONE
|
code/finetune/run_v10.sh
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# v10 = v9 recipe (format v2, head 768, 4 epochs) + the live data that failed silently in v9: goals on 421 pages,
|
| 3 |
+
# real DONE, step-2, DAgger with policy v9. Full logs kept per step under finetune/out/v10_*.log
|
| 4 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 5 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 6 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 7 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True CKPT=0 COMPILE=0
|
| 8 |
+
P=.venv/bin/python; O=finetune/out
|
| 9 |
+
bash $S/start_infra.sh
|
| 10 |
+
for i in 1 2 3 4 5 6; do
|
| 11 |
+
r=$(curl -s -m 60 http://127.0.0.1:30000/v1/chat/completions -H 'Content-Type: application/json' -d '{"model":"Qwen/Qwen3-8B-AWQ","max_tokens":20,"chat_template_kwargs":{"enable_thinking":false},"messages":[{"role":"user","content":"say ok"}]}' | head -c 300)
|
| 12 |
+
echo "sglang probe: $r" | cut -c1-120; echo "$r" | grep -q '"content"' && break; sleep 20
|
| 13 |
+
done
|
| 14 |
+
echo "== goals (421 pages)"; (cd ../jev-ultrafast && timeout 5400 .venv/bin/python ../laya/finetune/gen_goals.py ../laya/finetune/out/pages.jsonl ../laya/finetune/out/cases.jsonl 12 > ../laya/$O/v10_goals.log 2>&1); tail -3 $O/v10_goals.log; echo "cases: $(wc -l < $O/cases.jsonl)"
|
| 15 |
+
echo "== real DONE"; (cd ../jev-ultrafast && timeout 5400 .venv/bin/python ../laya/finetune/make_done_cases.py ../laya/finetune/out/pages.jsonl ../laya/finetune/out/cases.jsonl ../laya/finetune/out/done_cases.jsonl 700 > ../laya/$O/v10_done.log 2>&1); tail -1 $O/v10_done.log
|
| 16 |
+
echo "== step-2"; (cd ../jev-ultrafast && timeout 3600 .venv/bin/python ../laya/finetune/gen_step2.py ../laya/finetune/out/done_cases.jsonl ../laya/finetune/out/step2_cases.jsonl 2 > ../laya/$O/v10_step2.log 2>&1); tail -1 $O/v10_step2.log
|
| 17 |
+
echo "== dagger (policy v9)"
|
| 18 |
+
bash $S/restart_s1.sh $PWD/$O/laya-browser-v9 999 >/dev/null
|
| 19 |
+
(cd ../jev-ultrafast && timeout 3000 .venv/bin/python ../laya/finetune/dagger.py ../laya/finetune/out/dagger_cases.jsonl > ../laya/$O/v10_dagger.log 2>&1); tail -1 $O/v10_dagger.log
|
| 20 |
+
bash $S/stop_all.sh >/dev/null 2>&1; sleep 3
|
| 21 |
+
export LAYA_FMT=v2 LAYA_HEAD=768
|
| 22 |
+
echo "== build v10"; $P finetune/build_items.py $O/pages.jsonl $O/cases.jsonl $O/ $O/done_cases.jsonl $O/step2_cases.jsonl $O/m2w_cases.jsonl $O/dagger_cases.jsonl
|
| 23 |
+
echo "== train v10 (4 epochs)"; $P finetune/train.py $O/train_items.pt $O/laya-browser-v10 4 2>&1 | grep --line-buffered -E "=== epoch|saved|Error|Traceback"
|
| 24 |
+
echo "== calibrate"; $P finetune/calibrate.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v10 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 25 |
+
echo "== eval v10"; $P finetune/eval.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v10 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 26 |
+
echo "== eval v9 (same eval)"; $P finetune/eval.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v9 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 27 |
+
echo "== suite v10"
|
| 28 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 29 |
+
bash $S/restart_s1.sh $PWD/$O/laya-browser-v10 999 >/dev/null
|
| 30 |
+
(cd ../jev-ultrafast && SUITE_OUT=$PWD/../laya/$O/suite_v10.json timeout 1500 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -vE "Warn|TileLang")
|
| 31 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 32 |
+
echo V10_DONE
|
code/finetune/run_v10s.sh
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# v10s: same data as v10, 322M mmBERT-base (multilingual) backbone, format v3 -> target ~20 ms per browser step.
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 5 |
+
until grep -q "GATED_DONE" finetune/out/gated.log; do sleep 60; done
|
| 6 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/multilingual | head -1)
|
| 7 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True CKPT=0 COMPILE=0 LAYA_FMT=v3 LAYA_HEAD=768
|
| 8 |
+
P=.venv/bin/python; O=finetune/out
|
| 9 |
+
echo "== build v10s (format v3, mmBERT-base tokenizer)"
|
| 10 |
+
$P finetune/build_items.py $O/pages.jsonl $O/cases.jsonl $O/ $O/done_cases.jsonl $O/step2_cases.jsonl $O/m2w_cases.jsonl $O/dagger_cases.jsonl
|
| 11 |
+
echo "== train v10s (4 epochs)"; $P finetune/train.py $O/train_items.pt $O/laya-browser-v10s 4 2>&1 | grep --line-buffered -E "=== epoch|saved|Error|Traceback"
|
| 12 |
+
echo "== calibrate"; $P finetune/calibrate.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v10s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 13 |
+
echo "== eval v10s"; $P finetune/eval.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v10s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 14 |
+
echo "== profile step"; $P apps/profile_step.py $PWD/$O/laya-browser-v10s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 15 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 16 |
+
for tau in 0 0.7; do
|
| 17 |
+
echo "== suite v10s tau=$tau"
|
| 18 |
+
export ESCALATE_TAU=$tau ESCALATE_LOG=$PWD/$O/escalations_v10s_tau$tau.jsonl; rm -f $ESCALATE_LOG
|
| 19 |
+
bash $S/restart_s1.sh $PWD/$O/laya-browser-v10s 999 >/dev/null
|
| 20 |
+
(cd ../jev-ultrafast && SUITE_OUT=$PWD/../laya/$O/suite_v10s_tau$tau.json timeout 1500 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -vE "Warn|TileLang")
|
| 21 |
+
curl -s http://127.0.0.1:8791/ | python3 -c "import json,sys; d=json.load(sys.stdin); print(f\" server calls={d['calls']} escalated={d['escalated']} ({100*d['escalated']/max(1,d['calls']):.0f}%)\")"
|
| 22 |
+
done
|
| 23 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 24 |
+
echo V10S_DONE
|
code/finetune/run_v11.sh
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# v11s: v10s recipe + scripted scroll / search-submit / select trajectories (rollout_cases, x3), then 16 tasks x3.
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 5 |
+
O=finetune/out; P=.venv/bin/python
|
| 6 |
+
until ! pgrep -f "finetune/rollouts.py" >/dev/null; do sleep 60; done
|
| 7 |
+
echo "== rollout cases: $(wc -l < $O/rollout_cases.jsonl)"
|
| 8 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/multilingual | head -1)
|
| 9 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True CKPT=0 COMPILE=0 LAYA_FMT=v3 LAYA_HEAD=768
|
| 10 |
+
echo "== build v11s"
|
| 11 |
+
$P finetune/build_items.py $O/pages.jsonl $O/cases.jsonl $O/ $O/done_cases.jsonl $O/step2_cases.jsonl $O/m2w_cases.jsonl $O/dagger_cases.jsonl $O/rollout_cases.jsonl
|
| 12 |
+
echo "== train v11s (4 epochs)"; $P finetune/train.py $O/train_items.pt $O/laya-browser-v11s 4 2>&1 | grep --line-buffered -E "=== epoch|saved|Error|Traceback"
|
| 13 |
+
echo "== calibrate"; $P finetune/calibrate.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v11s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 14 |
+
echo "== eval v11s"; $P finetune/eval.py $O/pages.jsonl $O/eval_cases.jsonl $PWD/$O/laya-browser-v11s 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 15 |
+
echo "== suite v11s x3"
|
| 16 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 17 |
+
export ESCALATE_TAU=0
|
| 18 |
+
bash $S/restart_s1.sh $PWD/$O/laya-browser-v11s 999 >/dev/null
|
| 19 |
+
(cd ../jev-ultrafast && REPEATS=3 SUITE_OUT=$PWD/../laya/$O/suite_final_v11s.json timeout 3000 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -E "^==|per task")
|
| 20 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 21 |
+
echo V11_DONE
|
code/finetune/run_v2.sh
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 3 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 4 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
|
| 5 |
+
P=.venv/bin/python
|
| 6 |
+
for p in $(pgrep -f "sglang.launch_server|apps/systemone_server"); do kill $p; done; sleep 3
|
| 7 |
+
echo "== build"; $P finetune/build_items.py finetune/out/pages.jsonl finetune/out/cases.jsonl finetune/out/ finetune/out/done_cases.jsonl
|
| 8 |
+
echo "== zero-shot baseline (typed) on v2 eval"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl "$LAYA_BASE" 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 9 |
+
echo "== train v2"; $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser-v2 4 2>&1 | grep -E "=== epoch|saved|Error|Traceback"
|
| 10 |
+
echo "== eval v2"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v2 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 11 |
+
echo V2_DONE
|
code/finetune/run_v3.sh
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 3 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 4 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
|
| 5 |
+
P=.venv/bin/python
|
| 6 |
+
for p in $(pgrep -f "sglang.launch_server|apps/systemone_server"); do kill $p; done; sleep 3
|
| 7 |
+
echo "== build"; $P finetune/build_items.py finetune/out/pages.jsonl finetune/out/cases.jsonl finetune/out/ finetune/out/done_cases.jsonl finetune/out/m2w_cases.jsonl
|
| 8 |
+
echo "== zero-shot baseline (typed) on v2 eval"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl "$LAYA_BASE" 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 9 |
+
echo "== train v3"; $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser-v3 4 2>&1 | grep -E "=== epoch|saved|Error|Traceback"
|
| 10 |
+
echo "== eval v3"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v3 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 11 |
+
echo V3_DONE
|
code/finetune/run_v4.sh
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 3 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 4 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
|
| 5 |
+
P=.venv/bin/python
|
| 6 |
+
for p in $(pgrep -f "sglang.launch_server|apps/systemone_server"); do kill $p; done; sleep 3
|
| 7 |
+
echo "== wait for shards"; until [ "$(ls finetune/data/mind2web/data/train/train_*.json 2>/dev/null | wc -l)" -ge 11 ] && ! ls finetune/data/mind2web/data/train/*.incomplete >/dev/null 2>&1; do sleep 10; done
|
| 8 |
+
echo "== convert all shards"; $P finetune/convert_mind2web.py finetune/data/mind2web/data/train/train_*.json finetune/out/m2w_cases.jsonl 2>&1 | tail -1
|
| 9 |
+
echo "== build"; $P finetune/build_items.py finetune/out/pages.jsonl finetune/out/cases.jsonl finetune/out/ finetune/out/done_cases.jsonl finetune/out/m2w_cases.jsonl
|
| 10 |
+
echo "== train v4 (3 epochs)"; $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser-v4 3 2>&1 | grep -E "=== epoch|saved|Error|Traceback"
|
| 11 |
+
echo "== eval v4"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v4 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 12 |
+
echo "== eval v3 on same eval set"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v3 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 13 |
+
echo V4_DONE
|
code/finetune/run_v5.sh
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 3 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 4 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
|
| 5 |
+
P=.venv/bin/python
|
| 6 |
+
for p in $(pgrep -f "sglang.launch_server|apps/systemone_server"); do kill $p; done; sleep 3
|
| 7 |
+
|
| 8 |
+
echo "== convert all shards"; $P finetune/convert_mind2web.py finetune/data/mind2web/data/train/train_*.json finetune/out/m2w_cases.jsonl 2>&1 | tail -1
|
| 9 |
+
echo "== build"; $P finetune/build_items.py finetune/out/pages.jsonl finetune/out/cases.jsonl finetune/out/ finetune/out/done_cases.jsonl finetune/out/m2w_cases.jsonl
|
| 10 |
+
echo "== train v5 (3 epochs)"; $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser-v5 3 2>&1 | grep -E "=== epoch|saved|Error|Traceback"
|
| 11 |
+
echo "== calibrate"; $P finetune/calibrate.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v5 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 12 |
+
echo "== eval v5"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v5 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 13 |
+
|
| 14 |
+
echo V5_DONE
|
code/finetune/run_v6.sh
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 3 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 4 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
|
| 5 |
+
P=.venv/bin/python
|
| 6 |
+
for p in $(pgrep -f "sglang.launch_server|apps/systemone_server"); do kill $p; done; sleep 3
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
echo "== build"; $P finetune/build_items.py finetune/out/pages.jsonl finetune/out/cases.jsonl finetune/out/ finetune/out/done_cases.jsonl finetune/out/step2_cases.jsonl finetune/out/m2w_cases.jsonl
|
| 10 |
+
echo "== train v6 (3 epochs)"; $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser-v6 3 2>&1 | grep -E "=== epoch|saved|Error|Traceback"
|
| 11 |
+
echo "== calibrate"; $P finetune/calibrate.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v6 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 12 |
+
echo "== eval v6"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v6 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 13 |
+
|
| 14 |
+
echo V6_DONE
|
code/finetune/run_v7.sh
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# v7: best training config (no checkpointing, no compile, 3000-char page text) + DONE oversampling x4.
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 5 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 6 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True CKPT=0 COMPILE=0
|
| 7 |
+
P=.venv/bin/python
|
| 8 |
+
echo "== build (3000-char text, DONE x4)"
|
| 9 |
+
$P finetune/build_items.py finetune/out/pages.jsonl finetune/out/cases.jsonl finetune/out/ finetune/out/done_cases.jsonl finetune/out/step2_cases.jsonl finetune/out/m2w_cases.jsonl
|
| 10 |
+
echo "== speed: new items, CKPT=0 COMPILE=0 (256 items)"; LIMIT=256 $P finetune/train.py finetune/out/train_items.pt /tmp/bench_ck 1 2>&1 | grep -E "=== epoch|Error"
|
| 11 |
+
echo "== train v7 (3 epochs)"; $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser-v7 3 2>&1 | grep --line-buffered -E "=== epoch|saved|Error|Traceback"
|
| 12 |
+
echo "== calibrate v7"; $P finetune/calibrate.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v7 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 13 |
+
echo "== eval v7"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v7 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 14 |
+
echo "== eval v6 on same (truncated) eval"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v6 2>&1 | grep -vE "TileLang|Warn|warn|Fetch" | head -1
|
| 15 |
+
echo "== suite v7"
|
| 16 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 17 |
+
bash $S/restart_s1.sh $PWD/finetune/out/laya-browser-v7 999 >/dev/null
|
| 18 |
+
(cd ../jev-ultrafast && SUITE_OUT=$PWD/../laya/finetune/out/suite_v7.json timeout 1500 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -vE "Warn|TileLang")
|
| 19 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 20 |
+
echo V7_DONE
|
code/finetune/run_v8.sh
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# v8: after v7 -> DAgger harvest on real tasks with v7 as the policy and Qwen as teacher -> retrain -> eval -> suite.
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 5 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 6 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True CKPT=0 COMPILE=0
|
| 7 |
+
P=.venv/bin/python
|
| 8 |
+
echo "== build v8"
|
| 9 |
+
$P finetune/build_items.py finetune/out/pages.jsonl finetune/out/cases.jsonl finetune/out/ finetune/out/done_cases.jsonl finetune/out/step2_cases.jsonl finetune/out/m2w_cases.jsonl finetune/out/dagger_cases.jsonl
|
| 10 |
+
echo "== train v8 (3 epochs)"; $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser-v8 3 2>&1 | grep --line-buffered -E "=== epoch|saved|Error|Traceback"
|
| 11 |
+
echo "== calibrate v8"; $P finetune/calibrate.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v8 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 12 |
+
echo "== eval v8"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v8 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 13 |
+
echo "== suite v8"
|
| 14 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 15 |
+
bash $S/restart_s1.sh $PWD/finetune/out/laya-browser-v8 999 >/dev/null
|
| 16 |
+
(cd ../jev-ultrafast && SUITE_OUT=$PWD/../laya/finetune/out/suite_v8.json timeout 1500 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -vE "Warn|TileLang")
|
| 17 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 18 |
+
echo V8_DONE
|
code/finetune/run_v9.sh
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# v9 "one shot": 420 pages -> goals -> real DONE -> step-2 -> DAgger (policy v8) -> format v2 (head 768) -> 4 epochs -> eval -> suite
|
| 3 |
+
S=/tmp/claude-1000/-home-ckl-projects-S/8a7ce50a-5640-4606-837a-93df5ff48c83/scratchpad
|
| 4 |
+
cd /home/ckl/projects/S/laya && source env.sh
|
| 5 |
+
export LAYA_BASE=$(ls -d ~/.cache/huggingface/hub/models--convaiinnovations--laya/snapshots/*/typed-decisions | head -1)
|
| 6 |
+
export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True CKPT=0 COMPILE=0
|
| 7 |
+
P=.venv/bin/python; J=../jev-ultrafast/.venv/bin/python
|
| 8 |
+
until grep -q "V8_DONE" finetune/out/v8.log && grep -q "^done" finetune/out/collect2.log; do sleep 30; done
|
| 9 |
+
echo "== pages: $(wc -l < finetune/out/pages.jsonl)"
|
| 10 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 11 |
+
echo "== goals for all pages"; (cd ../jev-ultrafast && timeout 3600 .venv/bin/python ../laya/finetune/gen_goals.py ../laya/finetune/out/pages.jsonl ../laya/finetune/out/cases.jsonl 12 2>&1 | tail -2)
|
| 12 |
+
echo "== real DONE cases"; (cd ../jev-ultrafast && timeout 3600 .venv/bin/python ../laya/finetune/make_done_cases.py ../laya/finetune/out/pages.jsonl ../laya/finetune/out/cases.jsonl ../laya/finetune/out/done_cases.jsonl 700 2>&1 | tail -1)
|
| 13 |
+
echo "== step-2 negatives"; (cd ../jev-ultrafast && timeout 1800 .venv/bin/python ../laya/finetune/gen_step2.py ../laya/finetune/out/done_cases.jsonl ../laya/finetune/out/step2_cases.jsonl 2 2>&1 | tail -1)
|
| 14 |
+
echo "== dagger round 2 (policy v8)"
|
| 15 |
+
bash $S/restart_s1.sh $PWD/finetune/out/laya-browser-v8 999 >/dev/null
|
| 16 |
+
(cd ../jev-ultrafast && timeout 3000 .venv/bin/python ../laya/finetune/dagger.py ../laya/finetune/out/dagger_cases.jsonl 2>&1 | tail -1)
|
| 17 |
+
bash $S/stop_all.sh >/dev/null 2>&1; sleep 3
|
| 18 |
+
export LAYA_FMT=v2 LAYA_HEAD=768
|
| 19 |
+
echo "== build v9 (format v2, head 768)"
|
| 20 |
+
$P finetune/build_items.py finetune/out/pages.jsonl finetune/out/cases.jsonl finetune/out/ finetune/out/done_cases.jsonl finetune/out/step2_cases.jsonl finetune/out/m2w_cases.jsonl finetune/out/dagger_cases.jsonl
|
| 21 |
+
echo "== train v9 (4 epochs)"; $P finetune/train.py finetune/out/train_items.pt finetune/out/laya-browser-v9 4 2>&1 | grep --line-buffered -E "=== epoch|saved|Error|Traceback"
|
| 22 |
+
echo "== calibrate v9"; $P finetune/calibrate.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v9 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 23 |
+
echo "== eval v9"; $P finetune/eval.py finetune/out/pages.jsonl finetune/out/eval_cases.jsonl $PWD/finetune/out/laya-browser-v9 2>&1 | grep -vE "TileLang|Warn|warn|Fetch"
|
| 24 |
+
echo "== suite v9"
|
| 25 |
+
bash $S/start_infra.sh >/dev/null 2>&1
|
| 26 |
+
bash $S/restart_s1.sh $PWD/finetune/out/laya-browser-v9 999 >/dev/null
|
| 27 |
+
(cd ../jev-ultrafast && SUITE_OUT=$PWD/../laya/finetune/out/suite_v9.json timeout 1500 .venv/bin/python ../laya/apps/browser_suite.py 2>&1 | grep -vE "Warn|TileLang")
|
| 28 |
+
bash $S/stop_all.sh >/dev/null 2>&1
|
| 29 |
+
echo V9_DONE
|