diff --git a/.gitattributes b/.gitattributes index 3ca58a1ad6a9da6c86d6ddbd4fc5748550df03ce..a418bb30c4ab5355f134fbba8bd1e1f8824cb019 100644 --- a/.gitattributes +++ b/.gitattributes @@ -40,3 +40,4 @@ v11s/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text v14s/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text v15s/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text v17s/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text +v19s/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md index 4889d20eb874053a9fa6d651f5ae92f35109b1a2..d8bc2786613db4e2fa531b3627a502ff55f57233 100644 --- a/README.md +++ b/README.md @@ -16,100 +16,91 @@ tags: datasets: - osunlp/Mind2Web - stanfordnlp/nnetnav-live +- webagentlab/webchain --- # laya-browser — laya fine-tuned as a browser-agent decision head (drop-in replacement for TypeSafe Jev) -![laya driving a real browser: 6 tasks, 12 decisions, median 22 ms per decision](assets/laya_browser_demo.gif) +![laya driving a real browser on held-out sites: search, filters, open a result](assets/laya_browser_demo.gif) **laya** ([convaiinnovations/laya](https://huggingface.co/convaiinnovations/laya)) is a non-autoregressive "System 1" decision model: one bidirectional encoder pass answers several typed questions (`choice` / `score` / `noul`) with calibrated probabilities, no text -generation. Out of the box it is near chance at browser decisions ("which element should I click for this goal?"). +generation. This repo fine-tunes it into the decision head of [browser-use/jev-ultrafast](https://github.com/browser-use/jev-ultrafast), +whose `/v1/systemone` request is exactly laya's `predict(state, questions)`: every step, one forward pass (~22 ms at full GPU clock) picks the operation +(CLICK / TYPE_TEXT / SELECT / PRESS_ENTER / SCROLL / DONE …) and its target element. Trained and evaluated locally on one RTX 4070 +Ti SUPER (16 GB), no paid API; a local Qwen3-8B-AWQ only writes the text that TYPE_TEXT types and picks dropdown values. -This repo fine-tunes it into the decision head of [browser-use/jev-ultrafast](https://github.com/browser-use/jev-ultrafast), whose -`/v1/systemone` request is exactly laya's `predict(state, questions)`: every step, one ~20 ms forward pass picks the operation -(CLICK / TYPE_TEXT / SELECT / PRESS_ENTER / SCROLL / DONE …) and its target element. Everything was trained and evaluated locally on -one RTX 4070 Ti SUPER (16 GB), no paid API; a local Qwen3-8B-AWQ only writes the text that TYPE_TEXT types. +**One model, `v19s/`.** Earlier checkpoints are in the commit history. This repo is updated only when a new version is clearly better. -**One model, `v17s/`.** Earlier checkpoints (v10, v10s, v11s, v14s, v15s) were removed from the main branch; they remain in the -commit history. This repo is updated only when a new version is clearly better. +## Results (v19s, mmBERT-base 322M) -## Results (v17s, mmBERT-base 322M, 17–23 ms per step) +All numbers below were measured with the harness in `code/jev-ultrafast.patch`; the v17s column was measured with the harness it +shipped with, so part of the gain is the harness (see "What changed"). -| evaluation | result | -|---|---| -| **Suite B: 18 tasks on 18 sites that appear in no training source**, ×3 runs (`code/apps/browser_suite_b.py`) | **100 % (54/54)** | -| Suite A: the original 16 real-site tasks, ×3 (`code/apps/browser_suite.py`) | 85 % (41/48) | -| Held-out decisions (6,268: live pages, Mind2Web, NNetNav): operation / target top-1 | 0.761 / 0.606 | -| Held-out synthetic long-horizon forms (webgym, 10 each, 60-step budget): flight / hotel / shop | 2/10 · 3/10 · 9/10 | - -- **Suite B** is the headline number. At start-up it checks that none of its domains appear among the 682 domains of every - training source (`results/round3/train_domains.json`). Its tasks are mostly one to three steps: navigation, site search, man - pages, RFC / package search, a shop's search, page 2, sort by price. -- **Suite A** (the older suite) still fails `books-page2` (the "next" link is several scrolls down) and Google Flights. -- **Long multi-field forms are the open problem.** The model now plans the right sequence (trip type, origin with its - autocomplete suggestion, passengers popup, submit, "Modify search" after a wrong submission) but confuses a date field labelled - "Departure" with the origin, and runs out of steps. See "What still fails". +| evaluation | v17s | **v19s** | +|---|---|---| +| **Suite C: 27 multi-step tasks on held-out real sites** (search + filters + sort + open), ×2 | 2/54 | **11–14/54** (two ×2 runs; single ×1 runs: 5–7/27) | +| Suite B: 18 tasks on 18 held-out sites, ×3 | 54/54 | **54/54** | +| Suite A: 16 original real-site tasks, ×3 | 41/48 | 39/48 | +| webgym held-out synthetic forms, 7 kinds × 10 (flight, hotel, shop, car, restaurant, signup, filter) | 15/70 | **30/70** | +| held-out decisions on sites unseen in training (WebChain): click operation / target top-1 | — | 0.94 / 0.48 | + +- **Suite C** is the honest headline: real multi-step tasks on domains absent from every training source. v19s reaches + ~20–26 %; single runs vary by ±4 tasks (site load times, popups), so treat differences smaller than that as noise. +- Suite A lost `books-open-book` (0/3), gained nothing else; suite B unchanged. + +**Latency.** One decision is **22–27 ms** on an RTX 4070 Ti SUPER *while the GPU is at full clock* (requests back to back). +An agent waits seconds between decisions for pages to load, and the GPU drops to idle clocks (P5/P8, 210–850 MHz) within a +second or two: the next decision then takes **75–270 ms** (the demo above shows these live numbers). Locking the minimum SM +clock removes that — `sudo nvidia-smi -lgc 2100,3135` (undo: `sudo nvidia-smi -rgc`; costs some idle power). The first request +of each input-length bucket also compiles a kernel once (~100–500 ms); warm up with a few requests after starting the server. ## Use ```bash huggingface-cli download cklxx/laya-browser --local-dir laya-browser cd laya-browser/code && uv sync --extra fast # pinned uv.lock (Python 3.12, torch 2.11, tilelang 0.1.14) -uv run python verify.py # downloads v17s, answers one recorded browser step -uv run python verify.py --fast # same through the TileLang fast path +uv run python verify.py # downloads v19s, answers one recorded browser step ``` As a TypeSafe replacement for jev-ultrafast (apply `code/jev-ultrafast.patch` to jev-ultrafast `1231850`): ```bash -python code/apps/systemone_server.py 8791 /path/to/laya-browser/v17s 60 # 60 = split choices wider than 60 options +python code/apps/systemone_server.py 8791 /path/to/laya-browser/v19s 60 # 60 = split choices wider than 60 options # jev-ultrafast: TYPESAFE_BASE_URL=http://127.0.0.1:8791 ``` -The checkpoint records `laya_fmt` (v3) and `head_max_len_train` (768); the server applies the matching input format. -The chunk threshold matters: without it a 160-link page (Hacker News) leaves each option ~4 tokens and the choice becomes a coin -toss. - -## What went into v17s - -**Harness fixes** (`code/jev-ultrafast.patch`). Half of the original failures were harness bugs, not model errors: -a `PRESS_ENTER` control while a filled text field is focused (arXiv's search overlay has no submit button); elements covered by an -unrelated element are not offered, a target covered by its own ancestor is clicked through; observation retries while a page is -navigating; a choice that fails 3× or repeats on the same URL is excluded. - -**Training data** (v15s: a clean retrain from the mmBERT-base checkpoint, no suite start page and no DAgger data; v17s continues -it for one epoch): -- 421 crawled pages with reverse-generated goals, 700 real DONE states, step-2 negatives; -- [Mind2Web](https://huggingface.co/datasets/osunlp/Mind2Web) train (7.3k steps) plus 5.1k steps re-labelled with planner-style - sub-goals by a local Qwen (trajectory-conditioned, Plan-and-Act style); -- 8.9k steps from [NNetNav-live](https://huggingface.co/datasets/stanfordnlp/nnetnav-live); -- 2.3k scripted real-browser trajectories on 206 sites (search → Enter, open an article, scroll to page 2, `` option's name is kept when a label is shortened to 50 characters (before, "Please select an option -Option 1 Option 2 → Option 2" lost the part that matters: Mind2Web dropdown target 0.39 → 0.74), and synthetic category links -are real links that are sometimes the answer (dead distractor links had taught the model to skip sidebars). Sub-goals no longer -call the origin "departure". +The checkpoint records `laya_fmt` (**v5**) and `head_max_len_train` (768); the server applies the matching input format: +option labels without the duplicated `[key]`, `` value), never the model. +Because an observation only covers the viewport, the judge (`final_state`) scrolls to the top and then screen by screen, +merging the actions and visible text of the whole page before the check runs, so a check never depends on where the +agent left the scroll position. + +## Tasks + +Steps = number of browser actions of the scripted solution (a scroll counts as one action). + +| # | name | site | goal | scripted steps | check verifies | +|---|------|------|------|:--:|----------------| +| 1 | met-sunflowers | metmuseum.org collection search | search "sunflowers", tick "Has image", sort by date oldest first | 4: fill, Enter, checkbox, select | URL `q=sunflowers&showOnly=withImage&sortBy=Date` (newest gives `DateDesc`) | +| 2 | nuget-serilog-tool | nuget.org | search "serilog", package type ".NET tool", sort by downloads | 4: fill, Enter, radio, select | URL `q=serilog`, `packagetype=dotnettool`, `sortby=totalDownloads-desc` | +| 3 | alpine-curl-filter | pkgs.alpinelinux.org | package name curl, branch v3.20, repo main, arch aarch64 | 5: fill, 3× select, Enter | URL `name=curl&branch=v3.20&repo=main&arch=aarch64` | +| 4 | freesound-rain-cc0 | freesound.org | search "rain", license CC0 facet, category Soundscapes facet | 4: fill, Enter, 2× facet link | URL `q=rain` and `f=license:"Creative Commons 0" category:"Soundscapes"` | +| 5 | ats-shampoo-haircare | automationteststore.com | search "shampoo", category Hair Care, "search in descriptions", Search, sort price high→low | 6: fill, Enter, select, checkbox, button, select | URL `keyword=shampoo`, `category_id` contains 52, `description=1`, `sort=p.price-DESC` | +| 6 | bnf-hugo-printed-p2 | catalogue.bnf.fr | search "victor hugo", facet "Texte imprimé et livre numérique", page 2 | 4: fill, submit, facet, "Page suivante" | URL `motRecherche=victor hugo`, `listeAffinages` has `FacNatDoc_a`, `pageEnCours=2` | +| 7 | vsm-python-installs | marketplace.visualstudio.com | search "python", Sort By menu → Installs | 4: fill, Enter, menu, menuitem | URL `term=python&sortBy=Installs` | +| 8 | wp-cache-commercial-redis | wordpress.org/plugins | search "cache", "Commercial" filter, open Redis Object Cache | 4: fill, Enter, filter, link | URL `/plugins/redis-cache` | +| 9 | todomvc-active | demo.playwright.dev/todomvc | add "buy milk" and "walk the dog", show Active | 5: fill, Enter, fill, Enter, link | URL ends `#/active`, both todos in text, "2 items left"; localStorage is cleared before each run | +| 10 | setlist-radiohead-uk | setlist.fm | search "radiohead", Artist → Radiohead, Country → United Kingdom | 4: fill, Enter, 2× select | URL `query=radiohead&artist=bd6bd12&country=gb` | +| 11 | jetbrains-rust-free | plugins.jetbrains.com | search "rust", "free" filter, open the Rust plugin by JetBrains | 4: fill, Enter, button, link | URL `/plugin/22407-rust` | +| 12 | luarocks-rapidjson | luarocks.org | search "json", tick "Include non-root", Search, open rapidjson | 5: fill, Enter, checkbox, button, link | URL `/modules/xpol/rapidjson` | +| 13 | letcode-dropdowns | letcode.in/dropdowns | four `` has ~235 options, so the observation is truncated at 250 actions + and the "Words" text field is never offered. +- gitea.com explore: the custom Sort dropdown's options do not navigate when clicked (URL stays `sort=recentupdate`). +- rawg.io: Enter in the search box does not navigate (`rawg.io/?`); results are only rendered in-page. +- weather.gov: neither the "Go" nor the "Get Weather" form submits from a synthetic click (no navigation after 1.5 s). +- data.cityofnewyork.us: the search box is a combobox, so no Enter action is offered, and clicking the search button + drops the query when a filter is applied afterwards. +- uniprot.org: the facet sidebar ("Reviewed") is not rendered in the 1120×780 viewport. +- bstackdemo.com: vendor checkboxes and "Add to cart" buttons are custom elements that are not observed as actions. +- cms.demo.katalon.com: search only returns blog posts; the shop's sorting ` values), like suite A's dropdown/checkbox checks. +""" +import glob, json, os, re, sys, time +from urllib.parse import parse_qs, unquote_plus, urlparse +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import browser_suite # noqa: E402,F401 (sets up the jev env vars) +from jev_ultrafast import Agent # noqa: E402 +from jev_ultrafast.browser import Browser, StalePage # noqa: E402 + +HERE = os.path.dirname(os.path.abspath(__file__)) + + +def q(u): # decoded query parameters of a URL: {name: first value} + return {k: v[0] for k, v in parse_qs(urlparse(u).query).items()} + + +def selected(actions, label): # current value of the label contains `label` and whose option label equals/contains `option`.""" + hits = [a for a in page["actions"] if a["kind"] == "select" and label.lower() in a["label"].rsplit(" → ", 1)[0].lower()] + exact = [a for a in hits if a["label"].rsplit(" → ", 1)[1].strip().lower() == option.lower()] + part = [a for a in hits if option.lower() in a["label"].rsplit(" → ", 1)[1].lower()] + if not (exact or part): + raise LookupError(f"no select option {option!r} in dropdown {label!r} (have {[a['label'][-45:] for a in hits][:30]})") + return (exact or part)[0] + + +def run_steps(br, steps, verbose=False): + """Execute the steps; returns (final page, number of browser actions performed).""" + page = observe(br) + n = 0 + for step in steps: + op = step[0] + if op == "sleep": + page = observe(br, settle=step[1]); continue + if op == "scroll": + for _ in range(step[1] if len(step) > 1 else 1): + a = next((x for x in page["actions"] if x["id"] == "scroll_down"), None) + if a is None: break + try: + br.act(a, page) + except StalePage: # page changed under us: observe again, then scroll + page = observe(br, settle=0.5); continue + page = observe(br); n += 1 + if verbose: + print(f" {n:2d}. scroll") + continue + for attempt in range(4): # retry a stale/covered target after a fresh observation + try: + if op == "click": + a = find(page, "click", step[1]); br.act(a, page) + elif op == "fill": + a = find(page, "fill", step[1]); br.act(a, page, text=step[2]) + elif op == "select": + a = find_select(page, step[1], step[2]); br.act(a, page) + elif op == "enter": + a = next(x for x in page["actions"] if x["kind"] == "key"); br.act(a, page) + else: + raise ValueError(op) + break + except (StalePage, LookupError, StopIteration) as e: + if attempt == 3: + raise + time.sleep(0.8); page = observe(br) + n += 1 + if verbose: + print(f" {n:2d}. {op} {step[1:]}") + page = observe(br, settle=0.6) + time.sleep(1.5) # let a slow navigation land before judging (same as suite run()) + return observe(br), n + + +def show(page, limit=120): + print("URL:", page["url"]); print("TITLE:", page["title"]) + print("TEXT:", page["text"][:500].replace("\n", " | ")) + for a in page["actions"][:limit]: + extra = "" if a["kind"] != "select" else f" [current={a.get('current_value')}]" + chk = f" checked={a['checked']}" if "checked" in a else "" + print(f" {a['id']:5s} {a['kind']:6s} {a.get('role') or '':9s} {a['label'][:90]!r}{extra}{chk}") + if len(page["actions"]) > limit: + print(f" ... {len(page['actions'])-limit} more") + + +def parse_cli_step(s): + op, _, rest = s.partition(":") + if op in ("enter",): return ("enter",) + if op == "scroll": return ("scroll", int(rest or 1)) + if op == "sleep": return ("sleep", float(rest)) + if op == "fill": + lab, _, txt = rest.partition("="); return ("fill", lab, txt) + if op == "select": + lab, _, opt = rest.partition("="); return ("select", lab, opt) + return (op, rest) + + +# name -> steps (the start URL and the check live in browser_suite_c.TASKS) +SCRIPTS = { + # ---- multi-step ---- + "met-sunflowers": [("fill", "Search by subject", "sunflowers"), ("enter",), ("click", "Has image"), + ("select", "Relevance", "Date (oldest-newest)")], + "nuget-serilog-tool": [("fill", "Enter packages", "serilog"), ("enter",), ("click", "Package Type: .NET tool"), + ("select", "sort package", "Downloads")], + "alpine-curl-filter": [("fill", "Package name", "curl"), ("select", "Branch", "v3.20"), ("select", "Repository", "main"), + ("select", "Architecture", "aarch64"), ("enter",)], + "freesound-rain-cc0": [("fill", "Search sounds", "rain"), ("enter",), ("click", "Creative Commons 0"), ("click", "Soundscapes")], + "ats-shampoo-haircare": [("fill", "Search Keywords", "shampoo"), ("enter",), ("select", "All Categories", "Hair Care"), + ("click", "Search in product descriptions"), ("click", "@button re:^Search$"), ("select", "Date Old", "Price High > Low")], + "bnf-hugo-printed-p2": [("fill", "Rechercher une notice", "victor hugo"), ("click", "Submit"), ("click", "Texte imprimé"), + ("click", "Page suivante")], + "vsm-python-installs": [("fill", "Search Visual Studio Code extensions", "python"), ("enter",), ("click", "Sort By"), ("click", "re:^Installs")], + "wp-cache-commercial-redis": [("fill", "re:^Search$", "cache"), ("enter",), ("click", "Commercial"), ("click", "re:^Redis Object Cache")], + "todomvc-active": [("fill", "What needs to be done", "buy milk"), ("enter",), ("fill", "What needs to be done", "walk the dog"), + ("enter",), ("click", "re:^Active$")], + "setlist-radiohead-uk": [("fill", "Artist, Venue", "radiohead"), ("enter",), ("select", "Artist", "Radiohead ("), + ("select", "Country", "United Kingdom")], + "jetbrains-rust-free": [("fill", "re:^Search", "rust"), ("enter",), ("click", "re:^free$"), ("click", "re:^plugin icon Rust JetBrains")], + "luarocks-rapidjson": [("fill", "Search modules", "json"), ("enter",), ("click", "Include non-root"), ("click", "@button re:^Search$"), + ("click", "re:^rapidjson$")], + "letcode-dropdowns": [("select", "apple", "Apple"), ("select", "super hero", "Batman"), ("select", "programming language", "Swift"), + ("scroll", 1), ("select", "Select India", "India")], + "letcode-radio": [("click", "re:^Foo$"), ("click", "re:^Going$"), ("click", "I agree"), ("click", "Remember me")], + "clojars-ring-page3": [("fill", "Search projects", "ring"), ("enter",), ("scroll", 4), ("click", "re:^3$")], + "fred-unemployment-pop": [("fill", "re:^Search", "unemployment rate"), ("enter",), ("click", "Sort by Relevance"), ("click", "re:^Popularity")], + "modrinth-sodium": [("fill", "Search mods", "sodium"), ("enter",), ("click", "re:^1\\.21\\.11$"), ("click", "Sort by"), ("click", "re:^Downloads$")], + "tvmaze-friends-episodes": [("fill", "Search Shows", "friends"), ("enter",), ("click", "re:^Friends$"), ("click", "re:^Episodes$")], + "qaclickjet-form": [("click", "Round Trip"), ("click", "Senior Citizen"), ("select", "INR", "USD"), ("fill", "Type to Select", "India")], + # ---- shorter ---- + "cocktail-margarita": [("fill", "Search for a Cocktail", "margarita"), ("enter",), ("click", "re:^Margarita")], + "mealdb-arrabiata": [("fill", "Search for a Meal", "arrabiata"), ("enter",), ("click", "Spicy Arrabiata")], + "gentoo-openrc-talk": [("fill", "Search Gentoo Wiki", "OpenRC"), ("enter",), ("click", "re:^Discussion$")], + "webkit-css-bugs": [("click", "re:^Browse$"), ("click", "re:^WebKit \n"), ("click", "re:^CSS \n")], + "govdata-wetter-energie": [("fill", "Suchbegriff", "Wetter"), ("enter",), ("click", "Energie")], + "fedora-vim-common": [("fill", "re:^Search$", "vim"), ("enter",), ("click", "re:^vim-common$")], + "racket-argo": [("fill", "Search packages", "json"), ("enter",), ("click", "re:^argo$")], + "rdrr-ggplot": [("fill", "packages, doc text", "ggplot"), ("enter",)], +} + + +def main(): + if len(sys.argv) > 1 and sys.argv[1] == "--probe": + br = Browser(sys.argv[2]) + try: + page, n = run_steps(br, [parse_cli_step(s) for s in sys.argv[3:]], verbose=True) + show(page) + finally: + br.close() + return + sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) + from browser_suite_c import TASKS, check_page, final_state # noqa: E402 + flt = sys.argv[1] if len(sys.argv) > 1 else "" + repeats = int(os.environ.get("REPEATS", "1")) + rows = [] + shard, nshards = (int(v) for v in os.environ.get("SHARD", "0/1").split("/")) # SHARD=i/n runs every n-th task + for i, (name, url, goal, check, *extra) in enumerate(TASKS): + if (flt and flt not in name) or i % nshards != shard: continue + for rep in range(repeats): + t0 = time.time(); ok = start_ok = None; n = 0; err = "" + try: + if extra: extra[0](url) # per-task setup (e.g. clear localStorage) + br = Browser(url) + try: + start_ok = check_page(name, check, final_state(br)) + page, n = run_steps(br, SCRIPTS[name]) + page = final_state(br) + ok = check_page(name, check, page) + final = page["url"] + finally: + br.close() + except Exception as e: + err = f"{type(e).__name__}: {str(e)[:120]}"; final = "" + verdict = "PASS" if (ok and not start_ok) else "FAIL" + rows.append((name, verdict == "PASS", n)) + print(f"{verdict} {name:26s} steps={n:2d} start_check={start_ok!s:5s} final_check={ok!s:5s} {time.time()-t0:5.1f}s {final[:60]} {err}", flush=True) + good = sum(r[1] for r in rows) + print(f"\n== {good}/{len(rows)} scripted runs passed") + per = {}; steps = {} + for r in rows: per.setdefault(r[0], []).append(r[1]); steps[r[0]] = max(steps.get(r[0], 0), r[2]) + print(" per task: " + " ".join(f"{k}={sum(v)}/{len(v)}({steps[k]} steps)" for k, v in per.items())) + json.dump({k: {"pass": sum(v), "runs": len(v), "steps": steps[k]} for k, v in per.items()}, + open(os.environ.get("SCRIPTS_OUT", "/tmp/suite_c_scripts.json"), "w"), indent=1) + + +if __name__ == "__main__": + main() diff --git a/code/apps/systemone_server.py b/code/apps/systemone_server.py index 2f53b259ae50b7bbb0b8e1260d5e6e322f472565..bb06cd4a589dd70d4f1afac1434b1b8262921787 100644 --- a/code/apps/systemone_server.py +++ b/code/apps/systemone_server.py @@ -4,7 +4,7 @@ Request body: {"model": ..., "state": {...}, "questions": {...}} -> {"answers": ..., "model": ..., "usage": ...} """ -import json, os, sys, time, traceback +import json, os, re, sys, time, traceback from http.server import ThreadingHTTPServer, BaseHTTPRequestHandler from common import get_agent from fast_batch import predict_fast @@ -22,17 +22,28 @@ LOG = [] # ---- System 1 / System 2 gating: below ESCALATE_TAU confidence, ask the LLM teacher (same element table) and return its # decision in laya's answer format. Every escalation is also appended to ESCALATE_LOG as a DAgger case. TAU = float(os.environ.get("ESCALATE_TAU", "0")) # 0 = off -ESC_URL = os.environ.get("TEXT_MODEL_BASE_URL", "http://127.0.0.1:30000/v1") + "/chat/completions" -ESC_MODEL = os.environ.get("TEXT_MODEL", "Qwen/Qwen3-8B-AWQ") +# System 2 model: S2_BASE_URL / S2_API_KEY / S2_MODEL / S2_EXTRA_JSON (e.g. DeepSeek); defaults to the local text model +ESC_URL = os.environ.get("S2_BASE_URL", os.environ.get("TEXT_MODEL_BASE_URL", "http://127.0.0.1:30000/v1")).rstrip("/") + "/chat/completions" +ESC_MODEL = os.environ.get("S2_MODEL", os.environ.get("TEXT_MODEL", "Qwen/Qwen3-8B-AWQ")) +ESC_KEY = os.environ.get("S2_API_KEY", "") +ESC_EXTRA = json.loads(os.environ.get("S2_EXTRA_JSON", '{"chat_template_kwargs": {"enable_thinking": false}}')) ESC_LOG = os.environ.get("ESCALATE_LOG", "") STATS = {"calls": 0, "escalated": 0} -ESC_SYS = """You are the System-2 fallback for a browser agent. Given the goal, the actions so far, the page and a numbered list of -controls (each with the operations it supports), pick the single best NEXT step. Answer JSON: -{"operation": "CLICK"|"TYPE_TEXT"|"SELECT"|"DONE"|"BLOCKED"|"WAIT"|"SCROLL_DOWN"|"SCROLL_UP", "target": "