Feature Extraction
Transformers
Safetensors
English
multilingual
laya_browser
laya
custom_code
system-1
browser-agent
web-navigation
decision-model
mmbert
mind2web
tilelang
Instructions to use cklxx/laya-browser with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use cklxx/laya-browser with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("feature-extraction", model="cklxx/laya-browser", trust_remote_code=True)# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("cklxx/laya-browser", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download code/finetune/collect_pages.py from cklxx/laya-browser: direct link, hf CLI and curl.
- Browser
- Download file 9.44 kB
-
https://huggingface.co/cklxx/laya-browser/resolve/main/code/finetune/collect_pages.py
- Command line
-
hf download hf://cklxx/laya-browser/code/finetune/collect_pages.py
-
curl -L -o collect_pages.py https://huggingface.co/cklxx/laya-browser/resolve/main/code/finetune/collect_pages.py
9.44 kB
| """Crawl real pages with browser-harness and dump their jev-ultrafast observations (element tables + page text). | |
| python finetune/collect_pages.py out/pages.jsonl [max_pages=80] | |
| Starts from SEEDS, follows a few random same-site links from each page to diversify. One JSON line per page. | |
| """ | |
| import json, os, random, sys, time | |
| sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast") | |
| os.environ.setdefault("BU_CDP_URL", "http://127.0.0.1:9222") | |
| from jev_ultrafast.browser import Browser, StalePage | |
| from jev_ultrafast.model import action_space | |
| SEEDS = [ | |
| "https://en.wikipedia.org/wiki/Main_Page", "https://en.wikipedia.org/wiki/Special:Random", "https://en.wikipedia.org/wiki/Python_(programming_language)", | |
| "https://news.ycombinator.com/", "https://news.ycombinator.com/newest", "https://news.ycombinator.com/login", | |
| "https://github.com/", "https://github.com/tile-ai/tilelang", "https://github.com/browser-use/browser-use/issues", "https://github.com/login", | |
| "https://www.python.org/", "https://docs.python.org/3/", "https://pypi.org/", "https://pypi.org/project/laya/", | |
| "https://archlinux.org/", "https://wiki.archlinux.org/", "https://developer.mozilla.org/en-US/", "https://duckduckgo.com/", | |
| "https://www.bing.com/", "https://stackoverflow.com/questions", "https://www.reddit.com/", "https://arxiv.org/", | |
| "https://arxiv.org/list/cs.LG/recent", "https://huggingface.co/models", "https://huggingface.co/convaiinnovations/laya", | |
| "https://www.saucedemo.com/", "https://the-internet.herokuapp.com/", "https://the-internet.herokuapp.com/login", | |
| "https://demo.opencart.com/", "https://www.demoblaze.com/", "https://books.toscrape.com/", "https://quotes.toscrape.com/login", | |
| "https://www.google.com/travel/flights?hl=en", "https://www.booking.com/", "https://www.airbnb.com/", "https://www.amazon.com/", | |
| "https://www.ebay.com/", "https://www.imdb.com/", "https://www.nytimes.com/", "https://www.bbc.com/", | |
| "https://www.openstreetmap.org/", "https://weather.com/", "https://www.wolframalpha.com/", "https://translate.google.com/", | |
| "https://www.gnu.org/", "https://kernel.org/", "https://www.rust-lang.org/", "https://go.dev/", "https://nodejs.org/en", | |
| "https://www.npmjs.com/", "https://crates.io/", "https://docs.rs/", "https://www.kaggle.com/", "https://paperswithcode.com/", | |
| # round 2: more sites, more form-heavy pages | |
| "https://en.wikipedia.org/wiki/Special:Search", "https://en.wikipedia.org/wiki/Portal:Current_events", "https://de.wikipedia.org/", "https://zh.wikipedia.org/", | |
| "https://news.ycombinator.com/ask", "https://news.ycombinator.com/show", "https://news.ycombinator.com/jobs", "https://news.ycombinator.com/submit", | |
| "https://github.com/explore", "https://github.com/trending", "https://github.com/pytorch/pytorch", "https://github.com/pytorch/pytorch/pulls", | |
| "https://github.com/pytorch/pytorch/issues", "https://gitlab.com/explore", "https://gitee.com/explore", "https://about.gitlab.com/", | |
| "https://www.python.org/downloads/", "https://docs.python.org/3/tutorial/", "https://docs.python.org/3/library/", "https://peps.python.org/", | |
| "https://pypi.org/search/?q=torch", "https://pypi.org/project/torch/", "https://pypi.org/account/login/", "https://pypi.org/help/", | |
| "https://the-internet.herokuapp.com/dropdown", "https://the-internet.herokuapp.com/checkboxes", "https://the-internet.herokuapp.com/forgot_password", | |
| "https://the-internet.herokuapp.com/inputs", "https://the-internet.herokuapp.com/tables", "https://the-internet.herokuapp.com/javascript_alerts", | |
| "https://demo.opencart.com/index.php?route=product/category&path=20", "https://demo.opencart.com/index.php?route=account/login", | |
| "https://demo.opencart.com/index.php?route=account/register", "https://demo.opencart.com/index.php?route=product/search&search=mac", | |
| "https://www.demoblaze.com/cart.html", "https://www.demoblaze.com/prod.html?idp_=1", "https://books.toscrape.com/catalogue/page-2.html", | |
| "https://books.toscrape.com/catalogue/category/books/mystery_3/index.html", "https://quotes.toscrape.com/tag/love/", "https://quotes.toscrape.com/page/2/", | |
| "https://www.saucedemo.com/inventory.html", "https://automationexercise.com/", "https://automationexercise.com/login", "https://automationexercise.com/products", | |
| "https://practicetestautomation.com/practice-test-login/", "https://demoqa.com/", "https://demoqa.com/text-box", "https://demoqa.com/select-menu", | |
| "https://demoqa.com/webtables", "https://www.selenium.dev/selenium/web/web-form.html", "https://formy-project.herokuapp.com/", "https://formy-project.herokuapp.com/form", | |
| "https://parabank.parasoft.com/parabank/index.htm", "https://parabank.parasoft.com/parabank/register.htm", "https://www.globalsqa.com/angularJs-protractor/BankingProject/", | |
| "https://opensource-demo.orangehrmlive.com/", "https://magento.softwaretestingboard.com/", "https://magento.softwaretestingboard.com/women.html", | |
| "https://www.airbnb.com/s/London/homes", "https://www.booking.com/searchresults.html?ss=Paris", "https://www.google.com/travel/hotels?hl=en", | |
| "https://www.google.com/maps?hl=en", "https://www.google.com/search?q=tilelang&hl=en", "https://duckduckgo.com/?q=modernbert", "https://www.bing.com/search?q=laya", | |
| "https://arxiv.org/list/cs.CL/new", "https://arxiv.org/abs/2412.13663", "https://arxiv.org/search/?query=flash+attention&searchtype=all", | |
| "https://huggingface.co/datasets", "https://huggingface.co/spaces", "https://huggingface.co/docs", "https://huggingface.co/login", "https://huggingface.co/answerdotai/ModernBERT-base", | |
| "https://www.modelscope.cn/models", "https://www.modelscope.cn/datasets", "https://www.kaggle.com/datasets", "https://www.kaggle.com/competitions", | |
| "https://stackoverflow.com/", "https://stackoverflow.com/questions/tagged/python", "https://superuser.com/", "https://askubuntu.com/", | |
| "https://www.reddit.com/r/MachineLearning/", "https://old.reddit.com/", "https://old.reddit.com/r/python/", "https://lobste.rs/", | |
| "https://www.bbc.com/news", "https://www.bbc.com/sport", "https://www.theguardian.com/international", "https://www.reuters.com/", "https://apnews.com/", | |
| "https://www.imdb.com/chart/top/", "https://www.imdb.com/find/?q=inception", "https://www.rottentomatoes.com/", "https://www.goodreads.com/", | |
| "https://www.openstreetmap.org/search?query=Berlin", "https://www.wikidata.org/", "https://commons.wikimedia.org/", "https://www.wiktionary.org/", | |
| "https://developer.mozilla.org/en-US/docs/Web/JavaScript", "https://developer.mozilla.org/en-US/docs/Web/HTML/Element/select", "https://web.dev/", | |
| "https://www.rust-lang.org/learn", "https://doc.rust-lang.org/book/", "https://go.dev/doc/", "https://pkg.go.dev/", "https://nodejs.org/en/download", | |
| "https://www.npmjs.com/package/react", "https://react.dev/", "https://vuejs.org/", "https://tailwindcss.com/docs", "https://getbootstrap.com/docs/", | |
| "https://www.wolframalpha.com/input?i=2%2B2", "https://translate.google.com/?sl=en&tl=zh-CN&text=hello", "https://www.deepl.com/translator", | |
| "https://weather.com/weather/today/l/USNY0996", "https://www.timeanddate.com/", "https://www.xe.com/currencyconverter/", "https://www.calculator.net/", | |
| "https://archlinux.org/packages/", "https://aur.archlinux.org/", "https://wiki.archlinux.org/title/Installation_guide", "https://www.kernel.org/doc/", | |
| "https://www.debian.org/", "https://ubuntu.com/download", "https://www.gnu.org/software/", "https://www.fsf.org/", | |
| ] | |
| def observe(url, timeout=25): | |
| b = Browser(url) | |
| try: | |
| page = b.observe(screenshot=False) | |
| links = b.evaluate("""(() => { const out=[]; for (const a of document.querySelectorAll('a[href]')) { | |
| const h=a.href; if (h.startsWith(location.origin) && !h.includes('#') && h!==location.href) out.push(h); } return out.slice(0,400); })()""") or [] | |
| return page, links | |
| finally: | |
| b.close() | |
| def main(): | |
| out, max_pages = sys.argv[1], int(sys.argv[2]) if len(sys.argv) > 2 else 80 | |
| os.makedirs(os.path.dirname(out) or ".", exist_ok=True) | |
| seen = set() | |
| if os.path.exists(out): | |
| for line in open(out): | |
| seen.add(json.loads(line)["url"]) | |
| rng = random.Random(0) | |
| queue = list(SEEDS); rng.shuffle(queue) | |
| n = len(seen) | |
| with open(out, "a") as f: | |
| while queue and n < max_pages: | |
| url = queue.pop(0) | |
| if url in seen: | |
| continue | |
| t = time.time() | |
| try: | |
| page, links = observe(url) | |
| except Exception as e: | |
| print(f"skip {url}: {type(e).__name__}: {str(e)[:80]}", flush=True); continue | |
| elements, targets, controls = action_space(page["actions"]) | |
| if len(elements) < 5 or len(elements) > 160: | |
| print(f"skip {url}: {len(elements)} elements", flush=True); continue | |
| seen.add(page["url"]); n += 1 | |
| f.write(json.dumps({"url": page["url"], "title": page["title"], "text": page["text"], "actions": page["actions"], | |
| "scroll": page.get("scroll")}, ensure_ascii=False) + "\n"); f.flush() | |
| print(f"[{n}] {len(elements):3d} elements {time.time()-t:4.1f}s {page['title'][:60]}", flush=True) | |
| rng.shuffle(links) | |
| queue.extend(l for l in links[:3] if l not in seen) | |
| print("done", n, "pages") | |
| if __name__ == "__main__": | |
| main() | |