laya-browser / code /finetune /collect_pages.py
cklxx's picture
laya-browser v10 / v10s: laya fine-tuned as a browser-agent decision head + code + results
adf912b verified
Raw History Blame Contribute Delete
9.44 kB
"""Crawl real pages with browser-harness and dump their jev-ultrafast observations (element tables + page text).
python finetune/collect_pages.py out/pages.jsonl [max_pages=80]
Starts from SEEDS, follows a few random same-site links from each page to diversify. One JSON line per page.
"""
import json, os, random, sys, time
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast")
os.environ.setdefault("BU_CDP_URL", "http://127.0.0.1:9222")
from jev_ultrafast.browser import Browser, StalePage
from jev_ultrafast.model import action_space
SEEDS = [
"https://en.wikipedia.org/wiki/Main_Page", "https://en.wikipedia.org/wiki/Special:Random", "https://en.wikipedia.org/wiki/Python_(programming_language)",
"https://news.ycombinator.com/", "https://news.ycombinator.com/newest", "https://news.ycombinator.com/login",
"https://github.com/", "https://github.com/tile-ai/tilelang", "https://github.com/browser-use/browser-use/issues", "https://github.com/login",
"https://www.python.org/", "https://docs.python.org/3/", "https://pypi.org/", "https://pypi.org/project/laya/",
"https://archlinux.org/", "https://wiki.archlinux.org/", "https://developer.mozilla.org/en-US/", "https://duckduckgo.com/",
"https://www.bing.com/", "https://stackoverflow.com/questions", "https://www.reddit.com/", "https://arxiv.org/",
"https://arxiv.org/list/cs.LG/recent", "https://huggingface.co/models", "https://huggingface.co/convaiinnovations/laya",
"https://www.saucedemo.com/", "https://the-internet.herokuapp.com/", "https://the-internet.herokuapp.com/login",
"https://demo.opencart.com/", "https://www.demoblaze.com/", "https://books.toscrape.com/", "https://quotes.toscrape.com/login",
"https://www.google.com/travel/flights?hl=en", "https://www.booking.com/", "https://www.airbnb.com/", "https://www.amazon.com/",
"https://www.ebay.com/", "https://www.imdb.com/", "https://www.nytimes.com/", "https://www.bbc.com/",
"https://www.openstreetmap.org/", "https://weather.com/", "https://www.wolframalpha.com/", "https://translate.google.com/",
"https://www.gnu.org/", "https://kernel.org/", "https://www.rust-lang.org/", "https://go.dev/", "https://nodejs.org/en",
"https://www.npmjs.com/", "https://crates.io/", "https://docs.rs/", "https://www.kaggle.com/", "https://paperswithcode.com/",
# round 2: more sites, more form-heavy pages
"https://en.wikipedia.org/wiki/Special:Search", "https://en.wikipedia.org/wiki/Portal:Current_events", "https://de.wikipedia.org/", "https://zh.wikipedia.org/",
"https://news.ycombinator.com/ask", "https://news.ycombinator.com/show", "https://news.ycombinator.com/jobs", "https://news.ycombinator.com/submit",
"https://github.com/explore", "https://github.com/trending", "https://github.com/pytorch/pytorch", "https://github.com/pytorch/pytorch/pulls",
"https://github.com/pytorch/pytorch/issues", "https://gitlab.com/explore", "https://gitee.com/explore", "https://about.gitlab.com/",
"https://www.python.org/downloads/", "https://docs.python.org/3/tutorial/", "https://docs.python.org/3/library/", "https://peps.python.org/",
"https://pypi.org/search/?q=torch", "https://pypi.org/project/torch/", "https://pypi.org/account/login/", "https://pypi.org/help/",
"https://the-internet.herokuapp.com/dropdown", "https://the-internet.herokuapp.com/checkboxes", "https://the-internet.herokuapp.com/forgot_password",
"https://the-internet.herokuapp.com/inputs", "https://the-internet.herokuapp.com/tables", "https://the-internet.herokuapp.com/javascript_alerts",
"https://demo.opencart.com/index.php?route=product/category&path=20", "https://demo.opencart.com/index.php?route=account/login",
"https://demo.opencart.com/index.php?route=account/register", "https://demo.opencart.com/index.php?route=product/search&search=mac",
"https://www.demoblaze.com/cart.html", "https://www.demoblaze.com/prod.html?idp_=1", "https://books.toscrape.com/catalogue/page-2.html",
"https://books.toscrape.com/catalogue/category/books/mystery_3/index.html", "https://quotes.toscrape.com/tag/love/", "https://quotes.toscrape.com/page/2/",
"https://www.saucedemo.com/inventory.html", "https://automationexercise.com/", "https://automationexercise.com/login", "https://automationexercise.com/products",
"https://practicetestautomation.com/practice-test-login/", "https://demoqa.com/", "https://demoqa.com/text-box", "https://demoqa.com/select-menu",
"https://demoqa.com/webtables", "https://www.selenium.dev/selenium/web/web-form.html", "https://formy-project.herokuapp.com/", "https://formy-project.herokuapp.com/form",
"https://parabank.parasoft.com/parabank/index.htm", "https://parabank.parasoft.com/parabank/register.htm", "https://www.globalsqa.com/angularJs-protractor/BankingProject/",
"https://opensource-demo.orangehrmlive.com/", "https://magento.softwaretestingboard.com/", "https://magento.softwaretestingboard.com/women.html",
"https://www.airbnb.com/s/London/homes", "https://www.booking.com/searchresults.html?ss=Paris", "https://www.google.com/travel/hotels?hl=en",
"https://www.google.com/maps?hl=en", "https://www.google.com/search?q=tilelang&hl=en", "https://duckduckgo.com/?q=modernbert", "https://www.bing.com/search?q=laya",
"https://arxiv.org/list/cs.CL/new", "https://arxiv.org/abs/2412.13663", "https://arxiv.org/search/?query=flash+attention&searchtype=all",
"https://huggingface.co/datasets", "https://huggingface.co/spaces", "https://huggingface.co/docs", "https://huggingface.co/login", "https://huggingface.co/answerdotai/ModernBERT-base",
"https://www.modelscope.cn/models", "https://www.modelscope.cn/datasets", "https://www.kaggle.com/datasets", "https://www.kaggle.com/competitions",
"https://stackoverflow.com/", "https://stackoverflow.com/questions/tagged/python", "https://superuser.com/", "https://askubuntu.com/",
"https://www.reddit.com/r/MachineLearning/", "https://old.reddit.com/", "https://old.reddit.com/r/python/", "https://lobste.rs/",
"https://www.bbc.com/news", "https://www.bbc.com/sport", "https://www.theguardian.com/international", "https://www.reuters.com/", "https://apnews.com/",
"https://www.imdb.com/chart/top/", "https://www.imdb.com/find/?q=inception", "https://www.rottentomatoes.com/", "https://www.goodreads.com/",
"https://www.openstreetmap.org/search?query=Berlin", "https://www.wikidata.org/", "https://commons.wikimedia.org/", "https://www.wiktionary.org/",
"https://developer.mozilla.org/en-US/docs/Web/JavaScript", "https://developer.mozilla.org/en-US/docs/Web/HTML/Element/select", "https://web.dev/",
"https://www.rust-lang.org/learn", "https://doc.rust-lang.org/book/", "https://go.dev/doc/", "https://pkg.go.dev/", "https://nodejs.org/en/download",
"https://www.npmjs.com/package/react", "https://react.dev/", "https://vuejs.org/", "https://tailwindcss.com/docs", "https://getbootstrap.com/docs/",
"https://www.wolframalpha.com/input?i=2%2B2", "https://translate.google.com/?sl=en&tl=zh-CN&text=hello", "https://www.deepl.com/translator",
"https://weather.com/weather/today/l/USNY0996", "https://www.timeanddate.com/", "https://www.xe.com/currencyconverter/", "https://www.calculator.net/",
"https://archlinux.org/packages/", "https://aur.archlinux.org/", "https://wiki.archlinux.org/title/Installation_guide", "https://www.kernel.org/doc/",
"https://www.debian.org/", "https://ubuntu.com/download", "https://www.gnu.org/software/", "https://www.fsf.org/",
]
def observe(url, timeout=25):
b = Browser(url)
try:
page = b.observe(screenshot=False)
links = b.evaluate("""(() => { const out=[]; for (const a of document.querySelectorAll('a[href]')) {
const h=a.href; if (h.startsWith(location.origin) && !h.includes('#') && h!==location.href) out.push(h); } return out.slice(0,400); })()""") or []
return page, links
finally:
b.close()
def main():
out, max_pages = sys.argv[1], int(sys.argv[2]) if len(sys.argv) > 2 else 80
os.makedirs(os.path.dirname(out) or ".", exist_ok=True)
seen = set()
if os.path.exists(out):
for line in open(out):
seen.add(json.loads(line)["url"])
rng = random.Random(0)
queue = list(SEEDS); rng.shuffle(queue)
n = len(seen)
with open(out, "a") as f:
while queue and n < max_pages:
url = queue.pop(0)
if url in seen:
continue
t = time.time()
try:
page, links = observe(url)
except Exception as e:
print(f"skip {url}: {type(e).__name__}: {str(e)[:80]}", flush=True); continue
elements, targets, controls = action_space(page["actions"])
if len(elements) < 5 or len(elements) > 160:
print(f"skip {url}: {len(elements)} elements", flush=True); continue
seen.add(page["url"]); n += 1
f.write(json.dumps({"url": page["url"], "title": page["title"], "text": page["text"], "actions": page["actions"],
"scroll": page.get("scroll")}, ensure_ascii=False) + "\n"); f.flush()
print(f"[{n}] {len(elements):3d} elements {time.time()-t:4.1f}s {page['title'][:60]}", flush=True)
rng.shuffle(links)
queue.extend(l for l in links[:3] if l not in seen)
print("done", n, "pages")
if __name__ == "__main__":
main()