"""Crawl real pages with browser-harness and dump their jev-ultrafast observations (element tables + page text). python finetune/collect_pages.py out/pages.jsonl [max_pages=80] Starts from SEEDS, follows a few random same-site links from each page to diversify. One JSON line per page. """ import json, os, random, sys, time sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast") os.environ.setdefault("BU_CDP_URL", "http://127.0.0.1:9222") from jev_ultrafast.browser import Browser, StalePage from jev_ultrafast.model import action_space SEEDS = [ "https://en.wikipedia.org/wiki/Main_Page", "https://en.wikipedia.org/wiki/Special:Random", "https://en.wikipedia.org/wiki/Python_(programming_language)", "https://news.ycombinator.com/", "https://news.ycombinator.com/newest", "https://news.ycombinator.com/login", "https://github.com/", "https://github.com/tile-ai/tilelang", "https://github.com/browser-use/browser-use/issues", "https://github.com/login", "https://www.python.org/", "https://docs.python.org/3/", "https://pypi.org/", "https://pypi.org/project/laya/", "https://archlinux.org/", "https://wiki.archlinux.org/", "https://developer.mozilla.org/en-US/", "https://duckduckgo.com/", "https://www.bing.com/", "https://stackoverflow.com/questions", "https://www.reddit.com/", "https://arxiv.org/", "https://arxiv.org/list/cs.LG/recent", "https://huggingface.co/models", "https://huggingface.co/convaiinnovations/laya", "https://www.saucedemo.com/", "https://the-internet.herokuapp.com/", "https://the-internet.herokuapp.com/login", "https://demo.opencart.com/", "https://www.demoblaze.com/", "https://books.toscrape.com/", "https://quotes.toscrape.com/login", "https://www.google.com/travel/flights?hl=en", "https://www.booking.com/", "https://www.airbnb.com/", "https://www.amazon.com/", "https://www.ebay.com/", "https://www.imdb.com/", "https://www.nytimes.com/", "https://www.bbc.com/", "https://www.openstreetmap.org/", "https://weather.com/", "https://www.wolframalpha.com/", "https://translate.google.com/", "https://www.gnu.org/", "https://kernel.org/", "https://www.rust-lang.org/", "https://go.dev/", "https://nodejs.org/en", "https://www.npmjs.com/", "https://crates.io/", "https://docs.rs/", "https://www.kaggle.com/", "https://paperswithcode.com/", # round 2: more sites, more form-heavy pages "https://en.wikipedia.org/wiki/Special:Search", "https://en.wikipedia.org/wiki/Portal:Current_events", "https://de.wikipedia.org/", "https://zh.wikipedia.org/", "https://news.ycombinator.com/ask", "https://news.ycombinator.com/show", "https://news.ycombinator.com/jobs", "https://news.ycombinator.com/submit", "https://github.com/explore", "https://github.com/trending", "https://github.com/pytorch/pytorch", "https://github.com/pytorch/pytorch/pulls", "https://github.com/pytorch/pytorch/issues", "https://gitlab.com/explore", "https://gitee.com/explore", "https://about.gitlab.com/", "https://www.python.org/downloads/", "https://docs.python.org/3/tutorial/", "https://docs.python.org/3/library/", "https://peps.python.org/", "https://pypi.org/search/?q=torch", "https://pypi.org/project/torch/", "https://pypi.org/account/login/", "https://pypi.org/help/", "https://the-internet.herokuapp.com/dropdown", "https://the-internet.herokuapp.com/checkboxes", "https://the-internet.herokuapp.com/forgot_password", "https://the-internet.herokuapp.com/inputs", "https://the-internet.herokuapp.com/tables", "https://the-internet.herokuapp.com/javascript_alerts", "https://demo.opencart.com/index.php?route=product/category&path=20", "https://demo.opencart.com/index.php?route=account/login", "https://demo.opencart.com/index.php?route=account/register", "https://demo.opencart.com/index.php?route=product/search&search=mac", "https://www.demoblaze.com/cart.html", "https://www.demoblaze.com/prod.html?idp_=1", "https://books.toscrape.com/catalogue/page-2.html", "https://books.toscrape.com/catalogue/category/books/mystery_3/index.html", "https://quotes.toscrape.com/tag/love/", "https://quotes.toscrape.com/page/2/", "https://www.saucedemo.com/inventory.html", "https://automationexercise.com/", "https://automationexercise.com/login", "https://automationexercise.com/products", "https://practicetestautomation.com/practice-test-login/", "https://demoqa.com/", "https://demoqa.com/text-box", "https://demoqa.com/select-menu", "https://demoqa.com/webtables", "https://www.selenium.dev/selenium/web/web-form.html", "https://formy-project.herokuapp.com/", "https://formy-project.herokuapp.com/form", "https://parabank.parasoft.com/parabank/index.htm", "https://parabank.parasoft.com/parabank/register.htm", "https://www.globalsqa.com/angularJs-protractor/BankingProject/", "https://opensource-demo.orangehrmlive.com/", "https://magento.softwaretestingboard.com/", "https://magento.softwaretestingboard.com/women.html", "https://www.airbnb.com/s/London/homes", "https://www.booking.com/searchresults.html?ss=Paris", "https://www.google.com/travel/hotels?hl=en", "https://www.google.com/maps?hl=en", "https://www.google.com/search?q=tilelang&hl=en", "https://duckduckgo.com/?q=modernbert", "https://www.bing.com/search?q=laya", "https://arxiv.org/list/cs.CL/new", "https://arxiv.org/abs/2412.13663", "https://arxiv.org/search/?query=flash+attention&searchtype=all", "https://huggingface.co/datasets", "https://huggingface.co/spaces", "https://huggingface.co/docs", "https://huggingface.co/login", "https://huggingface.co/answerdotai/ModernBERT-base", "https://www.modelscope.cn/models", "https://www.modelscope.cn/datasets", "https://www.kaggle.com/datasets", "https://www.kaggle.com/competitions", "https://stackoverflow.com/", "https://stackoverflow.com/questions/tagged/python", "https://superuser.com/", "https://askubuntu.com/", "https://www.reddit.com/r/MachineLearning/", "https://old.reddit.com/", "https://old.reddit.com/r/python/", "https://lobste.rs/", "https://www.bbc.com/news", "https://www.bbc.com/sport", "https://www.theguardian.com/international", "https://www.reuters.com/", "https://apnews.com/", "https://www.imdb.com/chart/top/", "https://www.imdb.com/find/?q=inception", "https://www.rottentomatoes.com/", "https://www.goodreads.com/", "https://www.openstreetmap.org/search?query=Berlin", "https://www.wikidata.org/", "https://commons.wikimedia.org/", "https://www.wiktionary.org/", "https://developer.mozilla.org/en-US/docs/Web/JavaScript", "https://developer.mozilla.org/en-US/docs/Web/HTML/Element/select", "https://web.dev/", "https://www.rust-lang.org/learn", "https://doc.rust-lang.org/book/", "https://go.dev/doc/", "https://pkg.go.dev/", "https://nodejs.org/en/download", "https://www.npmjs.com/package/react", "https://react.dev/", "https://vuejs.org/", "https://tailwindcss.com/docs", "https://getbootstrap.com/docs/", "https://www.wolframalpha.com/input?i=2%2B2", "https://translate.google.com/?sl=en&tl=zh-CN&text=hello", "https://www.deepl.com/translator", "https://weather.com/weather/today/l/USNY0996", "https://www.timeanddate.com/", "https://www.xe.com/currencyconverter/", "https://www.calculator.net/", "https://archlinux.org/packages/", "https://aur.archlinux.org/", "https://wiki.archlinux.org/title/Installation_guide", "https://www.kernel.org/doc/", "https://www.debian.org/", "https://ubuntu.com/download", "https://www.gnu.org/software/", "https://www.fsf.org/", ] def observe(url, timeout=25): b = Browser(url) try: page = b.observe(screenshot=False) links = b.evaluate("""(() => { const out=[]; for (const a of document.querySelectorAll('a[href]')) { const h=a.href; if (h.startsWith(location.origin) && !h.includes('#') && h!==location.href) out.push(h); } return out.slice(0,400); })()""") or [] return page, links finally: b.close() def main(): out, max_pages = sys.argv[1], int(sys.argv[2]) if len(sys.argv) > 2 else 80 os.makedirs(os.path.dirname(out) or ".", exist_ok=True) seen = set() if os.path.exists(out): for line in open(out): seen.add(json.loads(line)["url"]) rng = random.Random(0) queue = list(SEEDS); rng.shuffle(queue) n = len(seen) with open(out, "a") as f: while queue and n < max_pages: url = queue.pop(0) if url in seen: continue t = time.time() try: page, links = observe(url) except Exception as e: print(f"skip {url}: {type(e).__name__}: {str(e)[:80]}", flush=True); continue elements, targets, controls = action_space(page["actions"]) if len(elements) < 5 or len(elements) > 160: print(f"skip {url}: {len(elements)} elements", flush=True); continue seen.add(page["url"]); n += 1 f.write(json.dumps({"url": page["url"], "title": page["title"], "text": page["text"], "actions": page["actions"], "scroll": page.get("scroll")}, ensure_ascii=False) + "\n"); f.flush() print(f"[{n}] {len(elements):3d} elements {time.time()-t:4.1f}s {page['title'][:60]}", flush=True) rng.shuffle(links) queue.extend(l for l in links[:3] if l not in seen) print("done", n, "pages") if __name__ == "__main__": main()