File size: 5,025 Bytes
adf912b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
"""Reverse-generate browser goals with a local LLM: pick an element as gold, ask Qwen to write the user goal for it.

    python finetune/gen_goals.py out/pages.jsonl out/cases.jsonl [per_page=12]

Each case: {url, title, goal, gold_op, gold_id, gold_node, kind, label}.  Also one DONE case per page.
"""
import json, os, random, sys, threading
from concurrent.futures import ThreadPoolExecutor
import httpx
sys.path.insert(0, "/home/ckl/projects/S/jev-ultrafast")
from jev_ultrafast.model import action_space

LLM = os.environ.get("TEXT_MODEL_BASE_URL", "http://127.0.0.1:30000/v1") + "/chat/completions"
MODEL = os.environ.get("TEXT_MODEL", "Qwen/Qwen3-8B-AWQ")
client = httpx.Client(timeout=120)

SYS = """You write realistic browser-automation goals. Given a web page and ONE target control on it, write the goal a user
would give to an assistant such that the assistant's NEXT step is to use exactly that control. Rules:
- One or two sentences, natural language, from the user's perspective, mention what they want (not the UI mechanics).
- The goal must single out the target among the other listed controls; do not mention element numbers.
- For a text field, the goal must imply typing a concrete value into it (include the value).
- For a dropdown option, the goal must imply choosing that option.
- Vary phrasing: sometimes terse ("open the login page"), sometimes contextual ("I want to read about X, take me there").
Return JSON: {"goal": "..."}"""

def ask(page, target, others):
    user = {"page": {"title": page["title"], "url": page["url"], "text_excerpt": page["text"][:700]},
            "target": {"kind": target["kind"], "label": target["label"], "role": target.get("role"),
                       "value": target.get("value", target.get("current_value", ""))},
            "other_controls_on_page": [o["label"][:60] for o in others]}
    body = {"model": MODEL, "max_tokens": 200, "temperature": 0.9, "response_format": {"type": "json_object"},
            "chat_template_kwargs": {"enable_thinking": False},
            "messages": [{"role": "system", "content": SYS}, {"role": "user", "content": json.dumps(user, ensure_ascii=False)}]}
    r = client.post(LLM, json=body).json()
    goal = json.loads(r["choices"][0]["message"]["content"])["goal"]
    return goal.strip()

def main():
    src, out, per_page = sys.argv[1], sys.argv[2], int(sys.argv[3]) if len(sys.argv) > 3 else 12
    pages = [json.loads(l) for l in open(src)]
    rng = random.Random(1)
    jobs = []
    for pi, page in enumerate(pages):
        elements, targets, controls = action_space(page["actions"])
        cands = [a for a in page["actions"] if a["kind"] in ("click", "fill", "select")]
        # de-duplicate by label, prefer informative labels
        seen, uniq = set(), []
        for a in cands:
            lab = a["label"].split(" → ")[0].strip()
            if len(lab) < 2 or lab.lower() in seen: continue
            seen.add(lab.lower()); uniq.append(a)
        rng.shuffle(uniq)
        fills = [a for a in uniq if a["kind"] == "fill"][:3]
        picks = fills + [a for a in uniq if a["kind"] != "fill"][: max(0, per_page - len(fills))]
        for a in picks:
            others = rng.sample([o for o in uniq if o is not a], min(10, len(uniq) - 1))
            jobs.append((pi, page, a, others))
    print(f"{len(pages)} pages -> {len(jobs)} goal jobs", flush=True)
    lock = threading.Lock(); done = [0]
    def work(job):
        pi, page, a, others = job
        try:
            goal = ask(page, a, others)
        except Exception as e:
            print("fail", type(e).__name__, str(e)[:60], flush=True); return None
        with lock:
            done[0] += 1
            if done[0] % 50 == 0: print(f"  {done[0]}/{len(jobs)}", flush=True)
        op = {"click": "CLICK", "fill": "TYPE_TEXT", "select": "SELECT"}[a["kind"]]
        return {"page": pi, "url": page["url"], "title": page["title"], "goal": goal, "gold_op": op, "gold_id": a["id"],
                "gold_node": a.get("node"), "kind": a["kind"], "label": a["label"]}
    with ThreadPoolExecutor(16) as ex:
        cases = [c for c in ex.map(work, jobs) if c]
    for pi, page in enumerate(pages):   # DONE cases: the goal is already satisfied by the current page
        cases.append({"page": pi, "url": page["url"], "title": page["title"], "gold_op": "DONE", "gold_id": "DONE", "kind": "done",
                      "label": "", "goal": rng.choice([f"Open the page titled '{page['title'][:70]}'. Stop once it is open.",
                                                       f"Go to {page['url']} and stop when it has loaded.",
                                                       f"Navigate to the '{page['title'][:50]}' page."])})
    with open(out, "w") as f:
        for c in cases: f.write(json.dumps(c, ensure_ascii=False) + "\n")
    print("wrote", len(cases), "cases ->", out)
    for c in rng.sample(cases, 8): print(f"  [{c['gold_op']:9s}] {c['label'][:35]:35s} <- {c['goal'][:90]}")

if __name__ == "__main__":
    main()