Download evaluation/code/swift15/evaluate.py from daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen: direct link, hf CLI and curl.
- Browser
- Download file 14.1 kB
-
https://huggingface.co/daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen/resolve/main/evaluation/code/swift15/evaluate.py
- Command line
-
hf download hf://daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen/evaluation/code/swift15/evaluate.py
-
curl -L -o evaluate.py https://huggingface.co/daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen/resolve/main/evaluation/code/swift15/evaluate.py
14.1 kB
| """Stream naturally terminated tasks, preserve raw responses and score failures too.""" | |
| import argparse | |
| import collections | |
| import concurrent.futures | |
| import copy | |
| import json | |
| import math | |
| import os | |
| import re | |
| import statistics | |
| import subprocess | |
| import time | |
| import urllib.request | |
| import uuid | |
| from pathlib import Path | |
| from common import ROOT, RUN, records, read_json, write_json, sha256, stamp | |
| def key(): | |
| return os.environ.get("VLLM_API_KEY") or (ROOT / "api_key.txt").read_text().strip() | |
| def post(api, path, body, timeout=600): | |
| req = urllib.request.Request(api + path, json.dumps(body).encode(), | |
| headers={"Content-Type": "application/json", "Authorization": "Bearer " + key()}) | |
| return urllib.request.urlopen(req, timeout=timeout) | |
| def stream(api, body): | |
| body = dict(body, stream=True, stream_options={"include_usage": True}) | |
| start = time.monotonic() | |
| first, first_answer, last = None, None, None | |
| content, reasoning, calls = [], [], {} | |
| usage, finish = {}, None | |
| with post(api, "/chat/completions", body) as response: | |
| for line in response: | |
| if not line.startswith(b"data: "): | |
| continue | |
| raw = line[6:].strip() | |
| if raw == b"[DONE]": | |
| break | |
| chunk = json.loads(raw) | |
| if "error" in chunk: | |
| raise RuntimeError(str(chunk["error"])) | |
| if chunk.get("usage"): | |
| usage = chunk["usage"] | |
| for choice in chunk.get("choices", []): | |
| delta = choice.get("delta", {}) | |
| now = time.monotonic() | |
| think = delta.get("reasoning_content") or delta.get("reasoning") or "" | |
| text = delta.get("content") or "" | |
| tool = delta.get("tool_calls") or [] | |
| if think or text or tool: | |
| first = first or now | |
| last = now | |
| if text: | |
| first_answer = first_answer or now | |
| content.append(text) | |
| reasoning.append(think) | |
| for call in tool: | |
| c = calls.setdefault(call["index"], {"id": "", "type": "function", "function": {"name": "", "arguments": ""}}) | |
| if call.get("id"): | |
| c["id"] = call["id"] | |
| for k in ["name", "arguments"]: | |
| c["function"][k] += call.get("function", {}).get(k) or "" | |
| finish = choice.get("finish_reason") or finish | |
| end = time.monotonic() | |
| if not usage: | |
| raise RuntimeError("Missing usage counters; refusing to report guessed token counts") | |
| n = usage.get("completion_tokens", 0) | |
| return {"content": "".join(content), "reasoning": "".join(reasoning), | |
| "tool_calls": [calls[i] for i in sorted(calls)], "usage": usage, "finish_reason": finish, | |
| "wall_seconds": end-start, "ttft_seconds": None if first is None else first-start, | |
| "time_to_answer_seconds": None if first_answer is None else first_answer-start, | |
| "decode_seconds": 0 if first is None or last is None else last-first, | |
| "decode_tps": (n-1)/(last-first) if n>1 and last is not None and last>first else None} | |
| def json_equal(a, b): | |
| if isinstance(b, bool): | |
| return isinstance(a, bool) and a == b | |
| if isinstance(b, dict): | |
| return isinstance(a, dict) and set(a) == set(b) and all(json_equal(a[k], v) for k,v in b.items()) | |
| if isinstance(b, list): | |
| return isinstance(a, list) and len(a) == len(b) and all(json_equal(x,y) for x,y in zip(a,b)) | |
| if isinstance(b, (int,float)) and not isinstance(b, bool): | |
| return isinstance(a, (int,float)) and not isinstance(a,bool) and math.isfinite(a) and abs(a-b)<1e-6 | |
| return a == b | |
| def gsm_score(text, expected): | |
| match = re.search(r"Final answer:\s*\**\s*\$?(-?[\d,]*\.?\d+)", text) | |
| numbers = re.findall(r"-?\d[\d,]*\.?\d*", text.replace("$", "")) | |
| pred = match.group(1) if match else (numbers[-1] if numbers else "") | |
| try: | |
| value, gold = float(pred.replace(",", "")), float(expected.replace(",", "")) | |
| return math.isfinite(value) and abs(value-gold)<1e-6 | |
| except ValueError: | |
| return False | |
| def code_score(text, task): | |
| blocks = re.findall(r"```(?:python|py)?\s*\n(.*?)```", text, re.S) | |
| if not blocks: | |
| return {"correct": False, "reason": "no_code_block"} | |
| # This image is pinned locally in the run manifest before evaluation begins. | |
| image = read_json(RUN / "evaluation/runtime.json")["code_image"] | |
| container = "swift15-test-" + uuid.uuid4().hex | |
| cmd = ["docker", "run", "--name", container, "--rm", "-i", "--network", "none", "--read-only", "--memory", "512m", "--cpus", "1", | |
| "--pids-limit", "64", "--cap-drop", "ALL", "--security-opt", "no-new-privileges", "--user", "65534:65534", | |
| "--tmpfs", "/tmp:rw,noexec,nosuid,size=64m", "-e", "SWIFT15_CODE_SANDBOX=1", | |
| "-v", str(ROOT / "swift15/code_runner.py") + ":/runner.py:ro", image, "python", "-I", "/runner.py"] | |
| try: | |
| r = subprocess.run(cmd, input=json.dumps({"code": blocks[-1], "tests": task["tests"]}), | |
| text=True, capture_output=True, timeout=180) | |
| if r.returncode: | |
| return {"correct": False, "reason": "sandbox_error", "detail": r.stderr[-1000:]} | |
| return json.loads(r.stdout) | |
| except subprocess.TimeoutExpired: | |
| return {"correct": False, "reason": "task_test_timeout"} | |
| finally: | |
| subprocess.run(["docker", "rm", "-f", container], capture_output=True, timeout=30) | |
| def execute(task, api): | |
| start = time.monotonic() | |
| result = {"id": task["id"], "suite": task["suite"], "calls": [], "correct": False, "error": None} | |
| body = {"model": "qwen3.8-27b", "messages": copy.deepcopy(task["messages"]), "max_tokens": task["max_tokens"], | |
| "temperature": 0, "seed": 15027, "top_p": 1.0, | |
| "chat_template_kwargs": {"enable_thinking": task["think"], "reasoning_effort": "xhigh"}} | |
| # HyperQwen evaluates thinking tasks at the model's recommended sampling. | |
| # Greedy remains the repository's GSM8K protocol and our deterministic tool test. | |
| if task["think"]: | |
| body.update(temperature=1.0, top_p=.95, top_k=20, min_p=0, | |
| presence_penalty=0, repetition_penalty=1.0) | |
| result["sampling"] = {k:body[k] for k in ["temperature","top_p","seed"]} | |
| if "top_k" in body: | |
| result["sampling"]["top_k"] = body["top_k"] | |
| if task.get("tools"): | |
| body.update(tools=task["tools"], tool_choice="auto") | |
| try: | |
| first = stream(api, body) | |
| result["calls"].append(first) | |
| response = first | |
| if task.get("tools"): | |
| calls = first["tool_calls"] | |
| expected = task["expected_call"] | |
| if len(calls)!=1 or calls[0]["function"]["name"]!=expected["name"] or not json_equal(json.loads(calls[0]["function"]["arguments"]),expected["arguments"]): | |
| result["error"] = "incorrect_tool_call" | |
| else: | |
| body["messages"].append({"role":"assistant","content":first["content"] or None,"tool_calls":calls}) | |
| body["messages"].append({"role":"tool","tool_call_id":calls[0]["id"],"content":json.dumps(task["tool_result"])}) | |
| body["tool_choice"] = "none" | |
| response = stream(api, body) | |
| result["calls"].append(response) | |
| result["model_seconds"] = sum(c["wall_seconds"] for c in result["calls"]) | |
| result["response"] = response["content"] | |
| if result["error"] is None and all(c["finish_reason"] != "length" for c in result["calls"]): | |
| if task["suite"] == "gsm8k": | |
| result["correct"] = gsm_score(response["content"],task["expected"]) | |
| elif task["suite"] == "tools": | |
| try: result["correct"] = json_equal(json.loads(response["content"]),task["expected"]) | |
| except ValueError: pass | |
| elif task["suite"] == "livecodebench": | |
| result["code_score"] = code_score(response["content"],task) | |
| result["correct"] = result["code_score"]["correct"] | |
| else: | |
| result["correct"] = None # official IFBench scoring, batched after generation | |
| except Exception as e: | |
| result["error"] = type(e).__name__ + ": " + str(e)[:500] | |
| result["model_seconds"] = time.monotonic()-start | |
| result["token_counts_incomplete"] = True | |
| result["task_seconds"] = time.monotonic()-start | |
| result.setdefault("model_seconds", sum(c["wall_seconds"] for c in result["calls"])) | |
| result["input_tokens"] = sum(c["usage"].get("prompt_tokens",0) for c in result["calls"]) | |
| result["output_tokens"] = sum(c["usage"].get("completion_tokens",0) for c in result["calls"]) | |
| result["total_tokens"] = result["input_tokens"] + result["output_tokens"] | |
| result["truncated"] = any(c["finish_reason"] == "length" for c in result["calls"]) | |
| if result["truncated"]: | |
| result["correct"] = False # token-limit truncation counts as a wrong answer | |
| return result | |
| def aggregate(rows): | |
| correct = sum(bool(r["correct"]) and not r["truncated"] for r in rows) | |
| truncated = sum(r["truncated"] for r in rows) | |
| return {"attempted":len(rows),"correct":correct,"accuracy":correct/len(rows), | |
| "truncation_policy": "count_as_wrong", | |
| "truncated_counted_as_wrong":truncated, | |
| "errors":sum(r["error"] is not None for r in rows),"truncated":sum(r["truncated"] for r in rows), | |
| "incomplete_token_counts":sum(r.get("token_counts_incomplete",False) for r in rows), | |
| "mean_output_tokens":statistics.mean(r["output_tokens"] for r in rows), | |
| "mean_input_tokens":statistics.mean(r["input_tokens"] for r in rows), | |
| "mean_total_tokens":statistics.mean(r["input_tokens"] + r["output_tokens"] for r in rows), | |
| "mean_model_seconds":statistics.mean(r["model_seconds"] for r in rows), | |
| "median_model_seconds":statistics.median(r["model_seconds"] for r in rows), | |
| "p95_model_seconds":sorted(r["model_seconds"] for r in rows)[math.ceil(.95*len(rows))-1], | |
| "summed_request_seconds_per_correct":sum(r["model_seconds"] for r in rows)/correct if correct else None, | |
| "note":"Summed request seconds are not GPU compute time when concurrency exceeds one."} | |
| def main(): | |
| ap=argparse.ArgumentParser() | |
| ap.add_argument("tag") | |
| ap.add_argument("--api",default="http://127.0.0.1:18021/v1") | |
| ap.add_argument("--pilot",action="store_true") | |
| ap.add_argument("--suites",default="gsm8k,ifbench,livecodebench,tools") | |
| ap.add_argument("--concurrency",type=int,default=1) | |
| args=ap.parse_args() | |
| tasks=records(RUN/"evaluation/tasks.jsonl") | |
| wanted=set(read_json(RUN/"evaluation/pilot-ids.json")) if args.pilot else None | |
| tasks=[t for t in tasks if t["suite"] in args.suites.split(",") and (wanted is None or t["id"] in wanted)] | |
| folder=RUN/"results"/args.tag | |
| folder.mkdir(parents=True,exist_ok=True) | |
| manifest={"tasks_sha256":sha256(RUN/"evaluation/tasks.jsonl"),"task_ids":[t["id"] for t in tasks], | |
| "concurrency":args.concurrency,"sampling":"thinking: temperature1/top_p0.95/top_k20/xhigh; nonthinking: greedy; seed15027", "api":args.api} | |
| identity = read_json(RUN/"active-server.json") | |
| assert not identity.get("stopped"), "No managed benchmark server is active" | |
| manifest["server"] = {k:v for k,v in identity.items() if k not in {"created", "pid"}} | |
| if (folder/"manifest.json").exists(): | |
| assert read_json(folder/"manifest.json")==manifest,"Cannot resume with different settings" | |
| write_json(folder/"manifest.json",manifest) | |
| output=folder/"tasks.jsonl" | |
| old=records(output) if output.exists() else [] | |
| done={r["id"] for r in old} | |
| todo=[t for t in tasks if t["id"] not in done] | |
| start=time.monotonic() | |
| with output.open("a") as f, concurrent.futures.ThreadPoolExecutor(args.concurrency) as pool: | |
| futures=[pool.submit(execute,t,args.api) for t in todo] | |
| for future in concurrent.futures.as_completed(futures): | |
| result=future.result() | |
| f.write(json.dumps(result)+"\n");f.flush() | |
| print(result["id"],"correct=",result["correct"],"tokens=",result["output_tokens"],"seconds=",round(result["model_seconds"],2),"error=",result["error"],flush=True) | |
| elapsed=time.monotonic()-start | |
| rows=records(output) | |
| for row in rows: | |
| if row["truncated"]: | |
| row["correct"] = False | |
| lookup={t["id"]:t for t in tasks} | |
| pending=[r for r in rows if r["suite"]=="ifbench" and r["correct"] is None and not r["truncated"]] | |
| if pending: | |
| env=dict(os.environ,NLTK_DATA=str(RUN/"nltk_data")) | |
| r=subprocess.run([str(RUN/"eval-venv/bin/python"),str(ROOT/"swift15/ifbench_score.py")], | |
| input=json.dumps([{"task":lookup[r["id"]],"response":r["response"]} for r in pending]), | |
| text=True,capture_output=True,check=True,env=env) | |
| scores=json.loads(r.stdout) | |
| for row,score in zip(pending,scores):row.update(score) | |
| from common import write_records | |
| write_records(folder/"scored.jsonl",rows) | |
| groups=collections.defaultdict(list) | |
| for row in rows:groups[row["suite"]].append(row) | |
| summary={"created":stamp(),"concurrency":args.concurrency,"resumed":bool(old), | |
| "new_run_wall_seconds":elapsed,"new_tasks":len(todo),"suites":{k:aggregate(v) for k,v in groups.items()}} | |
| summary["quality_comparison_ready"] = not any(r.get("token_counts_incomplete") for r in rows) | |
| summary["truncation_policy"] = "count_as_wrong" | |
| summary["truncated_task_ids"] = [r["id"] for r in rows if r["truncated"]] | |
| if not old: | |
| correct=sum(bool(r["correct"]) for r in rows) | |
| summary.update(suite_wall_seconds=elapsed,wall_seconds_per_correct=elapsed/correct if correct else None) | |
| write_json(folder/"summary.json",summary) | |
| print(json.dumps(summary,indent=2),flush=True) | |
| if __name__=="__main__":main() | |