daavidhauser's picture
Publish Swift HyperQwen collection with performance and quality comparisons
2bc6021 verified
Raw History Blame Contribute Delete
3.16 kB
"""Exercise real MTP serving, streaming, prompt logprobs, tools and scoring."""
import argparse
import json
import os
import subprocess
from pathlib import Path
from common import ROOT, RUN, records, write_json
from serve import server
from evaluate import execute, stream, post
def main():
ap = argparse.ArgumentParser()
ap.add_argument("model", type=Path)
ap.add_argument("--tag", default="smoke")
ap.add_argument("--batch-int8", action="store_true")
ap.add_argument("--check-harness", action="store_true")
args = ap.parse_args()
api = "http://127.0.0.1:18021/v1"
tasks = records(RUN / "evaluation/tasks.jsonl")
results = []
with server(args.model.resolve(), args.tag, batch=args.batch_int8, int8=args.batch_int8):
warm = stream(api, {"model": "qwen3.8-27b", "messages": [{"role":"user","content":"Reply with the word ready."}],
"max_tokens": 32, "temperature": 0, "chat_template_kwargs": {"enable_thinking": False}})
assert warm["usage"]["completion_tokens"] > 0
with post(api, "/completions", {"model":"qwen3.8-27b", "prompt":"A simple test of language model probabilities.",
"max_tokens":1,"temperature":0,"prompt_logprobs":0}) as response:
scored = json.load(response)["choices"][0]["prompt_logprobs"]
assert scored and all(len(entry)==1 for entry in scored[1:])
for suite in ["gsm8k", "tools", "ifbench", "livecodebench"]:
task = next(t for t in tasks if t["suite"] == suite)
result = execute(task, api)
assert result["output_tokens"] > 0, result
if result["error"] and result["error"] != "incorrect_tool_call":
raise RuntimeError(result["error"])
if result.get("code_score", {}).get("reason") == "sandbox_error":
raise RuntimeError(result["code_score"])
if suite == "ifbench" and result["correct"] is None:
scored = subprocess.run([str(RUN/"eval-venv/bin/python"),str(ROOT/"swift15/ifbench_score.py")],
input=json.dumps([{"task":task,"response":result["response"]}]), text=True,capture_output=True,check=True,
env=dict(os.environ,NLTK_DATA=str(RUN/"nltk_data")))
result.update(json.loads(scored.stdout)[0])
results.append(result)
print(suite, "correct:",result["correct"],"tokens:",result["output_tokens"],flush=True)
if args.check_harness:
subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/evaluate.py"),args.tag+"-harness",
"--pilot","--suites","tools","--concurrency","2"],check=True,cwd=ROOT)
subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/measure.py"),args.tag,
"--kind","speed","--concurrency","8" if args.batch_int8 else "1"],check=True,cwd=ROOT)
write_json(RUN/"results"/args.tag/"smoke.json",{"warmup":warm,"tasks":results,
"note":"Integration checks only; this small sample is not a quality benchmark."})
if __name__ == "__main__":
main()