File size: 3,162 Bytes
2bc6021 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 | """Exercise real MTP serving, streaming, prompt logprobs, tools and scoring."""
import argparse
import json
import os
import subprocess
from pathlib import Path
from common import ROOT, RUN, records, write_json
from serve import server
from evaluate import execute, stream, post
def main():
ap = argparse.ArgumentParser()
ap.add_argument("model", type=Path)
ap.add_argument("--tag", default="smoke")
ap.add_argument("--batch-int8", action="store_true")
ap.add_argument("--check-harness", action="store_true")
args = ap.parse_args()
api = "http://127.0.0.1:18021/v1"
tasks = records(RUN / "evaluation/tasks.jsonl")
results = []
with server(args.model.resolve(), args.tag, batch=args.batch_int8, int8=args.batch_int8):
warm = stream(api, {"model": "qwen3.8-27b", "messages": [{"role":"user","content":"Reply with the word ready."}],
"max_tokens": 32, "temperature": 0, "chat_template_kwargs": {"enable_thinking": False}})
assert warm["usage"]["completion_tokens"] > 0
with post(api, "/completions", {"model":"qwen3.8-27b", "prompt":"A simple test of language model probabilities.",
"max_tokens":1,"temperature":0,"prompt_logprobs":0}) as response:
scored = json.load(response)["choices"][0]["prompt_logprobs"]
assert scored and all(len(entry)==1 for entry in scored[1:])
for suite in ["gsm8k", "tools", "ifbench", "livecodebench"]:
task = next(t for t in tasks if t["suite"] == suite)
result = execute(task, api)
assert result["output_tokens"] > 0, result
if result["error"] and result["error"] != "incorrect_tool_call":
raise RuntimeError(result["error"])
if result.get("code_score", {}).get("reason") == "sandbox_error":
raise RuntimeError(result["code_score"])
if suite == "ifbench" and result["correct"] is None:
scored = subprocess.run([str(RUN/"eval-venv/bin/python"),str(ROOT/"swift15/ifbench_score.py")],
input=json.dumps([{"task":task,"response":result["response"]}]), text=True,capture_output=True,check=True,
env=dict(os.environ,NLTK_DATA=str(RUN/"nltk_data")))
result.update(json.loads(scored.stdout)[0])
results.append(result)
print(suite, "correct:",result["correct"],"tokens:",result["output_tokens"],flush=True)
if args.check_harness:
subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/evaluate.py"),args.tag+"-harness",
"--pilot","--suites","tools","--concurrency","2"],check=True,cwd=ROOT)
subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/measure.py"),args.tag,
"--kind","speed","--concurrency","8" if args.batch_int8 else "1"],check=True,cwd=ROOT)
write_json(RUN/"results"/args.tag/"smoke.json",{"warmup":warm,"tasks":results,
"note":"Integration checks only; this small sample is not a quality benchmark."})
if __name__ == "__main__":
main()
|