"""Exercise real MTP serving, streaming, prompt logprobs, tools and scoring.""" import argparse import json import os import subprocess from pathlib import Path from common import ROOT, RUN, records, write_json from serve import server from evaluate import execute, stream, post def main(): ap = argparse.ArgumentParser() ap.add_argument("model", type=Path) ap.add_argument("--tag", default="smoke") ap.add_argument("--batch-int8", action="store_true") ap.add_argument("--check-harness", action="store_true") args = ap.parse_args() api = "http://127.0.0.1:18021/v1" tasks = records(RUN / "evaluation/tasks.jsonl") results = [] with server(args.model.resolve(), args.tag, batch=args.batch_int8, int8=args.batch_int8): warm = stream(api, {"model": "qwen3.8-27b", "messages": [{"role":"user","content":"Reply with the word ready."}], "max_tokens": 32, "temperature": 0, "chat_template_kwargs": {"enable_thinking": False}}) assert warm["usage"]["completion_tokens"] > 0 with post(api, "/completions", {"model":"qwen3.8-27b", "prompt":"A simple test of language model probabilities.", "max_tokens":1,"temperature":0,"prompt_logprobs":0}) as response: scored = json.load(response)["choices"][0]["prompt_logprobs"] assert scored and all(len(entry)==1 for entry in scored[1:]) for suite in ["gsm8k", "tools", "ifbench", "livecodebench"]: task = next(t for t in tasks if t["suite"] == suite) result = execute(task, api) assert result["output_tokens"] > 0, result if result["error"] and result["error"] != "incorrect_tool_call": raise RuntimeError(result["error"]) if result.get("code_score", {}).get("reason") == "sandbox_error": raise RuntimeError(result["code_score"]) if suite == "ifbench" and result["correct"] is None: scored = subprocess.run([str(RUN/"eval-venv/bin/python"),str(ROOT/"swift15/ifbench_score.py")], input=json.dumps([{"task":task,"response":result["response"]}]), text=True,capture_output=True,check=True, env=dict(os.environ,NLTK_DATA=str(RUN/"nltk_data"))) result.update(json.loads(scored.stdout)[0]) results.append(result) print(suite, "correct:",result["correct"],"tokens:",result["output_tokens"],flush=True) if args.check_harness: subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/evaluate.py"),args.tag+"-harness", "--pilot","--suites","tools","--concurrency","2"],check=True,cwd=ROOT) subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/measure.py"),args.tag, "--kind","speed","--concurrency","8" if args.batch_int8 else "1"],check=True,cwd=ROOT) write_json(RUN/"results"/args.tag/"smoke.json",{"warmup":warm,"tasks":results, "note":"Integration checks only; this small sample is not a quality benchmark."}) if __name__ == "__main__": main()