File size: 3,162 Bytes
2bc6021
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
"""Exercise real MTP serving, streaming, prompt logprobs, tools and scoring."""
import argparse
import json
import os
import subprocess
from pathlib import Path
from common import ROOT, RUN, records, write_json
from serve import server
from evaluate import execute, stream, post


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("model", type=Path)
    ap.add_argument("--tag", default="smoke")
    ap.add_argument("--batch-int8", action="store_true")
    ap.add_argument("--check-harness", action="store_true")
    args = ap.parse_args()
    api = "http://127.0.0.1:18021/v1"
    tasks = records(RUN / "evaluation/tasks.jsonl")
    results = []
    with server(args.model.resolve(), args.tag, batch=args.batch_int8, int8=args.batch_int8):
        warm = stream(api, {"model": "qwen3.8-27b", "messages": [{"role":"user","content":"Reply with the word ready."}],
                            "max_tokens": 32, "temperature": 0, "chat_template_kwargs": {"enable_thinking": False}})
        assert warm["usage"]["completion_tokens"] > 0
        with post(api, "/completions", {"model":"qwen3.8-27b", "prompt":"A simple test of language model probabilities.",
                                       "max_tokens":1,"temperature":0,"prompt_logprobs":0}) as response:
            scored = json.load(response)["choices"][0]["prompt_logprobs"]
            assert scored and all(len(entry)==1 for entry in scored[1:])
        for suite in ["gsm8k", "tools", "ifbench", "livecodebench"]:
            task = next(t for t in tasks if t["suite"] == suite)
            result = execute(task, api)
            assert result["output_tokens"] > 0, result
            if result["error"] and result["error"] != "incorrect_tool_call":
                raise RuntimeError(result["error"])
            if result.get("code_score", {}).get("reason") == "sandbox_error":
                raise RuntimeError(result["code_score"])
            if suite == "ifbench" and result["correct"] is None:
                scored = subprocess.run([str(RUN/"eval-venv/bin/python"),str(ROOT/"swift15/ifbench_score.py")],
                    input=json.dumps([{"task":task,"response":result["response"]}]), text=True,capture_output=True,check=True,
                    env=dict(os.environ,NLTK_DATA=str(RUN/"nltk_data")))
                result.update(json.loads(scored.stdout)[0])
            results.append(result)
            print(suite, "correct:",result["correct"],"tokens:",result["output_tokens"],flush=True)
        if args.check_harness:
            subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/evaluate.py"),args.tag+"-harness",
                            "--pilot","--suites","tools","--concurrency","2"],check=True,cwd=ROOT)
            subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/measure.py"),args.tag,
                            "--kind","speed","--concurrency","8" if args.batch_int8 else "1"],check=True,cwd=ROOT)
    write_json(RUN/"results"/args.tag/"smoke.json",{"warmup":warm,"tasks":results,
               "note":"Integration checks only; this small sample is not a quality benchmark."})


if __name__ == "__main__":
    main()