Download evaluation/code/swift15/smoke.py from daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen: direct link, hf CLI and curl.
- Browser
- Download file 3.16 kB
-
https://huggingface.co/daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen/resolve/main/evaluation/code/swift15/smoke.py
- Command line
-
hf download hf://daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen/evaluation/code/swift15/smoke.py
-
curl -L -o smoke.py https://huggingface.co/daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen/resolve/main/evaluation/code/swift15/smoke.py
3.16 kB
| """Exercise real MTP serving, streaming, prompt logprobs, tools and scoring.""" | |
| import argparse | |
| import json | |
| import os | |
| import subprocess | |
| from pathlib import Path | |
| from common import ROOT, RUN, records, write_json | |
| from serve import server | |
| from evaluate import execute, stream, post | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("model", type=Path) | |
| ap.add_argument("--tag", default="smoke") | |
| ap.add_argument("--batch-int8", action="store_true") | |
| ap.add_argument("--check-harness", action="store_true") | |
| args = ap.parse_args() | |
| api = "http://127.0.0.1:18021/v1" | |
| tasks = records(RUN / "evaluation/tasks.jsonl") | |
| results = [] | |
| with server(args.model.resolve(), args.tag, batch=args.batch_int8, int8=args.batch_int8): | |
| warm = stream(api, {"model": "qwen3.8-27b", "messages": [{"role":"user","content":"Reply with the word ready."}], | |
| "max_tokens": 32, "temperature": 0, "chat_template_kwargs": {"enable_thinking": False}}) | |
| assert warm["usage"]["completion_tokens"] > 0 | |
| with post(api, "/completions", {"model":"qwen3.8-27b", "prompt":"A simple test of language model probabilities.", | |
| "max_tokens":1,"temperature":0,"prompt_logprobs":0}) as response: | |
| scored = json.load(response)["choices"][0]["prompt_logprobs"] | |
| assert scored and all(len(entry)==1 for entry in scored[1:]) | |
| for suite in ["gsm8k", "tools", "ifbench", "livecodebench"]: | |
| task = next(t for t in tasks if t["suite"] == suite) | |
| result = execute(task, api) | |
| assert result["output_tokens"] > 0, result | |
| if result["error"] and result["error"] != "incorrect_tool_call": | |
| raise RuntimeError(result["error"]) | |
| if result.get("code_score", {}).get("reason") == "sandbox_error": | |
| raise RuntimeError(result["code_score"]) | |
| if suite == "ifbench" and result["correct"] is None: | |
| scored = subprocess.run([str(RUN/"eval-venv/bin/python"),str(ROOT/"swift15/ifbench_score.py")], | |
| input=json.dumps([{"task":task,"response":result["response"]}]), text=True,capture_output=True,check=True, | |
| env=dict(os.environ,NLTK_DATA=str(RUN/"nltk_data"))) | |
| result.update(json.loads(scored.stdout)[0]) | |
| results.append(result) | |
| print(suite, "correct:",result["correct"],"tokens:",result["output_tokens"],flush=True) | |
| if args.check_harness: | |
| subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/evaluate.py"),args.tag+"-harness", | |
| "--pilot","--suites","tools","--concurrency","2"],check=True,cwd=ROOT) | |
| subprocess.run([str(ROOT/"venv/bin/python"),str(ROOT/"swift15/measure.py"),args.tag, | |
| "--kind","speed","--concurrency","8" if args.batch_int8 else "1"],check=True,cwd=ROOT) | |
| write_json(RUN/"results"/args.tag/"smoke.json",{"warmup":warm,"tasks":results, | |
| "note":"Integration checks only; this small sample is not a quality benchmark."}) | |
| if __name__ == "__main__": | |
| main() | |