daavidhauser's picture
Publish Swift HyperQwen collection with performance and quality comparisons
2bc6021 verified
Raw History Blame Contribute Delete
2.69 kB
{
"seed": 15027,
"sources": [
{
"dataset": "openai/gsm8k",
"split": "main/test",
"file_sha256": "ee7b8da9e381df27b9e3f7758a159ab2bdaa4dbaa910546cbbc47e0cb44e4f59",
"selection": "first 200, same as HyperQwen"
},
{
"dataset": "allenai/IFBench_test",
"file_sha256": "80037e4d99c39a55c1e2e7d5a863d8d9edeb5ebe136ba8c1e849f8c015027c6a",
"note": "Upstream names this split train, but it is the benchmark test set; never used for calibration."
},
{
"dataset": "livecodebench/code_generation_lite",
"file": "test.jsonl",
"sha256": "2bd02b38beb48e8c46b5b9987095d999ff38cd8efc255ea5d58974317c48f63f"
},
{
"dataset": "livecodebench/code_generation_lite",
"file": "test2.jsonl",
"sha256": "095df7c5daf15f882c51a9deb84085cff1e073495a5dbcf95015a564d485f3a3"
},
{
"dataset": "livecodebench/code_generation_lite",
"file": "test3.jsonl",
"sha256": "28ed26cc83363ce3f1fe2d5fad9f8393077beb1907b167a31bd3b32f80801b79"
},
{
"dataset": "livecodebench/code_generation_lite",
"file": "test4.jsonl",
"sha256": "d711138ddaebfcf5f8ec6a4283ee677298c0f5c5d374a235af92aaf0584510da"
},
{
"dataset": "livecodebench/code_generation_lite",
"file": "test5.jsonl",
"sha256": "7f77571c2a6df0c2a72a3277650309f67e01e0008e18117e624633df53f81214"
},
{
"dataset": "livecodebench/code_generation_lite",
"file": "test6.jsonl",
"sha256": "bb4c364f71921c4495a6ad15abe1a927350b720009f4933e2e71f8af0f6fd1f5"
}
],
"tasks": {
"gsm8k": 200,
"ifbench": 300,
"livecodebench": 100,
"tools": 30
},
"task_file_sha256": "809c2c6b124579d41c379a6a436e9649b5e1d1ddf44a147a3d70d9b4b19d68f4",
"ppl_file_sha256": "57c83ffe7c0dfba2b869f6a379d2df3f0bc0d055069b49569a2154336151a718",
"protocol": "Local single-user launcher and .env, FP8 KV, 150000 context, 128000 output ceiling. Full quality concurrency two; speed and latency pilot concurrency one. Truncated answers count as wrong.",
"lcb_scoring": "100 stratified stdin-only v6-era tasks; all supplied public/private tests; whitespace token comparison with numeric tolerance; not the full official LCB runner.",
"max_output_tokens_per_call": 128000,
"context_len": 150000,
"quality_concurrency": 2,
"supersedes": "<WORKSPACE>/runs/swift15/evaluation/tasks.jsonl",
"truncation_policy": "count_as_wrong",
"sampling": {
"thinking": {
"temperature": 1.0,
"top_p": 0.95,
"top_k": 20,
"reasoning_effort": "xhigh"
},
"nonthinking": "greedy",
"seed": 15027
},
"serving_profile": "local-single-user"
}