{ "seed": 15027, "sources": [ { "dataset": "openai/gsm8k", "split": "main/test", "file_sha256": "ee7b8da9e381df27b9e3f7758a159ab2bdaa4dbaa910546cbbc47e0cb44e4f59", "selection": "first 200, same as HyperQwen" }, { "dataset": "allenai/IFBench_test", "file_sha256": "80037e4d99c39a55c1e2e7d5a863d8d9edeb5ebe136ba8c1e849f8c015027c6a", "note": "Upstream names this split train, but it is the benchmark test set; never used for calibration." }, { "dataset": "livecodebench/code_generation_lite", "file": "test.jsonl", "sha256": "2bd02b38beb48e8c46b5b9987095d999ff38cd8efc255ea5d58974317c48f63f" }, { "dataset": "livecodebench/code_generation_lite", "file": "test2.jsonl", "sha256": "095df7c5daf15f882c51a9deb84085cff1e073495a5dbcf95015a564d485f3a3" }, { "dataset": "livecodebench/code_generation_lite", "file": "test3.jsonl", "sha256": "28ed26cc83363ce3f1fe2d5fad9f8393077beb1907b167a31bd3b32f80801b79" }, { "dataset": "livecodebench/code_generation_lite", "file": "test4.jsonl", "sha256": "d711138ddaebfcf5f8ec6a4283ee677298c0f5c5d374a235af92aaf0584510da" }, { "dataset": "livecodebench/code_generation_lite", "file": "test5.jsonl", "sha256": "7f77571c2a6df0c2a72a3277650309f67e01e0008e18117e624633df53f81214" }, { "dataset": "livecodebench/code_generation_lite", "file": "test6.jsonl", "sha256": "bb4c364f71921c4495a6ad15abe1a927350b720009f4933e2e71f8af0f6fd1f5" } ], "tasks": { "gsm8k": 200, "ifbench": 300, "livecodebench": 100, "tools": 30 }, "task_file_sha256": "809c2c6b124579d41c379a6a436e9649b5e1d1ddf44a147a3d70d9b4b19d68f4", "ppl_file_sha256": "57c83ffe7c0dfba2b869f6a379d2df3f0bc0d055069b49569a2154336151a718", "protocol": "Local single-user launcher and .env, FP8 KV, 150000 context, 128000 output ceiling. Full quality concurrency two; speed and latency pilot concurrency one. Truncated answers count as wrong.", "lcb_scoring": "100 stratified stdin-only v6-era tasks; all supplied public/private tests; whitespace token comparison with numeric tolerance; not the full official LCB runner.", "max_output_tokens_per_call": 128000, "context_len": 150000, "quality_concurrency": 2, "supersedes": "/runs/swift15/evaluation/tasks.jsonl", "truncation_policy": "count_as_wrong", "sampling": { "thinking": { "temperature": 1.0, "top_p": 0.95, "top_k": 20, "reasoning_effort": "xhigh" }, "nonthinking": "greedy", "seed": 15027 }, "serving_profile": "local-single-user" }