"""Freeze test data independently of calibration, including a reproducible LCB subset.""" import base64 import collections import io import json import pickle import random import zlib from pathlib import Path import pyarrow.parquet as pq from huggingface_hub import hf_hub_download from common import ROOT, RUN, write_json, write_records, sha256 class DataOnlyUnpickler(pickle.Unpickler): def find_class(self, module, name): raise ValueError("Executable object in benchmark test data") def private_tests(value): try: return json.loads(value) except (ValueError, TypeError): decoded = DataOnlyUnpickler(io.BytesIO(zlib.decompress(base64.b64decode(value)))).load() return json.loads(decoded) if isinstance(decoded, str) else decoded def download(repo, revision, filename): return Path(hf_hub_download(repo, filename, repo_type="dataset", revision=revision, local_dir=RUN / "evaluation/downloads" / repo.replace("/", "__"))) def tool_tasks(): rows = [] cities = ["Graz", "Linz", "Salzburg", "Innsbruck", "Klagenfurt", "Villach", "Wels", "Steyr", "Bregenz", "Eisenstadt"] tool = {"type": "function", "function": {"name": "get_weather", "description": "Get current weather for a city.", "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"], "additionalProperties": False}}} for i in range(30): if i < 20: city = cities[i % len(cities)] rows.append({"id": f"tool-{i}", "suite": "tools", "think": False, "max_tokens": 2048, "messages": [{"role": "user", "content": f"Use get_weather to check {city}. Then return only a JSON object with keys city and fahrenheit. Convert the tool's Celsius temperature to Fahrenheit."}], "tools": [tool], "expected_call": {"name": "get_weather", "arguments": {"city": city}}, "tool_result": {"city": city, "celsius": i - 7}, "expected": {"city": city, "fahrenheit": (i-7)*1.8+32}}) else: codes = [f"SKU-{i}-{j}" for j in range(4)] records = [{"sku": code, "stock": (i*j+3)%11} for j,code in enumerate(codes)] expected = [r["sku"] for r in records if r["stock"] >= 5] rows.append({"id": f"json-{i}", "suite": "tools", "think": False, "max_tokens": 2048, "messages": [{"role": "user", "content": f'Return only JSON with one key "available" containing the SKUs with stock >= 5, preserving input order. Records: {json.dumps(records)}'}], "expected": {"available": expected}}) return rows def main(): out = RUN / "evaluation" out.mkdir(parents=True, exist_ok=True) rng = random.Random(15027) tasks, sources = [], [] gsm = download("openai/gsm8k", "740312add88f781978c0658806c59bc2815b9866", "main/test-00000-of-00001.parquet") for i, r in enumerate(pq.read_table(gsm).to_pylist()[:200]): tasks.append({"id": f"gsm8k-{i}", "suite": "gsm8k", "think": False, "max_tokens": 768, "messages": [{"role": "user", "content": r["question"] + "\n\nSolve step by step, then give the final answer as 'Final answer: '."}], "expected": r["answer"].split("####")[-1].strip()}) sources.append({"dataset": "openai/gsm8k", "split": "main/test", "file_sha256": sha256(gsm), "selection": "first 200, same as HyperQwen"}) iff = download("allenai/IFBench_test", "2e8a48de45ff3bf41242f927254ca81b59ca3ae2", "data/train-00000-of-00001.parquet") for r in pq.read_table(iff).to_pylist(): tasks.append({"id": f"ifbench-{r['key']}", "suite": "ifbench", "think": True, "max_tokens": 4096, "messages": [{"role": "user", "content": r["prompt"]}], "instruction_id_list": r["instruction_id_list"], "kwargs": r["kwargs"]}) sources.append({"dataset": "allenai/IFBench_test", "file_sha256": sha256(iff), "note": "Upstream names this split train, but it is the benchmark test set; never used for calibration."}) pool = {} for filename in ["test.jsonl"] + [f"test{i}.jsonl" for i in range(2, 7)]: p = download("livecodebench/code_generation_lite", "0fe84c3912ea0c4d4a78037083943e8f0c4dd505", filename) sources.append({"dataset": "livecodebench/code_generation_lite", "file": filename, "sha256": sha256(p)}) for line in p.read_text().splitlines(): r = json.loads(line) # v6 timeframe; stdin programs only, to keep execution protocol explicit. if r["contest_date"][:10] > "2025-04-30" or r.get("starter_code"): continue public = json.loads(r["public_test_cases"]) private = private_tests(r["private_test_cases"]) tests = public + private if not tests or any(t.get("testtype") != "stdin" for t in tests): continue pool[(r["platform"], r["question_id"])] = (r, tests) grouped = collections.defaultdict(list) for value in pool.values(): grouped[value[0]["difficulty"]].append(value) for level, n in [("easy", 34), ("medium", 33), ("hard", 33)]: group = sorted(grouped[level], key=lambda x: (x[0]["platform"], x[0]["question_id"])) rng.shuffle(group) assert len(group) >= n, (level, len(group)) for r, tests in group[:n]: tasks.append({"id": f"lcb-{r['platform']}-{r['question_id']}", "suite": "livecodebench", "think": True, "difficulty": level, "max_tokens": 4096, "messages": [{"role": "user", "content": r["question_content"] + "\n\nWrite a complete Python 3 program that reads from standard input and writes to standard output. Put the final solution in a single ```python``` code block."}], "tests": tests}) tasks.extend(tool_tasks()) write_records(out / "tasks.jsonl", tasks) pilot = [] for suite in ["gsm8k", "ifbench", "livecodebench", "tools"]: candidates = [t for t in tasks if t["suite"] == suite] rng.shuffle(candidates) pilot.extend(t["id"] for t in candidates[:5]) write_json(out / "pilot-ids.json", pilot) # Freeze the old battery's input bytes, especially installed vLLM source text. texts = [] p = ROOT / "bench/quality-data/wikitext/wikitext-2-raw-v1/test-00000-of-00001.parquet" text = "".join(pq.read_table(p).column("text").to_pylist()) for i in range(0, min(len(text), 40*1200), 1200): texts.append({"language": "en", "text": text[i:i+1200]}) p = ROOT / "bench/quality-data/fineweb2/data/dan_Latn/test/000_00000.parquet" danish = pq.read_table(p, columns=["text"]).column("text").to_pylist() random.Random(0).shuffle(danish) texts.extend({"language": "da", "text": t[:1200]} for t in [t for t in danish if len(t)>1500][:40]) for p in sorted((ROOT / "venv/lib/python3.12/site-packages/vllm/v1/core").glob("*.py")): text = p.read_text() texts.extend({"language": "code", "text": text[i:i+1200]} for i in range(0,min(len(text),4800),1200) if len(text[i:i+1200])>800) if sum(t["language"] == "code" for t in texts) >= 40: break write_records(out / "perplexity.jsonl", texts) write_json(out / "manifest.json", {"seed": 15027, "sources": sources, "tasks": dict(collections.Counter(t["suite"] for t in tasks)), "task_file_sha256": sha256(out / "tasks.jsonl"), "ppl_file_sha256": sha256(out / "perplexity.jsonl"), "protocol": "Single-seed bounded-budget local comparison, not official full-release leaderboard scores.", "lcb_scoring": "100 stratified stdin-only v6-era tasks; all supplied public/private tests; whitespace token comparison with numeric tolerance; not the full official LCB runner."}) print("Frozen", len(tasks), "tasks and", len(texts), "perplexity windows", flush=True) if __name__ == "__main__": main()