muratcanlaloglu
Publish the v0.2 leadboard.
de8702b
Raw History Blame Contribute Delete
3.88 kB
"""Build dataset/v0.2/public.jsonl from the case modules in scripts/v02_cases/.
Each module defines CASES, a list of rows:
(id, domain, task, category, phenomena, difficulty, group, position, state, expected)
Ambiguous rows use a list of valid answers instead of expected; they are unscored.
Edit cases in the modules, not in the JSONL.
Labels from filled sheets in dataset/v0.2/annotations/*.csv (see scripts/annotation.py)
are merged into `annotations`; `?` is stored as null. Model sheets under
annotations/llm/ are not merged and never mark a case as reviewed.
--module builds a single module to --out, so one domain can be validated and gated alone:
python scripts/build_v02.py --module travel --out /tmp/travel.jsonl
python scripts/validate_dataset.py --version 0.2 --dataset /tmp/travel.jsonl
python scripts/baseline_report.py --version 0.2 --dataset /tmp/travel.jsonl
"""
import argparse
import importlib
import json
import sys
from pathlib import Path
HERE = Path(__file__).resolve().parent
ROOT = HERE.parent
sys.path.insert(0, str(HERE))
sys.path.insert(0, str(ROOT))
from annotation import UNSURE, read_sheet # noqa: E402
from benchmark.dataset import display_path, load_tasks, version_paths # noqa: E402
MODULES = ["pilot", "subscription", "telecom", "shipping", "health", "public_services", "travel"]
PUBLIC, TASKS = version_paths("0.2", "public")
SHEETS = PUBLIC.parent / "annotations"
def load_rows(modules):
rows = []
for name in modules:
if not (HERE / "v02_cases" / f"{name}.py").exists():
continue
for cid, domain, task, category, phenomena, difficulty, group, position, state, label in \
importlib.import_module(f"v02_cases.{name}").CASES:
scored = not isinstance(label, list)
rows.append({
"id": cid,
"version": "0.2",
"split": "public",
"domain": domain,
"task_id": task,
"category": category,
"phenomena": phenomena,
"difficulty": difficulty,
"group_id": group,
"answer_position": position,
"state": state,
"expected": label if scored else None,
"valid_answers": [label] if scored else label,
"scored": scored,
"review_status": "draft",
"annotations": {},
})
return rows
def merge_annotations(rows):
tasks = load_tasks(TASKS)
for sheet in sorted(SHEETS.glob("*.csv")):
labels, errors = read_sheet(sheet, rows, tasks, ignore_unknown=True)
if errors:
sys.exit(f"{sheet.name}: " + "; ".join(errors))
for row in rows:
label = labels.get(row["id"], (None,))[0]
if label:
row["annotations"][sheet.stem] = None if label == UNSURE else label
row["review_status"] = "reviewed"
def main():
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--module", choices=MODULES, help="build only this case module")
parser.add_argument("--out", type=Path, default=PUBLIC)
args = parser.parse_args()
if args.module and args.out == PUBLIC:
parser.error("--module needs --out, the public file must contain every module")
rows = load_rows([args.module] if args.module else MODULES)
merge_annotations(rows)
args.out.parent.mkdir(parents=True, exist_ok=True)
with args.out.open("w", encoding="utf-8") as f:
for row in rows:
f.write(json.dumps(row, ensure_ascii=False) + "\n")
reviewed = sum(r["review_status"] == "reviewed" for r in rows)
print(f"Wrote {len(rows)} cases ({reviewed} reviewed) to {display_path(args.out)}")
if __name__ == "__main__":
main()