import json import gradio as gr import pandas as pd try: import spaces except ImportError: spaces = None from benchmark.dataset import ( BENCHMARK_NAME, BENCHMARK_VERSION, display_path, load_cases, results_dir, sha256_file, version_paths, ) CATEGORY_LABELS = { "negation": "Negation", "correction": "Correction / Latest Intent", "implicit_intent": "Implicit Intent", "temporal_reasoning": "Temporal Reasoning", "distractor": "Distractor Resistance", "urgency": "Urgency", "coreference_scope": "Coreference / Scope", "noisy_turkish": "Noisy Turkish", "conditional": "Conditional", "reported_speech": "Reported Speech", "numeric_date": "Numbers / Dates", } DOMAIN_LABELS = { "subscription": "Subscription", "ecommerce": "E-commerce", "banking": "Banking", "telecom": "Telecom", "shipping": "Shipping", "health": "Health (admin)", "public_services": "Public Services", "travel": "Travel", } DIFFICULTY_ORDER = ["easy", "medium", "hard"] LEADERBOARD_COLUMNS = [ "Rank", "Model", "Accuracy %", "Correct", "Cases", "Group consistency %", "High-conf error %", "Mean gold p", ] ERROR_COLUMNS = [ "ID", "Domain", "Task", "Category", "Difficulty", "State", "Expected", "Predicted", "Predicted p", "Gold p", ] def version_info(version): public, tasks = version_paths(version, "public") cases = load_cases(public) return { "version": version, "public": public, "dataset_sha256": sha256_file(public) if public.exists() else None, "tasks_sha256": sha256_file(tasks), "scored": sum(1 for c in cases if c.get("scored", True)), "total": len(cases), "domains": len({c["domain"] for c in cases if c.get("domain")}), "results": results_dir(version), } def run_label(provider, model): return f"{model} ({provider})" def rejection_reason(data, info): if data.get("benchmark") != BENCHMARK_NAME: return "not a TurkishDecisionBenchmark result" if data.get("benchmark_version") != info["version"]: return f"benchmark version {data.get('benchmark_version')!r}" if data.get("split", "public") != "public": return f"split {data.get('split')!r}" if data.get("dataset_sha256") != info["dataset_sha256"]: return "dataset SHA-256 mismatch" if data.get("tasks_sha256") != info["tasks_sha256"]: return "tasks SHA-256 mismatch or missing" filters = data.get("filters", {}) if filters.get("category") or filters.get("difficulty") or filters.get("limit"): return "filtered / partial run" if data.get("report", {}).get("scored_total") != info["scored"]: return f"expected {info['scored']} scored cases" return None def load_result_files(info): accepted, rejected = [], [] for path in sorted(info["results"].glob("*.json")): try: data = json.loads(path.read_text(encoding="utf-8")) except (OSError, ValueError) as exc: rejected.append({"File": path.name, "Reason": f"unreadable: {exc}"}) continue reason = rejection_reason(data, info) if reason: rejected.append({"File": path.name, "Reason": reason}) continue provider = data.get("provider", "unknown") model = data.get("model", "unknown") if provider == "baseline": continue accepted.append({ "file": path.name, "label": run_label(provider, model), "provider": provider, "model": model, "interface": (data.get("adapter") or {}).get("protocol") or "", "created_at": data.get("created_at_utc", ""), "report": data.get("report", {}), "results": data.get("results", []), }) labels = [r["label"] for r in accepted] for r in accepted: if labels.count(r["label"]) > 1: r["label"] = f"{r['label']} [{r['file']}]" return accepted, pd.DataFrame(rejected, columns=["File", "Reason"]) def rank_key(item): r = item["report"] gold = r.get("mean_expected_probability") return ( -r.get("accuracy", 0), -(r.get("group_consistency") or 0), r.get("high_confidence_error_rate", 0), gold is None, -(gold or 0), ) def pct(value): return round(value * 100, 2) if isinstance(value, (int, float)) else None def make_leaderboard(records): rows = [] for rank, item in enumerate(records, 1): r = item["report"] rows.append({ "Rank": rank, "Model": item["model"], "Accuracy %": pct(r.get("accuracy", 0)), "Correct": r.get("correct", 0), "Cases": r.get("scored_total", 0), "Group consistency %": pct(r.get("group_consistency")), "High-conf error %": pct(r.get("high_confidence_error_rate", 0)), "Mean gold p": ( round(r["mean_expected_probability"], 4) if r.get("mean_expected_probability") is not None else None ), }) board = pd.DataFrame(rows, columns=LEADERBOARD_COLUMNS) if board["Group consistency %"].isna().all(): board = board.drop(columns="Group consistency %") return board def make_breakdown(records, key, column, labels=None): rows = [] for item in records: for name, stats in item["report"].get(key, {}).items(): rows.append({ "Model": item["label"], column: (labels or {}).get(name, name), "Accuracy": round(stats.get("accuracy", 0) * 100, 2), "Correct": stats.get("correct", 0), "Cases": stats.get("total", 0), }) return pd.DataFrame(rows, columns=["Model", column, "Accuracy", "Correct", "Cases"]) def pivot(df, column, order=None): if df.empty: return pd.DataFrame(columns=[column]) table = df.pivot_table(index=column, columns="Model", values="Accuracy", aggfunc="first") if order: table = table.reindex([o for o in order if o in table.index]) table.columns.name = None return table.reset_index() def for_model(df, label): return df[df["Model"] == label] if label else df.iloc[0:0] def make_error_table(records, label): item = next((r for r in records if r["label"] == label), None) if item is None: return pd.DataFrame(columns=ERROR_COLUMNS) def p(value): return round(value, 4) if isinstance(value, (int, float)) else None rows = [] for r in item["results"]: if not r.get("scored", True) or r.get("correct"): continue failed = bool(r.get("error")) rows.append({ "ID": r.get("id"), "Domain": DOMAIN_LABELS.get(r.get("domain"), r.get("domain")), "Task": r.get("task_id"), "Category": r.get("category"), "Difficulty": r.get("difficulty"), "State": r.get("state"), "Expected": r.get("expected"), "Predicted": f"ERROR: {r['error']}" if failed else r.get("predicted"), "Predicted p": p(r.get("predicted_probability")), "Gold p": p(r.get("expected_probability")), }) table = pd.DataFrame(rows, columns=ERROR_COLUMNS) return table.dropna(axis=1, how="all") if not table.empty else table def make_summary(state): info, records = state["info"], state["records"] domains = f" · {info['domains']} domains" if info["domains"] else "" lines = [ f"**{BENCHMARK_NAME} v{info['version']}** · {info['scored']} scored cases{domains} · " f"dataset `{(info['dataset_sha256'] or 'missing')[:12]}` · " f"tasks `{info['tasks_sha256'][:12]}`", ] if records: lines.append(f"Models on the leaderboard: **{len(records)}**") else: lines.append( f"No valid results yet. Run `runner.py` " f"and commit the JSON under `{display_path(info['results'])}/`." ) if len(state["rejected"]): lines.append( f"> ⚠️ {len(state['rejected'])} result file(s) excluded — see the Methodology tab." ) lines.append("Ambiguous control cases are not part of the headline accuracy.") return "\n\n".join(lines) def make_methodology(info): v = info["version"] return f""" ### Headline metrics - **Accuracy** over all {info['scored']} scored cases. Unanswered cases and invalid labels count as incorrect. - **Group consistency**: share of contrast groups (minimal edits of one message that flip the answer) where every variant is correct. Ranking tie-breaker after accuracy. - Category, domain and difficulty accuracy - **High-confidence error rate**: wrong answers with predicted probability `>= 0.90`, divided by scored cases - **Mean gold p**: mean probability assigned to the gold answer (answered cases only) ### Leaderboard admission A result file is shown only if it matches benchmark version `{v}`, the public split, dataset SHA-256 `{info['dataset_sha256']}`, tasks SHA-256 `{info['tasks_sha256']}`, is an unfiltered run, and covers all {info['scored']} scored cases. ### Interfaces - `decisions`: native typed `choice` question (Liquid / OpenRouter Decisions API) - `local`: local checkpoint (Laya, Julia-1) ### Adding a model ```bash cp .env.example .env # fill in API keys python runner.py --list-presets python runner.py --preset kev-4b --version {v} python scripts/run_all.py -- --version {v} # every preset with an available key ``` Then commit the generated `{display_path(info['results'])}/*.json`. """ def load_state(version): info = version_info(version) records, rejected = load_result_files(info) records.sort(key=rank_key) labels = [r["label"] for r in records] return { "info": info, "records": records, "rejected": rejected, "board": make_leaderboard(records), "cats": make_breakdown(records, "by_category", "Category", CATEGORY_LABELS), "domains": make_breakdown(records, "by_domain", "Domain", DOMAIN_LABELS), "diffs": make_breakdown(records, "by_difficulty", "Difficulty"), "labels": labels, "default": labels[0] if labels else None, } STATE = load_state(BENCHMARK_VERSION) def render(reload=False): global STATE if reload: STATE = load_state(BENCHMARK_VERSION) s = STATE dropdown = gr.Dropdown(choices=s["labels"], value=s["default"]) return ( make_summary(s), s["board"], pivot(s["cats"], "Category"), dropdown, for_model(s["cats"], s["default"]), pivot(s["domains"], "Domain"), dropdown, for_model(s["domains"], s["default"]), pivot(s["diffs"], "Difficulty", DIFFICULTY_ORDER), dropdown, for_model(s["diffs"], s["default"]), dropdown, make_error_table(s["records"], s["default"]), make_methodology(s["info"]), s["rejected"], ) def model_dropdown(state, label="Model"): return gr.Dropdown( choices=state["labels"], value=state["default"], label=label, interactive=True ) def bar_plot(df, x, title): return gr.BarPlot( value=df, x=x, y="Accuracy", title=title, y_lim=[0, 100], sort=None, x_label_angle=-30, ) INITIAL = STATE with gr.Blocks(title=BENCHMARK_NAME) as demo: gr.Markdown( """ # 🇹🇷 TurkishDecisionBenchmark ### Turkish typed-decision model leaderboard Negation • Latest Intent • Implicit Intent • Temporal Reasoning • Conditionals • Reported Speech • Numbers & Dates • Distractors • Coreference • Noisy Turkish """ ) summary = gr.Markdown(make_summary(INITIAL)) with gr.Tab("🏆 Leaderboard"): leaderboard = gr.Dataframe(value=INITIAL["board"], interactive=False, label="Leaderboard") gr.Markdown( "Ranking: higher **Accuracy** → higher **Group consistency** → lower " "**High-conf error %** → higher **Mean gold p**." ) with gr.Tab("🧩 Category Breakdown"): category_table = gr.Dataframe( value=pivot(INITIAL["cats"], "Category"), interactive=False, wrap=True, label="Accuracy % by category", ) category_model = model_dropdown(INITIAL) category_plot = bar_plot( for_model(INITIAL["cats"], INITIAL["default"]), "Category", "Category Accuracy" ) with gr.Tab("🗂️ Domain Breakdown"): domain_table = gr.Dataframe( value=pivot(INITIAL["domains"], "Domain"), interactive=False, wrap=True, label="Accuracy % by domain", ) domain_model = model_dropdown(INITIAL) domain_plot = bar_plot( for_model(INITIAL["domains"], INITIAL["default"]), "Domain", "Domain Accuracy" ) with gr.Tab("🔥 Difficulty"): difficulty_table = gr.Dataframe( value=pivot(INITIAL["diffs"], "Difficulty", DIFFICULTY_ORDER), interactive=False, wrap=True, label="Accuracy % by difficulty", ) difficulty_model = model_dropdown(INITIAL) difficulty_plot = bar_plot( for_model(INITIAL["diffs"], INITIAL["default"]), "Difficulty", "Difficulty Accuracy" ) with gr.Tab("🔎 Error Explorer"): error_model = model_dropdown(INITIAL) error_table = gr.Dataframe( value=make_error_table(INITIAL["records"], INITIAL["default"]), interactive=False, wrap=True, ) with gr.Tab("ℹ️ Methodology"): methodology = gr.Markdown(make_methodology(INITIAL["info"])) rejected_table = gr.Dataframe( value=INITIAL["rejected"], interactive=False, wrap=True, label="Excluded result files", ) refresh_btn = gr.Button("Refresh committed results") category_model.change( lambda m: for_model(STATE["cats"], m), category_model, category_plot ) domain_model.change( lambda m: for_model(STATE["domains"], m), domain_model, domain_plot ) difficulty_model.change( lambda m: for_model(STATE["diffs"], m), difficulty_model, difficulty_plot ) error_model.change( lambda m: make_error_table(STATE["records"], m), error_model, error_table ) outputs = [ summary, leaderboard, category_table, category_model, category_plot, domain_table, domain_model, domain_plot, difficulty_table, difficulty_model, difficulty_plot, error_model, error_table, methodology, rejected_table, ] refresh_btn.click(lambda: render(reload=True), None, outputs) # ZeroGPU Spaces refuse to start unless some function is marked with # @spaces.GPU. This leaderboard never calls it; the marker only satisfies # that check. The hardware was inherited from the Space template and cannot # be switched to CPU without a PRO plan. if spaces is not None: marker_out = gr.Textbox(visible=False) @spaces.GPU def _zero_gpu_marker(): return "" gr.Button(visible=False).click(_zero_gpu_marker, outputs=marker_out) if __name__ == "__main__": demo.launch()