blink-4b / eval /decision-index-0.2-local.json
thegovind's picture
Card: local Decision Index 0.2 run (descriptive; known training exposure not penalized) and its evidence file
bdbdcf2 verified
Raw History Blame
29.8 kB
{
"engine": "blink-4b",
"edition": "Decision Index 0.2 (local run, descriptive)",
"generated_utc": "2026-09-25T23:11:52+00:00",
"suite": {
"edition": "release-v2",
"requests": 121057,
"scoreable": 120615,
"excluded": 442,
"added_requests": 30419,
"benchmarks": 44,
"rows_sha256": "b2b56d6fb636837ca469e689087bdbf373dda8de7638aa2da6793e6eda0792d5",
"added_sha256": "7429f3c9cdddb772c1cfc42bb2a45e8516b0032152b746e6929f1c8b52f4ce89"
},
"completed": 151034,
"complete": true,
"counts": {
"ok": 151034
},
"latency_ms": {
"median": 30.6,
"p95": 717.1,
"mean": 144.9
},
"decision_index": 37.85,
"raw_index": 53.33,
"scores": {
"balanced_skill": 37.85,
"balanced_raw": 53.33,
"breadth_skill": 36.78
},
"areas": [
{
"id": "knowledge",
"label": "Knowledge & Reasoning",
"raw": 0.431,
"skill": 0.2642,
"coverage": 1.0,
"n": 10,
"benchmarks": [
25,
30,
31,
32,
33,
43,
44,
45,
57,
58
]
},
{
"id": "language",
"label": "Language Understanding",
"raw": 0.6271,
"skill": 0.4735,
"coverage": 1.0,
"n": 10,
"benchmarks": [
11,
12,
28,
29,
38,
39,
40,
41,
42,
59
]
},
{
"id": "retrieval",
"label": "Retrieval & Classification",
"raw": 0.5468,
"skill": 0.3676,
"coverage": 1.0,
"n": 7,
"benchmarks": [
4,
5,
10,
36,
37,
56,
61
]
},
{
"id": "tools",
"label": "Tools & Automation",
"raw": 0.6001,
"skill": 0.5155,
"coverage": 1.0,
"n": 6,
"benchmarks": [
1,
2,
3,
6,
9,
62
]
},
{
"id": "arts",
"label": "Arts & Human Taste",
"raw": 0.4614,
"skill": 0.2715,
"coverage": 1.0,
"n": 7,
"benchmarks": [
20,
21,
22,
23,
48,
50,
64
]
}
],
"index_benchmarks": {
"1": {
"raw": 0.9032,
"skill": 0.8693,
"coverage": 1.0,
"random": 0.2592,
"rule": "track",
"in_index": true,
"tracks": []
},
"2": {
"raw": 0.4126,
"skill": 0.3532,
"coverage": 1.0,
"random": 0.0918,
"rule": "track",
"in_index": true,
"tracks": []
},
"3": {
"raw": 0.748,
"skill": 0.7431,
"coverage": 1.0,
"random": 0.0189,
"rule": "chance",
"in_index": true
},
"4": {
"raw": 0.7149,
"skill": 0.7112,
"coverage": 1.0,
"random": 0.0127,
"rule": "chance",
"in_index": true
},
"5": {
"raw": 0.7391,
"skill": 0.7375,
"coverage": 1.0,
"random": 0.006,
"rule": "chance",
"in_index": true
},
"6": {
"raw": 0.7905,
"skill": 0.5052,
"coverage": 1.0,
"random": 0.5245,
"rule": "track",
"in_index": true,
"tracks": [
{
"track": "RouterBench-0shot",
"score": 0.7797,
"headline": false
},
{
"track": "RouterBench-5shot",
"score": 0.8014,
"headline": false
}
]
},
"9": {
"raw": 0.1187,
"skill": 0.1187,
"coverage": 1.0,
"random": 0.0,
"rule": "chance",
"in_index": true
},
"10": {
"raw": 0.5004,
"skill": 0.1687,
"coverage": 1.0,
"random": 0.399,
"rule": "chance",
"in_index": true
},
"11": {
"raw": 0.7612,
"skill": 0.6547,
"coverage": 1.0,
"random": 0.3085,
"rule": "track",
"in_index": true,
"tracks": []
},
"12": {
"raw": 0.6012,
"skill": 0.4026,
"coverage": 1.0,
"random": 0.3324,
"rule": "chance",
"in_index": true
},
"20": {
"raw": 0.8467,
"skill": 0.6933,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"21": {
"raw": 0.6054,
"skill": 0.2108,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"22": {
"raw": 0.0338,
"skill": 0.0262,
"coverage": 1.0,
"random": 0.0078,
"rule": "track",
"in_index": true,
"tracks": []
},
"23": {
"raw": 0.5808,
"skill": 0.1616,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"25": {
"raw": 0.3724,
"skill": 0.1633,
"coverage": 1.0,
"random": 0.25,
"rule": "track",
"in_index": true,
"tracks": []
},
"28": {
"raw": 0.7032,
"skill": 0.4064,
"coverage": 1.0,
"random": 0.5,
"rule": "chance",
"in_index": true
},
"29": {
"raw": 0.8194,
"skill": 0.7592,
"coverage": 1.0,
"random": 0.25,
"rule": "chance",
"in_index": true
},
"30": {
"raw": 0.5788,
"skill": 0.4872,
"coverage": 1.0,
"random": 0.25,
"rule": "track",
"in_index": true,
"tracks": [
{
"track": "GSM8K-4choice",
"score": 0.5967,
"headline": false
},
{
"track": "GSM8K-10choice",
"score": 0.561,
"headline": false
}
]
},
"31": {
"raw": 0.1348,
"skill": 0.0577,
"coverage": 1.0,
"random": 0.0819,
"rule": "track",
"in_index": true,
"tracks": []
},
"32": {
"raw": 0.5678,
"skill": 0.3129,
"coverage": 1.0,
"random": 0.371,
"rule": "chance",
"in_index": true
},
"33": {
"raw": 0.2594,
"skill": 0.2496,
"coverage": 1.0,
"random": 0.0131,
"rule": "chance",
"in_index": true
},
"36": {
"raw": 0.1751,
"skill": 0.135,
"coverage": 1.0,
"random": 0.0464,
"rule": "track",
"in_index": true,
"tracks": []
},
"37": {
"raw": 0.4309,
"skill": 0.2862,
"coverage": 1.0,
"random": 0.2027,
"rule": "track",
"in_index": true,
"tracks": []
},
"38": {
"raw": 0.03,
"skill": 0.03,
"coverage": 1.0,
"random": 0.0,
"rule": "chance",
"in_index": true
},
"39": {
"raw": 0.8603,
"skill": 0.7945,
"coverage": 1.0,
"random": 0.3201,
"rule": "chance",
"in_index": true
},
"40": {
"raw": 0.4523,
"skill": 0.2953,
"coverage": 1.0,
"random": 0.2227,
"rule": "track",
"in_index": true,
"tracks": [
{
"track": "A \u00b7 Arabic",
"score": 0.313,
"headline": false
},
{
"track": "A \u00b7 English",
"score": 0.4523,
"headline": true
},
{
"track": "C \u00b7 Arabic pairs",
"score": 0.645,
"headline": false
},
{
"track": "C \u00b7 English pairs",
"score": 0.915,
"headline": false
}
]
},
"41": {
"raw": 0.6169,
"skill": 0.4254,
"coverage": 1.0,
"random": 0.3333,
"rule": "track",
"in_index": true,
"tracks": []
},
"42": {
"raw": 0.7618,
"skill": 0.5366,
"coverage": 1.0,
"random": 0.486,
"rule": "chance",
"in_index": true
},
"43": {
"raw": 0.4719,
"skill": 0.1622,
"coverage": 1.0,
"random": 0.3697,
"rule": "track",
"in_index": true,
"tracks": []
},
"44": {
"raw": 0.6368,
"skill": 0.2736,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"45": {
"raw": 0.1297,
"skill": 0.0,
"coverage": 1.0,
"random": 0.1641,
"rule": "chance",
"in_index": true
},
"48": {
"raw": 0.1312,
"skill": 0.1312,
"coverage": 1.0,
"random": 0.25,
"rule": "vs baseline",
"in_index": true
},
"50": {
"raw": 0.4427,
"skill": 0.1912,
"coverage": 1.0,
"random": 0.311,
"rule": "track",
"in_index": true,
"tracks": []
},
"56": {
"raw": 0.6365,
"skill": 0.273,
"coverage": 1.0,
"random": 0.5,
"rule": "chance",
"in_index": true
},
"57": {
"raw": 0.5206,
"skill": 0.4608,
"coverage": 1.0,
"random": 0.1109,
"rule": "chance",
"in_index": true
},
"58": {
"raw": 0.6379,
"skill": 0.4751,
"coverage": 1.0,
"random": 0.3101,
"rule": "chance",
"in_index": true
},
"59": {
"raw": 0.6648,
"skill": 0.4306,
"coverage": 1.0,
"random": 0.4113,
"rule": "chance",
"in_index": true
},
"61": {
"raw": 0.6308,
"skill": 0.2616,
"coverage": 1.0,
"random": 0.5,
"rule": "chance",
"in_index": true
},
"62": {
"raw": 0.6276,
"skill": 0.5035,
"coverage": 1.0,
"random": 0.25,
"rule": "chance",
"in_index": true
},
"64": {
"raw": 0.589,
"skill": 0.4862,
"coverage": 1.0,
"random": 0.2,
"rule": "chance",
"in_index": true
},
"24": {
"raw": 0.7375,
"skill": 0.65,
"coverage": 1.0,
"random": 0.25,
"rule": "shown, not counted",
"in_index": false
},
"26": {
"raw": 0.9769,
"skill": 0.9692,
"coverage": 1.0,
"random": 0.2502,
"rule": "shown, not counted",
"in_index": false
},
"27": {
"raw": 0.9275,
"skill": 0.9033,
"coverage": 1.0,
"random": 0.2502,
"rule": "shown, not counted",
"in_index": false
},
"34": {
"raw": 0.1,
"skill": 0.0,
"coverage": 1.0,
"random": 0.1667,
"rule": "shown, not counted",
"in_index": false
}
},
"benchmarks": {
"1": {
"catalog_id": 1,
"dataset": "BFCL",
"requests": 1694,
"answered": 1694,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1694,
"metric": "case exact accuracy",
"score": 0.9032,
"reference_same_cases": null,
"median_ms": 94.5,
"index_raw": 0.9032,
"index_skill": 0.8693,
"coverage": 1.0,
"chance": 0.2592,
"in_index": true
},
"2": {
"catalog_id": 2,
"dataset": "ToolRet",
"requests": 1000,
"answered": 1000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1000,
"metric": "nDCG@10",
"score": 0.4126,
"reference_same_cases": null,
"median_ms": 802.3,
"index_raw": 0.4126,
"index_skill": 0.3532,
"coverage": 1.0,
"chance": 0.0918,
"in_index": true
},
"3": {
"catalog_id": 3,
"dataset": "API-Bank",
"requests": 508,
"answered": 508,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 508,
"metric": "accuracy",
"score": 0.748,
"reference_same_cases": null,
"median_ms": 425.6,
"index_raw": 0.748,
"index_skill": 0.7431,
"coverage": 1.0,
"chance": 0.0189,
"in_index": true
},
"4": {
"catalog_id": 4,
"dataset": "BANKING77",
"requests": 3080,
"answered": 3080,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 3080,
"metric": "macro-F1",
"score": 0.7149,
"reference_same_cases": null,
"median_ms": 68.5,
"index_raw": 0.7149,
"index_skill": 0.7112,
"coverage": 1.0,
"chance": 0.0127,
"in_index": true
},
"5": {
"catalog_id": 5,
"dataset": "CLINC150+OOS",
"requests": 5500,
"answered": 5500,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5500,
"metric": "macro-F1",
"score": 0.7391,
"reference_same_cases": null,
"median_ms": 115.0,
"index_raw": 0.7391,
"index_skill": 0.7375,
"coverage": 1.0,
"chance": 0.006,
"in_index": true
},
"6": {
"catalog_id": 6,
"dataset": "RouterBench",
"requests": 10000,
"answered": 10000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10000,
"metric": "selected quality (quality objective)",
"score": 0.7905,
"reference_same_cases": null,
"median_ms": 262.7,
"tracks": [
{
"track": "RouterBench-0shot",
"score": 0.7797,
"headline": false
},
{
"track": "RouterBench-5shot",
"score": 0.8014,
"headline": false
}
],
"index_raw": 0.7905,
"index_skill": 0.5052,
"coverage": 1.0,
"chance": 0.5245,
"in_index": true
},
"9": {
"catalog_id": 9,
"dataset": "Home appliance simulator",
"requests": 160,
"answered": 160,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 160,
"metric": "case exact accuracy",
"score": 0.1187,
"reference_same_cases": null,
"median_ms": 1148.5,
"index_raw": 0.1187,
"index_skill": 0.1187,
"coverage": 1.0,
"chance": 0.0,
"in_index": true
},
"10": {
"catalog_id": 10,
"dataset": "SGD/SGD-X",
"requests": 2500,
"answered": 2500,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2500,
"metric": "macro-F1",
"score": 0.5004,
"reference_same_cases": null,
"median_ms": 71.5,
"index_raw": 0.5004,
"index_skill": 0.1687,
"coverage": 1.0,
"chance": 0.399,
"in_index": true
},
"11": {
"catalog_id": 11,
"dataset": "ContractNLI",
"requests": 123,
"answered": 123,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 123,
"metric": "macro-F1",
"score": 0.7612,
"reference_same_cases": null,
"median_ms": 1993.0,
"index_raw": 0.7612,
"index_skill": 0.6547,
"coverage": 1.0,
"chance": 0.3085,
"in_index": true
},
"12": {
"catalog_id": 12,
"dataset": "ANLI",
"requests": 3200,
"answered": 3200,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 3200,
"metric": "macro-F1",
"score": 0.6012,
"reference_same_cases": null,
"median_ms": 13.0,
"index_raw": 0.6012,
"index_skill": 0.4026,
"coverage": 1.0,
"chance": 0.3324,
"in_index": true
},
"20": {
"catalog_id": 20,
"dataset": "BPoMP",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.8456,
"reference_same_cases": null,
"median_ms": 12.5,
"index_raw": 0.8467,
"index_skill": 0.6933,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"21": {
"catalog_id": 21,
"dataset": "Humicroedit",
"requests": 2628,
"answered": 2628,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2628,
"metric": "accuracy",
"score": 0.6054,
"reference_same_cases": null,
"median_ms": 8.0,
"index_raw": 0.6054,
"index_skill": 0.2108,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"22": {
"catalog_id": 22,
"dataset": "POP909-CL",
"requests": 2000,
"answered": 2000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2000,
"metric": "accuracy",
"score": 0.036,
"reference_same_cases": null,
"median_ms": 571.6,
"index_raw": 0.0338,
"index_skill": 0.0262,
"coverage": 1.0,
"chance": 0.0078,
"in_index": true
},
"23": {
"catalog_id": 23,
"dataset": "cfcolor",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.6016,
"reference_same_cases": null,
"median_ms": 33.4,
"index_raw": 0.5808,
"index_skill": 0.1616,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"24": {
"catalog_id": 24,
"dataset": "MMLU",
"requests": 14033,
"answered": 14033,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 14033,
"metric": "accuracy",
"score": 0.7375,
"reference_same_cases": null,
"median_ms": 13.0,
"index_raw": 0.7375,
"index_skill": 0.65,
"coverage": 1.0,
"chance": 0.25,
"in_index": false
},
"25": {
"catalog_id": 25,
"dataset": "GPQA Diamond",
"requests": 196,
"answered": 196,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 196,
"metric": "accuracy",
"score": 0.3724,
"reference_same_cases": null,
"median_ms": 23.5,
"index_raw": 0.3724,
"index_skill": 0.1633,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"26": {
"catalog_id": 26,
"dataset": "ARC-Easy",
"requests": 2376,
"answered": 2376,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2376,
"metric": "accuracy",
"score": 0.9769,
"reference_same_cases": null,
"median_ms": 12.3,
"index_raw": 0.9769,
"index_skill": 0.9692,
"coverage": 1.0,
"chance": 0.2502,
"in_index": false
},
"27": {
"catalog_id": 27,
"dataset": "ARC-Challenge",
"requests": 1172,
"answered": 1172,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1172,
"metric": "accuracy",
"score": 0.9275,
"reference_same_cases": null,
"median_ms": 12.1,
"index_raw": 0.9275,
"index_skill": 0.9033,
"coverage": 1.0,
"chance": 0.2502,
"in_index": false
},
"28": {
"catalog_id": 28,
"dataset": "WinoGrande",
"requests": 1267,
"answered": 1267,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1267,
"metric": "accuracy",
"score": 0.7032,
"reference_same_cases": null,
"median_ms": 8.4,
"index_raw": 0.7032,
"index_skill": 0.4064,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"29": {
"catalog_id": 29,
"dataset": "HellaSwag",
"requests": 10042,
"answered": 10042,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10042,
"metric": "accuracy",
"score": 0.8194,
"reference_same_cases": null,
"median_ms": 18.6,
"index_raw": 0.8194,
"index_skill": 0.7592,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"30": {
"catalog_id": 30,
"dataset": "GSM8K",
"requests": 2638,
"answered": 2638,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2638,
"metric": "accuracy",
"score": 0.5788,
"reference_same_cases": null,
"median_ms": 16.4,
"tracks": [
{
"track": "GSM8K-4choice",
"score": 0.5967,
"headline": false
},
{
"track": "GSM8K-10choice",
"score": 0.561,
"headline": false
}
],
"index_raw": 0.5788,
"index_skill": 0.4872,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"31": {
"catalog_id": 31,
"dataset": "ChessBench",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.1348,
"reference_same_cases": null,
"median_ms": 88.3,
"index_raw": 0.1348,
"index_skill": 0.0577,
"coverage": 1.0,
"chance": 0.0819,
"in_index": true
},
"32": {
"catalog_id": 32,
"dataset": "MuSR",
"requests": 752,
"answered": 752,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 752,
"metric": "accuracy",
"score": 0.5678,
"reference_same_cases": null,
"median_ms": 61.7,
"index_raw": 0.5678,
"index_skill": 0.3129,
"coverage": 1.0,
"chance": 0.371,
"in_index": true
},
"33": {
"catalog_id": 33,
"dataset": "SATA-Bench",
"requests": 1650,
"answered": 1650,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1650,
"metric": "case exact accuracy",
"score": 0.2594,
"reference_same_cases": null,
"median_ms": 213.7,
"index_raw": 0.2594,
"index_skill": 0.2496,
"coverage": 1.0,
"chance": 0.0131,
"in_index": true
},
"34": {
"catalog_id": 34,
"dataset": "SimpleBench",
"requests": 10,
"answered": 10,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10,
"metric": "accuracy",
"score": 0.1,
"reference_same_cases": null,
"median_ms": 14.3,
"index_raw": 0.1,
"index_skill": 0.0,
"coverage": 1.0,
"chance": 0.1667,
"in_index": false
},
"36": {
"catalog_id": 36,
"dataset": "BRIGHT",
"requests": 550,
"answered": 550,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 550,
"metric": "nDCG@10",
"score": 0.1751,
"reference_same_cases": null,
"median_ms": 1002.0,
"index_raw": 0.1751,
"index_skill": 0.135,
"coverage": 1.0,
"chance": 0.0464,
"in_index": true
},
"37": {
"catalog_id": 37,
"dataset": "Amazon ESCI",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "macro-F1",
"score": 0.4309,
"reference_same_cases": null,
"median_ms": 26.6,
"index_raw": 0.4309,
"index_skill": 0.2862,
"coverage": 1.0,
"chance": 0.2027,
"in_index": true
},
"38": {
"catalog_id": 38,
"dataset": "ACOS",
"requests": 1565,
"answered": 1565,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1565,
"metric": "case exact accuracy",
"score": 0.03,
"reference_same_cases": null,
"median_ms": 672.9,
"index_raw": 0.03,
"index_skill": 0.03,
"coverage": 1.0,
"chance": 0.0,
"in_index": true
},
"39": {
"catalog_id": 39,
"dataset": "FinEntity",
"requests": 979,
"answered": 979,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 979,
"metric": "macro-F1",
"score": 0.8603,
"reference_same_cases": null,
"median_ms": 24.9,
"index_raw": 0.8603,
"index_skill": 0.7945,
"coverage": 1.0,
"chance": 0.3201,
"in_index": true
},
"40": {
"catalog_id": 40,
"dataset": "iSarcasmEval",
"requests": 4600,
"answered": 4600,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 4600,
"metric": "Sarcasm F1 \u00b7 track A, English",
"score": 0.4523,
"reference_same_cases": null,
"median_ms": 8.6,
"tracks": [
{
"track": "A \u00b7 Arabic",
"score": 0.313,
"headline": false
},
{
"track": "A \u00b7 English",
"score": 0.4523,
"headline": true
},
{
"track": "C \u00b7 Arabic pairs",
"score": 0.645,
"headline": false
},
{
"track": "C \u00b7 English pairs",
"score": 0.915,
"headline": false
}
],
"index_raw": 0.4523,
"index_skill": 0.2953,
"coverage": 1.0,
"chance": 0.2227,
"in_index": true
},
"41": {
"catalog_id": 41,
"dataset": "VAST",
"requests": 3006,
"answered": 3006,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 3006,
"metric": "macro-F1",
"score": 0.6169,
"reference_same_cases": null,
"median_ms": 14.9,
"index_raw": 0.6169,
"index_skill": 0.4254,
"coverage": 1.0,
"chance": 0.3333,
"in_index": true
},
"42": {
"catalog_id": 42,
"dataset": "NLI4CT",
"requests": 5500,
"answered": 5500,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5500,
"metric": "macro-F1",
"score": 0.7618,
"reference_same_cases": null,
"median_ms": 37.0,
"index_raw": 0.7618,
"index_skill": 0.5366,
"coverage": 1.0,
"chance": 0.486,
"in_index": true
},
"43": {
"catalog_id": 43,
"dataset": "CRUXEval",
"requests": 570,
"answered": 570,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 570,
"metric": "accuracy",
"score": 0.4719,
"reference_same_cases": null,
"median_ms": 13.7,
"index_raw": 0.4719,
"index_skill": 0.1622,
"coverage": 1.0,
"chance": 0.3697,
"in_index": true
},
"44": {
"catalog_id": 44,
"dataset": "CLadder",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.6368,
"reference_same_cases": null,
"median_ms": 13.7,
"index_raw": 0.6368,
"index_skill": 0.2736,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"45": {
"catalog_id": 45,
"dataset": "HLE",
"requests": 501,
"answered": 501,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 501,
"metric": "accuracy",
"score": 0.1297,
"reference_same_cases": null,
"median_ms": 22.6,
"index_raw": 0.1297,
"index_skill": 0.0,
"coverage": 1.0,
"chance": 0.1641,
"in_index": true
},
"48": {
"catalog_id": 48,
"dataset": "ForecastBench",
"requests": 10139,
"answered": 10139,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10139,
"metric": "Brier (lower is better)",
"score": 0.2172,
"reference_same_cases": null,
"median_ms": 49.3,
"index_raw": 0.1312,
"index_skill": 0.1312,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"50": {
"catalog_id": 50,
"dataset": "Habermas Machine",
"requests": 1676,
"answered": 1676,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1676,
"metric": "accuracy",
"score": 0.4427,
"reference_same_cases": null,
"median_ms": 55.1,
"index_raw": 0.4427,
"index_skill": 0.1912,
"coverage": 1.0,
"chance": 0.311,
"in_index": true
},
"56": {
"catalog_id": 56,
"dataset": "PhishNChips phishing decisions",
"requests": 2000,
"answered": 2000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.6365,
"median_ms": 140.2,
"scored_requests": 2000,
"index_raw": 0.6365,
"index_skill": 0.273,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"57": {
"catalog_id": 57,
"dataset": "MMLU-Pro",
"requests": 12032,
"answered": 12032,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.5206,
"median_ms": 20.3,
"scored_requests": 12032,
"index_raw": 0.5206,
"index_skill": 0.4608,
"coverage": 1.0,
"chance": 0.1109,
"in_index": true
},
"58": {
"catalog_id": 58,
"dataset": "BBH fixed-option tasks",
"requests": 5507,
"answered": 5507,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.6379,
"median_ms": 15.0,
"scored_requests": 5507,
"index_raw": 0.6379,
"index_skill": 0.4751,
"coverage": 1.0,
"chance": 0.3101,
"in_index": true
},
"59": {
"catalog_id": 59,
"dataset": "RAGTruth response-level hallucination",
"requests": 2700,
"answered": 2700,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "F1 on hallucinated class",
"score": 0.6648,
"median_ms": 50.6,
"scored_requests": 2700,
"index_raw": 0.6648,
"index_skill": 0.4306,
"coverage": 1.0,
"chance": 0.4113,
"in_index": true
},
"61": {
"catalog_id": 61,
"dataset": "HoVer claim verification",
"requests": 4000,
"answered": 4000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.6308,
"median_ms": 29.9,
"scored_requests": 4000,
"index_raw": 0.6308,
"index_skill": 0.2616,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"62": {
"catalog_id": 62,
"dataset": "When2Call MCQ",
"requests": 3652,
"answered": 3652,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.6276,
"median_ms": 52.3,
"scored_requests": 3652,
"index_raw": 0.6276,
"index_skill": 0.5035,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"64": {
"catalog_id": 64,
"dataset": "New Yorker caption matching",
"requests": 528,
"answered": 528,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.589,
"median_ms": 16.6,
"scored_requests": 528,
"index_raw": 0.589,
"index_skill": 0.4862,
"coverage": 1.0,
"chance": 0.2,
"in_index": true
}
},
"panel_id": "decision-index-0.2",
"note": "Decision Index 0.2 averages 40 benchmarks in five equal-weight areas: each area is the plain mean of its benchmarks and the index is 100 x the mean of the five areas. Each benchmark is chance-corrected first, (score - chance) / (1 - chance) clipped to 0-1, so 0 means random guessing and 100 means perfect. Every score is coverage-adjusted, so an unanswered or unsupported request counts as wrong. ForecastBench enters against its baseline: clip((0.25 - Brier) / 0.25) x coverage, so always predicting 0.5 scores zero. MMLU, ARC-Easy, ARC-Challenge, SimpleBench stay on the board as non-index benchmarks. The six interactive environments are still unrun and stay out. Every entrant on the board has results on all 40 index benchmarks. Point estimates only, no uncertainty intervals yet.",
"local_run": {
"kit_commit": "19ad28ec9485493cc4f7fc07d91c178f948e6434",
"evaluated": "2026-09-25",
"note": "Scored locally with the official kit; not a leaderboard submission or result. Requests shared with 0.1 reuse this model's 0.1 predictions; the added requests ran with the same frozen evaluation setup at temperature 1.0. No leaderboard-style exposure penalty is applied: known training exposure (see the model card) stays in these scores.",
"without_mmlu_pro": 37.41,
"screening": [
{
"stage": "T3",
"rows": 23156,
"matched": 138,
"by_source": {
"MMLU-Pro": 131,
"SuperGPQA": 6,
"BoolQ": 1,
"MedMCQA": 0
}
},
{
"stage": "T4",
"rows": 42360,
"matched": 156,
"by_source": {
"MMLU-Pro": 148,
"SuperGPQA": 7,
"BoolQ": 1,
"MedMCQA": 0
}
}
],
"timing_note": "latency_ms and every benchmark's median_ms are each request's share of batched inference time, allocated by prompt tokens. They are not serial-request or HTTP-serving latency and shouldn't be used for serving-latency comparisons."
}
}