blink-mimo-9b / eval /decision-index-0.2-local.json
thegovind's picture
Card: local Decision Index 0.2 run (descriptive; known training exposure not penalized) and its evidence file
57ae92b verified
Raw History Blame
29.7 kB
{
"engine": "blink-mimo-9b",
"edition": "Decision Index 0.2 (local run, descriptive)",
"generated_utc": "2026-09-25T23:10:24+00:00",
"suite": {
"edition": "release-v2",
"requests": 121057,
"scoreable": 120615,
"excluded": 442,
"added_requests": 30419,
"benchmarks": 44,
"rows_sha256": "b2b56d6fb636837ca469e689087bdbf373dda8de7638aa2da6793e6eda0792d5",
"added_sha256": "7429f3c9cdddb772c1cfc42bb2a45e8516b0032152b746e6929f1c8b52f4ce89"
},
"completed": 151034,
"complete": true,
"counts": {
"ok": 151034
},
"latency_ms": {
"median": 45.8,
"p95": 1053.9,
"mean": 214.9
},
"decision_index": 43.36,
"raw_index": 57.24,
"scores": {
"balanced_skill": 43.36,
"balanced_raw": 57.24,
"breadth_skill": 42.38
},
"areas": [
{
"id": "knowledge",
"label": "Knowledge & Reasoning",
"raw": 0.4824,
"skill": 0.3319,
"coverage": 1.0,
"n": 10,
"benchmarks": [
25,
30,
31,
32,
33,
43,
44,
45,
57,
58
]
},
{
"id": "language",
"label": "Language Understanding",
"raw": 0.6719,
"skill": 0.5405,
"coverage": 1.0,
"n": 10,
"benchmarks": [
11,
12,
28,
29,
38,
39,
40,
41,
42,
59
]
},
{
"id": "retrieval",
"label": "Retrieval & Classification",
"raw": 0.5536,
"skill": 0.4048,
"coverage": 1.0,
"n": 7,
"benchmarks": [
4,
5,
10,
36,
37,
56,
61
]
},
{
"id": "tools",
"label": "Tools & Automation",
"raw": 0.6519,
"skill": 0.5702,
"coverage": 1.0,
"n": 6,
"benchmarks": [
1,
2,
3,
6,
9,
62
]
},
{
"id": "arts",
"label": "Arts & Human Taste",
"raw": 0.502,
"skill": 0.3204,
"coverage": 1.0,
"n": 7,
"benchmarks": [
20,
21,
22,
23,
48,
50,
64
]
}
],
"index_benchmarks": {
"1": {
"raw": 0.8926,
"skill": 0.855,
"coverage": 1.0,
"random": 0.2592,
"rule": "track",
"in_index": true,
"tracks": []
},
"2": {
"raw": 0.4296,
"skill": 0.3719,
"coverage": 1.0,
"random": 0.0918,
"rule": "track",
"in_index": true,
"tracks": []
},
"3": {
"raw": 0.8386,
"skill": 0.8355,
"coverage": 1.0,
"random": 0.0189,
"rule": "chance",
"in_index": true
},
"4": {
"raw": 0.8177,
"skill": 0.8154,
"coverage": 1.0,
"random": 0.0127,
"rule": "chance",
"in_index": true
},
"5": {
"raw": 0.8381,
"skill": 0.8371,
"coverage": 1.0,
"random": 0.006,
"rule": "chance",
"in_index": true
},
"6": {
"raw": 0.7989,
"skill": 0.5256,
"coverage": 1.0,
"random": 0.5245,
"rule": "track",
"in_index": true,
"tracks": [
{
"track": "RouterBench-0shot",
"score": 0.7864,
"headline": false
},
{
"track": "RouterBench-5shot",
"score": 0.8114,
"headline": false
}
]
},
"9": {
"raw": 0.3063,
"skill": 0.3063,
"coverage": 1.0,
"random": 0.0,
"rule": "chance",
"in_index": true
},
"10": {
"raw": 0.1982,
"skill": 0.0,
"coverage": 1.0,
"random": 0.399,
"rule": "chance",
"in_index": true
},
"11": {
"raw": 0.8171,
"skill": 0.7355,
"coverage": 1.0,
"random": 0.3085,
"rule": "track",
"in_index": true,
"tracks": []
},
"12": {
"raw": 0.6899,
"skill": 0.5355,
"coverage": 1.0,
"random": 0.3324,
"rule": "chance",
"in_index": true
},
"20": {
"raw": 0.8408,
"skill": 0.6817,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"21": {
"raw": 0.6381,
"skill": 0.2763,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"22": {
"raw": 0.0763,
"skill": 0.0691,
"coverage": 1.0,
"random": 0.0078,
"rule": "track",
"in_index": true,
"tracks": []
},
"23": {
"raw": 0.5965,
"skill": 0.193,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"25": {
"raw": 0.4082,
"skill": 0.2109,
"coverage": 1.0,
"random": 0.25,
"rule": "track",
"in_index": true,
"tracks": []
},
"28": {
"raw": 0.7419,
"skill": 0.4838,
"coverage": 1.0,
"random": 0.5,
"rule": "chance",
"in_index": true
},
"29": {
"raw": 0.867,
"skill": 0.8227,
"coverage": 1.0,
"random": 0.25,
"rule": "chance",
"in_index": true
},
"30": {
"raw": 0.6585,
"skill": 0.5882,
"coverage": 1.0,
"random": 0.25,
"rule": "track",
"in_index": true,
"tracks": [
{
"track": "GSM8K-4choice",
"score": 0.7096,
"headline": false
},
{
"track": "GSM8K-10choice",
"score": 0.6073,
"headline": false
}
]
},
"31": {
"raw": 0.2292,
"skill": 0.1605,
"coverage": 1.0,
"random": 0.0819,
"rule": "track",
"in_index": true,
"tracks": []
},
"32": {
"raw": 0.5705,
"skill": 0.3172,
"coverage": 1.0,
"random": 0.371,
"rule": "chance",
"in_index": true
},
"33": {
"raw": 0.3467,
"skill": 0.338,
"coverage": 1.0,
"random": 0.0131,
"rule": "chance",
"in_index": true
},
"36": {
"raw": 0.1795,
"skill": 0.1396,
"coverage": 1.0,
"random": 0.0464,
"rule": "track",
"in_index": true,
"tracks": []
},
"37": {
"raw": 0.5199,
"skill": 0.3978,
"coverage": 1.0,
"random": 0.2027,
"rule": "track",
"in_index": true,
"tracks": []
},
"38": {
"raw": 0.055,
"skill": 0.055,
"coverage": 1.0,
"random": 0.0,
"rule": "chance",
"in_index": true
},
"39": {
"raw": 0.8246,
"skill": 0.742,
"coverage": 1.0,
"random": 0.3201,
"rule": "chance",
"in_index": true
},
"40": {
"raw": 0.5057,
"skill": 0.3641,
"coverage": 1.0,
"random": 0.2227,
"rule": "track",
"in_index": true,
"tracks": [
{
"track": "A \u00b7 Arabic",
"score": 0.3205,
"headline": false
},
{
"track": "A \u00b7 English",
"score": 0.5057,
"headline": true
},
{
"track": "C \u00b7 Arabic pairs",
"score": 0.79,
"headline": false
},
{
"track": "C \u00b7 English pairs",
"score": 0.95,
"headline": false
}
]
},
"41": {
"raw": 0.7805,
"skill": 0.6707,
"coverage": 1.0,
"random": 0.3333,
"rule": "track",
"in_index": true,
"tracks": []
},
"42": {
"raw": 0.8038,
"skill": 0.6183,
"coverage": 1.0,
"random": 0.486,
"rule": "chance",
"in_index": true
},
"43": {
"raw": 0.5474,
"skill": 0.2819,
"coverage": 1.0,
"random": 0.3697,
"rule": "track",
"in_index": true,
"tracks": []
},
"44": {
"raw": 0.6614,
"skill": 0.3228,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"45": {
"raw": 0.1098,
"skill": 0.0,
"coverage": 1.0,
"random": 0.1641,
"rule": "chance",
"in_index": true
},
"48": {
"raw": 0.282,
"skill": 0.282,
"coverage": 1.0,
"random": 0.25,
"rule": "vs baseline",
"in_index": true
},
"50": {
"raw": 0.4553,
"skill": 0.2094,
"coverage": 1.0,
"random": 0.311,
"rule": "track",
"in_index": true,
"tracks": []
},
"56": {
"raw": 0.6655,
"skill": 0.331,
"coverage": 1.0,
"random": 0.5,
"rule": "chance",
"in_index": true
},
"57": {
"raw": 0.6137,
"skill": 0.5655,
"coverage": 1.0,
"random": 0.1109,
"rule": "chance",
"in_index": true
},
"58": {
"raw": 0.6784,
"skill": 0.5338,
"coverage": 1.0,
"random": 0.3101,
"rule": "chance",
"in_index": true
},
"59": {
"raw": 0.6336,
"skill": 0.3776,
"coverage": 1.0,
"random": 0.4113,
"rule": "chance",
"in_index": true
},
"61": {
"raw": 0.6565,
"skill": 0.313,
"coverage": 1.0,
"random": 0.5,
"rule": "chance",
"in_index": true
},
"62": {
"raw": 0.6451,
"skill": 0.5268,
"coverage": 1.0,
"random": 0.25,
"rule": "chance",
"in_index": true
},
"64": {
"raw": 0.625,
"skill": 0.5312,
"coverage": 1.0,
"random": 0.2,
"rule": "chance",
"in_index": true
},
"24": {
"raw": 0.788,
"skill": 0.7173,
"coverage": 1.0,
"random": 0.25,
"rule": "shown, not counted",
"in_index": false
},
"26": {
"raw": 0.9798,
"skill": 0.9731,
"coverage": 1.0,
"random": 0.2502,
"rule": "shown, not counted",
"in_index": false
},
"27": {
"raw": 0.9514,
"skill": 0.9352,
"coverage": 1.0,
"random": 0.2502,
"rule": "shown, not counted",
"in_index": false
},
"34": {
"raw": 0.1,
"skill": 0.0,
"coverage": 1.0,
"random": 0.1667,
"rule": "shown, not counted",
"in_index": false
}
},
"benchmarks": {
"1": {
"catalog_id": 1,
"dataset": "BFCL",
"requests": 1694,
"answered": 1694,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1694,
"metric": "case exact accuracy",
"score": 0.8926,
"reference_same_cases": null,
"median_ms": 139.8,
"index_raw": 0.8926,
"index_skill": 0.855,
"coverage": 1.0,
"chance": 0.2592,
"in_index": true
},
"2": {
"catalog_id": 2,
"dataset": "ToolRet",
"requests": 1000,
"answered": 1000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1000,
"metric": "nDCG@10",
"score": 0.4296,
"reference_same_cases": null,
"median_ms": 1187.9,
"index_raw": 0.4296,
"index_skill": 0.3719,
"coverage": 1.0,
"chance": 0.0918,
"in_index": true
},
"3": {
"catalog_id": 3,
"dataset": "API-Bank",
"requests": 508,
"answered": 508,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 508,
"metric": "accuracy",
"score": 0.8386,
"reference_same_cases": null,
"median_ms": 640.5,
"index_raw": 0.8386,
"index_skill": 0.8355,
"coverage": 1.0,
"chance": 0.0189,
"in_index": true
},
"4": {
"catalog_id": 4,
"dataset": "BANKING77",
"requests": 3080,
"answered": 3080,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 3080,
"metric": "macro-F1",
"score": 0.8177,
"reference_same_cases": null,
"median_ms": 103.1,
"index_raw": 0.8177,
"index_skill": 0.8154,
"coverage": 1.0,
"chance": 0.0127,
"in_index": true
},
"5": {
"catalog_id": 5,
"dataset": "CLINC150+OOS",
"requests": 5500,
"answered": 5500,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5500,
"metric": "macro-F1",
"score": 0.8381,
"reference_same_cases": null,
"median_ms": 172.3,
"index_raw": 0.8381,
"index_skill": 0.8371,
"coverage": 1.0,
"chance": 0.006,
"in_index": true
},
"6": {
"catalog_id": 6,
"dataset": "RouterBench",
"requests": 10000,
"answered": 10000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10000,
"metric": "selected quality (quality objective)",
"score": 0.7989,
"reference_same_cases": null,
"median_ms": 392.5,
"tracks": [
{
"track": "RouterBench-0shot",
"score": 0.7864,
"headline": false
},
{
"track": "RouterBench-5shot",
"score": 0.8114,
"headline": false
}
],
"index_raw": 0.7989,
"index_skill": 0.5256,
"coverage": 1.0,
"chance": 0.5245,
"in_index": true
},
"9": {
"catalog_id": 9,
"dataset": "Home appliance simulator",
"requests": 160,
"answered": 160,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 160,
"metric": "case exact accuracy",
"score": 0.3063,
"reference_same_cases": null,
"median_ms": 1713.5,
"index_raw": 0.3063,
"index_skill": 0.3063,
"coverage": 1.0,
"chance": 0.0,
"in_index": true
},
"10": {
"catalog_id": 10,
"dataset": "SGD/SGD-X",
"requests": 2500,
"answered": 2500,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2500,
"metric": "macro-F1",
"score": 0.1982,
"reference_same_cases": null,
"median_ms": 106.5,
"index_raw": 0.1982,
"index_skill": 0.0,
"coverage": 1.0,
"chance": 0.399,
"in_index": true
},
"11": {
"catalog_id": 11,
"dataset": "ContractNLI",
"requests": 123,
"answered": 123,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 123,
"metric": "macro-F1",
"score": 0.8171,
"reference_same_cases": null,
"median_ms": 2966.1,
"index_raw": 0.8171,
"index_skill": 0.7355,
"coverage": 1.0,
"chance": 0.3085,
"in_index": true
},
"12": {
"catalog_id": 12,
"dataset": "ANLI",
"requests": 3200,
"answered": 3200,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 3200,
"metric": "macro-F1",
"score": 0.6899,
"reference_same_cases": null,
"median_ms": 19.4,
"index_raw": 0.6899,
"index_skill": 0.5355,
"coverage": 1.0,
"chance": 0.3324,
"in_index": true
},
"20": {
"catalog_id": 20,
"dataset": "BPoMP",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.8398,
"reference_same_cases": null,
"median_ms": 18.7,
"index_raw": 0.8408,
"index_skill": 0.6817,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"21": {
"catalog_id": 21,
"dataset": "Humicroedit",
"requests": 2628,
"answered": 2628,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2628,
"metric": "accuracy",
"score": 0.6381,
"reference_same_cases": null,
"median_ms": 11.8,
"index_raw": 0.6381,
"index_skill": 0.2763,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"22": {
"catalog_id": 22,
"dataset": "POP909-CL",
"requests": 2000,
"answered": 2000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2000,
"metric": "accuracy",
"score": 0.0825,
"reference_same_cases": null,
"median_ms": 843.5,
"index_raw": 0.0763,
"index_skill": 0.0691,
"coverage": 1.0,
"chance": 0.0078,
"in_index": true
},
"23": {
"catalog_id": 23,
"dataset": "cfcolor",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.5886,
"reference_same_cases": null,
"median_ms": 50.0,
"index_raw": 0.5965,
"index_skill": 0.193,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"24": {
"catalog_id": 24,
"dataset": "MMLU",
"requests": 14033,
"answered": 14033,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 14033,
"metric": "accuracy",
"score": 0.788,
"reference_same_cases": null,
"median_ms": 18.2,
"index_raw": 0.788,
"index_skill": 0.7173,
"coverage": 1.0,
"chance": 0.25,
"in_index": false
},
"25": {
"catalog_id": 25,
"dataset": "GPQA Diamond",
"requests": 196,
"answered": 196,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 196,
"metric": "accuracy",
"score": 0.4082,
"reference_same_cases": null,
"median_ms": 34.6,
"index_raw": 0.4082,
"index_skill": 0.2109,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"26": {
"catalog_id": 26,
"dataset": "ARC-Easy",
"requests": 2376,
"answered": 2376,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2376,
"metric": "accuracy",
"score": 0.9798,
"reference_same_cases": null,
"median_ms": 16.1,
"index_raw": 0.9798,
"index_skill": 0.9731,
"coverage": 1.0,
"chance": 0.2502,
"in_index": false
},
"27": {
"catalog_id": 27,
"dataset": "ARC-Challenge",
"requests": 1172,
"answered": 1172,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1172,
"metric": "accuracy",
"score": 0.9514,
"reference_same_cases": null,
"median_ms": 16.9,
"index_raw": 0.9514,
"index_skill": 0.9352,
"coverage": 1.0,
"chance": 0.2502,
"in_index": false
},
"28": {
"catalog_id": 28,
"dataset": "WinoGrande",
"requests": 1267,
"answered": 1267,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1267,
"metric": "accuracy",
"score": 0.7419,
"reference_same_cases": null,
"median_ms": 12.0,
"index_raw": 0.7419,
"index_skill": 0.4838,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"29": {
"catalog_id": 29,
"dataset": "HellaSwag",
"requests": 10042,
"answered": 10042,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10042,
"metric": "accuracy",
"score": 0.867,
"reference_same_cases": null,
"median_ms": 27.6,
"index_raw": 0.867,
"index_skill": 0.8227,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"30": {
"catalog_id": 30,
"dataset": "GSM8K",
"requests": 2638,
"answered": 2638,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2638,
"metric": "accuracy",
"score": 0.6585,
"reference_same_cases": null,
"median_ms": 24.3,
"tracks": [
{
"track": "GSM8K-4choice",
"score": 0.7096,
"headline": false
},
{
"track": "GSM8K-10choice",
"score": 0.6073,
"headline": false
}
],
"index_raw": 0.6585,
"index_skill": 0.5882,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"31": {
"catalog_id": 31,
"dataset": "ChessBench",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.2292,
"reference_same_cases": null,
"median_ms": 132.4,
"index_raw": 0.2292,
"index_skill": 0.1605,
"coverage": 1.0,
"chance": 0.0819,
"in_index": true
},
"32": {
"catalog_id": 32,
"dataset": "MuSR",
"requests": 752,
"answered": 752,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 752,
"metric": "accuracy",
"score": 0.5705,
"reference_same_cases": null,
"median_ms": 93.7,
"index_raw": 0.5705,
"index_skill": 0.3172,
"coverage": 1.0,
"chance": 0.371,
"in_index": true
},
"33": {
"catalog_id": 33,
"dataset": "SATA-Bench",
"requests": 1650,
"answered": 1650,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1650,
"metric": "case exact accuracy",
"score": 0.3467,
"reference_same_cases": null,
"median_ms": 321.2,
"index_raw": 0.3467,
"index_skill": 0.338,
"coverage": 1.0,
"chance": 0.0131,
"in_index": true
},
"34": {
"catalog_id": 34,
"dataset": "SimpleBench",
"requests": 10,
"answered": 10,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10,
"metric": "accuracy",
"score": 0.1,
"reference_same_cases": null,
"median_ms": 21.2,
"index_raw": 0.1,
"index_skill": 0.0,
"coverage": 1.0,
"chance": 0.1667,
"in_index": false
},
"36": {
"catalog_id": 36,
"dataset": "BRIGHT",
"requests": 550,
"answered": 550,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 550,
"metric": "nDCG@10",
"score": 0.1795,
"reference_same_cases": null,
"median_ms": 1508.7,
"index_raw": 0.1795,
"index_skill": 0.1396,
"coverage": 1.0,
"chance": 0.0464,
"in_index": true
},
"37": {
"catalog_id": 37,
"dataset": "Amazon ESCI",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "macro-F1",
"score": 0.5199,
"reference_same_cases": null,
"median_ms": 39.0,
"index_raw": 0.5199,
"index_skill": 0.3978,
"coverage": 1.0,
"chance": 0.2027,
"in_index": true
},
"38": {
"catalog_id": 38,
"dataset": "ACOS",
"requests": 1565,
"answered": 1565,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1565,
"metric": "case exact accuracy",
"score": 0.055,
"reference_same_cases": null,
"median_ms": 982.6,
"index_raw": 0.055,
"index_skill": 0.055,
"coverage": 1.0,
"chance": 0.0,
"in_index": true
},
"39": {
"catalog_id": 39,
"dataset": "FinEntity",
"requests": 979,
"answered": 979,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 979,
"metric": "macro-F1",
"score": 0.8246,
"reference_same_cases": null,
"median_ms": 36.0,
"index_raw": 0.8246,
"index_skill": 0.742,
"coverage": 1.0,
"chance": 0.3201,
"in_index": true
},
"40": {
"catalog_id": 40,
"dataset": "iSarcasmEval",
"requests": 4600,
"answered": 4600,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 4600,
"metric": "Sarcasm F1 \u00b7 track A, English",
"score": 0.5057,
"reference_same_cases": null,
"median_ms": 12.9,
"tracks": [
{
"track": "A \u00b7 Arabic",
"score": 0.3205,
"headline": false
},
{
"track": "A \u00b7 English",
"score": 0.5057,
"headline": true
},
{
"track": "C \u00b7 Arabic pairs",
"score": 0.79,
"headline": false
},
{
"track": "C \u00b7 English pairs",
"score": 0.95,
"headline": false
}
],
"index_raw": 0.5057,
"index_skill": 0.3641,
"coverage": 1.0,
"chance": 0.2227,
"in_index": true
},
"41": {
"catalog_id": 41,
"dataset": "VAST",
"requests": 3006,
"answered": 3006,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 3006,
"metric": "macro-F1",
"score": 0.7805,
"reference_same_cases": null,
"median_ms": 23.6,
"index_raw": 0.7805,
"index_skill": 0.6707,
"coverage": 1.0,
"chance": 0.3333,
"in_index": true
},
"42": {
"catalog_id": 42,
"dataset": "NLI4CT",
"requests": 5500,
"answered": 5500,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5500,
"metric": "macro-F1",
"score": 0.8038,
"reference_same_cases": null,
"median_ms": 54.9,
"index_raw": 0.8038,
"index_skill": 0.6183,
"coverage": 1.0,
"chance": 0.486,
"in_index": true
},
"43": {
"catalog_id": 43,
"dataset": "CRUXEval",
"requests": 570,
"answered": 570,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 570,
"metric": "accuracy",
"score": 0.5474,
"reference_same_cases": null,
"median_ms": 20.4,
"index_raw": 0.5474,
"index_skill": 0.2819,
"coverage": 1.0,
"chance": 0.3697,
"in_index": true
},
"44": {
"catalog_id": 44,
"dataset": "CLadder",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.6614,
"reference_same_cases": null,
"median_ms": 19.9,
"index_raw": 0.6614,
"index_skill": 0.3228,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"45": {
"catalog_id": 45,
"dataset": "HLE",
"requests": 501,
"answered": 501,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 501,
"metric": "accuracy",
"score": 0.1098,
"reference_same_cases": null,
"median_ms": 30.4,
"index_raw": 0.1098,
"index_skill": 0.0,
"coverage": 1.0,
"chance": 0.1641,
"in_index": true
},
"48": {
"catalog_id": 48,
"dataset": "ForecastBench",
"requests": 10139,
"answered": 10139,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10139,
"metric": "Brier (lower is better)",
"score": 0.1795,
"reference_same_cases": null,
"median_ms": 73.2,
"index_raw": 0.282,
"index_skill": 0.282,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"50": {
"catalog_id": 50,
"dataset": "Habermas Machine",
"requests": 1676,
"answered": 1676,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1676,
"metric": "accuracy",
"score": 0.4553,
"reference_same_cases": null,
"median_ms": 81.4,
"index_raw": 0.4553,
"index_skill": 0.2094,
"coverage": 1.0,
"chance": 0.311,
"in_index": true
},
"56": {
"catalog_id": 56,
"dataset": "PhishNChips phishing decisions",
"requests": 2000,
"answered": 2000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.6655,
"median_ms": 208.8,
"scored_requests": 2000,
"index_raw": 0.6655,
"index_skill": 0.331,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"57": {
"catalog_id": 57,
"dataset": "MMLU-Pro",
"requests": 12032,
"answered": 12032,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.6137,
"median_ms": 30.6,
"scored_requests": 12032,
"index_raw": 0.6137,
"index_skill": 0.5655,
"coverage": 1.0,
"chance": 0.1109,
"in_index": true
},
"58": {
"catalog_id": 58,
"dataset": "BBH fixed-option tasks",
"requests": 5507,
"answered": 5507,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.6784,
"median_ms": 23.7,
"scored_requests": 5507,
"index_raw": 0.6784,
"index_skill": 0.5338,
"coverage": 1.0,
"chance": 0.3101,
"in_index": true
},
"59": {
"catalog_id": 59,
"dataset": "RAGTruth response-level hallucination",
"requests": 2700,
"answered": 2700,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "F1 on hallucinated class",
"score": 0.6336,
"median_ms": 77.1,
"scored_requests": 2700,
"index_raw": 0.6336,
"index_skill": 0.3776,
"coverage": 1.0,
"chance": 0.4113,
"in_index": true
},
"61": {
"catalog_id": 61,
"dataset": "HoVer claim verification",
"requests": 4000,
"answered": 4000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.6565,
"median_ms": 45.6,
"scored_requests": 4000,
"index_raw": 0.6565,
"index_skill": 0.313,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"62": {
"catalog_id": 62,
"dataset": "When2Call MCQ",
"requests": 3652,
"answered": 3652,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.6451,
"median_ms": 79.8,
"scored_requests": 3652,
"index_raw": 0.6451,
"index_skill": 0.5268,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"64": {
"catalog_id": 64,
"dataset": "New Yorker caption matching",
"requests": 528,
"answered": 528,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.625,
"median_ms": 25.2,
"scored_requests": 528,
"index_raw": 0.625,
"index_skill": 0.5312,
"coverage": 1.0,
"chance": 0.2,
"in_index": true
}
},
"panel_id": "decision-index-0.2",
"note": "Decision Index 0.2 averages 40 benchmarks in five equal-weight areas: each area is the plain mean of its benchmarks and the index is 100 x the mean of the five areas. Each benchmark is chance-corrected first, (score - chance) / (1 - chance) clipped to 0-1, so 0 means random guessing and 100 means perfect. Every score is coverage-adjusted, so an unanswered or unsupported request counts as wrong. ForecastBench enters against its baseline: clip((0.25 - Brier) / 0.25) x coverage, so always predicting 0.5 scores zero. MMLU, ARC-Easy, ARC-Challenge, SimpleBench stay on the board as non-index benchmarks. The six interactive environments are still unrun and stay out. Every entrant on the board has results on all 40 index benchmarks. Point estimates only, no uncertainty intervals yet.",
"local_run": {
"kit_commit": "19ad28ec9485493cc4f7fc07d91c178f948e6434",
"evaluated": "2026-09-25",
"note": "Scored locally with the official kit; not a leaderboard submission or result. Requests shared with 0.1 reuse this model's 0.1 predictions; the added requests ran with the same frozen evaluation setup at temperature 1.0. No leaderboard-style exposure penalty is applied: known training exposure (see the model card) stays in these scores.",
"without_mmlu_pro": 42.84,
"screening": [
{
"stage": "MiMo",
"rows": 123195,
"matched": 180,
"by_source": {
"MMLU-Pro": 0,
"SuperGPQA": 176,
"BoolQ": 3,
"MedMCQA": 1
}
}
],
"timing_note": "latency_ms and every benchmark's median_ms are each request's share of batched inference time, allocated by prompt tokens. They are not serial-request or HTTP-serving latency and shouldn't be used for serving-latency comparisons."
}
}