Wald-4B / evaluation /scores.json
Harry19081's picture
Release Wald-Q4B22D0-f7: full DI0.2.1 54.59 and clean calibration
503dfd6 verified
Raw History Blame
33.1 kB
{
"engine": "run",
"edition": "0.2.1",
"generated_utc": "2026-09-28T23:16:58+00:00",
"suite": {
"edition": "release-v2.1",
"requests": 120340,
"scoreable": 119898,
"excluded": 442,
"added_requests": 30419,
"benchmarks": 44,
"rows_sha256": "b2b56d6fb636837ca469e689087bdbf373dda8de7638aa2da6793e6eda0792d5",
"added_sha256": "7429f3c9cdddb772c1cfc42bb2a45e8516b0032152b746e6929f1c8b52f4ce89"
},
"completed": 150317,
"complete": true,
"counts": {
"ok": 150317
},
"latency_ms": {
"median": 2359.7,
"p95": 16111.3,
"mean": 5418.8
},
"decision_index": 54.59,
"raw_index": 65.9,
"scores": {
"balanced_skill": 54.59,
"balanced_raw": 65.9,
"breadth_skill": 52.62
},
"areas": [
{
"id": "knowledge",
"label": "Knowledge & Reasoning",
"raw": 0.539,
"skill": 0.4253,
"coverage": 1.0,
"n": 10,
"benchmarks": [
25,
30,
31,
32,
33,
43,
44,
45,
57,
58
]
},
{
"id": "language",
"label": "Language Understanding",
"raw": 0.7533,
"skill": 0.6283,
"coverage": 1.0,
"n": 10,
"benchmarks": [
11,
12,
28,
29,
38,
39,
40,
41,
42,
59
]
},
{
"id": "retrieval",
"label": "Retrieval & Classification",
"raw": 0.6454,
"skill": 0.5067,
"coverage": 1.0,
"n": 6,
"benchmarks": [
4,
5,
36,
37,
56,
61
]
},
{
"id": "tools",
"label": "Tools & Automation",
"raw": 0.821,
"skill": 0.7949,
"coverage": 1.0,
"n": 5,
"benchmarks": [
1,
2,
3,
9,
62
]
},
{
"id": "arts",
"label": "Arts & Human Taste",
"raw": 0.4569,
"skill": 0.268,
"coverage": 1.0,
"n": 7,
"benchmarks": [
20,
21,
22,
23,
48,
50,
64
]
}
],
"index_benchmarks": {
"1": {
"raw": 0.938,
"skill": 0.9163,
"coverage": 1.0,
"random": 0.2592,
"rule": "track",
"in_index": true,
"tracks": []
},
"2": {
"raw": 0.66,
"skill": 0.6073,
"coverage": 1.0,
"random": 0.1341,
"rule": "track",
"in_index": true,
"tracks": []
},
"3": {
"raw": 0.7874,
"skill": 0.7833,
"coverage": 1.0,
"random": 0.0189,
"rule": "chance",
"in_index": true
},
"4": {
"raw": 0.7745,
"skill": 0.7716,
"coverage": 1.0,
"random": 0.0127,
"rule": "chance",
"in_index": true
},
"5": {
"raw": 0.8488,
"skill": 0.8479,
"coverage": 1.0,
"random": 0.006,
"rule": "chance",
"in_index": true
},
"9": {
"raw": 0.875,
"skill": 0.875,
"coverage": 1.0,
"random": 0.0,
"rule": "chance",
"in_index": true
},
"11": {
"raw": 0.761,
"skill": 0.6543,
"coverage": 1.0,
"random": 0.3085,
"rule": "track",
"in_index": true,
"tracks": []
},
"12": {
"raw": 0.6881,
"skill": 0.5328,
"coverage": 1.0,
"random": 0.3324,
"rule": "chance",
"in_index": true
},
"20": {
"raw": 0.8322,
"skill": 0.6644,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"21": {
"raw": 0.6005,
"skill": 0.2009,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"22": {
"raw": 0.042,
"skill": 0.0346,
"coverage": 1.0,
"random": 0.0078,
"rule": "track",
"in_index": true,
"tracks": []
},
"23": {
"raw": 0.5803,
"skill": 0.1606,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"25": {
"raw": 0.5102,
"skill": 0.3469,
"coverage": 1.0,
"random": 0.25,
"rule": "track",
"in_index": true,
"tracks": []
},
"28": {
"raw": 0.7916,
"skill": 0.5832,
"coverage": 1.0,
"random": 0.5,
"rule": "chance",
"in_index": true
},
"29": {
"raw": 0.8992,
"skill": 0.8656,
"coverage": 1.0,
"random": 0.25,
"rule": "chance",
"in_index": true
},
"30": {
"raw": 0.9151,
"skill": 0.8982,
"coverage": 1.0,
"random": 0.25,
"rule": "track",
"in_index": true,
"tracks": [
{
"track": "GSM8K-4choice",
"score": 0.9333,
"headline": false
},
{
"track": "GSM8K-10choice",
"score": 0.8969,
"headline": false
}
]
},
"31": {
"raw": 0.1076,
"skill": 0.028,
"coverage": 1.0,
"random": 0.0819,
"rule": "track",
"in_index": true,
"tracks": []
},
"32": {
"raw": 0.5758,
"skill": 0.3256,
"coverage": 1.0,
"random": 0.371,
"rule": "chance",
"in_index": true
},
"33": {
"raw": 0.2503,
"skill": 0.2403,
"coverage": 1.0,
"random": 0.0131,
"rule": "chance",
"in_index": true
},
"36": {
"raw": 0.4021,
"skill": 0.3236,
"coverage": 1.0,
"random": 0.116,
"rule": "track",
"in_index": true,
"tracks": []
},
"37": {
"raw": 0.5261,
"skill": 0.4056,
"coverage": 1.0,
"random": 0.2027,
"rule": "track",
"in_index": true,
"tracks": []
},
"38": {
"raw": 0.5652,
"skill": 0.5513,
"coverage": 1.0,
"random": 0.031,
"rule": "chance",
"in_index": true
},
"39": {
"raw": 0.8918,
"skill": 0.8409,
"coverage": 1.0,
"random": 0.3201,
"rule": "chance",
"in_index": true
},
"40": {
"raw": 0.5714,
"skill": 0.4487,
"coverage": 1.0,
"random": 0.2227,
"rule": "track",
"in_index": true,
"tracks": [
{
"track": "A · Arabic",
"score": 0.4873,
"headline": false
},
{
"track": "A · English",
"score": 0.5714,
"headline": true
},
{
"track": "C · Arabic pairs",
"score": 0.675,
"headline": false
},
{
"track": "C · English pairs",
"score": 0.905,
"headline": false
}
]
},
"41": {
"raw": 0.7784,
"skill": 0.6675,
"coverage": 1.0,
"random": 0.3333,
"rule": "track",
"in_index": true,
"tracks": []
},
"42": {
"raw": 0.7894,
"skill": 0.5903,
"coverage": 1.0,
"random": 0.486,
"rule": "chance",
"in_index": true
},
"43": {
"raw": 0.7825,
"skill": 0.6548,
"coverage": 1.0,
"random": 0.3697,
"rule": "track",
"in_index": true,
"tracks": []
},
"44": {
"raw": 0.7442,
"skill": 0.4884,
"coverage": 1.0,
"random": 0.5,
"rule": "track",
"in_index": true,
"tracks": []
},
"45": {
"raw": 0.0998,
"skill": 0.0,
"coverage": 1.0,
"random": 0.1641,
"rule": "chance",
"in_index": true
},
"48": {
"raw": 0.1868,
"skill": 0.1868,
"coverage": 1.0,
"random": 0.25,
"rule": "vs baseline",
"in_index": true
},
"50": {
"raw": 0.4123,
"skill": 0.147,
"coverage": 1.0,
"random": 0.311,
"rule": "track",
"in_index": true,
"tracks": []
},
"56": {
"raw": 0.5225,
"skill": 0.045,
"coverage": 1.0,
"random": 0.5,
"rule": "chance",
"in_index": true
},
"57": {
"raw": 0.6506,
"skill": 0.607,
"coverage": 1.0,
"random": 0.1109,
"rule": "chance",
"in_index": true
},
"58": {
"raw": 0.7774,
"skill": 0.6773,
"coverage": 1.0,
"random": 0.3101,
"rule": "chance",
"in_index": true
},
"59": {
"raw": 0.773,
"skill": 0.5293,
"coverage": 1.0,
"random": 0.5177,
"rule": "chance",
"in_index": true
},
"61": {
"raw": 0.7808,
"skill": 0.5616,
"coverage": 1.0,
"random": 0.5,
"rule": "chance",
"in_index": true
},
"62": {
"raw": 0.828,
"skill": 0.7707,
"coverage": 1.0,
"random": 0.25,
"rule": "chance",
"in_index": true
},
"64": {
"raw": 0.5985,
"skill": 0.4981,
"coverage": 1.0,
"random": 0.2,
"rule": "chance",
"in_index": true
},
"6": {
"raw": 0.7667,
"skill": 0.453,
"coverage": 1.0,
"random": 0.5735,
"rule": "shown, not counted",
"in_index": false
},
"10": {
"raw": 0.0748,
"skill": 0.0,
"coverage": 1.0,
"random": 0.399,
"rule": "shown, not counted",
"in_index": false
},
"24": {
"raw": 0.7918,
"skill": 0.7224,
"coverage": 1.0,
"random": 0.25,
"rule": "shown, not counted",
"in_index": false
},
"26": {
"raw": 0.9827,
"skill": 0.9769,
"coverage": 1.0,
"random": 0.2502,
"rule": "shown, not counted",
"in_index": false
},
"27": {
"raw": 0.9608,
"skill": 0.9477,
"coverage": 1.0,
"random": 0.2502,
"rule": "shown, not counted",
"in_index": false
},
"34": {
"raw": 0.1,
"skill": 0.0,
"coverage": 1.0,
"random": 0.1667,
"rule": "shown, not counted",
"in_index": false
}
},
"benchmarks": {
"1": {
"catalog_id": 1,
"dataset": "BFCL",
"requests": 1694,
"answered": 1694,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1694,
"metric": "case exact accuracy",
"score": 0.938,
"reference_same_cases": null,
"median_ms": 3427.0,
"index_raw": 0.938,
"index_skill": 0.9163,
"coverage": 1.0,
"chance": 0.2592,
"in_index": true
},
"2": {
"catalog_id": 2,
"dataset": "ToolRet",
"requests": 685,
"answered": 685,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 685,
"metric": "nDCG@10",
"score": 0.66,
"reference_same_cases": null,
"median_ms": 48200.3,
"index_raw": 0.66,
"index_skill": 0.6073,
"coverage": 1.0,
"chance": 0.1341,
"in_index": true
},
"3": {
"catalog_id": 3,
"dataset": "API-Bank",
"requests": 508,
"answered": 508,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 508,
"metric": "accuracy",
"score": 0.7874,
"reference_same_cases": null,
"median_ms": 893.8,
"index_raw": 0.7874,
"index_skill": 0.7833,
"coverage": 1.0,
"chance": 0.0189,
"in_index": true
},
"4": {
"catalog_id": 4,
"dataset": "BANKING77",
"requests": 3080,
"answered": 3080,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 3080,
"metric": "macro-F1",
"score": 0.7745,
"reference_same_cases": null,
"median_ms": 347.4,
"index_raw": 0.7745,
"index_skill": 0.7716,
"coverage": 1.0,
"chance": 0.0127,
"in_index": true
},
"5": {
"catalog_id": 5,
"dataset": "CLINC150+OOS",
"requests": 5500,
"answered": 5500,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5500,
"metric": "macro-F1",
"score": 0.8488,
"reference_same_cases": null,
"median_ms": 578.9,
"index_raw": 0.8488,
"index_skill": 0.8479,
"coverage": 1.0,
"chance": 0.006,
"in_index": true
},
"6": {
"catalog_id": 6,
"dataset": "RouterBench",
"requests": 10000,
"answered": 10000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10000,
"metric": "selected quality (quality objective)",
"score": 0.7667,
"reference_same_cases": null,
"median_ms": 14357.0,
"tracks": {
"RouterBench-0shot": {
"metric": "selected quality (quality objective)",
"score": 0.7308614831101339,
"scored_requests": 5003
},
"RouterBench-5shot": {
"metric": "selected quality (quality objective)",
"score": 0.8026115669401641,
"scored_requests": 4997
}
},
"index_raw": 0.7667,
"index_skill": 0.453,
"coverage": 1.0,
"chance": 0.5735,
"in_index": false
},
"9": {
"catalog_id": 9,
"dataset": "Home appliance simulator",
"requests": 88,
"answered": 88,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 88,
"metric": "case exact accuracy",
"score": 0.875,
"reference_same_cases": null,
"median_ms": 23359.2,
"index_raw": 0.875,
"index_skill": 0.875,
"coverage": 1.0,
"chance": 0.0,
"in_index": true
},
"10": {
"catalog_id": 10,
"dataset": "SGD/SGD-X",
"requests": 2500,
"answered": 2500,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2500,
"metric": "macro-F1",
"score": 0.0748,
"reference_same_cases": null,
"median_ms": 1551.0,
"index_raw": 0.0748,
"index_skill": 0.0,
"coverage": 1.0,
"chance": 0.399,
"in_index": false
},
"11": {
"catalog_id": 11,
"dataset": "ContractNLI",
"requests": 123,
"answered": 123,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 123,
"metric": "macro-F1",
"score": 0.761,
"reference_same_cases": null,
"median_ms": 48119.1,
"index_raw": 0.761,
"index_skill": 0.6543,
"coverage": 1.0,
"chance": 0.3085,
"in_index": true
},
"12": {
"catalog_id": 12,
"dataset": "ANLI",
"requests": 3200,
"answered": 3200,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 3200,
"metric": "macro-F1",
"score": 0.6881,
"reference_same_cases": null,
"median_ms": 1871.7,
"index_raw": 0.6881,
"index_skill": 0.5328,
"coverage": 1.0,
"chance": 0.3324,
"in_index": true
},
"20": {
"catalog_id": 20,
"dataset": "BPoMP",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.8328,
"reference_same_cases": null,
"median_ms": 3517.3,
"index_raw": 0.8322,
"index_skill": 0.6644,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"21": {
"catalog_id": 21,
"dataset": "Humicroedit",
"requests": 2628,
"answered": 2628,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2628,
"metric": "accuracy",
"score": 0.6005,
"reference_same_cases": null,
"median_ms": 1887.3,
"index_raw": 0.6005,
"index_skill": 0.2009,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"22": {
"catalog_id": 22,
"dataset": "POP909-CL",
"requests": 2000,
"answered": 2000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2000,
"metric": "accuracy",
"score": 0.0415,
"reference_same_cases": null,
"median_ms": 951.7,
"index_raw": 0.042,
"index_skill": 0.0346,
"coverage": 1.0,
"chance": 0.0078,
"in_index": true
},
"23": {
"catalog_id": 23,
"dataset": "cfcolor",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.5986,
"reference_same_cases": null,
"median_ms": 5303.8,
"index_raw": 0.5803,
"index_skill": 0.1606,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"24": {
"catalog_id": 24,
"dataset": "MMLU",
"requests": 14033,
"answered": 14033,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 14033,
"metric": "accuracy",
"score": 0.7918,
"reference_same_cases": null,
"median_ms": 1615.5,
"index_raw": 0.7918,
"index_skill": 0.7224,
"coverage": 1.0,
"chance": 0.25,
"in_index": false
},
"25": {
"catalog_id": 25,
"dataset": "GPQA Diamond",
"requests": 196,
"answered": 196,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 196,
"metric": "accuracy",
"score": 0.5102,
"reference_same_cases": null,
"median_ms": 5373.5,
"index_raw": 0.5102,
"index_skill": 0.3469,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"26": {
"catalog_id": 26,
"dataset": "ARC-Easy",
"requests": 2376,
"answered": 2376,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2376,
"metric": "accuracy",
"score": 0.9827,
"reference_same_cases": null,
"median_ms": 915.8,
"index_raw": 0.9827,
"index_skill": 0.9769,
"coverage": 1.0,
"chance": 0.2502,
"in_index": false
},
"27": {
"catalog_id": 27,
"dataset": "ARC-Challenge",
"requests": 1172,
"answered": 1172,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1172,
"metric": "accuracy",
"score": 0.9608,
"reference_same_cases": null,
"median_ms": 1117.7,
"index_raw": 0.9608,
"index_skill": 0.9477,
"coverage": 1.0,
"chance": 0.2502,
"in_index": false
},
"28": {
"catalog_id": 28,
"dataset": "WinoGrande",
"requests": 1267,
"answered": 1267,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1267,
"metric": "accuracy",
"score": 0.7916,
"reference_same_cases": null,
"median_ms": 1430.2,
"index_raw": 0.7916,
"index_skill": 0.5832,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"29": {
"catalog_id": 29,
"dataset": "HellaSwag",
"requests": 10042,
"answered": 10042,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10042,
"metric": "accuracy",
"score": 0.8992,
"reference_same_cases": null,
"median_ms": 2731.9,
"index_raw": 0.8992,
"index_skill": 0.8656,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"30": {
"catalog_id": 30,
"dataset": "GSM8K",
"requests": 2638,
"answered": 2638,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 2638,
"metric": "accuracy",
"score": 0.9151,
"reference_same_cases": null,
"median_ms": 1297.7,
"tracks": [
{
"track": "GSM8K-4choice",
"score": 0.9333,
"headline": false
},
{
"track": "GSM8K-10choice",
"score": 0.8969,
"headline": false
}
],
"index_raw": 0.9151,
"index_skill": 0.8982,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"31": {
"catalog_id": 31,
"dataset": "ChessBench",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.1076,
"reference_same_cases": null,
"median_ms": 367.5,
"index_raw": 0.1076,
"index_skill": 0.028,
"coverage": 1.0,
"chance": 0.0819,
"in_index": true
},
"32": {
"catalog_id": 32,
"dataset": "MuSR",
"requests": 752,
"answered": 752,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 752,
"metric": "accuracy",
"score": 0.5758,
"reference_same_cases": null,
"median_ms": 2307.2,
"index_raw": 0.5758,
"index_skill": 0.3256,
"coverage": 1.0,
"chance": 0.371,
"in_index": true
},
"33": {
"catalog_id": 33,
"dataset": "SATA-Bench",
"requests": 1650,
"answered": 1650,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1650,
"metric": "case exact accuracy",
"score": 0.2503,
"reference_same_cases": null,
"median_ms": 20847.2,
"index_raw": 0.2503,
"index_skill": 0.2403,
"coverage": 1.0,
"chance": 0.0131,
"in_index": true
},
"34": {
"catalog_id": 34,
"dataset": "SimpleBench",
"requests": 10,
"answered": 10,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10,
"metric": "accuracy",
"score": 0.1,
"reference_same_cases": null,
"median_ms": 6311.8,
"index_raw": 0.1,
"index_skill": 0.0,
"coverage": 1.0,
"chance": 0.1667,
"in_index": false
},
"36": {
"catalog_id": 36,
"dataset": "BRIGHT",
"requests": 220,
"answered": 220,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 220,
"metric": "nDCG@10",
"score": 0.4021,
"reference_same_cases": null,
"median_ms": 51918.0,
"index_raw": 0.4021,
"index_skill": 0.3236,
"coverage": 1.0,
"chance": 0.116,
"in_index": true
},
"37": {
"catalog_id": 37,
"dataset": "Amazon ESCI",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "macro-F1",
"score": 0.5261,
"reference_same_cases": null,
"median_ms": 1674.8,
"index_raw": 0.5261,
"index_skill": 0.4056,
"coverage": 1.0,
"chance": 0.2027,
"in_index": true
},
"38": {
"catalog_id": 38,
"dataset": "ACOS",
"requests": 1565,
"answered": 1565,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1565,
"metric": "per-review F1",
"score": 0.5652,
"reference_same_cases": null,
"median_ms": 92273.0,
"index_raw": 0.5652,
"index_skill": 0.5513,
"coverage": 1.0,
"chance": 0.031,
"in_index": true
},
"39": {
"catalog_id": 39,
"dataset": "FinEntity",
"requests": 979,
"answered": 979,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 979,
"metric": "macro-F1",
"score": 0.8918,
"reference_same_cases": null,
"median_ms": 2899.9,
"index_raw": 0.8918,
"index_skill": 0.8409,
"coverage": 1.0,
"chance": 0.3201,
"in_index": true
},
"40": {
"catalog_id": 40,
"dataset": "iSarcasmEval",
"requests": 4600,
"answered": 4600,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 4600,
"metric": "Sarcasm F1 · track A, English",
"score": 0.5714,
"reference_same_cases": null,
"median_ms": 1845.7,
"tracks": [
{
"track": "A · Arabic",
"score": 0.4873,
"headline": false
},
{
"track": "A · English",
"score": 0.5714,
"headline": true
},
{
"track": "C · Arabic pairs",
"score": 0.675,
"headline": false
},
{
"track": "C · English pairs",
"score": 0.905,
"headline": false
}
],
"index_raw": 0.5714,
"index_skill": 0.4487,
"coverage": 1.0,
"chance": 0.2227,
"in_index": true
},
"41": {
"catalog_id": 41,
"dataset": "VAST",
"requests": 3006,
"answered": 3006,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 3006,
"metric": "macro-F1",
"score": 0.7784,
"reference_same_cases": null,
"median_ms": 1187.7,
"index_raw": 0.7784,
"index_skill": 0.6675,
"coverage": 1.0,
"chance": 0.3333,
"in_index": true
},
"42": {
"catalog_id": 42,
"dataset": "NLI4CT",
"requests": 5500,
"answered": 5500,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5500,
"metric": "macro-F1",
"score": 0.7894,
"reference_same_cases": null,
"median_ms": 2523.8,
"index_raw": 0.7894,
"index_skill": 0.5903,
"coverage": 1.0,
"chance": 0.486,
"in_index": true
},
"43": {
"catalog_id": 43,
"dataset": "CRUXEval",
"requests": 570,
"answered": 570,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 570,
"metric": "accuracy",
"score": 0.7825,
"reference_same_cases": null,
"median_ms": 2111.4,
"index_raw": 0.7825,
"index_skill": 0.6548,
"coverage": 1.0,
"chance": 0.3697,
"in_index": true
},
"44": {
"catalog_id": 44,
"dataset": "CLadder",
"requests": 5000,
"answered": 5000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 5000,
"metric": "accuracy",
"score": 0.7442,
"reference_same_cases": null,
"median_ms": 3131.1,
"index_raw": 0.7442,
"index_skill": 0.4884,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"45": {
"catalog_id": 45,
"dataset": "HLE",
"requests": 501,
"answered": 501,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 501,
"metric": "accuracy",
"score": 0.0998,
"reference_same_cases": null,
"median_ms": 5403.5,
"index_raw": 0.0998,
"index_skill": 0.0,
"coverage": 1.0,
"chance": 0.1641,
"in_index": true
},
"48": {
"catalog_id": 48,
"dataset": "ForecastBench",
"requests": 10139,
"answered": 10139,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 10139,
"metric": "Brier (lower is better)",
"score": 0.2033,
"reference_same_cases": null,
"median_ms": 5139.2,
"index_raw": 0.1868,
"index_skill": 0.1868,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"50": {
"catalog_id": 50,
"dataset": "Habermas Machine",
"requests": 1676,
"answered": 1676,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"scored_requests": 1676,
"metric": "accuracy",
"score": 0.4123,
"reference_same_cases": null,
"median_ms": 3961.4,
"index_raw": 0.4123,
"index_skill": 0.147,
"coverage": 1.0,
"chance": 0.311,
"in_index": true
},
"56": {
"catalog_id": 56,
"dataset": "PhishNChips phishing decisions",
"requests": 2000,
"answered": 2000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.5225,
"median_ms": 13023.1,
"scored_requests": 2000,
"index_raw": 0.5225,
"index_skill": 0.045,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"57": {
"catalog_id": 57,
"dataset": "MMLU-Pro",
"requests": 12032,
"answered": 12032,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.6506,
"median_ms": 2305.7,
"scored_requests": 12032,
"index_raw": 0.6506,
"index_skill": 0.607,
"coverage": 1.0,
"chance": 0.1109,
"in_index": true
},
"58": {
"catalog_id": 58,
"dataset": "BBH fixed-option tasks",
"requests": 5507,
"answered": 5507,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.7774,
"median_ms": 1827.8,
"scored_requests": 5507,
"index_raw": 0.7774,
"index_skill": 0.6773,
"coverage": 1.0,
"chance": 0.3101,
"in_index": true
},
"59": {
"catalog_id": 59,
"dataset": "RAGTruth response-level hallucination",
"requests": 2700,
"answered": 2700,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "F1 on hallucinated class",
"score": 0.773,
"median_ms": 2469.0,
"scored_requests": 2700,
"index_raw": 0.773,
"index_skill": 0.5293,
"coverage": 1.0,
"chance": 0.5177,
"in_index": true
},
"61": {
"catalog_id": 61,
"dataset": "HoVer claim verification",
"requests": 4000,
"answered": 4000,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.7808,
"median_ms": 2007.0,
"scored_requests": 4000,
"index_raw": 0.7808,
"index_skill": 0.5616,
"coverage": 1.0,
"chance": 0.5,
"in_index": true
},
"62": {
"catalog_id": 62,
"dataset": "When2Call MCQ",
"requests": 3652,
"answered": 3652,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.828,
"median_ms": 1970.4,
"scored_requests": 3652,
"index_raw": 0.828,
"index_skill": 0.7707,
"coverage": 1.0,
"chance": 0.25,
"in_index": true
},
"64": {
"catalog_id": 64,
"dataset": "New Yorker caption matching",
"requests": 528,
"answered": 528,
"unsupported": 0,
"errors": 0,
"abstained": 0,
"pending": 0,
"metric": "accuracy",
"score": 0.5985,
"median_ms": 2952.1,
"scored_requests": 528,
"index_raw": 0.5985,
"index_skill": 0.4981,
"coverage": 1.0,
"chance": 0.2,
"in_index": true
}
},
"panel_id": "decision-index-0.2.1",
"note": "Decision Index 0.2.1 averages 38 benchmarks in five areas. Arts & Human Taste weighs 10%; the other four share 90% in proportion to the square root of their benchmark count (knowledge 25.8%, language 25.8%, retrieval 20.0%, tools 18.3%). Inside an area, gold ★ benchmarks weigh 1.2 and the rest 1.0; the index is 100 x the weighted mean of the five areas. Each benchmark is chance-corrected first, (score - chance) / (1 - chance) clipped to 0-1, so 0 means random guessing and 100 means perfect. Every score is coverage-adjusted, so an unanswered or unsupported request counts as wrong. ForecastBench enters against its baseline: clip((0.25 - Brier) / 0.25) x coverage, so always predicting 0.5 scores zero. MMLU, ARC-Easy, ARC-Challenge, RouterBench, SGD stay on the board as non-index benchmarks. The six interactive environments are still unrun and stay out. Every entrant on the board has results on all 38 index benchmarks. Point estimates only, no uncertainty intervals yet."
}