ZefanCai's picture
Release audited Open-Jev 27B v1.1 adapter and decision head
28cf730 verified
Raw History Blame Contribute Delete
6.69 kB
{
"schema_version": 1,
"status": "complete_independently_audited",
"model": "New Open-Jev 27B on community-hard-mix-v2",
"optimizer_steps": 37160,
"checkpoint": {
"model": "Qwen/Qwen3.8-27B",
"method": "lora_decision_head",
"checkpoint_sha256": "c49994563c3c4f04a99d9130203c4e526f4ae5086c84deec57698d18cb652e71",
"base_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0",
"temperature": 2.5343690298472983,
"code_commit": "7ac6bab261bd97efc3d446152999e94ae432e5b0",
"max_length": 16384
},
"scope": "231 public JevBench tasks only: 72 original, 48 easy, 111 hard. The private and judge tiers are unavailable; this is not the full 534-task benchmark and no full-benchmark composite is computed. ZefanCai Open-Jev uses Qwen LoRA weights and a decision head; it is a different system from the Kotoba and Codiv projects also called OpenJev.",
"benchmark": {
"upstream_commit": "f8ce71361165846101d02ebc83ad44e47ae44fc3",
"input_sha256": "2e20ff5f94dd04d015b3a7b9abd955757b6c5be25b55b97fcd141cb52f88b4aa",
"dataset_sha256": "dc3995d8ae1e2fc8e81ce38431add509eb8bb39b85aadfd0c7c32079382dde51"
},
"metrics": {
"n_planned": 231,
"n_attempted": 231,
"n_scorable": 231,
"n_valid": 231,
"n_correct": 197,
"accuracy": 0.8528138528138528,
"schema_validity": 1.0,
"schema_validity_strict": 1.0,
"n_renormalized": 0,
"operational_success": 1.0,
"brier_mean": 0.2419831167860275,
"ordinal_mae": 0.3798003972534468
},
"per_public_tier": {
"original": {
"n_planned": 72,
"n_attempted": 72,
"n_valid": 72,
"n_correct": 69,
"accuracy": 0.9583333333333334
},
"easy": {
"n_planned": 48,
"n_attempted": 48,
"n_valid": 48,
"n_correct": 48,
"accuracy": 1.0
},
"hard": {
"n_planned": 111,
"n_attempted": 111,
"n_valid": 111,
"n_correct": 80,
"accuracy": 0.7207207207207207
}
},
"per_family": {
"policy": {
"n_planned": 12,
"n_correct": 11,
"accuracy": 0.9166666666666666
},
"intent": {
"n_planned": 24,
"n_correct": 23,
"accuracy": 0.9583333333333334
},
"ordinal": {
"n_planned": 12,
"n_correct": 12,
"accuracy": 1.0
},
"extraction": {
"n_planned": 24,
"n_correct": 23,
"accuracy": 0.9583333333333334
},
"adequacy": {
"n_planned": 12,
"n_correct": 12,
"accuracy": 1.0
},
"routing": {
"n_planned": 12,
"n_correct": 12,
"accuracy": 1.0
},
"fact": {
"n_planned": 12,
"n_correct": 12,
"accuracy": 1.0
},
"tool_selection": {
"n_planned": 12,
"n_correct": 12,
"accuracy": 1.0
},
"long_policy": {
"n_planned": 19,
"n_correct": 11,
"accuracy": 0.5789473684210527
},
"probability": {
"n_planned": 10,
"n_correct": 7,
"accuracy": 0.7
},
"temporal_numeric": {
"n_planned": 15,
"n_correct": 2,
"accuracy": 0.13333333333333333
},
"ambiguous": {
"n_planned": 7,
"n_correct": 5,
"accuracy": 0.7142857142857143
},
"multi_hop": {
"n_planned": 18,
"n_correct": 16,
"accuracy": 0.8888888888888888
},
"tradeoff": {
"n_planned": 6,
"n_correct": 5,
"accuracy": 0.8333333333333334
},
"adversarial": {
"n_planned": 6,
"n_correct": 6,
"accuracy": 1.0
},
"trap": {
"n_planned": 8,
"n_correct": 8,
"accuracy": 1.0
},
"judge_hard": {
"n_planned": 17,
"n_correct": 15,
"accuracy": 0.8823529411764706
},
"routing_hard": {
"n_planned": 5,
"n_correct": 5,
"accuracy": 1.0
}
},
"comparison_baselines": [
{
"label": "Released Open-Jev 2B",
"overall": {
"n_correct": 150,
"n_planned": 231,
"accuracy": 0.6493506493506493
},
"hard": {
"n_correct": 46,
"n_planned": 111,
"accuracy": 0.4144144144144144
}
},
{
"label": "Released Open-Jev 9B",
"overall": {
"n_correct": 179,
"n_planned": 231,
"accuracy": 0.7748917748917749
},
"hard": {
"n_correct": 66,
"n_planned": 111,
"accuracy": 0.5945945945945946
}
},
{
"label": "Jev 1.13.0",
"overall": {
"n_correct": 200,
"n_planned": 231,
"accuracy": 0.8658008658008658
},
"hard": {
"n_correct": 81,
"n_planned": 111,
"accuracy": 0.7297297297297297
}
}
],
"protocol": {
"sum_tolerance_strict": 0.001,
"sum_tolerance_rounding": 0.02,
"argmax_tie_break": "lexicographically_smallest_label",
"score_accuracy": "argmax_level_equals_gold_level",
"probabilities": "native",
"concurrency": 1,
"retries": 0,
"warmups": 0,
"prefix_cache": false,
"latency_note": "Four-rank frozen-base FSDP2 execution on a shared node; one observation per task. Timing starts after dispatch and includes collective prediction, rank validation and response validation. No HTTP/network timing, warmups or retries. These diagnostics are not comparable to the old single-GPU HTTP latency, matched-hardware speedups, throughput or energy efficiency.",
"transport": "torch_distributed_collective",
"execution": {
"mode": "frozen_base_fsdp2",
"world_size": 4,
"backend": "nccl",
"node_shared_with_other_jobs": true,
"candidate_batch_size": 1,
"http_server_started": false
}
},
"execution": {
"mode": "frozen_base_fsdp2",
"world_size": 4,
"backend": "nccl",
"node_shared_with_other_jobs": true,
"candidate_batch_size": 1,
"http_server_started": false
},
"failed_requests": 0,
"pending_requests": 0,
"in_flight_requests": 0,
"audit": {
"file": "audit.json",
"sha256": "0759d6d9d457e0b823f823d8d9e7b7c9f0f4f3a9278f6cd32aac1e7b442392fc"
},
"source_summary_sha256": "ede7fda4e114c22bed02ec5a389a3bd4b9952eeeeed4a64e39cdce88d6589b8f",
"limitations": [
"231 public tasks only; unavailable private/judge tasks were not scored.",
"Four-rank shared-node direct-execution timings cannot be compared as speedups against the old single-GPU HTTP or hosted HTTPS timings.",
"Checkpoint hash verification establishes artifact identity; this offline audit does not rerun model inference or establish benchmark-training semantic separation.",
"Raw requests, labels, task IDs and responses remain in an ignored local run directory."
]
}