{ "schema_version": 1, "status": "complete_independently_audited", "model": "New Open-Jev 27B on community-hard-mix-v2", "optimizer_steps": 37160, "checkpoint": { "model": "Qwen/Qwen3.8-27B", "method": "lora_decision_head", "checkpoint_sha256": "c49994563c3c4f04a99d9130203c4e526f4ae5086c84deec57698d18cb652e71", "base_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", "temperature": 2.5343690298472983, "code_commit": "7ac6bab261bd97efc3d446152999e94ae432e5b0", "max_length": 16384 }, "scope": "231 public JevBench tasks only: 72 original, 48 easy, 111 hard. The private and judge tiers are unavailable; this is not the full 534-task benchmark and no full-benchmark composite is computed. ZefanCai Open-Jev uses Qwen LoRA weights and a decision head; it is a different system from the Kotoba and Codiv projects also called OpenJev.", "benchmark": { "upstream_commit": "f8ce71361165846101d02ebc83ad44e47ae44fc3", "input_sha256": "2e20ff5f94dd04d015b3a7b9abd955757b6c5be25b55b97fcd141cb52f88b4aa", "dataset_sha256": "dc3995d8ae1e2fc8e81ce38431add509eb8bb39b85aadfd0c7c32079382dde51" }, "metrics": { "n_planned": 231, "n_attempted": 231, "n_scorable": 231, "n_valid": 231, "n_correct": 197, "accuracy": 0.8528138528138528, "schema_validity": 1.0, "schema_validity_strict": 1.0, "n_renormalized": 0, "operational_success": 1.0, "brier_mean": 0.2419831167860275, "ordinal_mae": 0.3798003972534468 }, "per_public_tier": { "original": { "n_planned": 72, "n_attempted": 72, "n_valid": 72, "n_correct": 69, "accuracy": 0.9583333333333334 }, "easy": { "n_planned": 48, "n_attempted": 48, "n_valid": 48, "n_correct": 48, "accuracy": 1.0 }, "hard": { "n_planned": 111, "n_attempted": 111, "n_valid": 111, "n_correct": 80, "accuracy": 0.7207207207207207 } }, "per_family": { "policy": { "n_planned": 12, "n_correct": 11, "accuracy": 0.9166666666666666 }, "intent": { "n_planned": 24, "n_correct": 23, "accuracy": 0.9583333333333334 }, "ordinal": { "n_planned": 12, "n_correct": 12, "accuracy": 1.0 }, "extraction": { "n_planned": 24, "n_correct": 23, "accuracy": 0.9583333333333334 }, "adequacy": { "n_planned": 12, "n_correct": 12, "accuracy": 1.0 }, "routing": { "n_planned": 12, "n_correct": 12, "accuracy": 1.0 }, "fact": { "n_planned": 12, "n_correct": 12, "accuracy": 1.0 }, "tool_selection": { "n_planned": 12, "n_correct": 12, "accuracy": 1.0 }, "long_policy": { "n_planned": 19, "n_correct": 11, "accuracy": 0.5789473684210527 }, "probability": { "n_planned": 10, "n_correct": 7, "accuracy": 0.7 }, "temporal_numeric": { "n_planned": 15, "n_correct": 2, "accuracy": 0.13333333333333333 }, "ambiguous": { "n_planned": 7, "n_correct": 5, "accuracy": 0.7142857142857143 }, "multi_hop": { "n_planned": 18, "n_correct": 16, "accuracy": 0.8888888888888888 }, "tradeoff": { "n_planned": 6, "n_correct": 5, "accuracy": 0.8333333333333334 }, "adversarial": { "n_planned": 6, "n_correct": 6, "accuracy": 1.0 }, "trap": { "n_planned": 8, "n_correct": 8, "accuracy": 1.0 }, "judge_hard": { "n_planned": 17, "n_correct": 15, "accuracy": 0.8823529411764706 }, "routing_hard": { "n_planned": 5, "n_correct": 5, "accuracy": 1.0 } }, "comparison_baselines": [ { "label": "Released Open-Jev 2B", "overall": { "n_correct": 150, "n_planned": 231, "accuracy": 0.6493506493506493 }, "hard": { "n_correct": 46, "n_planned": 111, "accuracy": 0.4144144144144144 } }, { "label": "Released Open-Jev 9B", "overall": { "n_correct": 179, "n_planned": 231, "accuracy": 0.7748917748917749 }, "hard": { "n_correct": 66, "n_planned": 111, "accuracy": 0.5945945945945946 } }, { "label": "Jev 1.13.0", "overall": { "n_correct": 200, "n_planned": 231, "accuracy": 0.8658008658008658 }, "hard": { "n_correct": 81, "n_planned": 111, "accuracy": 0.7297297297297297 } } ], "protocol": { "sum_tolerance_strict": 0.001, "sum_tolerance_rounding": 0.02, "argmax_tie_break": "lexicographically_smallest_label", "score_accuracy": "argmax_level_equals_gold_level", "probabilities": "native", "concurrency": 1, "retries": 0, "warmups": 0, "prefix_cache": false, "latency_note": "Four-rank frozen-base FSDP2 execution on a shared node; one observation per task. Timing starts after dispatch and includes collective prediction, rank validation and response validation. No HTTP/network timing, warmups or retries. These diagnostics are not comparable to the old single-GPU HTTP latency, matched-hardware speedups, throughput or energy efficiency.", "transport": "torch_distributed_collective", "execution": { "mode": "frozen_base_fsdp2", "world_size": 4, "backend": "nccl", "node_shared_with_other_jobs": true, "candidate_batch_size": 1, "http_server_started": false } }, "execution": { "mode": "frozen_base_fsdp2", "world_size": 4, "backend": "nccl", "node_shared_with_other_jobs": true, "candidate_batch_size": 1, "http_server_started": false }, "failed_requests": 0, "pending_requests": 0, "in_flight_requests": 0, "audit": { "file": "audit.json", "sha256": "0759d6d9d457e0b823f823d8d9e7b7c9f0f4f3a9278f6cd32aac1e7b442392fc" }, "source_summary_sha256": "ede7fda4e114c22bed02ec5a389a3bd4b9952eeeeed4a64e39cdce88d6589b8f", "limitations": [ "231 public tasks only; unavailable private/judge tasks were not scored.", "Four-rank shared-node direct-execution timings cannot be compared as speedups against the old single-GPU HTTP or hosted HTTPS timings.", "Checkpoint hash verification establishes artifact identity; this offline audit does not rerun model inference or establish benchmark-training semantic separation.", "Raw requests, labels, task IDs and responses remain in an ignored local run directory." ] }