{ "schema": "br-voice-reasoner-evaluation-protocol.v1", "evaluation_date_utc": "2026-09-02", "benchmark": { "name": "VoiceBench", "upstream_repository": "https://github.com/matthewcym/VoiceBench", "snapshot_identity": "SHA256 manifest", "git_revision": null, "revision_note": "The evaluated local snapshot did not retain Git metadata; release-relevant files are identified by SHA256." }, "subset_samples": { "wildvoice": 1000, "bbh": 1000, "alpacaeval_full": 636, "mmsu": 3074, "openbookqa": 455, "ifeval": 345, "advbench": 520, "commoneval": 200, "sdqa_usa": 553, "total": 7783 }, "generation": { "reasoning_enabled": true, "temperature": 0.6, "top_p": 0.95, "top_k": 20, "maximum_tokens": 16384, "seed_policy": "A deterministic per-sample primary seed is derived from the input and base seed 1234.", "empty_final_policy": { "primary_attempts": 3, "maximum_consecutive_fallback_seeds": 8, "selection": "First non-empty visible final answer; no ranking or best-of selection.", "samples_entering_seed_fallback": 5, "samples_recovered": 4, "samples_exhausted": 1, "exhausted_record": "No answer.", "exhausted_scoring": "incorrect" } }, "model_judge": { "requested_model": "openai/gpt-4o-mini", "model_kind": "alias", "accessed_utc": "2026-09-02", "gateway": "OpenRouter", "provider_only": "OpenAI", "allow_provider_fallbacks": false, "judgments_per_sample": 3, "vote_mode": "three separate requests without a seed", "temperature": 0.5, "top_p": 0.95, "maximum_tokens": 1024 }, "aggregation": "Open-ended ratings are normalized to 0-100. Overall is the arithmetic mean of nine 0-100 subset scores.", "sha256": { "main.py": "cb7c86d7149b76d50e3316d09a377fb7cfa6549855ad2922f86b7dd980416f3e", "evaluate.py": "0b610247b1373f2c256a4f183c2dfcbe7c562b576823cf4ce1396bf036ba1f42", "src/models/qwen3_omni_thinking.py": "f44bcbc3c9b866c13b4db33ff156026dcf136f7fcfbd223ea006e600986543d8", "qwen3omni_thinking/run_voicebench.sh": "4f040e0f695a12ee6a9fe4e1164c6fe3e57582160d5afca99d975f6ce7a7226c", "qwen3omni_thinking/audit_outputs.py": "3ae9335dff1d568d19dc395bdf2b352b6a81685a9128edccdeca3882bdee7fb0", "qwen3omni_thinking/score_local.sh": "332d701640ee1d76b3d7566e7250c55055546b3aa8186cd443531886635511ac", "qwen3omni_thinking/score_thinking_mcq.py": "505018a7ae166f71f1a1d19b99887f69adc7fc802f91ceca930dbf2349f858eb", "qwen3omni_thinking/api_judge_openrouter.py": "5b721566c1095e5dcf83b909f6ea667d127aed2533315b73f0819d168071ec33", "qwen3omni_thinking/repro_utils.py": "9b36582a40c858f9ae6f5c64f7230dd7cf7db2e868d4f2bd77c446e70ea1a046" } }