{ "scope": "Inference support only; model weights unchanged", "unit_tests": "PASS", "selected_cases": 9, "tests": [ { "id": "MMLU-Pro:001", "thinking": false, "completed": true, "finish_reason": "eos", "output_tokens": 50, "seconds": 2.9331285804510117 }, { "id": "HumanEval-Plus:001", "thinking": false, "completed": true, "finish_reason": "eos", "output_tokens": 233, "seconds": 10.614582307636738 }, { "id": "MATH-Level-5:002", "thinking": false, "completed": true, "finish_reason": "eos", "output_tokens": 116, "seconds": 5.350349590182304 }, { "id": "GSM8K:001", "thinking": false, "completed": true, "finish_reason": "eos", "output_tokens": 111, "seconds": 5.108682680875063 }, { "id": "Q36-Hermes-Tool-Format:001", "thinking": false, "completed": true, "finish_reason": "eos", "output_tokens": 41, "seconds": 1.973704420030117 }, { "id": "Q36-Output-Integrity:001", "thinking": false, "completed": true, "finish_reason": "eos", "output_tokens": 2, "seconds": 0.2203267514705658 }, { "id": "MMLU-Pro:001", "thinking": true, "completed": false, "finish_reason": "repeated_block", "output_tokens": 432, "seconds": 19.451829474419355 }, { "id": "HumanEval-Plus:001", "thinking": true, "completed": false, "finish_reason": "time_limit", "output_tokens": 1005, "seconds": 45.0059671998024 }, { "id": "MATH-Level-5:002", "thinking": true, "completed": false, "finish_reason": "repeated_block", "output_tokens": 352, "seconds": 15.891684971749783 } ], "limitations": "Selected known failures plus regressions; not independent benchmark accuracy; raw diagnostic outputs are kept privately." }