{ "generated_utc": "2026-09-04T23:47:51.958393+00:00", "profile": { "model": "Q36-v1.3-Q4_K_M.gguf", "backend": "llama.cpp", "llama_cpp_commit": "427291b5b34cd914a31b3fd3b61a68f6184f4b9f", "manifest_sha256": "9490bd427291856b11b8a77efb20f69a0fa24d38831c2b2d8fb9735584754bfd", "thinking": false, "temperature": 0, "seed": 42, "max_new_tokens": 4096, "context": 16384, "wall_budget_minutes": 20.0, "per_prompt_seconds": 45, "note": "User-requested llama.cpp Q4 evaluation; not directly comparable to the earlier thinking-enabled BF16 run. Selected adapters still require official evaluators.", "gpu_inventory": "NVIDIA H200, 570.124.06, 143771 MiB", "llama_binary_version": "", "cpu_threads": 8 }, "available_manifest_prompts": 204, "attempted": 204, "completed": 186, "incomplete_or_error": 18, "request_errors": 0, "strict_auto_scored": 108, "strict_auto_correct": 61, "review_or_official_evaluator_pending": 78, "elapsed_seconds": 474.9204415604472, "by_benchmark": { "MMLU-Pro": { "attempted": 1, "completed": 1, "strict_scored": 1, "strict_correct": 0, "review_pending": 0 }, "BBH": { "attempted": 23, "completed": 20, "strict_scored": 20, "strict_correct": 3, "review_pending": 0 }, "ARC-Challenge": { "attempted": 10, "completed": 8, "strict_scored": 8, "strict_correct": 8, "review_pending": 0 }, "GSM8K": { "attempted": 10, "completed": 9, "strict_scored": 9, "strict_correct": 5, "review_pending": 0 }, "MATH-Level-5": { "attempted": 10, "completed": 8, "strict_scored": 0, "strict_correct": 0, "review_pending": 8 }, "GPQA-Diamond": { "attempted": 5, "completed": 5, "strict_scored": 5, "strict_correct": 2, "review_pending": 0 }, "TruthfulQA": { "attempted": 10, "completed": 9, "strict_scored": 9, "strict_correct": 6, "review_pending": 0 }, "HumanEval-Plus": { "attempted": 5, "completed": 5, "strict_scored": 0, "strict_correct": 0, "review_pending": 5 }, "MBPP-Plus": { "attempted": 5, "completed": 5, "strict_scored": 0, "strict_correct": 0, "review_pending": 5 }, "LiveCodeBench": { "attempted": 9, "completed": 9, "strict_scored": 0, "strict_correct": 0, "review_pending": 9 }, "Q36-JSON-Schema": { "attempted": 5, "completed": 5, "strict_scored": 5, "strict_correct": 0, "review_pending": 0 }, "Q36-Hermes-Tool-Format": { "attempted": 5, "completed": 5, "strict_scored": 5, "strict_correct": 5, "review_pending": 0 }, "Q36-Agent-Function-Calling": { "attempted": 5, "completed": 5, "strict_scored": 5, "strict_correct": 0, "review_pending": 0 }, "MMMU": { "attempted": 6, "completed": 6, "strict_scored": 6, "strict_correct": 4, "review_pending": 0 }, "MathVista": { "attempted": 5, "completed": 5, "strict_scored": 5, "strict_correct": 2, "review_pending": 0 }, "ChartQA": { "attempted": 5, "completed": 5, "strict_scored": 5, "strict_correct": 4, "review_pending": 0 }, "Q36-Benign-Compliance-No-Overrefusal": { "attempted": 20, "completed": 18, "strict_scored": 0, "strict_correct": 0, "review_pending": 18 }, "Q36-Contradiction-and-Anti-Sycophancy": { "attempted": 10, "completed": 8, "strict_scored": 0, "strict_correct": 0, "review_pending": 8 }, "Q36-Answer-Termination-No-Looping": { "attempted": 10, "completed": 10, "strict_scored": 10, "strict_correct": 7, "review_pending": 0 }, "Q36-CAPTCHA-Detection-and-Handoff": { "attempted": 10, "completed": 10, "strict_scored": 10, "strict_correct": 10, "review_pending": 0 }, "IFEval": { "attempted": 10, "completed": 6, "strict_scored": 0, "strict_correct": 0, "review_pending": 6 }, "Q36-Output-Integrity": { "attempted": 5, "completed": 5, "strict_scored": 5, "strict_correct": 5, "review_pending": 0 }, "Q36-Legal-Alternatives-and-Boundaries": { "attempted": 10, "completed": 10, "strict_scored": 0, "strict_correct": 0, "review_pending": 10 }, "Q36-Direct-Style-and-Personality": { "attempted": 10, "completed": 9, "strict_scored": 0, "strict_correct": 0, "review_pending": 9 } }, "median_tps_including_prefill_for_outputs_ge64_tokens": 156.35807976215648, "speed_sample_count": 104, "limitations": [ "Compact mixed diagnostic, not official benchmark leaderboard results.", "No meaningful overall accuracy is claimed across heterogeneous tasks and incomplete evaluators.", "Strict scores combine required answer content and output format; semantically correct but misformatted outputs can fail.", "Coding execution tests, several mathematics tasks and subjective rubrics still require their evaluators/review.", "Greedy decoding, thinking disabled. Repetition/token-limit failures remain visible; the Transformers helper is not active in llama.cpp.", "Speed includes prefill and is workload dependent, not a pure decode microbenchmark.", "Only Q4_K_M on H200 was benchmarked here. This does not establish BF16/other-quant or other-device quality/performance.", "No completed original-Huihui comparison; no claim of outperforming the source model.", "204 available prompts from the larger proposed matrix; unavailable adapters are documented in the private preparation report." ] }