oktayd's picture
Document completed Q4 llama.cpp diagnostic, including failures and unscored tasks
61cb76f verified
Raw History Blame Contribute Delete
6 kB
{
"generated_utc": "2026-09-04T23:47:51.958393+00:00",
"profile": {
"model": "Q36-v1.3-Q4_K_M.gguf",
"backend": "llama.cpp",
"llama_cpp_commit": "427291b5b34cd914a31b3fd3b61a68f6184f4b9f",
"manifest_sha256": "9490bd427291856b11b8a77efb20f69a0fa24d38831c2b2d8fb9735584754bfd",
"thinking": false,
"temperature": 0,
"seed": 42,
"max_new_tokens": 4096,
"context": 16384,
"wall_budget_minutes": 20.0,
"per_prompt_seconds": 45,
"note": "User-requested llama.cpp Q4 evaluation; not directly comparable to the earlier thinking-enabled BF16 run. Selected adapters still require official evaluators.",
"gpu_inventory": "NVIDIA H200, 570.124.06, 143771 MiB",
"llama_binary_version": "",
"cpu_threads": 8
},
"available_manifest_prompts": 204,
"attempted": 204,
"completed": 186,
"incomplete_or_error": 18,
"request_errors": 0,
"strict_auto_scored": 108,
"strict_auto_correct": 61,
"review_or_official_evaluator_pending": 78,
"elapsed_seconds": 474.9204415604472,
"by_benchmark": {
"MMLU-Pro": {
"attempted": 1,
"completed": 1,
"strict_scored": 1,
"strict_correct": 0,
"review_pending": 0
},
"BBH": {
"attempted": 23,
"completed": 20,
"strict_scored": 20,
"strict_correct": 3,
"review_pending": 0
},
"ARC-Challenge": {
"attempted": 10,
"completed": 8,
"strict_scored": 8,
"strict_correct": 8,
"review_pending": 0
},
"GSM8K": {
"attempted": 10,
"completed": 9,
"strict_scored": 9,
"strict_correct": 5,
"review_pending": 0
},
"MATH-Level-5": {
"attempted": 10,
"completed": 8,
"strict_scored": 0,
"strict_correct": 0,
"review_pending": 8
},
"GPQA-Diamond": {
"attempted": 5,
"completed": 5,
"strict_scored": 5,
"strict_correct": 2,
"review_pending": 0
},
"TruthfulQA": {
"attempted": 10,
"completed": 9,
"strict_scored": 9,
"strict_correct": 6,
"review_pending": 0
},
"HumanEval-Plus": {
"attempted": 5,
"completed": 5,
"strict_scored": 0,
"strict_correct": 0,
"review_pending": 5
},
"MBPP-Plus": {
"attempted": 5,
"completed": 5,
"strict_scored": 0,
"strict_correct": 0,
"review_pending": 5
},
"LiveCodeBench": {
"attempted": 9,
"completed": 9,
"strict_scored": 0,
"strict_correct": 0,
"review_pending": 9
},
"Q36-JSON-Schema": {
"attempted": 5,
"completed": 5,
"strict_scored": 5,
"strict_correct": 0,
"review_pending": 0
},
"Q36-Hermes-Tool-Format": {
"attempted": 5,
"completed": 5,
"strict_scored": 5,
"strict_correct": 5,
"review_pending": 0
},
"Q36-Agent-Function-Calling": {
"attempted": 5,
"completed": 5,
"strict_scored": 5,
"strict_correct": 0,
"review_pending": 0
},
"MMMU": {
"attempted": 6,
"completed": 6,
"strict_scored": 6,
"strict_correct": 4,
"review_pending": 0
},
"MathVista": {
"attempted": 5,
"completed": 5,
"strict_scored": 5,
"strict_correct": 2,
"review_pending": 0
},
"ChartQA": {
"attempted": 5,
"completed": 5,
"strict_scored": 5,
"strict_correct": 4,
"review_pending": 0
},
"Q36-Benign-Compliance-No-Overrefusal": {
"attempted": 20,
"completed": 18,
"strict_scored": 0,
"strict_correct": 0,
"review_pending": 18
},
"Q36-Contradiction-and-Anti-Sycophancy": {
"attempted": 10,
"completed": 8,
"strict_scored": 0,
"strict_correct": 0,
"review_pending": 8
},
"Q36-Answer-Termination-No-Looping": {
"attempted": 10,
"completed": 10,
"strict_scored": 10,
"strict_correct": 7,
"review_pending": 0
},
"Q36-CAPTCHA-Detection-and-Handoff": {
"attempted": 10,
"completed": 10,
"strict_scored": 10,
"strict_correct": 10,
"review_pending": 0
},
"IFEval": {
"attempted": 10,
"completed": 6,
"strict_scored": 0,
"strict_correct": 0,
"review_pending": 6
},
"Q36-Output-Integrity": {
"attempted": 5,
"completed": 5,
"strict_scored": 5,
"strict_correct": 5,
"review_pending": 0
},
"Q36-Legal-Alternatives-and-Boundaries": {
"attempted": 10,
"completed": 10,
"strict_scored": 0,
"strict_correct": 0,
"review_pending": 10
},
"Q36-Direct-Style-and-Personality": {
"attempted": 10,
"completed": 9,
"strict_scored": 0,
"strict_correct": 0,
"review_pending": 9
}
},
"median_tps_including_prefill_for_outputs_ge64_tokens": 156.35807976215648,
"speed_sample_count": 104,
"limitations": [
"Compact mixed diagnostic, not official benchmark leaderboard results.",
"No meaningful overall accuracy is claimed across heterogeneous tasks and incomplete evaluators.",
"Strict scores combine required answer content and output format; semantically correct but misformatted outputs can fail.",
"Coding execution tests, several mathematics tasks and subjective rubrics still require their evaluators/review.",
"Greedy decoding, thinking disabled. Repetition/token-limit failures remain visible; the Transformers helper is not active in llama.cpp.",
"Speed includes prefill and is workload dependent, not a pure decode microbenchmark.",
"Only Q4_K_M on H200 was benchmarked here. This does not establish BF16/other-quant or other-device quality/performance.",
"No completed original-Huihui comparison; no claim of outperforming the source model.",
"204 available prompts from the larger proposed matrix; unavailable adapters are documented in the private preparation report."
]
}