{ "teacher": "Qwen/Qwen3.8-27B-FP8", "revision": "017b9c7af6b5689d5dd426a76e0bc077eb5ca20a", "server": { "vllm": "0.28.0", "hardware": "RTX PRO 6000 Blackwell Max-Q Workstation Edition, 96 GB", "context": 8192, "max_num_seqs": 32, "gpu_memory_utilization": 0.85, "language_model_only": true, "fp8_backend": "CUTLASS" }, "phases": [ { "name": "language_templates", "requests": 45, "concurrency": 16, "temperature": 0.8, "top_p": 0.95, "top_k": 20, "thinking": false, "mtp": false, "max_tokens": 4096 }, { "name": "direct_paraphrases", "requests": 1500, "concurrency": 32, "temperature": 0.7, "top_p": 0.95, "top_k": 20, "thinking": false, "mtp": false, "max_tokens": 500 }, { "name": "round_trip_verification", "requests": 1500, "temperature": 0, "thinking": false, "mtp": true, "num_speculative_tokens": 3, "max_tokens": 700 }, { "name": "training_template_semantic_audit", "requests": 330, "template_instances": 1313, "concurrency": 32, "temperature": 0, "thinking": false, "mtp": true, "num_speculative_tokens": 3, "max_tokens": 700, "accepted_instances": 1276, "flagged_instances": 37 } ], "matched_benchmark": { "concurrency": 32, "mtp_off_tps": [ 688.80098, 693.58296 ], "mtp_3_tps": [ 926.42637, 935.17359 ], "warmup": "one full request round per configuration before measured rounds", "measured_rounds_per_configuration": 2, "requests_per_round": 64, "valid_json_per_configuration": 128, "truncated_per_configuration": 0, "temperature": 0.7, "top_p": 0.8, "top_k": 20, "presence_penalty": 0, "repetition_penalty": 1.0 }, "scope": "MTP speeds inference; it does not certify semantic correctness. Initial template/paraphrase generation preceded MTP; verification used MTP. No claim of a globally optimal configuration." }