TinyQuery-140M / provenance /teacher-runtime.json
karmx's picture
Release TinyQuery 139.7M from scratch with frozen weights, reproducible Mac evaluations and runtime source
b296ad4 verified
Raw
History Blame Contribute Delete
2.1 kB
{
"teacher": "Qwen/Qwen3.8-27B-FP8",
"revision": "017b9c7af6b5689d5dd426a76e0bc077eb5ca20a",
"server": {
"vllm": "0.28.0",
"hardware": "RTX PRO 6000 Blackwell Max-Q Workstation Edition, 96 GB",
"context": 8192,
"max_num_seqs": 32,
"gpu_memory_utilization": 0.85,
"language_model_only": true,
"fp8_backend": "CUTLASS"
},
"phases": [
{
"name": "language_templates",
"requests": 45,
"concurrency": 16,
"temperature": 0.8,
"top_p": 0.95,
"top_k": 20,
"thinking": false,
"mtp": false,
"max_tokens": 4096
},
{
"name": "direct_paraphrases",
"requests": 1500,
"concurrency": 32,
"temperature": 0.7,
"top_p": 0.95,
"top_k": 20,
"thinking": false,
"mtp": false,
"max_tokens": 500
},
{
"name": "round_trip_verification",
"requests": 1500,
"temperature": 0,
"thinking": false,
"mtp": true,
"num_speculative_tokens": 3,
"max_tokens": 700
},
{
"name": "training_template_semantic_audit",
"requests": 330,
"template_instances": 1313,
"concurrency": 32,
"temperature": 0,
"thinking": false,
"mtp": true,
"num_speculative_tokens": 3,
"max_tokens": 700,
"accepted_instances": 1276,
"flagged_instances": 37
}
],
"matched_benchmark": {
"concurrency": 32,
"mtp_off_tps": [
688.80098,
693.58296
],
"mtp_3_tps": [
926.42637,
935.17359
],
"warmup": "one full request round per configuration before measured rounds",
"measured_rounds_per_configuration": 2,
"requests_per_round": 64,
"valid_json_per_configuration": 128,
"truncated_per_configuration": 0,
"temperature": 0.7,
"top_p": 0.8,
"top_k": 20,
"presence_penalty": 0,
"repetition_penalty": 1.0
},
"scope": "MTP speeds inference; it does not certify semantic correctness. Initial template/paraphrase generation preceded MTP; verification used MTP. No claim of a globally optimal configuration."
}