diff --git "a/benchmark/results/full-sweep-r5.raw.jsonl" "b/benchmark/results/full-sweep-r5.raw.jsonl" new file mode 100644--- /dev/null +++ "b/benchmark/results/full-sweep-r5.raw.jsonl" @@ -0,0 +1,350 @@ +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 12722.21091325, "output_tokens": 2584451, "output_toks_per_s": 203.14480066576567, "sample_count": 1319, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 12506.582732429997, "output_tokens": 2584451, "output_toks_per_s": 206.64725571266004, "sample_count": 1319, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 12481.399236264995, "output_tokens": 2584451, "output_toks_per_s": 207.06420418720504, "sample_count": 1319, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 12770.057838322999, "output_tokens": 2584451, "output_toks_per_s": 202.38365657546603, "sample_count": 1319, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 12697.651651029002, "output_tokens": 2584451, "output_toks_per_s": 203.53771477031813, "sample_count": 1319, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 999.5675302349991, "output_tokens": 2595268, "output_toks_per_s": 2596.3908605453103, "sample_count": 1319, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 1000.1154154360011, "output_tokens": 2593577, "output_toks_per_s": 2593.2776957240762, "sample_count": 1319, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 1000.0563899379995, "output_tokens": 2597254, "output_toks_per_s": 2597.107549266319, "sample_count": 1319, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 1002.5276272070005, "output_tokens": 2603399, "output_toks_per_s": 2596.835168775308, "sample_count": 1319, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 998.4887380389991, "output_tokens": 2600752, "output_toks_per_s": 2604.688366448475, "sample_count": 1319, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 8133.358541389003, "output_tokens": 1641100, "output_toks_per_s": 201.77396479557328, "sample_count": 500, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 7982.634443843999, "output_tokens": 1641100, "output_toks_per_s": 205.58375954013198, "sample_count": 500, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 7954.144333613001, "output_tokens": 1641100, "output_toks_per_s": 206.3201183142933, "sample_count": 500, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 8108.434943158001, "output_tokens": 1641100, "output_toks_per_s": 202.3941748937359, "sample_count": 500, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 7965.49525855, "output_tokens": 1641100, "output_toks_per_s": 206.02610970591903, "sample_count": 500, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1880.6364478540017, "output_tokens": 4913569, "output_toks_per_s": 2612.7160332380477, "sample_count": 1500, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1872.7447239020003, "output_tokens": 4913541, "output_toks_per_s": 2623.7110361535442, "sample_count": 1500, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1874.0331250880008, "output_tokens": 4897547, "output_toks_per_s": 2613.3726957307763, "sample_count": 1500, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1875.7614257550013, "output_tokens": 4904269, "output_toks_per_s": 2614.5483816130895, "sample_count": 1500, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1881.8707376310012, "output_tokens": 4917547, "output_toks_per_s": 2613.116247394584, "sample_count": 1500, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 703.8517636910001, "output_tokens": 143806, "output_toks_per_s": 204.31290708981254, "sample_count": 164, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 702.3662257769975, "output_tokens": 143806, "output_toks_per_s": 204.7450385885421, "sample_count": 164, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 721.5811043880003, "output_tokens": 143806, "output_toks_per_s": 199.2929126407311, "sample_count": 164, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 715.481437081995, "output_tokens": 143806, "output_toks_per_s": 200.99193710251308, "sample_count": 164, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 701.686428560999, "output_tokens": 143806, "output_toks_per_s": 204.94339657518208, "sample_count": 164, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 446.34307147300206, "output_tokens": 1092698, "output_toks_per_s": 2448.112382239799, "sample_count": 1148, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 445.1544392169999, "output_tokens": 1094255, "output_toks_per_s": 2458.1468892565226, "sample_count": 1148, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 441.4764698280014, "output_tokens": 1079393, "output_toks_per_s": 2444.9615636831786, "sample_count": 1148, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 441.53533940599664, "output_tokens": 1086834, "output_toks_per_s": 2461.4881369679997, "sample_count": 1148, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 447.2572925120003, "output_tokens": 1096197, "output_toks_per_s": 2450.931529463185, "sample_count": 1148, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 1452.3421888949997, "output_tokens": 292967, "output_toks_per_s": 201.72036744515498, "sample_count": 257, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 1416.5146230989994, "output_tokens": 292967, "output_toks_per_s": 206.82243248506492, "sample_count": 257, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 1420.4301267539995, "output_tokens": 292967, "output_toks_per_s": 206.25231363509243, "sample_count": 257, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 1449.5309404729996, "output_tokens": 292967, "output_toks_per_s": 202.1115878384778, "sample_count": 257, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 1420.5681903099976, "output_tokens": 292967, "output_toks_per_s": 206.2322681856395, "sample_count": 257, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 460.71360322099645, "output_tokens": 1181780, "output_toks_per_s": 2565.1076758702093, "sample_count": 1028, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 458.9379220619994, "output_tokens": 1181640, "output_toks_per_s": 2574.727306671268, "sample_count": 1028, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 454.732926785, "output_tokens": 1182988, "output_toks_per_s": 2601.500639867503, "sample_count": 1028, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 453.398811938001, "output_tokens": 1178403, "output_toks_per_s": 2599.0429815266875, "sample_count": 1028, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 459.939580774997, "output_tokens": 1182634, "output_toks_per_s": 2571.281206125519, "sample_count": 1028, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 1840.1863868889996, "output_tokens": 372583, "output_toks_per_s": 202.4702511955243, "sample_count": 80, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 1840.9798910300015, "output_tokens": 372583, "output_toks_per_s": 202.382981919235, "sample_count": 80, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 1847.3243596889988, "output_tokens": 372583, "output_toks_per_s": 201.68791584750454, "sample_count": 80, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 1857.880140759, "output_tokens": 372583, "output_toks_per_s": 200.54200043700808, "sample_count": 80, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "baseline", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 1808.1864082360007, "output_tokens": 372583, "output_toks_per_s": 206.05342364202266, "sample_count": 80, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 1034.0258575810003, "output_tokens": 2615845, "output_toks_per_s": 2529.767491617189, "sample_count": 560, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 1034.976310029997, "output_tokens": 2630253, "output_toks_per_s": 2541.365415333774, "sample_count": 560, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 1037.2291160709974, "output_tokens": 2635678, "output_toks_per_s": 2541.075987129916, "sample_count": 560, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 1044.7758614669947, "output_tokens": 2653525, "output_toks_per_s": 2539.8031270306365, "sample_count": 560, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"kind": "baseline"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "baseline", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 1027.3922799569991, "output_tokens": 2603811, "output_toks_per_s": 2534.3883254690027, "sample_count": 560, "spec_accept_length": null, "spec_verify_ct_sum": 0}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 5354.180563369999, "output_tokens": 2584201, "output_toks_per_s": 482.65107413065397, "sample_count": 1319, "spec_accept_length": 3.5782057762591424, "spec_verify_ct_sum": 741348}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 5333.134350584998, "output_tokens": 2584201, "output_toks_per_s": 484.55576591962955, "sample_count": 1319, "spec_accept_length": 3.5782057762591424, "spec_verify_ct_sum": 741348}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 5330.768447383998, "output_tokens": 2584201, "output_toks_per_s": 484.770821600432, "sample_count": 1319, "spec_accept_length": 3.5782057762591424, "spec_verify_ct_sum": 741348}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 5324.604112124001, "output_tokens": 2584201, "output_toks_per_s": 485.3320445206121, "sample_count": 1319, "spec_accept_length": 3.5782057762591424, "spec_verify_ct_sum": 741348}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 5325.310943021001, "output_tokens": 2584201, "output_toks_per_s": 485.2676261818444, "sample_count": 1319, "spec_accept_length": 3.5782057762591424, "spec_verify_ct_sum": 741348}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 583.8059294960003, "output_tokens": 2610594, "output_toks_per_s": 4471.681201069208, "sample_count": 1319, "spec_accept_length": 3.574944096945126, "spec_verify_ct_sum": 749603}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 585.0870216010007, "output_tokens": 2618974, "output_toks_per_s": 4476.21277401365, "sample_count": 1319, "spec_accept_length": 3.5738960817603576, "spec_verify_ct_sum": 752086}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 581.7585581009989, "output_tokens": 2582921, "output_toks_per_s": 4439.850456916836, "sample_count": 1319, "spec_accept_length": 3.577482366150014, "spec_verify_ct_sum": 740829}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 585.6102675430011, "output_tokens": 2617682, "output_toks_per_s": 4470.0070082152115, "sample_count": 1319, "spec_accept_length": 3.574557513555571, "spec_verify_ct_sum": 752099}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 573.9988309589971, "output_tokens": 2556383, "output_toks_per_s": 4453.637990392723, "sample_count": 1319, "spec_accept_length": 3.5789975360699815, "spec_verify_ct_sum": 732778}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 3248.724171488, "output_tokens": 1638735, "output_toks_per_s": 504.42417192020855, "sample_count": 500, "spec_accept_length": 3.6360816058326244, "spec_verify_ct_sum": 452103}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 3246.597874975996, "output_tokens": 1638735, "output_toks_per_s": 504.7545347796164, "sample_count": 500, "spec_accept_length": 3.6360816058326244, "spec_verify_ct_sum": 452103}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 3246.538869027998, "output_tokens": 1638735, "output_toks_per_s": 504.7637087094637, "sample_count": 500, "spec_accept_length": 3.6360816058326244, "spec_verify_ct_sum": 452103}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 3232.6523394979995, "output_tokens": 1638735, "output_toks_per_s": 506.93202605711696, "sample_count": 500, "spec_accept_length": 3.6360816058326244, "spec_verify_ct_sum": 452103}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 3243.56899549, "output_tokens": 1638735, "output_toks_per_s": 505.22587997313104, "sample_count": 500, "spec_accept_length": 3.6360816058326244, "spec_verify_ct_sum": 452103}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1032.2871816900006, "output_tokens": 5004115, "output_toks_per_s": 4847.599668735161, "sample_count": 1500, "spec_accept_length": 3.622878372167905, "spec_verify_ct_sum": 1386075}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1028.5204133379957, "output_tokens": 4977064, "output_toks_per_s": 4839.052230229699, "sample_count": 1500, "spec_accept_length": 3.62411954350306, "spec_verify_ct_sum": 1378167}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1021.2755576160016, "output_tokens": 4949186, "output_toks_per_s": 4846.082884381423, "sample_count": 1500, "spec_accept_length": 3.625299287483131, "spec_verify_ct_sum": 1370238}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1030.898002078, "output_tokens": 4965378, "output_toks_per_s": 4816.5560414233, "sample_count": 1500, "spec_accept_length": 3.626102913153403, "spec_verify_ct_sum": 1374723}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1017.8732070610022, "output_tokens": 4930415, "output_toks_per_s": 4843.840043924562, "sample_count": 1500, "spec_accept_length": 3.629116022651079, "spec_verify_ct_sum": 1363501}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 290.618083545, "output_tokens": 140780, "output_toks_per_s": 484.41582947195127, "sample_count": 164, "spec_accept_length": 3.642721763935972, "spec_verify_ct_sum": 39031}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 291.8244459960006, "output_tokens": 140780, "output_toks_per_s": 482.41332051369466, "sample_count": 164, "spec_accept_length": 3.642721763935972, "spec_verify_ct_sum": 39031}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 291.8714304779969, "output_tokens": 140780, "output_toks_per_s": 482.33566323858776, "sample_count": 164, "spec_accept_length": 3.642721763935972, "spec_verify_ct_sum": 39031}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 291.6532769319999, "output_tokens": 140780, "output_toks_per_s": 482.69644517940185, "sample_count": 164, "spec_accept_length": 3.642721763935972, "spec_verify_ct_sum": 39031}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 291.4707623960021, "output_tokens": 140780, "output_toks_per_s": 482.99870231488774, "sample_count": 164, "spec_accept_length": 3.642721763935972, "spec_verify_ct_sum": 39031}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 279.57132642499346, "output_tokens": 1076436, "output_toks_per_s": 3850.30902047388, "sample_count": 1148, "spec_accept_length": 3.643797616469542, "spec_verify_ct_sum": 297721}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 271.983096502001, "output_tokens": 1047773, "output_toks_per_s": 3852.34602251207, "sample_count": 1148, "spec_accept_length": 3.6429972296172006, "spec_verify_ct_sum": 289761}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 268.3650091290001, "output_tokens": 1033061, "output_toks_per_s": 3849.462354846041, "sample_count": 1148, "spec_accept_length": 3.6331350208532927, "spec_verify_ct_sum": 286101}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 279.11101170399706, "output_tokens": 1057479, "output_toks_per_s": 3788.7398047966594, "sample_count": 1148, "spec_accept_length": 3.640198362458388, "spec_verify_ct_sum": 292465}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 272.3255532650019, "output_tokens": 1048378, "output_toks_per_s": 3849.7231986886522, "sample_count": 1148, "spec_accept_length": 3.638702056536006, "spec_verify_ct_sum": 289619}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 591.7887732980016, "output_tokens": 287735, "output_toks_per_s": 486.2123328167767, "sample_count": 257, "spec_accept_length": 3.5323876083202626, "spec_verify_ct_sum": 81640}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 590.062971465999, "output_tokens": 287735, "output_toks_per_s": 487.63439482590894, "sample_count": 257, "spec_accept_length": 3.5323876083202626, "spec_verify_ct_sum": 81640}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 588.644635487999, "output_tokens": 287735, "output_toks_per_s": 488.8093471903664, "sample_count": 257, "spec_accept_length": 3.5323876083202626, "spec_verify_ct_sum": 81640}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 591.6887871829967, "output_tokens": 287735, "output_toks_per_s": 486.2944950670659, "sample_count": 257, "spec_accept_length": 3.5323876083202626, "spec_verify_ct_sum": 81640}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 589.6375679109988, "output_tokens": 287735, "output_toks_per_s": 487.98620654278153, "sample_count": 257, "spec_accept_length": 3.5323876083202626, "spec_verify_ct_sum": 81640}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 284.30833948200006, "output_tokens": 1177589, "output_toks_per_s": 4141.943223141207, "sample_count": 1028, "spec_accept_length": 3.5282962980493227, "spec_verify_ct_sum": 334037}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 284.1059694810001, "output_tokens": 1183536, "output_toks_per_s": 4165.825878850991, "sample_count": 1028, "spec_accept_length": 3.529977750277039, "spec_verify_ct_sum": 335772}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 283.46316691500033, "output_tokens": 1178761, "output_toks_per_s": 4158.427399329328, "sample_count": 1028, "spec_accept_length": 3.5287504177994324, "spec_verify_ct_sum": 334612}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 288.7872632159997, "output_tokens": 1182308, "output_toks_per_s": 4094.044823284632, "sample_count": 1028, "spec_accept_length": 3.5289650518054123, "spec_verify_ct_sum": 335365}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 289.99631136099924, "output_tokens": 1185721, "output_toks_per_s": 4088.7451100161275, "sample_count": 1028, "spec_accept_length": 3.5300240865618524, "spec_verify_ct_sum": 336383}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 793.948311741, "output_tokens": 352194, "output_toks_per_s": 443.5981471233255, "sample_count": 80, "spec_accept_length": 3.2436865756814495, "spec_verify_ct_sum": 110843}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 795.1692330570004, "output_tokens": 352194, "output_toks_per_s": 442.91703622133673, "sample_count": 80, "spec_accept_length": 3.2436865756814495, "spec_verify_ct_sum": 110843}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 795.5970919309998, "output_tokens": 352194, "output_toks_per_s": 442.67884281123656, "sample_count": 80, "spec_accept_length": 3.2436865756814495, "spec_verify_ct_sum": 110843}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 796.5986804819986, "output_tokens": 352194, "output_toks_per_s": 442.1222487927016, "sample_count": 80, "spec_accept_length": 3.2436865756814495, "spec_verify_ct_sum": 110843}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s3", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 794.8836558469993, "output_tokens": 352194, "output_toks_per_s": 443.0761626677489, "sample_count": 80, "spec_accept_length": 3.2436865756814495, "spec_verify_ct_sum": 110843}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 649.0264763209998, "output_tokens": 2589497, "output_toks_per_s": 3989.817202340555, "sample_count": 560, "spec_accept_length": 3.244127125926367, "spec_verify_ct_sum": 814514}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 644.770616763999, "output_tokens": 2613011, "output_toks_per_s": 4052.621090449633, "sample_count": 560, "spec_accept_length": 3.2478528389696253, "spec_verify_ct_sum": 820878}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 646.054375484, "output_tokens": 2588509, "output_toks_per_s": 4006.6426267305824, "sample_count": 560, "spec_accept_length": 3.2498232286952504, "spec_verify_ct_sum": 812801}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 650.470943103006, "output_tokens": 2628396, "output_toks_per_s": 4040.758511766109, "sample_count": 560, "spec_accept_length": 3.242813036625915, "spec_verify_ct_sum": 826528}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 3}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s3", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 640.6562864649968, "output_tokens": 2583794, "output_toks_per_s": 4033.0424512913437, "sample_count": 560, "spec_accept_length": 3.247589663539011, "spec_verify_ct_sum": 812584}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 4920.405756715998, "output_tokens": 2640209, "output_toks_per_s": 536.5835930088298, "sample_count": 1319, "spec_accept_length": 5.612090632830061, "spec_verify_ct_sum": 503221}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 4923.994516095001, "output_tokens": 2640209, "output_toks_per_s": 536.1925143031701, "sample_count": 1319, "spec_accept_length": 5.612090632830061, "spec_verify_ct_sum": 503221}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 4925.412250297, "output_tokens": 2640209, "output_toks_per_s": 536.0381762644937, "sample_count": 1319, "spec_accept_length": 5.612090632830061, "spec_verify_ct_sum": 503221}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 4910.333115772, "output_tokens": 2640209, "output_toks_per_s": 537.6842950877697, "sample_count": 1319, "spec_accept_length": 5.612090632830061, "spec_verify_ct_sum": 503221}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 4915.543710987, "output_tokens": 2640209, "output_toks_per_s": 537.1143367311992, "sample_count": 1319, "spec_accept_length": 5.612090632830061, "spec_verify_ct_sum": 503221}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 533.230014462999, "output_tokens": 2621796, "output_toks_per_s": 4916.820000540175, "sample_count": 1319, "spec_accept_length": 5.630051268155856, "spec_verify_ct_sum": 498182}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 546.3212425519998, "output_tokens": 2674924, "output_toks_per_s": 4896.247466975249, "sample_count": 1319, "spec_accept_length": 5.617811502777725, "spec_verify_ct_sum": 508851}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 534.3949004890019, "output_tokens": 2625370, "output_toks_per_s": 4912.79014376379, "sample_count": 1319, "spec_accept_length": 5.634709609361745, "spec_verify_ct_sum": 497816}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 538.9801835419998, "output_tokens": 2642942, "output_toks_per_s": 4903.597721592392, "sample_count": 1319, "spec_accept_length": 5.6222930411358565, "spec_verify_ct_sum": 503162}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 539.4720620700002, "output_tokens": 2635299, "output_toks_per_s": 4884.959176362411, "sample_count": 1319, "spec_accept_length": 5.625317541316592, "spec_verify_ct_sum": 501946}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2790.3101730810013, "output_tokens": 1643236, "output_toks_per_s": 588.9080059460106, "sample_count": 500, "spec_accept_length": 5.810442271539047, "spec_verify_ct_sum": 285934}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2789.288893265999, "output_tokens": 1643236, "output_toks_per_s": 589.1236307458719, "sample_count": 500, "spec_accept_length": 5.810442271539047, "spec_verify_ct_sum": 285934}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2777.7756029169996, "output_tokens": 1643236, "output_toks_per_s": 591.5654231660772, "sample_count": 500, "spec_accept_length": 5.810442271539047, "spec_verify_ct_sum": 285934}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2790.7259448070035, "output_tokens": 1643236, "output_toks_per_s": 588.8202684529957, "sample_count": 500, "spec_accept_length": 5.810442271539047, "spec_verify_ct_sum": 285934}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2792.8323830910003, "output_tokens": 1643236, "output_toks_per_s": 588.376162475361, "sample_count": 500, "spec_accept_length": 5.810442271539047, "spec_verify_ct_sum": 285934}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 891.0372662119989, "output_tokens": 4983510, "output_toks_per_s": 5592.931057963522, "sample_count": 1500, "spec_accept_length": 5.801760808540504, "spec_verify_ct_sum": 868645}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 885.0889599929999, "output_tokens": 4980424, "output_toks_per_s": 5627.032112161234, "sample_count": 1500, "spec_accept_length": 5.802931116635095, "spec_verify_ct_sum": 867972}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 892.807988949, "output_tokens": 4990019, "output_toks_per_s": 5589.128974836095, "sample_count": 1500, "spec_accept_length": 5.802603210339732, "spec_verify_ct_sum": 869568}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 888.1124704260001, "output_tokens": 4981704, "output_toks_per_s": 5609.316574071335, "sample_count": 1500, "spec_accept_length": 5.807372246903123, "spec_verify_ct_sum": 866992}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 889.2622857859997, "output_tokens": 4970616, "output_toks_per_s": 5589.594970404688, "sample_count": 1500, "spec_accept_length": 5.8023251041535815, "spec_verify_ct_sum": 866390}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 255.41948906999824, "output_tokens": 142501, "output_toks_per_s": 557.9096588081707, "sample_count": 164, "spec_accept_length": 5.892961907896723, "spec_verify_ct_sum": 24985}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 254.91092904999823, "output_tokens": 142501, "output_toks_per_s": 559.0227164095026, "sample_count": 164, "spec_accept_length": 5.892961907896723, "spec_verify_ct_sum": 24985}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 255.50906617099827, "output_tokens": 142501, "output_toks_per_s": 557.7140652403773, "sample_count": 164, "spec_accept_length": 5.892961907896723, "spec_verify_ct_sum": 24985}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 256.3401526890011, "output_tokens": 142501, "output_toks_per_s": 555.9058871782998, "sample_count": 164, "spec_accept_length": 5.892961907896723, "spec_verify_ct_sum": 24985}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 254.99527598200075, "output_tokens": 142501, "output_toks_per_s": 558.8378037640928, "sample_count": 164, "spec_accept_length": 5.892961907896723, "spec_verify_ct_sum": 24985}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 242.47895728799995, "output_tokens": 1053021, "output_toks_per_s": 4342.731475660767, "sample_count": 1148, "spec_accept_length": 5.904713806029675, "spec_verify_ct_sum": 182043}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 243.77045758999884, "output_tokens": 1046868, "output_toks_per_s": 4294.482647116915, "sample_count": 1148, "spec_accept_length": 5.877946209420865, "spec_verify_ct_sum": 181633}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 241.8981787900011, "output_tokens": 1051539, "output_toks_per_s": 4347.03148762799, "sample_count": 1148, "spec_accept_length": 5.883520932719905, "spec_verify_ct_sum": 181803}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 238.82196508000015, "output_tokens": 1052067, "output_toks_per_s": 4405.235505233283, "sample_count": 1148, "spec_accept_length": 5.866618091556483, "spec_verify_ct_sum": 182540}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 241.1943826409988, "output_tokens": 1048608, "output_toks_per_s": 4347.563937924627, "sample_count": 1148, "spec_accept_length": 5.865947023460028, "spec_verify_ct_sum": 181940}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 547.3375120020028, "output_tokens": 297657, "output_toks_per_s": 543.8271513882843, "sample_count": 257, "spec_accept_length": 5.35728336005909, "spec_verify_ct_sum": 55728}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 545.1661157260023, "output_tokens": 297657, "output_toks_per_s": 545.9932145702189, "sample_count": 257, "spec_accept_length": 5.35728336005909, "spec_verify_ct_sum": 55728}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 549.7132237350015, "output_tokens": 297657, "output_toks_per_s": 541.4768776664731, "sample_count": 257, "spec_accept_length": 5.35728336005909, "spec_verify_ct_sum": 55728}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 548.5233630549992, "output_tokens": 297657, "output_toks_per_s": 542.6514530615437, "sample_count": 257, "spec_accept_length": 5.35728336005909, "spec_verify_ct_sum": 55728}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 546.7907023689986, "output_tokens": 297657, "output_toks_per_s": 544.3709973677055, "sample_count": 257, "spec_accept_length": 5.35728336005909, "spec_verify_ct_sum": 55728}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 260.762322054994, "output_tokens": 1174125, "output_toks_per_s": 4502.663539529229, "sample_count": 1028, "spec_accept_length": 5.342423628299345, "spec_verify_ct_sum": 220927}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 266.44918018299995, "output_tokens": 1177849, "output_toks_per_s": 4420.539028084236, "sample_count": 1028, "spec_accept_length": 5.342117427340458, "spec_verify_ct_sum": 221644}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 266.6318452009982, "output_tokens": 1181977, "output_toks_per_s": 4432.992612375227, "sample_count": 1028, "spec_accept_length": 5.351551094322617, "spec_verify_ct_sum": 222014}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 263.6230149959997, "output_tokens": 1171874, "output_toks_per_s": 4445.2643864109605, "sample_count": 1028, "spec_accept_length": 5.344124574882167, "spec_verify_ct_sum": 220901}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 262.3023706959998, "output_tokens": 1181293, "output_toks_per_s": 4503.554416475638, "sample_count": 1028, "spec_accept_length": 5.348702599438006, "spec_verify_ct_sum": 222303}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 826.9652363810001, "output_tokens": 364560, "output_toks_per_s": 440.84078019458565, "sample_count": 80, "spec_accept_length": 4.566831861126655, "spec_verify_ct_sum": 85056}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 824.6043641599999, "output_tokens": 364560, "output_toks_per_s": 442.102923347206, "sample_count": 80, "spec_accept_length": 4.566831861126655, "spec_verify_ct_sum": 85056}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 829.0414189829999, "output_tokens": 364560, "output_toks_per_s": 439.7367750904561, "sample_count": 80, "spec_accept_length": 4.566831861126655, "spec_verify_ct_sum": 85056}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 825.1122672340061, "output_tokens": 364560, "output_toks_per_s": 441.83078409693417, "sample_count": 80, "spec_accept_length": 4.566831861126655, "spec_verify_ct_sum": 85056}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s7", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 822.6332684790032, "output_tokens": 364560, "output_toks_per_s": 443.16223762023185, "sample_count": 80, "spec_accept_length": 4.566831861126655, "spec_verify_ct_sum": 85056}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 652.6803587600007, "output_tokens": 2614309, "output_toks_per_s": 4005.4966645033, "sample_count": 560, "spec_accept_length": 4.5462857610925695, "spec_verify_ct_sum": 608467}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 639.1581992209976, "output_tokens": 2572432, "output_toks_per_s": 4024.7187678031282, "sample_count": 560, "spec_accept_length": 4.541140409587448, "spec_verify_ct_sum": 598145}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 648.046446204, "output_tokens": 2601538, "output_toks_per_s": 4014.4313964512603, "sample_count": 560, "spec_accept_length": 4.538706990188629, "spec_verify_ct_sum": 605881}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 646.6196402810019, "output_tokens": 2564062, "output_toks_per_s": 3965.332693708057, "sample_count": 560, "spec_accept_length": 4.547014391265291, "spec_verify_ct_sum": 598630}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 7}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s7", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 650.3192075379993, "output_tokens": 2567933, "output_toks_per_s": 3948.726979357981, "sample_count": 560, "spec_accept_length": 4.536286333067362, "spec_verify_ct_sum": 600333}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 6016.504327284005, "output_tokens": 2633316, "output_toks_per_s": 437.6820586762118, "sample_count": 1319, "spec_accept_length": 7.0062492275330905, "spec_verify_ct_sum": 420009}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 5981.800727970003, "output_tokens": 2633316, "output_toks_per_s": 440.2212844849561, "sample_count": 1319, "spec_accept_length": 7.0062492275330905, "spec_verify_ct_sum": 420009}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 5977.275908796004, "output_tokens": 2633316, "output_toks_per_s": 440.5545335668512, "sample_count": 1319, "spec_accept_length": 7.0062492275330905, "spec_verify_ct_sum": 420009}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 6008.2401835519995, "output_tokens": 2633316, "output_toks_per_s": 438.28407646034134, "sample_count": 1319, "spec_accept_length": 7.0062492275330905, "spec_verify_ct_sum": 420009}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 5993.46458077, "output_tokens": 2633316, "output_toks_per_s": 439.36457194541214, "sample_count": 1319, "spec_accept_length": 7.0062492275330905, "spec_verify_ct_sum": 420009}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 635.195493772997, "output_tokens": 2588924, "output_toks_per_s": 4075.7908791544683, "sample_count": 1319, "spec_accept_length": 7.028758151669042, "spec_verify_ct_sum": 409949}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 631.6296095159996, "output_tokens": 2569433, "output_toks_per_s": 4067.9426063779465, "sample_count": 1319, "spec_accept_length": 7.047938079492356, "spec_verify_ct_sum": 407095}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 631.3477863080007, "output_tokens": 2587774, "output_toks_per_s": 4098.809017978507, "sample_count": 1319, "spec_accept_length": 7.027212467026291, "spec_verify_ct_sum": 408819}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 636.2576454850005, "output_tokens": 2596187, "output_toks_per_s": 4080.402048483053, "sample_count": 1319, "spec_accept_length": 7.018942724844907, "spec_verify_ct_sum": 411828}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 632.5290150629999, "output_tokens": 2577260, "output_toks_per_s": 4074.5324540460883, "sample_count": 1319, "spec_accept_length": 7.021066070014281, "spec_verify_ct_sum": 408670}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 3287.2208207260046, "output_tokens": 1645406, "output_toks_per_s": 500.5462333487536, "sample_count": 500, "spec_accept_length": 7.293583213431485, "spec_verify_ct_sum": 231130}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 3294.935278861998, "output_tokens": 1645406, "output_toks_per_s": 499.3743004773948, "sample_count": 500, "spec_accept_length": 7.293583213431485, "spec_verify_ct_sum": 231130}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 3296.732258230997, "output_tokens": 1645406, "output_toks_per_s": 499.1021020563293, "sample_count": 500, "spec_accept_length": 7.293583213431485, "spec_verify_ct_sum": 231130}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 3283.654683666, "output_tokens": 1645406, "output_toks_per_s": 501.08983998372344, "sample_count": 500, "spec_accept_length": 7.293583213431485, "spec_verify_ct_sum": 231130}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 3298.307403537001, "output_tokens": 1645406, "output_toks_per_s": 498.86375000569035, "sample_count": 500, "spec_accept_length": 7.293583213431485, "spec_verify_ct_sum": 231130}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1013.8140113550035, "output_tokens": 4915907, "output_toks_per_s": 4848.923910047062, "sample_count": 1500, "spec_accept_length": 7.319652469882943, "spec_verify_ct_sum": 687309}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1007.5195624040025, "output_tokens": 4890928, "output_toks_per_s": 4854.424849409326, "sample_count": 1500, "spec_accept_length": 7.316763966618117, "spec_verify_ct_sum": 683402}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1010.0378447040057, "output_tokens": 4915444, "output_toks_per_s": 4866.5938863315405, "sample_count": 1500, "spec_accept_length": 7.306245394453256, "spec_verify_ct_sum": 688465}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1005.9169547140045, "output_tokens": 4888917, "output_toks_per_s": 4860.1596554160715, "sample_count": 1500, "spec_accept_length": 7.3161087173814945, "spec_verify_ct_sum": 683448}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 1016.6760639940003, "output_tokens": 4925737, "output_toks_per_s": 4844.942429990236, "sample_count": 1500, "spec_accept_length": 7.300664273487231, "spec_verify_ct_sum": 690652}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 294.24955546600177, "output_tokens": 141275, "output_toks_per_s": 480.1196718080453, "sample_count": 164, "spec_accept_length": 7.541115689183139, "spec_verify_ct_sum": 19796}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 292.83024030699744, "output_tokens": 141275, "output_toks_per_s": 482.44675772519287, "sample_count": 164, "spec_accept_length": 7.541115689183139, "spec_verify_ct_sum": 19796}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 293.48970198799907, "output_tokens": 141275, "output_toks_per_s": 481.3627157718018, "sample_count": 164, "spec_accept_length": 7.541115689183139, "spec_verify_ct_sum": 19796}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 294.46221144599986, "output_tokens": 141275, "output_toks_per_s": 479.77293692881136, "sample_count": 164, "spec_accept_length": 7.541115689183139, "spec_verify_ct_sum": 19796}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 295.1616778049938, "output_tokens": 141275, "output_toks_per_s": 478.6359836771797, "sample_count": 164, "spec_accept_length": 7.541115689183139, "spec_verify_ct_sum": 19796}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 249.1991581789989, "output_tokens": 991423, "output_toks_per_s": 3978.436392982773, "sample_count": 1148, "spec_accept_length": 7.62654306537956, "spec_verify_ct_sum": 134707}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 256.86359636999987, "output_tokens": 997695, "output_toks_per_s": 3884.143234383698, "sample_count": 1148, "spec_accept_length": 7.641649859587302, "spec_verify_ct_sum": 135308}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 260.5610697079974, "output_tokens": 1010724, "output_toks_per_s": 3879.0292085179367, "sample_count": 1148, "spec_accept_length": 7.59629136775389, "spec_verify_ct_sum": 137805}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 260.6396870689996, "output_tokens": 1016589, "output_toks_per_s": 3900.3614968693414, "sample_count": 1148, "spec_accept_length": 7.597399809656241, "spec_verify_ct_sum": 139733}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 250.26644255500287, "output_tokens": 987010, "output_toks_per_s": 3943.836776211328, "sample_count": 1148, "spec_accept_length": 7.646486552545969, "spec_verify_ct_sum": 133829}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 658.6616551929983, "output_tokens": 292957, "output_toks_per_s": 444.7761573643739, "sample_count": 257, "spec_accept_length": 6.397939410117218, "spec_verify_ct_sum": 46043}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 656.5153794209982, "output_tokens": 292957, "output_toks_per_s": 446.2302166605268, "sample_count": 257, "spec_accept_length": 6.397939410117218, "spec_verify_ct_sum": 46043}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 659.0126916259997, "output_tokens": 292957, "output_toks_per_s": 444.53923835241983, "sample_count": 257, "spec_accept_length": 6.397939410117218, "spec_verify_ct_sum": 46043}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 657.5017437500001, "output_tokens": 292957, "output_toks_per_s": 445.5607955184225, "sample_count": 257, "spec_accept_length": 6.397939410117218, "spec_verify_ct_sum": 46043}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 655.6599357230043, "output_tokens": 292957, "output_toks_per_s": 446.81241606893775, "sample_count": 257, "spec_accept_length": 6.397939410117218, "spec_verify_ct_sum": 46043}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 304.0589184820001, "output_tokens": 1164410, "output_toks_per_s": 3829.553843752594, "sample_count": 1028, "spec_accept_length": 6.401538593212767, "spec_verify_ct_sum": 183614}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 313.4707432130017, "output_tokens": 1173981, "output_toks_per_s": 3745.1054856570336, "sample_count": 1028, "spec_accept_length": 6.402095217121619, "spec_verify_ct_sum": 185150}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 306.6029368159998, "output_tokens": 1168136, "output_toks_per_s": 3809.9308901957065, "sample_count": 1028, "spec_accept_length": 6.394534807528577, "spec_verify_ct_sum": 184283}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 315.4809098110054, "output_tokens": 1165023, "output_toks_per_s": 3692.8478515480647, "sample_count": 1028, "spec_accept_length": 6.389780037776488, "spec_verify_ct_sum": 184034}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 303.67762412799857, "output_tokens": 1166496, "output_toks_per_s": 3841.2313167608554, "sample_count": 1028, "spec_accept_length": 6.401807547637614, "spec_verify_ct_sum": 183678}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 1113.5440721150007, "output_tokens": 378138, "output_toks_per_s": 339.58063220774613, "sample_count": 80, "spec_accept_length": 5.256258044609895, "spec_verify_ct_sum": 79620}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 1124.0493021580041, "output_tokens": 378138, "output_toks_per_s": 336.4069523232054, "sample_count": 80, "spec_accept_length": 5.256258044609895, "spec_verify_ct_sum": 79620}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 1123.5837407490035, "output_tokens": 378138, "output_toks_per_s": 336.5463438870392, "sample_count": 80, "spec_accept_length": 5.256258044609895, "spec_verify_ct_sum": 79620}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 1116.8964136519999, "output_tokens": 378138, "output_toks_per_s": 338.5613879478526, "sample_count": 80, "spec_accept_length": 5.256258044609895, "spec_verify_ct_sum": 79620}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "mtp_s15", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 1116.1255011559988, "output_tokens": 378138, "output_toks_per_s": 338.7952336976022, "sample_count": 80, "spec_accept_length": 5.256258044609895, "spec_verify_ct_sum": 79620}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 829.5628774390007, "output_tokens": 2578944, "output_toks_per_s": 3108.7987060867904, "sample_count": 560, "spec_accept_length": 5.272645273632353, "spec_verify_ct_sum": 544297}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 831.7048222730009, "output_tokens": 2572759, "output_toks_per_s": 3093.355877111304, "sample_count": 560, "spec_accept_length": 5.218276967999593, "spec_verify_ct_sum": 543864}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 821.6955492929992, "output_tokens": 2553406, "output_toks_per_s": 3107.4842771109006, "sample_count": 560, "spec_accept_length": 5.21814503037147, "spec_verify_ct_sum": 540109}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 820.5228651349998, "output_tokens": 2531843, "output_toks_per_s": 3085.6458821332644, "sample_count": 560, "spec_accept_length": 5.241437740695722, "spec_verify_ct_sum": 532362}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"eagle_topk": 1, "kind": "mtp", "num_steps": 15}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "mtp_s15", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 822.6793424509997, "output_tokens": 2555321, "output_toks_per_s": 3106.0959819253026, "sample_count": 560, "spec_accept_length": 5.267506782717704, "spec_verify_ct_sum": 537731}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 5002.228082689002, "output_tokens": 2640735, "output_toks_per_s": 527.911753792011, "sample_count": 1319, "spec_accept_length": 3.5482318363047147, "spec_verify_ct_sum": 764746}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 4946.024679867995, "output_tokens": 2640735, "output_toks_per_s": 533.9105991016768, "sample_count": 1319, "spec_accept_length": 3.5482318363047147, "spec_verify_ct_sum": 764746}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 5011.7886997099995, "output_tokens": 2640735, "output_toks_per_s": 526.9046957531954, "sample_count": 1319, "spec_accept_length": 3.5482318363047147, "spec_verify_ct_sum": 764746}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 4981.201094913995, "output_tokens": 2640735, "output_toks_per_s": 530.1402111021569, "sample_count": 1319, "spec_accept_length": 3.5482318363047147, "spec_verify_ct_sum": 764746}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 4973.129655956, "output_tokens": 2640735, "output_toks_per_s": 531.0006339443333, "sample_count": 1319, "spec_accept_length": 3.5482318363047147, "spec_verify_ct_sum": 764746}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 502.9623741889991, "output_tokens": 2547414, "output_toks_per_s": 5064.820214648409, "sample_count": 1319, "spec_accept_length": 3.5560856475508396, "spec_verify_ct_sum": 736774}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 511.5801702250028, "output_tokens": 2588735, "output_toks_per_s": 5060.272369160487, "sample_count": 1319, "spec_accept_length": 3.5522818615842224, "spec_verify_ct_sum": 749174}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 517.342013042995, "output_tokens": 2631400, "output_toks_per_s": 5086.383733890391, "sample_count": 1319, "spec_accept_length": 3.551258572272507, "spec_verify_ct_sum": 761867}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 509.46113169899763, "output_tokens": 2599111, "output_toks_per_s": 5101.686543450815, "sample_count": 1319, "spec_accept_length": 3.552649665187453, "spec_verify_ct_sum": 751559}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 515.2678955259998, "output_tokens": 2615881, "output_toks_per_s": 5076.739736190309, "sample_count": 1319, "spec_accept_length": 3.548763303349016, "spec_verify_ct_sum": 757640}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2982.394056227, "output_tokens": 1649943, "output_toks_per_s": 553.2276985849845, "sample_count": 500, "spec_accept_length": 3.621766695901931, "spec_verify_ct_sum": 457367}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2990.262761877002, "output_tokens": 1649943, "output_toks_per_s": 551.771911497277, "sample_count": 500, "spec_accept_length": 3.621766695901931, "spec_verify_ct_sum": 457367}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2994.3438534899997, "output_tokens": 1649943, "output_toks_per_s": 551.0198830628423, "sample_count": 500, "spec_accept_length": 3.621766695901931, "spec_verify_ct_sum": 457367}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2999.8259309940004, "output_tokens": 1649943, "output_toks_per_s": 550.0129134003741, "sample_count": 500, "spec_accept_length": 3.621766695901931, "spec_verify_ct_sum": 457367}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2996.7647016870033, "output_tokens": 1649943, "output_toks_per_s": 550.5747578617629, "sample_count": 500, "spec_accept_length": 3.621766695901931, "spec_verify_ct_sum": 457367}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 928.5700291419998, "output_tokens": 4928321, "output_toks_per_s": 5307.430614095715, "sample_count": 1500, "spec_accept_length": 3.6152087637761547, "spec_verify_ct_sum": 1368726}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 934.2962486410033, "output_tokens": 4966032, "output_toks_per_s": 5315.264839416221, "sample_count": 1500, "spec_accept_length": 3.6151281573264313, "spec_verify_ct_sum": 1379235}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 927.3149531090003, "output_tokens": 4942979, "output_toks_per_s": 5330.420892521704, "sample_count": 1500, "spec_accept_length": 3.6145566813781285, "spec_verify_ct_sum": 1372606}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 930.0203836029978, "output_tokens": 4956799, "output_toks_per_s": 5329.774580635356, "sample_count": 1500, "spec_accept_length": 3.615236428394392, "spec_verify_ct_sum": 1375937}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 926.3971295679949, "output_tokens": 4919043, "output_toks_per_s": 5309.864250436407, "sample_count": 1500, "spec_accept_length": 3.6150400575508264, "spec_verify_ct_sum": 1366105}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 256.83608952100076, "output_tokens": 140150, "output_toks_per_s": 545.6787644656158, "sample_count": 164, "spec_accept_length": 3.6969498177945153, "spec_verify_ct_sum": 38313}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 256.4049518669999, "output_tokens": 140150, "output_toks_per_s": 546.596307830659, "sample_count": 164, "spec_accept_length": 3.6969498177945153, "spec_verify_ct_sum": 38313}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 257.9977331819973, "output_tokens": 140150, "output_toks_per_s": 543.2218270737095, "sample_count": 164, "spec_accept_length": 3.6969498177945153, "spec_verify_ct_sum": 38313}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 258.5389400439999, "output_tokens": 140150, "output_toks_per_s": 542.0846854874098, "sample_count": 164, "spec_accept_length": 3.6969498177945153, "spec_verify_ct_sum": 38313}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 259.2384923940008, "output_tokens": 140150, "output_toks_per_s": 540.6218756549262, "sample_count": 164, "spec_accept_length": 3.6969498177945153, "spec_verify_ct_sum": 38313}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 207.25496872700023, "output_tokens": 1053984, "output_toks_per_s": 5085.446233080789, "sample_count": 1148, "spec_accept_length": 3.674919766002785, "spec_verify_ct_sum": 288890}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 210.03027938400191, "output_tokens": 1057556, "output_toks_per_s": 5035.254931344696, "sample_count": 1148, "spec_accept_length": 3.671930873992907, "spec_verify_ct_sum": 289826}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 209.66416981699876, "output_tokens": 1066203, "output_toks_per_s": 5085.2894938158215, "sample_count": 1148, "spec_accept_length": 3.677224857312975, "spec_verify_ct_sum": 292072}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 206.75046306899458, "output_tokens": 1057711, "output_toks_per_s": 5115.882132979948, "sample_count": 1148, "spec_accept_length": 3.676739898163083, "spec_verify_ct_sum": 289417}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 209.42504113999894, "output_tokens": 1064964, "output_toks_per_s": 5085.179853387639, "sample_count": 1148, "spec_accept_length": 3.674691793763517, "spec_verify_ct_sum": 291281}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 535.2551270509975, "output_tokens": 294380, "output_toks_per_s": 549.9807197025734, "sample_count": 257, "spec_accept_length": 3.5875877331737334, "spec_verify_ct_sum": 82300}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 536.8345133100011, "output_tokens": 294380, "output_toks_per_s": 548.3626568361244, "sample_count": 257, "spec_accept_length": 3.5875877331737334, "spec_verify_ct_sum": 82300}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 535.7174886849989, "output_tokens": 294380, "output_toks_per_s": 549.5060479033474, "sample_count": 257, "spec_accept_length": 3.5875877331737334, "spec_verify_ct_sum": 82300}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 535.0112383029991, "output_tokens": 294380, "output_toks_per_s": 550.2314323970899, "sample_count": 257, "spec_accept_length": 3.5875877331737334, "spec_verify_ct_sum": 82300}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 532.1017651000002, "output_tokens": 294380, "output_toks_per_s": 553.2400366021636, "sample_count": 257, "spec_accept_length": 3.5875877331737334, "spec_verify_ct_sum": 82300}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 225.05048260199874, "output_tokens": 1181544, "output_toks_per_s": 5250.128710408312, "sample_count": 1028, "spec_accept_length": 3.5916357455985546, "spec_verify_ct_sum": 330411}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 223.2900623559981, "output_tokens": 1175826, "output_toks_per_s": 5265.912811316005, "sample_count": 1028, "spec_accept_length": 3.5908322475436396, "spec_verify_ct_sum": 328698}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 224.64051230900077, "output_tokens": 1178965, "output_toks_per_s": 5248.229662057986, "sample_count": 1028, "spec_accept_length": 3.5918704531385357, "spec_verify_ct_sum": 329553}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 226.26513355300267, "output_tokens": 1187651, "output_toks_per_s": 5248.935093757132, "sample_count": 1028, "spec_accept_length": 3.593856875262959, "spec_verify_ct_sum": 331735}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 221.39313421500083, "output_tokens": 1176803, "output_toks_per_s": 5315.4448721845865, "sample_count": 1028, "spec_accept_length": 3.5919613131629045, "spec_verify_ct_sum": 329067}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 774.4455520980009, "output_tokens": 366934, "output_toks_per_s": 473.80219178218863, "sample_count": 80, "spec_accept_length": 3.187624530804912, "spec_verify_ct_sum": 118198}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 781.5975648149997, "output_tokens": 366934, "output_toks_per_s": 469.46666228016136, "sample_count": 80, "spec_accept_length": 3.187624530804912, "spec_verify_ct_sum": 118198}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 776.4341739529991, "output_tokens": 366934, "output_toks_per_s": 472.588678228648, "sample_count": 80, "spec_accept_length": 3.187624530804912, "spec_verify_ct_sum": 118198}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 769.4569382059999, "output_tokens": 366934, "output_toks_per_s": 476.8739896679754, "sample_count": 80, "spec_accept_length": 3.187624530804912, "spec_verify_ct_sum": 118198}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b4", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 771.9417481069995, "output_tokens": 366934, "output_toks_per_s": 475.33897590047036, "sample_count": 80, "spec_accept_length": 3.187624530804912, "spec_verify_ct_sum": 118198}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 592.3435003699997, "output_tokens": 2624409, "output_toks_per_s": 4430.5525398028285, "sample_count": 560, "spec_accept_length": 3.197113079364059, "spec_verify_ct_sum": 841835}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 586.3370903140003, "output_tokens": 2583610, "output_toks_per_s": 4406.356075165572, "sample_count": 560, "spec_accept_length": 3.1933225730293207, "spec_verify_ct_sum": 830414}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 588.877918172002, "output_tokens": 2627689, "output_toks_per_s": 4462.196524802435, "sample_count": 560, "spec_accept_length": 3.1926628813303988, "spec_verify_ct_sum": 842642}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 595.1836989389994, "output_tokens": 2617765, "output_toks_per_s": 4398.247137256183, "sample_count": 560, "spec_accept_length": 3.1903410353961745, "spec_verify_ct_sum": 840120}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 4, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b4", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 594.1730976789986, "output_tokens": 2631788, "output_toks_per_s": 4429.328776884175, "sample_count": 560, "spec_accept_length": 3.1957613791801864, "spec_verify_ct_sum": 844213}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 3763.2739457120006, "output_tokens": 2594203, "output_toks_per_s": 689.3473707796163, "sample_count": 1319, "spec_accept_length": 5.737918415362598, "spec_verify_ct_sum": 486689}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 3768.5925314189953, "output_tokens": 2594203, "output_toks_per_s": 688.3745001275581, "sample_count": 1319, "spec_accept_length": 5.737918415362598, "spec_verify_ct_sum": 486689}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 3766.252309411997, "output_tokens": 2594203, "output_toks_per_s": 688.8022327970422, "sample_count": 1319, "spec_accept_length": 5.737918415362598, "spec_verify_ct_sum": 486689}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 3763.9062722200033, "output_tokens": 2594203, "output_toks_per_s": 689.2315622062246, "sample_count": 1319, "spec_accept_length": 5.737918415362598, "spec_verify_ct_sum": 486689}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 3764.512501355999, "output_tokens": 2594203, "output_toks_per_s": 689.1205698123072, "sample_count": 1319, "spec_accept_length": 5.737918415362598, "spec_verify_ct_sum": 486689}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 419.8601949590011, "output_tokens": 2587204, "output_toks_per_s": 6162.060683682191, "sample_count": 1319, "spec_accept_length": 5.743924302158809, "spec_verify_ct_sum": 485146}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 421.6339740619951, "output_tokens": 2603157, "output_toks_per_s": 6173.9735413665785, "sample_count": 1319, "spec_accept_length": 5.74038802649249, "spec_verify_ct_sum": 488665}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 429.1704336020048, "output_tokens": 2639077, "output_toks_per_s": 6149.251657087293, "sample_count": 1319, "spec_accept_length": 5.730022543069771, "spec_verify_ct_sum": 497266}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 426.7668007370012, "output_tokens": 2625148, "output_toks_per_s": 6151.246993595855, "sample_count": 1319, "spec_accept_length": 5.735796118302449, "spec_verify_ct_sum": 493328}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 428.14422113099863, "output_tokens": 2636503, "output_toks_per_s": 6157.978713423562, "sample_count": 1319, "spec_accept_length": 5.730201551383567, "spec_verify_ct_sum": 495268}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2154.562639076001, "output_tokens": 1644162, "output_toks_per_s": 763.107078058826, "sample_count": 500, "spec_accept_length": 6.002647542213242, "spec_verify_ct_sum": 277367}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2154.803338828, "output_tokens": 1644162, "output_toks_per_s": 763.0218360874926, "sample_count": 500, "spec_accept_length": 6.002647542213242, "spec_verify_ct_sum": 277367}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2154.7363428829995, "output_tokens": 1644162, "output_toks_per_s": 763.0455602749709, "sample_count": 500, "spec_accept_length": 6.002647542213242, "spec_verify_ct_sum": 277367}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2158.747667600999, "output_tokens": 1644162, "output_toks_per_s": 761.6276902928379, "sample_count": 500, "spec_accept_length": 6.002647542213242, "spec_verify_ct_sum": 277367}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 2157.732461921005, "output_tokens": 1644162, "output_toks_per_s": 761.9860334938007, "sample_count": 500, "spec_accept_length": 6.002647542213242, "spec_verify_ct_sum": 277367}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 712.4250778429996, "output_tokens": 4902255, "output_toks_per_s": 6881.08146732074, "sample_count": 1500, "spec_accept_length": 5.992415763568392, "spec_verify_ct_sum": 828845}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 708.2706855159995, "output_tokens": 4898608, "output_toks_per_s": 6916.293586866716, "sample_count": 1500, "spec_accept_length": 6.00277407583608, "spec_verify_ct_sum": 827001}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 714.4235714360038, "output_tokens": 4931578, "output_toks_per_s": 6902.876944677851, "sample_count": 1500, "spec_accept_length": 5.993282524870867, "spec_verify_ct_sum": 834202}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 709.2289774600067, "output_tokens": 4919667, "output_toks_per_s": 6936.641277150043, "sample_count": 1500, "spec_accept_length": 5.998130623458094, "spec_verify_ct_sum": 831304}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 712.9126497710022, "output_tokens": 4930393, "output_toks_per_s": 6915.8444608658765, "sample_count": 1500, "spec_accept_length": 6.003305276326551, "spec_verify_ct_sum": 831471}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 186.99408931400103, "output_tokens": 140489, "output_toks_per_s": 751.3018219741184, "sample_count": 164, "spec_accept_length": 6.292280290267946, "spec_verify_ct_sum": 23074}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 187.2393588940031, "output_tokens": 140489, "output_toks_per_s": 750.3176726829713, "sample_count": 164, "spec_accept_length": 6.292280290267946, "spec_verify_ct_sum": 23074}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 186.4297813669982, "output_tokens": 140489, "output_toks_per_s": 753.5759521352384, "sample_count": 164, "spec_accept_length": 6.292280290267946, "spec_verify_ct_sum": 23074}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 186.58097255999746, "output_tokens": 140489, "output_toks_per_s": 752.9653108374918, "sample_count": 164, "spec_accept_length": 6.292280290267946, "spec_verify_ct_sum": 23074}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 186.18967487600094, "output_tokens": 140489, "output_toks_per_s": 754.5477486523525, "sample_count": 164, "spec_accept_length": 6.292280290267946, "spec_verify_ct_sum": 23074}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 160.51378221199957, "output_tokens": 1050880, "output_toks_per_s": 6546.976748775652, "sample_count": 1148, "spec_accept_length": 6.244970028909135, "spec_verify_ct_sum": 172597}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 159.203115848999, "output_tokens": 1065152, "output_toks_per_s": 6690.522319992, "sample_count": 1148, "spec_accept_length": 6.27554013296705, "spec_verify_ct_sum": 173584}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 159.93619859900355, "output_tokens": 1061916, "output_toks_per_s": 6639.62260765285, "sample_count": 1148, "spec_accept_length": 6.260427574286677, "spec_verify_ct_sum": 173223}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 159.21482427800038, "output_tokens": 1076358, "output_toks_per_s": 6760.413201980505, "sample_count": 1148, "spec_accept_length": 6.261577665937855, "spec_verify_ct_sum": 175310}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 157.78645773999597, "output_tokens": 1055978, "output_toks_per_s": 6692.450132444598, "sample_count": 1148, "spec_accept_length": 6.2625952454219975, "spec_verify_ct_sum": 171986}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 401.74988761400164, "output_tokens": 301551, "output_toks_per_s": 750.5938627411092, "sample_count": 257, "spec_accept_length": 5.8598800768392305, "spec_verify_ct_sum": 52097}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 401.68704967699887, "output_tokens": 301551, "output_toks_per_s": 750.7112819357274, "sample_count": 257, "spec_accept_length": 5.8598800768392305, "spec_verify_ct_sum": 52097}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 401.3851704379995, "output_tokens": 301551, "output_toks_per_s": 751.2758871259283, "sample_count": 257, "spec_accept_length": 5.8598800768392305, "spec_verify_ct_sum": 52097}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 401.9211115299986, "output_tokens": 301551, "output_toks_per_s": 750.2740994422554, "sample_count": 257, "spec_accept_length": 5.8598800768392305, "spec_verify_ct_sum": 52097}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 400.3679461049978, "output_tokens": 301551, "output_toks_per_s": 753.1846715843661, "sample_count": 257, "spec_accept_length": 5.8598800768392305, "spec_verify_ct_sum": 52097}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 175.19603549699968, "output_tokens": 1163649, "output_toks_per_s": 6641.982489494907, "sample_count": 1028, "spec_accept_length": 5.834161955477256, "spec_verify_ct_sum": 202246}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 179.07147010100016, "output_tokens": 1176960, "output_toks_per_s": 6572.571271884735, "sample_count": 1028, "spec_accept_length": 5.8431018318070045, "spec_verify_ct_sum": 203925}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 177.21554937399924, "output_tokens": 1181154, "output_toks_per_s": 6665.069764884282, "sample_count": 1028, "spec_accept_length": 5.846507789958617, "spec_verify_ct_sum": 204389}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 173.35317415199825, "output_tokens": 1170559, "output_toks_per_s": 6752.452072055163, "sample_count": 1028, "spec_accept_length": 5.843352932560099, "spec_verify_ct_sum": 203090}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 171.90101226499974, "output_tokens": 1164580, "output_toks_per_s": 6774.712869082487, "sample_count": 1028, "spec_accept_length": 5.83986898939138, "spec_verify_ct_sum": 202188}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 672.9240064219994, "output_tokens": 367574, "output_toks_per_s": 546.2340420197309, "sample_count": 80, "spec_accept_length": 4.583530246016292, "spec_verify_ct_sum": 86934}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 671.5057589659991, "output_tokens": 367574, "output_toks_per_s": 547.3877105179253, "sample_count": 80, "spec_accept_length": 4.583530246016292, "spec_verify_ct_sum": 86934}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 674.9773897959967, "output_tokens": 367574, "output_toks_per_s": 544.5723153942898, "sample_count": 80, "spec_accept_length": 4.583530246016292, "spec_verify_ct_sum": 86934}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 672.9695194450032, "output_tokens": 367574, "output_toks_per_s": 546.1971001348436, "sample_count": 80, "spec_accept_length": 4.583530246016292, "spec_verify_ct_sum": 86934}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b8", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 674.191114358, "output_tokens": 367574, "output_toks_per_s": 545.2074228982136, "sample_count": 80, "spec_accept_length": 4.583530246016292, "spec_verify_ct_sum": 86934}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 545.5921214689879, "output_tokens": 2616522, "output_toks_per_s": 4795.747403674205, "sample_count": 560, "spec_accept_length": 4.5587327721121, "spec_verify_ct_sum": 618879}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 545.8678008850002, "output_tokens": 2598913, "output_toks_per_s": 4761.066682787398, "sample_count": 560, "spec_accept_length": 4.564403111869214, "spec_verify_ct_sum": 614030}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 545.6740177620013, "output_tokens": 2601898, "output_toks_per_s": 4768.22776109313, "sample_count": 560, "spec_accept_length": 4.559024478308863, "spec_verify_ct_sum": 615602}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 542.4903077380004, "output_tokens": 2581859, "output_toks_per_s": 4759.272125552752, "sample_count": 560, "spec_accept_length": 4.564412356290379, "spec_verify_ct_sum": 611224}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 8, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b8", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 545.9896704429993, "output_tokens": 2583396, "output_toks_per_s": 4731.584020452093, "sample_count": 560, "spec_accept_length": 4.566122861441735, "spec_verify_ct_sum": 610517}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 3644.640504670999, "output_tokens": 2594195, "output_toks_per_s": 711.7835069536378, "sample_count": 1319, "spec_accept_length": 7.695530331008224, "spec_verify_ct_sum": 385387}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 3649.7434926589995, "output_tokens": 2594195, "output_toks_per_s": 710.7883075119929, "sample_count": 1319, "spec_accept_length": 7.695530331008224, "spec_verify_ct_sum": 385387}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 3649.961563705001, "output_tokens": 2594195, "output_toks_per_s": 710.7458406676168, "sample_count": 1319, "spec_accept_length": 7.695530331008224, "spec_verify_ct_sum": 385387}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 3642.726196923002, "output_tokens": 2594195, "output_toks_per_s": 712.1575599591612, "sample_count": 1319, "spec_accept_length": 7.695530331008224, "spec_verify_ct_sum": 385387}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 3648.2162889730025, "output_tokens": 2594195, "output_toks_per_s": 711.0858552551125, "sample_count": 1319, "spec_accept_length": 7.695530331008224, "spec_verify_ct_sum": 385387}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 473.84875266900053, "output_tokens": 2675504, "output_toks_per_s": 5646.3248767248115, "sample_count": 1319, "spec_accept_length": 7.640898005338085, "spec_verify_ct_sum": 401748}, "run_index": 0, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 465.912559862998, "output_tokens": 2635268, "output_toks_per_s": 5656.142862460936, "sample_count": 1319, "spec_accept_length": 7.676446661010194, "spec_verify_ct_sum": 394124}, "run_index": 1, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 453.6312395549976, "output_tokens": 2570464, "output_toks_per_s": 5666.417512430514, "sample_count": 1319, "spec_accept_length": 7.70244392711985, "spec_verify_ct_sum": 383576}, "run_index": 2, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 469.28337179200025, "output_tokens": 2639095, "output_toks_per_s": 5623.670384745109, "sample_count": 1319, "spec_accept_length": 7.669061206479703, "spec_verify_ct_sum": 395979}, "run_index": 3, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "gsm8k"}, "kind": "result", "metrics": {"generation_turn_count": 1319, "latency_s": 465.45831610999994, "output_tokens": 2630399, "output_toks_per_s": 5651.202071075186, "sample_count": 1319, "spec_accept_length": 7.65932118955639, "spec_verify_ct_sum": 393584}, "run_index": 4, "source_generation_turn_count": 1319, "source_sample_count": 1319, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 1990.2699761860003, "output_tokens": 1653563, "output_toks_per_s": 830.8234660549723, "sample_count": 500, "spec_accept_length": 8.212200109481499, "spec_verify_ct_sum": 208019}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 1984.9407135540023, "output_tokens": 1653563, "output_toks_per_s": 833.0541001596586, "sample_count": 500, "spec_accept_length": 8.212200109481499, "spec_verify_ct_sum": 208019}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 1987.130993674, "output_tokens": 1653563, "output_toks_per_s": 832.1358809580705, "sample_count": 500, "spec_accept_length": 8.212200109481499, "spec_verify_ct_sum": 208019}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 1987.3889001949938, "output_tokens": 1653563, "output_toks_per_s": 832.0278934021217, "sample_count": 500, "spec_accept_length": 8.212200109481499, "spec_verify_ct_sum": 208019}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 500, "latency_s": 1988.2367855740013, "output_tokens": 1653563, "output_toks_per_s": 831.6730743529718, "sample_count": 500, "spec_accept_length": 8.212200109481499, "spec_verify_ct_sum": 208019}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 718.7151603599996, "output_tokens": 4926479, "output_toks_per_s": 6854.563910315124, "sample_count": 1500, "spec_accept_length": 8.253933871803715, "spec_verify_ct_sum": 616140}, "run_index": 0, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 719.75256480799, "output_tokens": 4936121, "output_toks_per_s": 6858.080458965534, "sample_count": 1500, "spec_accept_length": 8.238531726727771, "spec_verify_ct_sum": 618904}, "run_index": 1, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 722.5212979349999, "output_tokens": 4959104, "output_toks_per_s": 6863.609438466872, "sample_count": 1500, "spec_accept_length": 8.23539063796915, "spec_verify_ct_sum": 621726}, "run_index": 2, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 720.243996348001, "output_tokens": 4949690, "output_toks_per_s": 6872.240553336668, "sample_count": 1500, "spec_accept_length": 8.25476688606018, "spec_verify_ct_sum": 619111}, "run_index": 3, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "math500"}, "kind": "result", "metrics": {"generation_turn_count": 1500, "latency_s": 718.8008418719983, "output_tokens": 4916869, "output_toks_per_s": 6840.377352918544, "sample_count": 1500, "spec_accept_length": 8.270536864937004, "spec_verify_ct_sum": 614755}, "run_index": 4, "source_generation_turn_count": 500, "source_sample_count": 500, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 168.27081695599918, "output_tokens": 147016, "output_toks_per_s": 873.6868499214747, "sample_count": 164, "spec_accept_length": 9.344167953537664, "spec_verify_ct_sum": 16775}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 168.51539606799997, "output_tokens": 147016, "output_toks_per_s": 872.4188022599167, "sample_count": 164, "spec_accept_length": 9.344167953537664, "spec_verify_ct_sum": 16775}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 168.0079108389982, "output_tokens": 147016, "output_toks_per_s": 875.054033264453, "sample_count": 164, "spec_accept_length": 9.344167953537664, "spec_verify_ct_sum": 16775}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 168.02363117499954, "output_tokens": 147016, "output_toks_per_s": 874.9721629743871, "sample_count": 164, "spec_accept_length": 9.344167953537664, "spec_verify_ct_sum": 16775}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 164, "latency_s": 167.6331245750007, "output_tokens": 147016, "output_toks_per_s": 877.0104379592567, "sample_count": 164, "spec_accept_length": 9.344167953537664, "spec_verify_ct_sum": 16775}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 154.9234565900042, "output_tokens": 1048513, "output_toks_per_s": 6767.942202418242, "sample_count": 1148, "spec_accept_length": 9.132273628006024, "spec_verify_ct_sum": 120486}, "run_index": 0, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 150.97487579500012, "output_tokens": 1025881, "output_toks_per_s": 6795.044503914567, "sample_count": 1148, "spec_accept_length": 9.122333045837035, "spec_verify_ct_sum": 117988}, "run_index": 1, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 149.2242534370016, "output_tokens": 1021249, "output_toks_per_s": 6843.7199481861935, "sample_count": 1148, "spec_accept_length": 9.072872989855977, "spec_verify_ct_sum": 117486}, "run_index": 2, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 153.40302577599505, "output_tokens": 1039236, "output_toks_per_s": 6774.546947447648, "sample_count": 1148, "spec_accept_length": 9.10303261729889, "spec_verify_ct_sum": 120261}, "run_index": 3, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "humaneval"}, "kind": "result", "metrics": {"generation_turn_count": 1148, "latency_s": 154.28460226600146, "output_tokens": 1039488, "output_toks_per_s": 6737.470782779885, "sample_count": 1148, "spec_accept_length": 9.104096544692375, "spec_verify_ct_sum": 118806}, "run_index": 4, "source_generation_turn_count": 164, "source_sample_count": 164, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 360.54568844499954, "output_tokens": 291768, "output_toks_per_s": 809.2400196445799, "sample_count": 257, "spec_accept_length": 7.80149519208513, "spec_verify_ct_sum": 38256}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 361.45280303000254, "output_tokens": 291768, "output_toks_per_s": 807.2091226133932, "sample_count": 257, "spec_accept_length": 7.80149519208513, "spec_verify_ct_sum": 38256}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 361.7411674460018, "output_tokens": 291768, "output_toks_per_s": 806.5656504068012, "sample_count": 257, "spec_accept_length": 7.80149519208513, "spec_verify_ct_sum": 38256}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 361.25918036599614, "output_tokens": 291768, "output_toks_per_s": 807.6417593164172, "sample_count": 257, "spec_accept_length": 7.80149519208513, "spec_verify_ct_sum": 38256}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 257, "latency_s": 360.91860221899697, "output_tokens": 291768, "output_toks_per_s": 808.4038844386358, "sample_count": 257, "spec_accept_length": 7.80149519208513, "spec_verify_ct_sum": 38256}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 1, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 187.0395874149981, "output_tokens": 1177392, "output_toks_per_s": 6294.8812936998, "sample_count": 1028, "spec_accept_length": 7.812158535987461, "spec_verify_ct_sum": 154914}, "run_index": 0, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 183.61466368399851, "output_tokens": 1176084, "output_toks_per_s": 6405.174708835046, "sample_count": 1028, "spec_accept_length": 7.821201910853448, "spec_verify_ct_sum": 154651}, "run_index": 1, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 184.0353872150008, "output_tokens": 1176084, "output_toks_per_s": 6390.531830848545, "sample_count": 1028, "spec_accept_length": 7.821201910853448, "spec_verify_ct_sum": 154651}, "run_index": 2, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 186.56107529299334, "output_tokens": 1182289, "output_toks_per_s": 6337.275866057378, "sample_count": 1028, "spec_accept_length": 7.840147201158057, "spec_verify_ct_sum": 154741}, "run_index": 3, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "mbpp"}, "kind": "result", "metrics": {"generation_turn_count": 1028, "latency_s": 188.56117148699923, "output_tokens": 1186559, "output_toks_per_s": 6292.7006161595145, "sample_count": 1028, "spec_accept_length": 7.8095345586023095, "spec_verify_ct_sum": 156378}, "run_index": 4, "source_generation_turn_count": 257, "source_sample_count": 257, "warmdown_generation_turn_count": 32, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 715.7991668760005, "output_tokens": 368704, "output_toks_per_s": 515.094201085975, "sample_count": 80, "spec_accept_length": 5.523386645644664, "spec_verify_ct_sum": 76747}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 714.8396381099956, "output_tokens": 368704, "output_toks_per_s": 515.7856116860519, "sample_count": 80, "spec_accept_length": 5.523386645644664, "spec_verify_ct_sum": 76747}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 715.2812581019971, "output_tokens": 368704, "output_toks_per_s": 515.4671617964075, "sample_count": 80, "spec_accept_length": 5.523386645644664, "spec_verify_ct_sum": 76747}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 715.3516129399977, "output_tokens": 368704, "output_toks_per_s": 515.4164655960958, "sample_count": 80, "spec_accept_length": 5.523386645644664, "spec_verify_ct_sum": 76747}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 1, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 1, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 1, "mode": "dflash_b16", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 160, "latency_s": 716.0124552750058, "output_tokens": 368704, "output_toks_per_s": 514.9407629485835, "sample_count": 80, "spec_accept_length": 5.523386645644664, "spec_verify_ct_sum": 76747}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 2, "warmup_generation_turn_count": 8} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 657.208404818004, "output_tokens": 2615462, "output_toks_per_s": 3979.6539131666777, "sample_count": 560, "spec_accept_length": 5.504454883340155, "spec_verify_ct_sum": 542968}, "run_index": 0, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 648.3174016070006, "output_tokens": 2624722, "output_toks_per_s": 4048.5138814630545, "sample_count": 560, "spec_accept_length": 5.525158595260345, "spec_verify_ct_sum": 544458}, "run_index": 1, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 656.382369687999, "output_tokens": 2621119, "output_toks_per_s": 3993.280625812524, "sample_count": 560, "spec_accept_length": 5.545766435460611, "spec_verify_ct_sum": 543165}, "run_index": 2, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 643.7072747460006, "output_tokens": 2594724, "output_toks_per_s": 4030.906751867684, "sample_count": 560, "spec_accept_length": 5.550628711462469, "spec_verify_ct_sum": 536773}, "run_index": 3, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64} +{"deployment": {"mode_config": {"block_size": 16, "draft_attention_backend": "fa4", "draft_model": "modal-labs/Qwen3.5-397B-A17B-DFlash", "kind": "dflash"}, "shared_config": {"attention_backend": "trtllm_mha", "cuda_graph_max_bs": 32, "dtype": "bfloat16", "enable_flashinfer_allreduce_fusion": true, "enable_piecewise_cuda_graph": true, "linear_attn_backend": "flashinfer", "mamba_scheduler_strategy": "extra_buffer", "mamba_ssm_dtype": "bfloat16", "max_running_requests": 32, "mem_fraction_static": 0.8, "page_size": null, "tp_size": 8}}, "key": {"backend": "trtllm_mha", "concurrency": 32, "mode": "dflash_b16", "tp": 8, "workload": "mt-bench"}, "kind": "result", "metrics": {"generation_turn_count": 1120, "latency_s": 646.1519307499984, "output_tokens": 2602453, "output_toks_per_s": 4027.6177724630384, "sample_count": 560, "spec_accept_length": 5.546500999126334, "spec_verify_ct_sum": 538723}, "run_index": 4, "source_generation_turn_count": 160, "source_sample_count": 80, "warmdown_generation_turn_count": 64, "warmup_generation_turn_count": 64}